{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:07:42.039419Z","iopub.execute_input":"2025-05-28T20:07:42.039676Z","iopub.status.idle":"2025-05-28T20:07:42.403946Z","shell.execute_reply.started":"2025-05-28T20:07:42.03963Z","shell.execute_reply":"2025-05-28T20:07:42.403334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport gc  # For garbage collection to manage memory\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\n# Create time-based sample weights\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\n# Function to run the full model pipeline with given features\ndef run_model_pipeline(train_data, test_data, features_to_use, verbose=False):\n    \"\"\"\n    Run the complete 3-model ensemble pipeline with given features\n    Returns the weighted ensemble score\n    \"\"\"\n    # Select features\n    train = train_data[features_to_use + [\"label\"]].copy()\n    test = test_data[features_to_use].copy()\n    \n    # Define cross-validation\n    FOLDS = 5\n    kf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n    \n    # XGBoost parameters (same for all models)\n    xgb_params = {\n        \"tree_method\": \"gpu_hist\",\n        \"colsample_bylevel\": 0.4778015829774066,\n        \"colsample_bynode\": 0.362764358742407,\n        \"colsample_bytree\": 0.7107423488010493,\n        \"gamma\": 1.7094857725240398,\n        \"learning_rate\": 0.02213323588455387,\n        \"max_depth\": 20,\n        \"max_leaves\": 12,\n        \"min_child_weight\": 16,\n        \"n_estimators\": 1667,\n        \"n_jobs\": -1,\n        \"random_state\": 42,\n        \"reg_alpha\": 39.352415706891264,\n        \"reg_lambda\": 75.44843704068275,\n        \"subsample\": 0.06566669853471274,\n        \"verbosity\": 0\n    }\n    \n    # Initialize predictions for all three models\n    oof_preds_model1 = np.zeros(len(train))\n    test_preds_model1 = np.zeros(len(test))\n    oof_preds_model2 = np.zeros(len(train))\n    test_preds_model2 = np.zeros(len(test))\n    oof_preds_model3 = np.zeros(len(train))\n    test_preds_model3 = np.zeros(len(test))\n    \n    # Generate sample weights for Model 1 (full data)\n    sample_weights_full = create_time_weights(len(train), decay_factor=0.95)\n    \n    # Calculate the cutoff for 75% most recent data\n    cutoff_idx_75 = int(len(train) * 0.25)\n    \n    # Calculate the cutoff for 50% most recent data\n    cutoff_idx_50 = int(len(train) * 0.50)\n    \n    # Cross-validation loop\n    for i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n        if verbose:\n            print(f\"\\n### Fold {i + 1}\")\n        \n        # ========== MODEL 1: FULL DATA WITH TIME WEIGHTS ==========\n        X_train_m1 = train.iloc[train_idx][features_to_use]\n        y_train_m1 = train.iloc[train_idx][\"label\"]\n        X_valid = train.iloc[valid_idx][features_to_use]\n        y_valid = train.iloc[valid_idx][\"label\"]\n        X_test = test[features_to_use]\n        \n        # Extract sample weights for this fold's training data\n        train_weights_m1 = sample_weights_full[train_idx]\n        \n        model1 = XGBRegressor(**xgb_params)\n        model1.fit(\n            X_train_m1, y_train_m1,\n            sample_weight=train_weights_m1,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=0\n        )\n        \n        oof_preds_model1[valid_idx] = model1.predict(X_valid)\n        test_preds_model1 += model1.predict(X_test)\n        \n        # ========== MODEL 2: 75% MOST RECENT DATA ==========\n        # Filter train indices to only include those from the recent 75% of data\n        train_idx_recent_75 = train_idx[train_idx >= cutoff_idx_75]\n        \n        # Adjust indices to start from 0 for the recent subset\n        train_idx_recent_adjusted_75 = train_idx_recent_75 - cutoff_idx_75\n        \n        # Get the recent subset of training data\n        train_recent_75 = train.iloc[cutoff_idx_75:].reset_index(drop=True)\n        \n        X_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][features_to_use]\n        y_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][\"label\"]\n        \n        # Create time weights for the recent data subset\n        sample_weights_recent_75 = create_time_weights(len(train_recent_75), decay_factor=0.95)\n        train_weights_m2 = sample_weights_recent_75[train_idx_recent_adjusted_75]\n        \n        model2 = XGBRegressor(**xgb_params)\n        model2.fit(\n            X_train_m2, y_train_m2,\n            sample_weight=train_weights_m2,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=0\n        )\n        \n        # For validation predictions, we need to handle cases where validation indices\n        # might be from the older 25% of data\n        valid_idx_in_range_75 = valid_idx[valid_idx >= cutoff_idx_75]\n        if len(valid_idx_in_range_75) > 0:\n            X_valid_m2 = train.iloc[valid_idx_in_range_75][features_to_use]\n            oof_preds_model2[valid_idx_in_range_75] = model2.predict(X_valid_m2)\n        \n        # For indices before cutoff, use Model 1 predictions\n        valid_idx_out_range_75 = valid_idx[valid_idx < cutoff_idx_75]\n        if len(valid_idx_out_range_75) > 0:\n            oof_preds_model2[valid_idx_out_range_75] = oof_preds_model1[valid_idx_out_range_75]\n        \n        test_preds_model2 += model2.predict(X_test)\n        \n        # ========== MODEL 3: 50% MOST RECENT DATA ==========\n        # Filter train indices to only include those from the recent 50% of data\n        train_idx_recent_50 = train_idx[train_idx >= cutoff_idx_50]\n        \n        # Adjust indices to start from 0 for the recent subset\n        train_idx_recent_adjusted_50 = train_idx_recent_50 - cutoff_idx_50\n        \n        # Get the recent subset of training data\n        train_recent_50 = train.iloc[cutoff_idx_50:].reset_index(drop=True)\n        \n        X_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][features_to_use]\n        y_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][\"label\"]\n        \n        # Create time weights for the recent data subset\n        sample_weights_recent_50 = create_time_weights(len(train_recent_50), decay_factor=0.95)\n        train_weights_m3 = sample_weights_recent_50[train_idx_recent_adjusted_50]\n        \n        model3 = XGBRegressor(**xgb_params)\n        model3.fit(\n            X_train_m3, y_train_m3,\n            sample_weight=train_weights_m3,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=0\n        )\n        \n        # For validation predictions, we need to handle cases where validation indices\n        # might be from the older 50% of data\n        valid_idx_in_range_50 = valid_idx[valid_idx >= cutoff_idx_50]\n        if len(valid_idx_in_range_50) > 0:\n            X_valid_m3 = train.iloc[valid_idx_in_range_50][features_to_use]\n            oof_preds_model3[valid_idx_in_range_50] = model3.predict(X_valid_m3)\n        \n        # For indices before cutoff, use Model 1 predictions\n        valid_idx_out_range_50 = valid_idx[valid_idx < cutoff_idx_50]\n        if len(valid_idx_out_range_50) > 0:\n            oof_preds_model3[valid_idx_out_range_50] = oof_preds_model1[valid_idx_out_range_50]\n        \n        test_preds_model3 += model3.predict(X_test)\n    \n    # Average test predictions across folds\n    test_preds_model1 /= FOLDS\n    test_preds_model2 /= FOLDS\n    test_preds_model3 /= FOLDS\n    \n    # Calculate individual model scores\n    pearson_score_model1 = pearsonr(train[\"label\"], oof_preds_model1)[0]\n    pearson_score_model2 = pearsonr(train[\"label\"], oof_preds_model2)[0]\n    pearson_score_model3 = pearsonr(train[\"label\"], oof_preds_model3)[0]\n    \n    # Performance-weighted ensemble\n    total_score = pearson_score_model1 + pearson_score_model2 + pearson_score_model3\n    weight_model1 = pearson_score_model1 / total_score\n    weight_model2 = pearson_score_model2 / total_score\n    weight_model3 = pearson_score_model3 / total_score\n    \n    weighted_ensemble_oof = (weight_model1 * oof_preds_model1 + \n                            weight_model2 * oof_preds_model2 + \n                            weight_model3 * oof_preds_model3)\n    weighted_ensemble_test = (weight_model1 * test_preds_model1 + \n                             weight_model2 * test_preds_model2 + \n                             weight_model3 * test_preds_model3)\n    \n    weighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n    \n    # Clean up memory\n    del model1, model2, model3\n    gc.collect()\n    \n    return weighted_ensemble_score, weighted_ensemble_test\n\n# Main execution\nprint(\"Loading data...\")\ntrain_full = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest_full = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n# Base features\nbase_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\",\n    # \"X17\" - Version 3\n    # \"X23\", \"X24\", \"X25\", \"X22\", \"X30\" - Version 4\n    # \"X35\", \"X33\", \"X31\", \"X37\", \"X36\", \"X32\", \"X38\", \"X39\", \"X40\", \"X34\"\n]\n\n# Get all feature columns (X_1 to X_890)\nall_X_features = [col for col in train_full.columns if col.startswith('X') and col.replace('X', '').isdigit()]\nunused_features = [f for f in all_X_features if f not in base_features]\n\n# Limit to first 10 unused features\nfeatures_to_test = unused_features[340:440]\n\nprint(f\"Total features available: {len(all_X_features)}\")\nprint(f\"Base features: {len(base_features)}\")\nprint(f\"Total unused features: {len(unused_features)}\")\nprint(f\"Testing first 10 unused features: {features_to_test}\")\n\n# Save the list of features being tested\nwith open('tested_features.txt', 'w') as f:\n    f.write(\"Base features (25):\\n\")\n    for feat in base_features:\n        f.write(f\"  {feat}\\n\")\n    f.write(f\"\\nFirst 10 unused features tested:\\n\")\n    for feat in features_to_test:\n        f.write(f\"  {feat}\\n\")\n\n# Reduce memory for the full dataset\ntrain_full = reduce_mem_usage(train_full, \"train_full\")\ntest_full = reduce_mem_usage(test_full, \"test_full\")\n\n# First, get the baseline score with just the base features\nprint(\"\\n\" + \"=\"*60)\nprint(\"RUNNING BASELINE MODEL WITH 25 FEATURES\")\nprint(\"=\"*60)\nbaseline_score, baseline_predictions = run_model_pipeline(train_full, test_full, base_features, verbose=True)\nprint(f\"\\nBaseline Weighted Ensemble Score: {baseline_score:.6f}\")\n\n# Initialize tracking variables\nbest_score = baseline_score\nbest_feature = None\nbest_predictions = baseline_predictions\nresults_list = []\n\n# Test each unused feature\nprint(\"\\n\" + \"=\"*60)\nprint(\"TESTING ADDITIONAL FEATURES (First 10)\")\nprint(\"=\"*60)\n\nfor idx, feature in enumerate(features_to_test):\n    print(f\"\\nTesting feature {idx+1}/10: {feature}\")\n    print(\"-\" * 40)\n    \n    # Create feature list with the additional feature\n    test_features = base_features + [feature]\n    \n    try:\n        # Run model with additional feature\n        score, predictions = run_model_pipeline(train_full, test_full, test_features, verbose=False)\n        \n        # Track results\n        improvement = score - baseline_score\n        results_list.append({\n            'feature': feature,\n            'score': score,\n            'improvement': improvement,\n            'percentage_improvement': (improvement / baseline_score) * 100\n        })\n        \n        print(f\"Score: {score:.6f}\")\n        print(f\"Improvement: {improvement:+.6f} ({(improvement/baseline_score)*100:+.2f}%)\")\n        \n        # Update best if improved\n        if score > best_score:\n            best_score = score\n            best_feature = feature\n            best_predictions = predictions\n            print(f\"  *** NEW BEST FEATURE! ***\")\n    \n    except Exception as e:\n        print(f\"Error: {str(e)}\")\n        results_list.append({\n            'feature': feature,\n            'score': np.nan,\n            'improvement': np.nan,\n            'percentage_improvement': np.nan\n        })\n\n# Save results summary\nresults_df = pd.DataFrame(results_list)\nresults_df = results_df.sort_values('score', ascending=False)\nresults_df.to_csv(\"feature_search_results_top10.csv\", index=False)\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"FEATURE SEARCH COMPLETE - SUMMARY\")\nprint(\"=\"*60)\nprint(f\"\\nBaseline score (25 features): {baseline_score:.6f}\")\nprint(f\"Best score achieved: {best_score:.6f}\")\n\nif best_feature:\n    print(f\"\\nBest additional feature: {best_feature}\")\n    print(f\"Improvement: {best_score - baseline_score:+.6f} ({((best_score - baseline_score)/baseline_score)*100:+.2f}%)\")\n    \n    # Create the best feature list for easy copy-paste\n    best_feature_list = base_features + [best_feature]\n    \n    # Save best feature configuration\n    with open('best_feature_config.txt', 'w') as f:\n        f.write(\"# Best feature configuration\\n\")\n        f.write(f\"# Score: {best_score:.6f}\\n\")\n        f.write(f\"# Improvement over baseline: {best_score - baseline_score:+.6f}\\n\")\n        f.write(f\"# Best additional feature: {best_feature}\\n\\n\")\n        f.write(\"selected_features = [\\n\")\n        for i, feat in enumerate(best_feature_list):\n            if i < len(best_feature_list) - 1:\n                f.write(f'    \"{feat}\",\\n')\n            else:\n                f.write(f'    \"{feat}\"\\n')\n        f.write(\"]\\n\")\n    \n    print(\"\\nBest feature configuration saved to 'best_feature_config.txt'\")\nelse:\n    print(\"\\nNo improvement found over baseline\")\n\n# Print detailed results table\nprint(\"\\n\" + \"=\"*60)\nprint(\"DETAILED RESULTS\")\nprint(\"=\"*60)\nprint(f\"\\n{'Feature':<10} {'Score':<10} {'Improvement':<12} {'% Improvement':<15}\")\nprint(\"-\" * 60)\nprint(f\"{'BASELINE':<10} {baseline_score:<10.6f} {0:<12.6f} {0:<15.2f}%\")\nprint(\"-\" * 60)\n\nfor _, row in results_df.iterrows():\n    if not pd.isna(row['score']):\n        print(f\"{row['feature']:<10} {row['score']:<10.6f} {row['improvement']:<+12.6f} {row['percentage_improvement']:<+15.2f}%\")\n\n# Save final best submission\nsample[\"prediction\"] = best_predictions\nsample.to_csv(\"submission.csv\", index=False)\nprint(f\"\\nFinal submission saved to 'submission.csv'\")\n\n# If best feature was found, you can use this configuration next time\nif best_feature:\n    print(\"\\n\" + \"=\"*60)\n    print(\"TO USE BEST CONFIGURATION IN FUTURE RUNS:\")\n    print(\"=\"*60)\n    print(\"Copy the content from 'best_feature_config.txt' to replace your selected_features list\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:44:46.516605Z","iopub.execute_input":"2025-05-28T20:44:46.51729Z","iopub.status.idle":"2025-05-28T20:45:29.481637Z","shell.execute_reply.started":"2025-05-28T20:44:46.517262Z","shell.execute_reply":"2025-05-28T20:45:29.477685Z"}},"outputs":[],"execution_count":null}]}