{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:07:42.039419Z","iopub.execute_input":"2025-05-28T20:07:42.039676Z","iopub.status.idle":"2025-05-28T20:07:42.403946Z","shell.execute_reply.started":"2025-05-28T20:07:42.03963Z","shell.execute_reply":"2025-05-28T20:07:42.403334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\n# Create time-based sample weights\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\n# Load data\ntrain = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n# Select features\nselected_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\ntrain = train[selected_features + [\"label\"]]\ntest = test[selected_features]\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nprint(\"Train=\", train.shape)\nprint(\"Test=\", test.shape)\nprint(\"Sample=\", sample.shape)\n\nRMV = [\"label\"]\nFEATURES = [c for c in train.columns if c not in RMV]\nprint(f\"There are {len(FEATURES)} FEATURES: {FEATURES}\")\n\n# Define cross-validation\nFOLDS = 5\nkf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n\n# XGBoost parameters (same for both models)\nxgb_params = {\n    \"tree_method\": \"gpu_hist\",\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\n# Initialize predictions for both models\noof_preds_model1 = np.zeros(len(train))\ntest_preds_model1 = np.zeros(len(test))\noof_preds_model2 = np.zeros(len(train))\ntest_preds_model2 = np.zeros(len(test))\n\n# Generate sample weights for Model 1 (full data)\nsample_weights_full = create_time_weights(len(train), decay_factor=0.95)\nprint(f\"\\nModel 1 - Full data sample weights range: [{sample_weights_full.min():.4f}, {sample_weights_full.max():.4f}]\")\nprint(f\"Model 1 - Full data sample weights mean: {sample_weights_full.mean():.4f}\")\n\n# Calculate the cutoff for 75% most recent data\ncutoff_idx = int(len(train) * 0.25)\nprint(f\"\\nModel 2 - Using most recent {len(train) - cutoff_idx} samples (75% of data)\")\n\n# Cross-validation loop\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"\\n\" + \"#\" * 50)\n    print(f\"### Fold {i + 1}\")\n    print(\"#\" * 50)\n    \n    # ========== MODEL 1: FULL DATA WITH TIME WEIGHTS ==========\n    print(\"\\n--- Model 1: Full Data with Time Weights ---\")\n    \n    X_train_m1 = train.iloc[train_idx][FEATURES]\n    y_train_m1 = train.iloc[train_idx][\"label\"]\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n    \n    # Extract sample weights for this fold's training data\n    train_weights_m1 = sample_weights_full[train_idx]\n    \n    model1 = XGBRegressor(**xgb_params)\n    model1.fit(\n        X_train_m1, y_train_m1,\n        sample_weight=train_weights_m1,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    oof_preds_model1[valid_idx] = model1.predict(X_valid)\n    test_preds_model1 += model1.predict(X_test)\n    \n    # ========== MODEL 2: 75% MOST RECENT DATA ==========\n    print(\"\\n--- Model 2: 75% Most Recent Data ---\")\n    \n    # Filter train indices to only include those from the recent 75% of data\n    train_idx_recent = train_idx[train_idx >= cutoff_idx]\n    \n    # Adjust indices to start from 0 for the recent subset\n    train_idx_recent_adjusted = train_idx_recent - cutoff_idx\n    \n    # Get the recent subset of training data\n    train_recent = train.iloc[cutoff_idx:].reset_index(drop=True)\n    \n    X_train_m2 = train_recent.iloc[train_idx_recent_adjusted][FEATURES]\n    y_train_m2 = train_recent.iloc[train_idx_recent_adjusted][\"label\"]\n    \n    # Create time weights for the recent data subset\n    sample_weights_recent = create_time_weights(len(train_recent), decay_factor=0.95)\n    train_weights_m2 = sample_weights_recent[train_idx_recent_adjusted]\n    \n    model2 = XGBRegressor(**xgb_params)\n    model2.fit(\n        X_train_m2, y_train_m2,\n        sample_weight=train_weights_m2,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 25% of data\n    valid_idx_in_range = valid_idx[valid_idx >= cutoff_idx]\n    if len(valid_idx_in_range) > 0:\n        X_valid_m2 = train.iloc[valid_idx_in_range][FEATURES]\n        oof_preds_model2[valid_idx_in_range] = model2.predict(X_valid_m2)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range = valid_idx[valid_idx < cutoff_idx]\n    if len(valid_idx_out_range) > 0:\n        oof_preds_model2[valid_idx_out_range] = oof_preds_model1[valid_idx_out_range]\n    \n    test_preds_model2 += model2.predict(X_test)\n\n# Average test predictions across folds\ntest_preds_model1 /= FOLDS\ntest_preds_model2 /= FOLDS\n\n# Calculate individual model scores\npearson_score_model1 = pearsonr(train[\"label\"], oof_preds_model1)[0]\npearson_score_model2 = pearsonr(train[\"label\"], oof_preds_model2)[0]\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"INDIVIDUAL MODEL PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Model 1 (Full Data) Pearson Correlation: {pearson_score_model1:.4f}\")\nprint(f\"Model 2 (75% Recent) Pearson Correlation: {pearson_score_model2:.4f}\")\n\n# Create ensemble predictions\n# Simple average ensemble\nensemble_oof_preds = 0.5 * oof_preds_model1 + 0.5 * oof_preds_model2\nensemble_test_preds = 0.5 * test_preds_model1 + 0.5 * test_preds_model2\n\n# Calculate ensemble score\nensemble_pearson_score = pearsonr(train[\"label\"], ensemble_oof_preds)[0]\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"ENSEMBLE PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Ensemble (50/50) Pearson Correlation: {ensemble_pearson_score:.4f}\")\n\n# Performance-weighted ensemble\nweight_model1 = pearson_score_model1 / (pearson_score_model1 + pearson_score_model2)\nweight_model2 = pearson_score_model2 / (pearson_score_model1 + pearson_score_model2)\n\nweighted_ensemble_oof = weight_model1 * oof_preds_model1 + weight_model2 * oof_preds_model2\nweighted_ensemble_test = weight_model1 * test_preds_model1 + weight_model2 * test_preds_model2\n\nweighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n\nprint(f\"\\nWeighted Ensemble Performance:\")\nprint(f\"  Model 1 weight: {weight_model1:.3f}\")\nprint(f\"  Model 2 weight: {weight_model2:.3f}\")\nprint(f\"  Weighted Ensemble Pearson Correlation: {weighted_ensemble_score:.4f}\")\n\n# Use the better ensemble for final predictions\nif weighted_ensemble_score > ensemble_pearson_score:\n    final_test_preds = weighted_ensemble_test\n    print(\"\\nUsing weighted ensemble for final predictions\")\nelse:\n    final_test_preds = ensemble_test_preds\n    print(\"\\nUsing simple average ensemble for final predictions\")\n\n# SHAP analysis (using Model 1 as representative)\nprint(\"\\nGenerating SHAP analysis...\")\nexplainer = shap.TreeExplainer(model1, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\nshap_values = explainer.shap_values(X_test)\nshap.summary_plot(shap_values, X_test)\n\n# Save predictions\nsample[\"prediction\"] = final_test_preds\nsample.to_csv(\"submission.csv\", index=False)\nprint(\"\\nPredictions saved to submission.csv\")\nprint(sample.head())\n\n# Save detailed results\nensemble_results = pd.DataFrame({\n    'model': ['Model 1 (Full Data)', 'Model 2 (75% Recent)', 'Simple Ensemble', 'Weighted Ensemble'],\n    'pearson_correlation': [pearson_score_model1, pearson_score_model2, ensemble_pearson_score, weighted_ensemble_score],\n    'weight_in_final': [weight_model1 if weighted_ensemble_score > ensemble_pearson_score else 0.5,\n                        weight_model2 if weighted_ensemble_score > ensemble_pearson_score else 0.5,\n                        np.nan, np.nan]\n})\nensemble_results.to_csv(\"ensemble_results.csv\", index=False)\nprint(\"\\nEnsemble results saved to ensemble_results.csv\")\nprint(ensemble_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-28T20:44:46.516605Z","iopub.execute_input":"2025-05-28T20:44:46.51729Z","iopub.status.idle":"2025-05-28T20:45:29.481637Z","shell.execute_reply.started":"2025-05-28T20:44:46.517262Z","shell.execute_reply":"2025-05-28T20:45:29.477685Z"}},"outputs":[],"execution_count":null}]}