{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip uninstall -y torch torchvision torchaudio\n# !pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:13:09.013612Z","iopub.execute_input":"2025-07-21T10:13:09.014446Z","iopub.status.idle":"2025-07-21T10:13:58.505740Z","shell.execute_reply.started":"2025-07-21T10:13:09.014413Z","shell.execute_reply":"2025-07-21T10:13:58.504896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# all needed libraries are loaded\nimport gc\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing\n\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nimport shap\n\nfrom sklearn.model_selection import KFold\nfrom scipy.stats import pearsonr\n\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:43:59.232841Z","iopub.execute_input":"2025-07-21T09:43:59.233099Z","iopub.status.idle":"2025-07-21T09:44:16.016695Z","shell.execute_reply.started":"2025-07-21T09:43:59.233078Z","shell.execute_reply":"2025-07-21T09:44:16.015897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check torch version (should be 11.8)\nprint(torch.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:44:16.018170Z","iopub.execute_input":"2025-07-21T09:44:16.018794Z","iopub.status.idle":"2025-07-21T09:44:16.023448Z","shell.execute_reply.started":"2025-07-21T09:44:16.018773Z","shell.execute_reply":"2025-07-21T09:44:16.022697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# configuration settings\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        \n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            \n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            \n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            \n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            \n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            \n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\n\n# Create time-based sample weights\ndef create_time_weights(n_samples, decay_factor = 0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    \n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    \n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    \n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    \n    return weights","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:44:16.024244Z","iopub.execute_input":"2025-07-21T09:44:16.024521Z","iopub.status.idle":"2025-07-21T09:44:16.044546Z","shell.execute_reply.started":"2025-07-21T09:44:16.024497Z","shell.execute_reply":"2025-07-21T09:44:16.043688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = [\n    \"buy_qty\", \"sell_qty\", \"volume\", \"bid_qty\", \"ask_qty\",\n    'X22', 'X28', 'X40', 'X52', 'X55', 'X97', 'X137', 'X138', 'X168', 'X169', 'X174', 'X175', 'X178',\n    'X179', 'X180', 'X181', 'X173', 'X197', 'X198', 'X272', 'X288', 'X297', 'X302', 'X321', 'X333',\n    'X338', 'X341', 'X343', 'X344', 'X345', 'X363', 'X379', 'X385', 'X386', 'X415', 'X421', 'X427',\n    'X428', 'X435', 'X438', 'X444', 'X445', 'X450', 'X452', 'X459', 'X466', 'X586', 'X587', 'X593',\n    'X598', 'X572', 'X603', 'X605', 'X612', 'X674', 'X680', 'X683', 'X686', 'X692', 'X695', 'X696', 'X532'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:44:16.045873Z","iopub.execute_input":"2025-07-21T09:44:16.046168Z","iopub.status.idle":"2025-07-21T09:44:16.061287Z","shell.execute_reply.started":"2025-07-21T09:44:16.046119Z","shell.execute_reply":"2025-07-21T09:44:16.060535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ntrain = pd.read_parquet(CFG.train_path).reset_index(drop = True)\ntest = pd.read_parquet(CFG.test_path).reset_index(drop = True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n\n# Select features\n# selected_features = [\n#     \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n#     \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n#     \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\",\"X888\", \"X421\", \"X333\"\n# ]\n\n# based data: \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n# based: final_selected\n#selected_features = final_selected + [\"bid_qty\"] + [\"ask_qty\"] + [\"buy_qty\"] + [\"sell_qty\"] + [\"volume\"]\n\n\n# selected_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume',\n#                      'X177', 'X214', 'X97', 'X285', 'X465', 'X473', 'X344', 'X680',\n#                      'X174', 'X592', 'X32', 'X779', 'X754', 'X508', 'X731', 'X376',\n#                      'X356', 'X331', 'X501', 'X651', 'X86', 'X24', 'X94', 'X421',\n#                      'X417', 'X168', 'X340', 'X368', 'X85', 'X283', 'X160', 'X178',\n#                      'X180', 'X40', 'X84', 'X291', 'X182', 'X581', 'X269', 'X95',\n#                      'X612', 'X740', 'X766', 'X91', 'X588', 'X609', 'X161', 'X166',\n#                      'X171', 'X451', 'X136', 'X413', 'X133', 'X88', 'X33', 'X584',\n#                      'X561', 'X576', 'X445', 'X379', 'X332', 'X613', 'X573', 'X422',\n#                      'X338', 'X564', 'X98', 'X362', 'X370', 'X20', 'X757', 'X778',\n#                      'X284', 'X137', 'X758', 'X750', 'X655', 'X287', 'X131', 'X345',\n#                      'X429', 'X770', 'X605', 'X499', 'X167', 'X135', 'X604', 'X607',\n#                      'X44', 'X751', 'X271', 'X653', 'X708', 'X27', 'X458', 'X768',\n#                      'X120', 'X427', 'X611', 'X586', 'X741', 'X35', 'X610', 'X162',\n#                      'X608', 'X385', 'X780', 'X333', 'X761', 'X424', 'X383', 'X777',\n#                      'X169', 'X444', 'X38', 'X302', 'X684', 'X652', 'X657', 'X342',\n#                      'X387', 'X96', 'X730', 'X270', 'X138', 'X414', 'X299', 'X759',\n#                      'X140', 'X654', 'X281', 'X752', 'X125', 'X375', 'X648', 'X415',\n#                      'X425', 'X466', 'X374', 'X570', 'X181', 'X587', 'X28', 'X22',\n#                      'X301', 'X300', 'X614', 'X682', 'X767', 'X341', 'X21', 'X280',\n#                      'X765', 'X582', 'X756', 'X298', 'X500']\n\n\ntrain = train[selected_features + [\"label\"]]\ntest = test[selected_features]\n\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\nprint(\"Train =\", train.shape)\nprint(\"Test =\", test.shape)\nprint(\"Sample =\", sample.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:44:16.062103Z","iopub.execute_input":"2025-07-21T09:44:16.062365Z","iopub.status.idle":"2025-07-21T09:45:09.077607Z","shell.execute_reply.started":"2025-07-21T09:44:16.062349Z","shell.execute_reply":"2025-07-21T09:45:09.076703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Simple EDA**","metadata":{}},{"cell_type":"code","source":"# histogram\nplt.figure(figsize=(8, 4))\nplt.hist(train[\"label\"], bins = 100)\n\nplt.title(\"Target Distribution\")\nplt.xlabel(\"Target\")\nplt.ylabel(\"Frequency\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T10:40:18.622502Z","iopub.status.idle":"2025-07-20T10:40:18.622855Z","shell.execute_reply.started":"2025-07-20T10:40:18.622668Z","shell.execute_reply":"2025-07-20T10:40:18.622685Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot graph\nplt.figure(figsize=(8, 4))\nplt.plot(train[\"label\"].values)\n\nplt.title(\"Target Over Samples\")\nplt.xlabel(\"Temp Samples for Date\")\nplt.ylabel(\"Target\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T10:40:18.623849Z","iopub.status.idle":"2025-07-20T10:40:18.624067Z","shell.execute_reply.started":"2025-07-20T10:40:18.623966Z","shell.execute_reply":"2025-07-20T10:40:18.623976Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nulls = train.isnull().sum()\n\nnulls = nulls[nulls > 0]\n\n# 결측수\nprint(f\"결측치: {nulls.values}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T10:40:18.625170Z","iopub.status.idle":"2025-07-20T10:40:18.625509Z","shell.execute_reply.started":"2025-07-20T10:40:18.625332Z","shell.execute_reply":"2025-07-20T10:40:18.625347Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import zscore\n\nz = zscore(train['label'])\n\noutliers_z = train[np.abs(z) > 3] # threshold = 3\n\nprint(f\"이상치 개수 {len(outliers_z)}\")\n\nprint(\"\\n이상치 목록\")\nprint(outliers_z.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T10:40:18.627109Z","iopub.status.idle":"2025-07-20T10:40:18.627408Z","shell.execute_reply.started":"2025-07-20T10:40:18.627281Z","shell.execute_reply":"2025-07-20T10:40:18.627295Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMV = [\"label\"]\nFEATURES = [c for c in train.columns if c not in RMV]\nprint(f\"There are {len(FEATURES)} FEATURES: {FEATURES}\")\n\n# Define cross-validation\nFOLDS = 10\nkf = KFold(n_splits = FOLDS, shuffle = True, random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:20.373594Z","iopub.execute_input":"2025-07-21T09:45:20.374409Z","iopub.status.idle":"2025-07-21T09:45:20.379358Z","shell.execute_reply.started":"2025-07-21T09:45:20.374379Z","shell.execute_reply":"2025-07-21T09:45:20.378485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# XGBoost parameters (same for all models)\nxgb_params = {\n    \"tree_method\": \"gpu_hist\",\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:23.452325Z","iopub.execute_input":"2025-07-21T09:45:23.452933Z","iopub.status.idle":"2025-07-21T09:45:23.457488Z","shell.execute_reply.started":"2025-07-21T09:45:23.452906Z","shell.execute_reply":"2025-07-21T09:45:23.456569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LGBM parameters (same for all models)\nlgbm_params = {\n    \"boosting_type\": \"gbdt\",\n    \"colsample_bytree\": 0.5625888953382505,\n    \"learning_rate\": 0.029312951475451557,\n    \"min_child_samples\": 63,\n    \"min_child_weight\": 0.11456572852335424,\n    \"n_estimators\": 126,\n    \"n_jobs\": -1,\n    \"num_leaves\": 37,\n    \"random_state\": 42,\n    \"reg_alpha\": 85.2476527854083,\n    \"reg_lambda\": 99.38305361388907,\n    \"subsample\": 0.450669817684892,\n    \"verbose\": 200\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:25.838756Z","iopub.execute_input":"2025-07-21T09:45:25.839055Z","iopub.status.idle":"2025-07-21T09:45:25.843603Z","shell.execute_reply.started":"2025-07-21T09:45:25.839032Z","shell.execute_reply":"2025-07-21T09:45:25.842723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **단일 모델 테스트**","metadata":{}},{"cell_type":"code","source":"# # 단일 모델 테스트\n# oof_preds_model1 = np.zeros(len(train))\n# test_preds_model1 = np.zeros(len(test))\n\n# # Generate sample weights for Model 1 (full data)\n# sample_weights_full = create_time_weights(len(train), decay_factor = 0.95)\n# print(f\"\\nModel 1 - Full data sample weights range: [{sample_weights_full.min():.4f}, {sample_weights_full.max():.4f}]\")\n# print(f\"Model 1 - Full data sample weights mean: {sample_weights_full.mean():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T06:17:41.281976Z","iopub.execute_input":"2025-07-13T06:17:41.282639Z","iopub.status.idle":"2025-07-13T06:17:41.313250Z","shell.execute_reply.started":"2025-07-13T06:17:41.282617Z","shell.execute_reply":"2025-07-13T06:17:41.312518Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Cross-validation loop\n# for i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n#     print(\"\\n\" + \"#\" * 50)\n#     print(f\"### Fold {i + 1}\")\n#     print(\"#\" * 50)\n    \n#     # ========== MODEL 1: FULL DATA WITH TIME WEIGHTS ==========\n#     print(\"\\n--- Model 1: Full Data with Time Weights ---\")\n    \n#     X_train_m1 = train.iloc[train_idx][FEATURES]\n#     y_train_m1 = train.iloc[train_idx][\"label\"]\n    \n#     X_valid = train.iloc[valid_idx][FEATURES]\n#     y_valid = train.iloc[valid_idx][\"label\"]\n    \n#     X_test = test[FEATURES]\n    \n#     # Extract sample weights for this fold's training data\n#     train_weights_m1 = sample_weights_full[train_idx]\n    \n#     # model1 = XGBRegressor(**xgb_params)\n    \n#     # model1.fit(\n#     #     X_train_m1, y_train_m1,\n#     #     sample_weight = train_weights_m1,\n#     #     eval_set = [(X_valid, y_valid)],\n#     #     early_stopping_rounds = 25,\n#     #     verbose = 200\n#     # )\n\n#     model1 = LGBMRegressor(**lgbm_params)\n\n#     model1.fit(\n#         X_train_m1, y_train_m1,\n#         sample_weight = train_weights_m1,\n#         eval_set = [(X_valid, y_valid)],\n#     )\n    \n#     oof_preds_model1[valid_idx] = model1.predict(X_valid)\n#     test_preds_model1 += model1.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T06:44:23.689620Z","iopub.execute_input":"2025-07-13T06:44:23.690351Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Average test predictions across folds\n# test_preds_model1 /= FOLDS\n\n\n# # Calculate individual model scores\n# pearson_score_model1 = pearsonr(train[\"label\"], oof_preds_model1)[0]\n\n# # print\n# print(\"\\n\" + \"=\" * 50)\n# print(\"INDIVIDUAL MODEL PERFORMANCE\")\n# print(\"=\" * 50)\n# print(f\"Model 1 (Full Data) Pearson Correlation: {pearson_score_model1:.4f}\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-07-20T10:40:18.636465Z","iopub.status.idle":"2025-07-20T10:40:18.636736Z","shell.execute_reply.started":"2025-07-20T10:40:18.636585Z","shell.execute_reply":"2025-07-20T10:40:18.636600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create ensemble predictions\n# # Simple average ensemble (now with 3 models)\n# ensemble_oof_preds = oof_preds_model1\n# ensemble_test_preds = test_preds_model1\n\n# # Calculate ensemble score\n# ensemble_pearson_score = pearsonr(train[\"label\"], ensemble_oof_preds)[0]\n\n# print(\"\\n\" + \"=\" * 50)\n# print(\"ENSEMBLE PERFORMANCE\")\n# print(\"=\" * 50)\n# print(f\"Ensemble (Equal Weight) Pearson Correlation: {ensemble_pearson_score:.4f}\")\n\n# # Performance-weighted ensemble (now with 3 models)\n# total_score = pearson_score_model1 + pearson_score_model2 + pearson_score_model3+ pearson_score_model3_lb  # UPDATED\n# weight_model1 = pearson_score_model1 / total_score  # UPDATED\n# weight_model2 = pearson_score_model2 / total_score  # UPDATED\n# weight_model3 = pearson_score_model3 / total_score  # NEW\n# weight_model3_lb = pearson_score_model3_lb / total_score  # NEW\n\n# weighted_ensemble_oof = (weight_model1 * oof_preds_model1 + \n#                         weight_model2 * oof_preds_model2 + \n#                         weight_model3 * oof_preds_model3 + \n#                         weight_model3_lb * oof_preds_model3_lb)  # UPDATED\n\n\n# weighted_ensemble_test = (weight_model1 * test_preds_model1 + \n#                          weight_model2 * test_preds_model2 + \n#                          weight_model3 * test_preds_model3  + \n#                          weight_model3_lb * test_preds_model3_lb)  # UPDATED\n\n# weighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n\n# print(f\"\\nWeighted Ensemble Performance:\")\n# print(f\"  Model 1 weight: {weight_model1:.3f}\")\n\n# print(f\"  Weighted Ensemble Pearson Correlation: {weighted_ensemble_score:.4f}\")\n\n# # Use the better ensemble for final predictions\n# if weighted_ensemble_score > ensemble_pearson_score:\n#     final_test_preds = weighted_ensemble_test\n#     print(\"\\nUsing weighted ensemble for final predictions\")\n# else:\n#     final_test_preds = ensemble_test_preds\n#     print(\"\\nUsing simple average ensemble for final predictions\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"-----------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# **Ensemble Model Testing**","metadata":{}},{"cell_type":"code","source":"INPUT_DIM = train.shape[1]\nEPOCHS = 30\nBATCH_SIZE = 64\nLR = 1e-3\nN_SPLITS = 5\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nprint(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:28.639394Z","iopub.execute_input":"2025-07-21T09:45:28.640070Z","iopub.status.idle":"2025-07-21T09:45:28.774575Z","shell.execute_reply.started":"2025-07-21T09:45:28.640044Z","shell.execute_reply":"2025-07-21T09:45:28.773410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\n\nclass RegressionDataset(Dataset):\n    def __init__(self, X, y):\n        self.X = torch.tensor(X, dtype = torch.float32)\n        self.y = torch.tensor(y, dtype = torch.float32)\n    \n    def __len__(self):\n        return len(self.X)\n    \n    def __getitem__(self, idx):\n        return self.X[idx], self.y[idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:30.746725Z","iopub.execute_input":"2025-07-21T09:45:30.747547Z","iopub.status.idle":"2025-07-21T09:45:30.752602Z","shell.execute_reply.started":"2025-07-21T09:45:30.747519Z","shell.execute_reply":"2025-07-21T09:45:30.751700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DNNModel(nn.Module):\n    def __init__(self, input_dim):\n        super().__init__()\n        self.model = nn.Sequential(\n            nn.Linear(input_dim, 128),\n            nn.BatchNorm1d(128),\n            nn.GELU(),\n            nn.Dropout(0.1),\n            \n            nn.Linear(128, 64),\n            nn.BatchNorm1d(64),\n            nn.GELU(),\n            nn.Dropout(0.2),\n\n            nn.Linear(64, 32),\n            nn.BatchNorm1d(32),\n            nn.GELU(),\n            nn.Dropout(0.2),\n\n            nn.Linear(32, 16),\n            nn.BatchNorm1d(16),\n            nn.GELU(),\n            nn.Dropout(0.2),\n\n            nn.Linear(16, 8),\n            nn.BatchNorm1d(8),\n            nn.GELU(),\n            nn.Dropout(0.2),\n\n            nn.Linear(8, 1)\n        )\n\n    def forward(self, x):\n        return self.model(x).squeeze(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T10:41:20.358082Z","iopub.execute_input":"2025-07-20T10:41:20.358875Z","iopub.status.idle":"2025-07-20T10:41:20.364098Z","shell.execute_reply.started":"2025-07-20T10:41:20.358846Z","shell.execute_reply":"2025-07-20T10:41:20.363449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class ResidualBlock(nn.Module):\n#     def __init__(self, dim, dropout=0.3):\n#         super().__init__()\n#         self.block = nn.Sequential(\n#             nn.Linear(dim, dim),\n#             nn.BatchNorm1d(dim),\n#             nn.GELU(),\n#             nn.Dropout(dropout),\n#             nn.Linear(dim, dim),\n#             nn.BatchNorm1d(dim),\n#             nn.GELU(),\n#             nn.Dropout(dropout)\n#         )\n#         self.shortcut = nn.Identity()\n\n#     def forward(self, x):\n#         return self.block(x) + self.shortcut(x)\n\n# class DNNModel(nn.Module):\n#     def __init__(self, input_dim):\n#         super().__init__()\n#         self.input_layer = nn.Sequential(\n#             nn.Linear(input_dim, 512),\n#             nn.BatchNorm1d(512),\n#             nn.GELU(),\n#             nn.Dropout(0.2)\n#         )\n\n#         self.res_blocks = nn.Sequential(\n#             ResidualBlock(512, dropout = 0.2),\n#             ResidualBlock(512, dropout = 0.3)\n#         )\n\n#         self.output_layers = nn.Sequential(\n#             nn.Linear(512, 256),\n#             nn.BatchNorm1d(256),\n#             nn.GELU(),\n#             nn.Dropout(0.3),\n#             nn.Linear(256, 128),\n#             nn.BatchNorm1d(128),\n#             nn.GELU(),\n#             nn.Dropout(0.3),\n#             nn.Linear(128, 64),\n#             nn.BatchNorm1d(64),\n#             nn.GELU(),\n#             nn.Dropout(0.3),\n#             nn.Linear(64, 1)\n#         )\n\n#     def forward(self, x):\n#         x = self.input_layer(x)\n#         x = self.res_blocks(x)\n#         x = self.output_layers(x)\n#         return x.squeeze(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:32.541439Z","iopub.execute_input":"2025-07-21T09:45:32.542056Z","iopub.status.idle":"2025-07-21T09:45:32.549449Z","shell.execute_reply.started":"2025-07-21T09:45:32.542032Z","shell.execute_reply":"2025-07-21T09:45:32.548736Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize predictions for all three models\noof_preds_model1 = np.zeros(len(train))\ntest_preds_model1 = np.zeros(len(test))\n\noof_preds_model2 = np.zeros(len(train))\ntest_preds_model2 = np.zeros(len(test))\n\noof_preds_model3 = np.zeros(len(train))\ntest_preds_model3 = np.zeros(len(test))\n\noof_preds_model3_lb = np.zeros(len(train))\ntest_preds_model3_lb = np.zeros(len(test))\n\noof_preds_dnn = np.zeros(len(train))\ntest_preds_dnn = np.zeros(len(test))\n\n\n# Generate sample weights for Model 1 (full data)\nsample_weights_full = create_time_weights(len(train), decay_factor = 0.95)\nprint(f\"\\nModel 1 - Full data sample weights range: [{sample_weights_full.min():.4f}, {sample_weights_full.max():.4f}]\")\nprint(f\"Model 1 - Full data sample weights mean: {sample_weights_full.mean():.4f}\")\n\n# Calculate the cutoff for 75% most recent data\ncutoff_idx_75 = int(len(train) * 0.25)\nprint(f\"\\nModel 2 - Using most recent {len(train) - cutoff_idx_75} samples (75% of data)\")\n\n# Calculate the cutoff for 50% most recent data\ncutoff_idx_50 = int(len(train) * 0.50)  # NEW: 50% cutoff\nprint(f\"\\nModel 3 - Using most recent {len(train) - cutoff_idx_50} samples (50% of data)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T09:45:35.360653Z","iopub.execute_input":"2025-07-21T09:45:35.361329Z","iopub.status.idle":"2025-07-21T09:45:35.400877Z","shell.execute_reply.started":"2025-07-21T09:45:35.361302Z","shell.execute_reply":"2025-07-21T09:45:35.400178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cross-validation loop\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print(\"\\n\" + \"#\" * 50)\n    print(f\"### Fold {i + 1}\")\n    print(\"#\" * 50)\n    \n    # ========== MODEL 1: FULL DATA WITH TIME WEIGHTS ==========\n    print(\"\\n--- Model 1: Full Data with Time Weights ---\")\n    \n    X_train_m1 = train.iloc[train_idx][FEATURES]\n    y_train_m1 = train.iloc[train_idx][\"label\"]\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx][\"label\"]\n    X_test = test[FEATURES]\n    \n    # Extract sample weights for this fold's training data\n    train_weights_m1 = sample_weights_full[train_idx]\n    \n    model1 = XGBRegressor(**xgb_params)\n    model1.fit(\n        X_train_m1, y_train_m1,\n        sample_weight=train_weights_m1,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds = 25,\n        verbose = 200\n    )\n    \n    oof_preds_model1[valid_idx] = model1.predict(X_valid)\n    test_preds_model1 += model1.predict(X_test)\n    \n    # ========== MODEL 2: 75% MOST RECENT DATA ==========\n    print(\"\\n--- Model 2: 75% Most Recent Data ---\")\n    \n    # Filter train indices to only include those from the recent 75% of data\n    train_idx_recent_75 = train_idx[train_idx >= cutoff_idx_75]\n    \n    # Adjust indices to start from 0 for the recent subset\n    train_idx_recent_adjusted_75 = train_idx_recent_75 - cutoff_idx_75\n    \n    # Get the recent subset of training data\n    train_recent_75 = train.iloc[cutoff_idx_75:].reset_index(drop=True)\n    \n    X_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][FEATURES]\n    y_train_m2 = train_recent_75.iloc[train_idx_recent_adjusted_75][\"label\"]\n    \n    # Create time weights for the recent data subset\n    sample_weights_recent_75 = create_time_weights(len(train_recent_75), decay_factor=0.95)\n    train_weights_m2 = sample_weights_recent_75[train_idx_recent_adjusted_75]\n    \n    model2 = XGBRegressor(**xgb_params)\n    model2.fit(\n        X_train_m2, y_train_m2,\n        sample_weight=train_weights_m2,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 25% of data\n    valid_idx_in_range_75 = valid_idx[valid_idx >= cutoff_idx_75]\n    if len(valid_idx_in_range_75) > 0:\n        X_valid_m2 = train.iloc[valid_idx_in_range_75][FEATURES]\n        oof_preds_model2[valid_idx_in_range_75] = model2.predict(X_valid_m2)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_75 = valid_idx[valid_idx < cutoff_idx_75]\n    if len(valid_idx_out_range_75) > 0:\n        oof_preds_model2[valid_idx_out_range_75] = oof_preds_model1[valid_idx_out_range_75]\n    \n    test_preds_model2 += model2.predict(X_test)\n\n    \n    # ========== MODEL 3: 50% MOST RECENT DATA ==========\n    print(\"\\n--- Model 3: 50% Most Recent Data ---\")\n    \n    # Filter train indices to only include those from the recent 50% of data\n    train_idx_recent_50 = train_idx[train_idx >= cutoff_idx_50]\n    \n    # Adjust indices to start from 0 for the recent subset\n    train_idx_recent_adjusted_50 = train_idx_recent_50 - cutoff_idx_50\n    \n    # Get the recent subset of training data\n    train_recent_50 = train.iloc[cutoff_idx_50:].reset_index(drop=True)\n    \n    X_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][FEATURES]\n    y_train_m3 = train_recent_50.iloc[train_idx_recent_adjusted_50][\"label\"]\n    \n    # Create time weights for the recent data subset\n    sample_weights_recent_50 = create_time_weights(len(train_recent_50), decay_factor=0.95)\n    train_weights_m3 = sample_weights_recent_50[train_idx_recent_adjusted_50]\n    \n    model3 = XGBRegressor(**xgb_params)\n    model3.fit(\n        X_train_m3, y_train_m3,\n        sample_weight=train_weights_m3,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 50% of data\n    valid_idx_in_range_50 = valid_idx[valid_idx >= cutoff_idx_50]\n    if len(valid_idx_in_range_50) > 0:\n        X_valid_m3 = train.iloc[valid_idx_in_range_50][FEATURES]\n        oof_preds_model3[valid_idx_in_range_50] = model3.predict(X_valid_m3)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_50 = valid_idx[valid_idx < cutoff_idx_50]\n    if len(valid_idx_out_range_50) > 0:\n        oof_preds_model3[valid_idx_out_range_50] = oof_preds_model1[valid_idx_out_range_50]\n    \n    test_preds_model3 += model3.predict(X_test)\n\n    # ========== MODEL 3: 50% MOST RECENT DATA  LGBM==========\n    print(\"\\n--- Model 3: 50% Most Recent Data with LGBM ---\")\n    \n    model3_lb = LGBMRegressor(**lgbm_params)\n    model3_lb.fit(\n        X_train_m3, y_train_m3,\n        sample_weight=train_weights_m3,\n        eval_set=[(X_valid, y_valid)]\n    )\n    \n    # For validation predictions, we need to handle cases where validation indices\n    # might be from the older 50% of data\n    valid_idx_in_range_50 = valid_idx[valid_idx >= cutoff_idx_50]\n    if len(valid_idx_in_range_50) > 0:\n        X_valid_m3 = train.iloc[valid_idx_in_range_50][FEATURES]\n        oof_preds_model3_lb[valid_idx_in_range_50] = model3_lb.predict(X_valid_m3)\n    \n    # For indices before cutoff, use Model 1 predictions\n    valid_idx_out_range_50 = valid_idx[valid_idx < cutoff_idx_50]\n    if len(valid_idx_out_range_50) > 0:\n        oof_preds_model3_lb[valid_idx_out_range_50] = oof_preds_model1[valid_idx_out_range_50]\n    \n    test_preds_model3_lb += model3_lb.predict(X_test)\n\n    # ========== MODEL 3: 50% MOST RECENT DATA DNN ==========\n    print(\"\\n--- Model 3: Full Data with DNN ---\")\n    \n    # Prepare dataset and loader\n    train_dataset = RegressionDataset(X_train_m1.values, y_train_m1.values)\n    valid_dataset = RegressionDataset(X_valid.values, y_valid.values)\n\n    train_loader = DataLoader(train_dataset, batch_size = 128, shuffle = True)\n    valid_loader = DataLoader(valid_dataset, batch_size = 256, shuffle = False)\n\n    # DNN model init\n    model_dnn = DNNModel(input_dim = X_train_m1.shape[1]).to(DEVICE)\n    optimizer = torch.optim.AdamW(model_dnn.parameters(), lr = 1e-3)\n    criterion = nn.MSELoss()\n\n    # Train DNN\n    for epoch in range(50):\n        model_dnn.train()\n        \n        for xb, yb in train_loader:\n            xb, yb = xb.to(DEVICE), yb.to(DEVICE)\n            optimizer.zero_grad()\n            loss = criterion(model_dnn(xb), yb)\n            loss.backward()\n            optimizer.step()\n    \n    # Predict with DNN\n    model_dnn.eval()\n    with torch.no_grad():\n        # Validation prediction\n        val_preds_dnn = model_dnn(torch.tensor(X_valid.values, dtype = torch.float32).to(DEVICE)).cpu().numpy()\n        oof_preds_dnn[valid_idx] = val_preds_dnn\n\n        # Test prediction\n        test_preds_dnn += model_dnn(torch.tensor(X_test.values, dtype = torch.float32).to(DEVICE)).cpu().numpy()\n\n\n# Average test predictions across folds\ntest_preds_model1 /= FOLDS\ntest_preds_model2 /= FOLDS\ntest_preds_model3 /= FOLDS\ntest_preds_model3_lb /= FOLDS\ntest_preds_dnn /= FOLDS\n\n\n# Calculate individual model scores\npearson_score_model1 = pearsonr(train[\"label\"], oof_preds_model1)[0]\npearson_score_model2 = pearsonr(train[\"label\"], oof_preds_model2)[0]\npearson_score_model3 = pearsonr(train[\"label\"], oof_preds_model3)[0]  \npearson_score_model3_lb = pearsonr(train[\"label\"], oof_preds_model3_lb)[0]  # NEW\npearson_score_model_dnn = pearsonr(train[\"label\"], oof_preds_dnn)[0]\n\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"INDIVIDUAL MODEL PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Model 1 (Full Data) Pearson Correlation: {pearson_score_model1:.4f}\")\nprint(f\"Model 2 (75% Recent) Pearson Correlation: {pearson_score_model2:.4f}\")\nprint(f\"Model 3 XB (50% Recent) Pearson Correlation: {pearson_score_model3:.4f}\")  # NEW\nprint(f\"Model 3 LB (50% Recent) Pearson Correlation: {pearson_score_model3_lb:.4f}\")  # NEW\nprint(f\"Model DNN Pearson Correlation: {pearson_score_model_dnn:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:04:00.274932Z","iopub.execute_input":"2025-07-21T10:04:00.275569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"All trains are done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T11:07:21.232829Z","iopub.execute_input":"2025-07-20T11:07:21.233095Z","iopub.status.idle":"2025-07-20T11:07:21.237392Z","shell.execute_reply.started":"2025-07-20T11:07:21.233067Z","shell.execute_reply":"2025-07-20T11:07:21.236740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create ensemble predictions\n# Simple average ensemble (5 models totla)\nensemble_oof_preds = (oof_preds_model1 + oof_preds_model2 + oof_preds_model3 + oof_preds_model3_lb + oof_preds_dnn) / 5  # UPDATED\nensemble_test_preds = (test_preds_model1 + test_preds_model2 + test_preds_model3 + test_preds_model3_lb + test_preds_dnn) / 5  # UPDATED\n\n# Calculate ensemble score\nensemble_pearson_score = pearsonr(train[\"label\"], ensemble_oof_preds)[0]\n\n\nprint(\"\\n\" + \"=\" * 50)\nprint(\"ENSEMBLE PERFORMANCE\")\nprint(\"=\" * 50)\nprint(f\"Ensemble (Equal Weight) Pearson Correlation: {ensemble_pearson_score:.4f}\")\n\n\ntotal_score = pearson_score_model1 + pearson_score_model2 + pearson_score_model3 + pearson_score_model3_lb + pearson_score_model_dnn # UPDATED\n\nweight_model1 = pearson_score_model1 / total_score\nweight_model2 = pearson_score_model2 / total_score\nweight_model3 = pearson_score_model3 / total_score\nweight_model3_lb = pearson_score_model3_lb / total_score\nweight_model_dnn = pearson_score_model_dnn / total_score\n\n\nweighted_ensemble_oof = (weight_model1 * oof_preds_model1 + \n                        weight_model2 * oof_preds_model2 + \n                        weight_model3 * oof_preds_model3 + \n                        weight_model3_lb * oof_preds_model3_lb +\n                        weight_model_dnn * oof_preds_dnn)\n\nweighted_ensemble_test = (weight_model1 * test_preds_model1 + \n                         weight_model2 * test_preds_model2 + \n                         weight_model3 * test_preds_model3  + \n                         weight_model3_lb * test_preds_model3_lb +\n                         weight_model_dnn * test_preds_dnn)\n\nweighted_ensemble_score = pearsonr(train[\"label\"], weighted_ensemble_oof)[0]\n\nprint(f\"\\nWeighted Ensemble Performance:\")\nprint(f\"  Model 1 weight: {weight_model1:.3f}\")\nprint(f\"  Model 2 weight: {weight_model2:.3f}\")\nprint(f\"  Model 3 weight: {weight_model3:.3f}\")  # NEW\nprint(f\"  Model 3 weight: {weight_model3_lb:.3f}\")  # NEW\nprint(f\"  Model DNN weight: {weight_model_dnn:.3f}\")  # NEW\nprint(f\"  Weighted Ensemble Pearson Correlation: {weighted_ensemble_score:.4f}\")\n\n# Use the better ensemble for final predictions\nif weighted_ensemble_score > ensemble_pearson_score:\n    final_test_preds = weighted_ensemble_test\n    print(\"\\nUsing weighted ensemble for final predictions\")\n    \nelse:\n    final_test_preds = ensemble_test_preds\n    print(\"\\nUsing simple average ensemble for final predictions\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T11:48:40.035834Z","iopub.execute_input":"2025-07-20T11:48:40.036162Z","iopub.status.idle":"2025-07-20T11:48:40.089247Z","shell.execute_reply.started":"2025-07-20T11:48:40.036138Z","shell.execute_reply":"2025-07-20T11:48:40.088369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # SHAP analysis (using Model 1 as representative)\n# print(\"\\nGenerating SHAP analysis for model 1...\")\n# # 1) 학습 데이터(또는 CV의 각 트레이닝 폴드)만으로 explainer 생성\n# explainer = shap.TreeExplainer(model1, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\n# shap_values = explainer.shap_values(X_test)  # 테스트가 아닌 학습 데이터 사용\n# #shap_values = explainer.shap_values(X_train)\n# shap.summary_plot(shap_values, X_test, max_display=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T10:00:18.887701Z","iopub.execute_input":"2025-07-12T10:00:18.888318Z","iopub.status.idle":"2025-07-12T10:02:06.986431Z","shell.execute_reply.started":"2025-07-12T10:00:18.888295Z","shell.execute_reply":"2025-07-12T10:02:06.985712Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- 준비: 평균 절대 SHAP 값 계산 ---\n\n# feature_names = X_train_m1.columns.tolist()\n# mean_abs_shap = np.mean(np.abs(shap_values_train), axis=0)\n# feat_imp = pd.Series(mean_abs_shap, index = feature_names).sort_values(ascending=False)\n\n\n# # --- 1) 퍼센타일(percentile) 방식 ---\n# p = 80  # 예: 상위 25% 컷오프 → 75th percentile\n# threshold_pct = np.percentile(mean_abs_shap, p)\n# selected_pct = feat_imp[feat_imp >= threshold_pct].index.tolist()\n\n# print(f\"퍼센타일 기준 (상위 {100-p}% 피처) 선택 개수: {len(selected_pct)}\")\n# print(selected_pct)\n\n\n# # --- 2) 누적 기여(cumulative contribution) 방식 ---\n# c = 0.80  # 예: 상위 피처들이 전체 기여의 80% 채우기\n# cum_pct = feat_imp.cumsum() / feat_imp.sum()\n# selected_cum = cum_pct[cum_pct <= c].index.tolist()\n\n# print(f\"누적 기여 기준 (전체 기여의 {int(c*100)}%까지) 선택 개수: {len(selected_cum)}\")\n# print(selected_cum)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T09:00:20.495531Z","iopub.execute_input":"2025-07-12T09:00:20.496038Z","iopub.status.idle":"2025-07-12T09:00:21.323266Z","shell.execute_reply.started":"2025-07-12T09:00:20.496016Z","shell.execute_reply":"2025-07-12T09:00:21.322598Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final_selected = list(set(selected_pct) & set(selected_cum))\n# print(f\"두 기준 교집합 선택 개수: {len(final_selected)}\")\n# print(final_selected)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T09:00:25.845988Z","iopub.execute_input":"2025-07-12T09:00:25.846640Z","iopub.status.idle":"2025-07-12T09:00:25.851525Z","shell.execute_reply.started":"2025-07-12T09:00:25.846610Z","shell.execute_reply":"2025-07-12T09:00:25.850877Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # SHAP analysis (using Model 1 as representative)\n# print(\"\\nGenerating SHAP analysis for model 2...\")\n# explainer = shap.TreeExplainer(model2, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\n# shap_values = explainer.shap_values(X_test)\n# shap.summary_plot(shap_values, X_test, max_display=30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T08:12:35.843404Z","iopub.execute_input":"2025-07-12T08:12:35.843667Z","iopub.status.idle":"2025-07-12T08:14:51.011793Z","shell.execute_reply.started":"2025-07-12T08:12:35.843645Z","shell.execute_reply":"2025-07-12T08:14:51.010851Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save predictions\nsample[\"prediction\"] = final_test_preds\nsample.to_csv(\"submission.csv\", index = False)\nprint(\"\\nPredictions saved to submission.csv\")\nprint(sample.head(20))\n\n# Save detailed results (now with 3 models)\nensemble_results = pd.DataFrame({\n    'model': ['Model 1 (Full Data)', 'Model 2 (75% Recent)', 'Model 3 (50% Recent)', \n              'Model 4 (DNN - 50% Recent)', 'Simple Ensemble', 'Weighted Ensemble'],\n    'pearson_correlation': [pearson_score_model1, pearson_score_model2, pearson_score_model3, pearson_score_model_dnn,\n                           ensemble_pearson_score, weighted_ensemble_score],\n    'weight_in_final': [weight_model1 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        weight_model2 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        weight_model3 if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        weight_model_dnn if weighted_ensemble_score > ensemble_pearson_score else 1/3,\n                        np.nan, np.nan]  # UPDATED\n})\n\nensemble_results.to_csv(\"ensemble_results.csv\", index = False)\n\nprint(\"\\nEnsemble results saved to ensemble_results.csv\")\nprint(ensemble_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-20T11:49:20.134818Z","iopub.execute_input":"2025-07-20T11:49:20.135576Z","iopub.status.idle":"2025-07-20T11:49:21.199475Z","shell.execute_reply.started":"2025-07-20T11:49:20.135550Z","shell.execute_reply":"2025-07-20T11:49:21.198780Z"}},"outputs":[],"execution_count":null}]}