{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nimport numpy as np\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:56:08.220336Z","iopub.execute_input":"2025-07-08T04:56:08.222335Z","iopub.status.idle":"2025-07-08T04:56:11.239637Z","shell.execute_reply.started":"2025-07-08T04:56:08.222276Z","shell.execute_reply":"2025-07-08T04:56:11.238580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ntrain = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\nprint('done')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:56:11.241357Z","iopub.execute_input":"2025-07-08T04:56:11.241848Z","iopub.status.idle":"2025-07-08T04:57:10.185473Z","shell.execute_reply.started":"2025-07-08T04:56:11.241821Z","shell.execute_reply":"2025-07-08T04:57:10.184435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:57:10.186407Z","iopub.execute_input":"2025-07-08T04:57:10.186703Z","iopub.status.idle":"2025-07-08T04:57:10.214618Z","shell.execute_reply.started":"2025-07-08T04:57:10.186679Z","shell.execute_reply":"2025-07-08T04:57:10.213712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def drop_inf_columns(df):\n    \"\"\"Drop columns from DataFrame if any value is inf or -inf in that column.\"\"\"\n    return df.loc[:, ~df.isin([float('inf'), float('-inf')]).any()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:57:10.216387Z","iopub.execute_input":"2025-07-08T04:57:10.216777Z","iopub.status.idle":"2025-07-08T04:57:10.222366Z","shell.execute_reply.started":"2025-07-08T04:57:10.216753Z","shell.execute_reply":"2025-07-08T04:57:10.221501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = drop_inf_columns(train)\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:57:10.223353Z","iopub.execute_input":"2025-07-08T04:57:10.223960Z","iopub.status.idle":"2025-07-08T04:58:08.525732Z","shell.execute_reply.started":"2025-07-08T04:57:10.223930Z","shell.execute_reply":"2025-07-08T04:58:08.524843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop columns with all NaNs\ntrain = train.dropna(axis=1, how='all')\n\n# Drop rows with any NaNs\ntrain = train.dropna(axis=0, how='any')\n\n# Drop same columns in test as in train\ntest = test[train.drop(columns=[\"label\"]).columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:58:08.526674Z","iopub.execute_input":"2025-07-08T04:58:08.526986Z","iopub.status.idle":"2025-07-08T04:58:16.591295Z","shell.execute_reply.started":"2025-07-08T04:58:08.526952Z","shell.execute_reply":"2025-07-08T04:58:16.590311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop(columns = ['label'])\ny = train['label']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:03:58.242112Z","iopub.execute_input":"2025-07-08T05:03:58.242457Z","iopub.status.idle":"2025-07-08T05:03:58.256606Z","shell.execute_reply.started":"2025-07-08T05:03:58.242434Z","shell.execute_reply":"2025-07-08T05:03:58.255255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_fit, X_val, y_fit, y_val = train_test_split(X, y, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:58:17.753805Z","iopub.execute_input":"2025-07-08T04:58:17.754128Z","iopub.status.idle":"2025-07-08T04:58:21.295146Z","shell.execute_reply.started":"2025-07-08T04:58:17.754102Z","shell.execute_reply":"2025-07-08T04:58:21.294213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ndel X,y,train\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:05:38.644568Z","iopub.execute_input":"2025-07-08T05:05:38.645014Z","iopub.status.idle":"2025-07-08T05:05:38.659435Z","shell.execute_reply.started":"2025-07-08T05:05:38.644986Z","shell.execute_reply":"2025-07-08T05:05:38.658260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = drop_inf_columns(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:05:59.197243Z","iopub.execute_input":"2025-07-08T05:05:59.197577Z","iopub.status.idle":"2025-07-08T05:07:00.299244Z","shell.execute_reply.started":"2025-07-08T05:05:59.197548Z","shell.execute_reply":"2025-07-08T05:07:00.298225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # X_all = train.drop(columns=[\"label\"])\n# # y = train[\"label\"]\n# # # Compute correlation of each column with the target (faster than .corr())\n# # correlations = X_all.corrwith(y).abs().sort_values(ascending=False)\n\n# # top_features = correlations.index\n\n# # Final training and test sets\n# X = traintrain.drop(columns=[\"label\"])\n# y = train[\"label\"]\n# X_test = test[train.columns]\n\n# print('done')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:58:21.422074Z","iopub.execute_input":"2025-07-08T04:58:21.422810Z","iopub.status.idle":"2025-07-08T04:58:21.426994Z","shell.execute_reply.started":"2025-07-08T04:58:21.422780Z","shell.execute_reply":"2025-07-08T04:58:21.425993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Train-validation split\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # Split data\n# X_fit, X_val, y_fit, y_val = train_test_split(X_train, y_train, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:58:21.428087Z","iopub.execute_input":"2025-07-08T04:58:21.428453Z","iopub.status.idle":"2025-07-08T04:58:21.445333Z","shell.execute_reply.started":"2025-07-08T04:58:21.428422Z","shell.execute_reply":"2025-07-08T04:58:21.444201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import RidgeCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import pearsonr\nimport numpy as np\n\ntop_features = X_fit.shape[1]\nmodels = []\nalphas = [1e-4, 1e-3, 1e-2, 1e-1, 1, 10]\n\n# Train per-feature RidgeCV\nfor i in range(top_features):\n    model = make_pipeline(\n        StandardScaler(),\n        RidgeCV(alphas=alphas, scoring='neg_mean_squared_error', cv=5)\n    )\n    Xi = X_fit.iloc[:, [i]]     # ✅ Retain feature name\n    model.fit(Xi, y_fit)\n    models.append(model)\n\n# Predict on validation set\npreds = np.zeros((X_val.shape[0], top_features))\nfor i, model in enumerate(models):\n    Xi_val = X_val.iloc[:, [i]]\n    preds[:, i] = model.predict(Xi_val)\n\n# Average predictions\ny_pred = preds.mean(axis=1)\n\n# Pearson correlation\ncorr, _ = pearsonr(y_pred, y_val)\nprint(f\"📈 Pearson Correlation on Validation Set: {corr:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T04:58:21.446493Z","iopub.execute_input":"2025-07-08T04:58:21.446843Z","iopub.status.idle":"2025-07-08T05:03:04.211915Z","shell.execute_reply.started":"2025-07-08T04:58:21.446812Z","shell.execute_reply":"2025-07-08T05:03:04.210909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = np.zeros((test.shape[0], top_features))\n\nfor i, model in enumerate(models):\n    i_test = X_test.iloc[:, [i]]  \n    preds[:, i] = model.predict(i_test)\n\ny_pred = preds.mean(axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:07:00.300689Z","iopub.execute_input":"2025-07-08T05:07:00.301054Z","iopub.status.idle":"2025-07-08T05:07:23.750053Z","shell.execute_reply.started":"2025-07-08T05:07:00.301025Z","shell.execute_reply":"2025-07-08T05:07:23.749310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsubmission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\nsubmission[\"prediction\"] = y_pred\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"📁 Submission file saved as 'submission.csv'\")\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:07:35.925478Z","iopub.execute_input":"2025-07-08T05:07:35.925844Z","iopub.status.idle":"2025-07-08T05:07:37.619935Z","shell.execute_reply.started":"2025-07-08T05:07:35.925817Z","shell.execute_reply":"2025-07-08T05:07:37.619027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.metrics import r2_score\n\n# feature_scores = []\n# for i, model in enumerate(models):\n#     Xi_val = X_val.iloc[:, [i]]\n#     y_pred_i = model.predict(Xi_val)\n#     score = r2_score(y_val, y_pred_i)\n#     feature_scores.append((X_val.columns[i], score))\n\n# # Sort by R²\n# top_features = sorted(feature_scores, key=lambda x: x[1], reverse=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:07:48.317014Z","iopub.execute_input":"2025-07-08T05:07:48.317369Z","iopub.status.idle":"2025-07-08T05:07:52.475680Z","shell.execute_reply.started":"2025-07-08T05:07:48.317342Z","shell.execute_reply":"2025-07-08T05:07:52.474614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Top 20 features by R² score:\")\n# for name, score in top_features[:20]:\n#     print(f\"{name}: {score:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:07:55.241619Z","iopub.execute_input":"2025-07-08T05:07:55.241989Z","iopub.status.idle":"2025-07-08T05:07:55.247577Z","shell.execute_reply.started":"2025-07-08T05:07:55.241960Z","shell.execute_reply":"2025-07-08T05:07:55.246399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from scipy.stats import pearsonr\n\n# feature_scores = []\n# for i, model in enumerate(models):\n#     Xi_val = X_val.iloc[:, [i]]\n#     y_pred_i = model.predict(Xi_val)\n#     score, _ = pearsonr(y_val, y_pred_i)\n#     feature_scores.append((X_val.columns[i], score))\n\n# # Sort by absolute Pearson\n# top_features = sorted(feature_scores, key=lambda x: abs(x[1]), reverse=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:07:55.730292Z","iopub.execute_input":"2025-07-08T05:07:55.731046Z","iopub.status.idle":"2025-07-08T05:08:01.031816Z","shell.execute_reply.started":"2025-07-08T05:07:55.731013Z","shell.execute_reply":"2025-07-08T05:08:01.030767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Top 20 features by corr score:\")\n# for name, score in top_features[:20]:\n#     print(f\"{name}: {score:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:08:01.033587Z","iopub.execute_input":"2025-07-08T05:08:01.033968Z","iopub.status.idle":"2025-07-08T05:08:01.039679Z","shell.execute_reply.started":"2025-07-08T05:08:01.033937Z","shell.execute_reply":"2025-07-08T05:08:01.038883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.metrics import mean_squared_error\n\n# feature_scores = []\n# for i, model in enumerate(models):\n#     Xi_val = X_val.iloc[:, [i]]\n#     y_pred_i = model.predict(Xi_val)\n#     mse = mean_squared_error(y_val, y_pred_i)\n#     feature_scores.append((X_val.columns[i], mse))\n\n# # Sort by ascending MSE (lower is better)\n# top_features = sorted(feature_scores, key=lambda x: x[1])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:08:01.040468Z","iopub.execute_input":"2025-07-08T05:08:01.040705Z","iopub.status.idle":"2025-07-08T05:08:04.312007Z","shell.execute_reply.started":"2025-07-08T05:08:01.040687Z","shell.execute_reply":"2025-07-08T05:08:04.311191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Top 20 features by corr score:\")\n# for name, score in top_features[:20]:\n#     print(f\"{name}: {score:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T05:08:04.313373Z","iopub.execute_input":"2025-07-08T05:08:04.313816Z","iopub.status.idle":"2025-07-08T05:08:04.318550Z","shell.execute_reply.started":"2025-07-08T05:08:04.313786Z","shell.execute_reply":"2025-07-08T05:08:04.317663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}