{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:38:58.272195Z","iopub.execute_input":"2025-06-26T05:38:58.272515Z","iopub.status.idle":"2025-06-26T05:38:59.340238Z","shell.execute_reply.started":"2025-06-26T05:38:58.272489Z","shell.execute_reply":"2025-06-26T05:38:59.339385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\n\n# Load data\ntrain = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\nprint('done')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:38:59.341796Z","iopub.execute_input":"2025-06-26T05:38:59.342270Z","iopub.status.idle":"2025-06-26T05:39:48.940087Z","shell.execute_reply.started":"2025-06-26T05:38:59.342244Z","shell.execute_reply":"2025-06-26T05:39:48.939113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop columns with all NaNs\ntrain = train.dropna(axis=1, how='all')\n\n# Drop rows with any NaNs\ntrain = train.dropna(axis=0, how='any')\n\n# Drop same columns in test as in train\ntest = test[train.drop(columns=[\"label\"]).columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:39:48.941332Z","iopub.execute_input":"2025-06-26T05:39:48.941767Z","iopub.status.idle":"2025-06-26T05:39:58.082427Z","shell.execute_reply.started":"2025-06-26T05:39:48.941743Z","shell.execute_reply":"2025-06-26T05:39:58.081294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_all = train.drop(columns=[\"label\"])\ny = train[\"label\"]\n# Compute correlation of each column with the target (faster than .corr())\ncorrelations = X_all.corrwith(y).abs().sort_values(ascending=False)\n\n# Select top 30 features\ntop_features = correlations.head(500).index\n\n# Final training and test sets\nX = train[top_features]\ny = train[\"label\"]\nX_test = test[top_features]\n\nprint('done')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:39:58.084435Z","iopub.execute_input":"2025-06-26T05:39:58.084763Z","iopub.status.idle":"2025-06-26T05:40:10.932171Z","shell.execute_reply.started":"2025-06-26T05:39:58.084736Z","shell.execute_reply":"2025-06-26T05:40:10.931186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train-validation split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:40:10.933029Z","iopub.execute_input":"2025-06-26T05:40:10.933305Z","iopub.status.idle":"2025-06-26T05:40:15.088299Z","shell.execute_reply.started":"2025-06-26T05:40:10.933278Z","shell.execute_reply":"2025-06-26T05:40:15.087405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import RidgeCV\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\n\nalphas = [0.0001, 0.001, 0.01, 0.1, 1, 10, 100]\n\nmodel = make_pipeline(\n    StandardScaler(),\n    RidgeCV(alphas=alphas, scoring='neg_mean_squared_error', cv=5)\n)\n\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:40:15.089406Z","iopub.execute_input":"2025-06-26T05:40:15.089669Z","iopub.status.idle":"2025-06-26T05:42:39.974323Z","shell.execute_reply.started":"2025-06-26T05:40:15.089648Z","shell.execute_reply":"2025-06-26T05:42:39.973410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate\ny_pred = model.predict(X_val)\nmse = mean_squared_error(y_val, y_pred)\nbest_alpha = model.named_steps['ridgecv'].alpha_\nprint(f\"✅ Validation MSE: {mse:.4f}\")\nprint(f\"🔍 Best Alpha: {best_alpha}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:42:39.975041Z","iopub.execute_input":"2025-06-26T05:42:39.975620Z","iopub.status.idle":"2025-06-26T05:42:40.439550Z","shell.execute_reply.started":"2025-06-26T05:42:39.975590Z","shell.execute_reply":"2025-06-26T05:42:40.438532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import pearsonr\n\n# y_val: actual target values from validation set\n# y_pred: predicted target values from Ridge model\n\npearson_corr, _ = pearsonr(y_val, y_pred)\nprint(f\"📈 Pearson Correlation (Predictions vs Actual): {pearson_corr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:42:40.440488Z","iopub.execute_input":"2025-06-26T05:42:40.440905Z","iopub.status.idle":"2025-06-26T05:42:40.455956Z","shell.execute_reply.started":"2025-06-26T05:42:40.440872Z","shell.execute_reply":"2025-06-26T05:42:40.455054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nsubmission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\nsubmission[\"prediction\"] = model.predict(X_test)\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(\"📁 Submission file saved as 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-26T05:44:16.573179Z","iopub.execute_input":"2025-06-26T05:44:16.573526Z","iopub.status.idle":"2025-06-26T05:44:20.000441Z","shell.execute_reply.started":"2025-06-26T05:44:16.573502Z","shell.execute_reply":"2025-06-26T05:44:19.999465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}