{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom scipy.stats import pearsonr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:10:04.454085Z","iopub.execute_input":"2025-06-25T09:10:04.454785Z","iopub.status.idle":"2025-06-25T09:10:05.569194Z","shell.execute_reply.started":"2025-06-25T09:10:04.454757Z","shell.execute_reply":"2025-06-25T09:10:05.568256Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: Load the Data","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\ntest_path = '/kaggle/input/drw-crypto-market-prediction/test.parquet'\n\ntrain = pd.read_parquet(train_path)\ntest = pd.read_parquet(test_path)\n\nprint(\"Train shape:\", train.shape)\nprint(\"Test shape:\", test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:10:14.207703Z","iopub.execute_input":"2025-06-25T09:10:14.208214Z","iopub.status.idle":"2025-06-25T09:11:13.359773Z","shell.execute_reply.started":"2025-06-25T09:10:14.208186Z","shell.execute_reply":"2025-06-25T09:11:13.358826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:16:02.261087Z","iopub.execute_input":"2025-06-25T09:16:02.262159Z","iopub.status.idle":"2025-06-25T09:16:02.294502Z","shell.execute_reply.started":"2025-06-25T09:16:02.262127Z","shell.execute_reply":"2025-06-25T09:16:02.293611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:18:34.561550Z","iopub.execute_input":"2025-06-25T09:18:34.561886Z","iopub.status.idle":"2025-06-25T09:18:34.581078Z","shell.execute_reply.started":"2025-06-25T09:18:34.561860Z","shell.execute_reply":"2025-06-25T09:18:34.580295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(list(test.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:18:38.547644Z","iopub.execute_input":"2025-06-25T09:18:38.547992Z","iopub.status.idle":"2025-06-25T09:18:38.552871Z","shell.execute_reply.started":"2025-06-25T09:18:38.547943Z","shell.execute_reply":"2025-06-25T09:18:38.552024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Quick Exploration","metadata":{}},{"cell_type":"code","source":"print(train.head())\nprint(train.columns)\n\n# Check target distribution\nprint(train['label'].describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:11:13.361303Z","iopub.execute_input":"2025-06-25T09:11:13.361769Z","iopub.status.idle":"2025-06-25T09:11:13.414791Z","shell.execute_reply.started":"2025-06-25T09:11:13.361741Z","shell.execute_reply":"2025-06-25T09:11:13.414090Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## What Do I Think About It?\n\n* Reasonable target column: Log-return makes sense for a financial task.\n\n* Outliers are extreme: A few timestamps have massive jumps — this could:\n\n* Skew the model\n\n* Be artifacts (e.g. from corrupted features or divisions by zero → inf)\n\n* Modeling tip:\n\n  * Consider clipping label values to remove extreme outliers. Example:\n\n  * Many top competitors clip to ±5 or ±10 to reduce noise and improve performance.\n\n","metadata":{}},{"cell_type":"markdown","source":"# Step 4: Check for problematic values","metadata":{}},{"cell_type":"code","source":"print(\"Any NaNs?\", train.isna().sum().sum(), test.isna().sum().sum())\nprint(\"Any Infs?\", np.isinf(train.to_numpy()).sum(), np.isinf(test.to_numpy()).sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:11:28.807355Z","iopub.execute_input":"2025-06-25T09:11:28.807653Z","iopub.status.idle":"2025-06-25T09:11:36.150703Z","shell.execute_reply.started":"2025-06-25T09:11:28.807628Z","shell.execute_reply":"2025-06-25T09:11:36.149895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Those are huge numbers of inf values:\n\n* 11+ million infs in train\n\n* 11+ million infs in test\n\nThat’s a very significant portion of the data and likely coming from specific features.","metadata":{}},{"cell_type":"markdown","source":"### Why This Is a Problem\n\n* These values will break most models, or make them behave unpredictably.\n\n* If the infs come from only a few columns, it’s best to drop those columns entirely.\n\n* If they are spread everywhere, you’ll need to decide whether to fill or clip them.","metadata":{}},{"cell_type":"code","source":"# Count inf values per column\ninf_counts = pd.DataFrame({\n    'train_inf': np.isinf(train).sum(),\n    'test_inf': np.isinf(test).sum()\n})\n\n# Filter only columns with inf values\ninf_cols = inf_counts[(inf_counts['train_inf'] > 0) | (inf_counts['test_inf'] > 0)]\nprint(\"Columns with inf values:\")\nprint(inf_cols.sort_values(by='train_inf', ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:14:20.955512Z","iopub.execute_input":"2025-06-25T09:14:20.955808Z","iopub.status.idle":"2025-06-25T09:14:25.863845Z","shell.execute_reply.started":"2025-06-25T09:14:20.955787Z","shell.execute_reply":"2025-06-25T09:14:25.863002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## What I Found?\n\nThe following 21 features (X697 through X717) contain only inf values in every row:\n\n* train_inf = 525,887 (every row in train)\n\n* test_inf = 538,150 (every row in test)\n\nThis means:\n\n* These columns are completely unusable — they carry no real signal, just garbage (infinite values).\n\n* They will crash or severely bias your model if not removed.","metadata":{}},{"cell_type":"markdown","source":"# Step 5: Prepare Features and Target","metadata":{}},{"cell_type":"code","source":"# Step 2: Prepare cleaned features and target from train\nX = train.drop(columns=['label'] + list(inf_cols.index))\ny = train['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:14:10.411831Z","iopub.execute_input":"2025-06-25T09:14:10.412692Z","iopub.status.idle":"2025-06-25T09:14:13.757003Z","shell.execute_reply.started":"2025-06-25T09:14:10.412633Z","shell.execute_reply":"2025-06-25T09:14:13.756045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Prepare cleaned test features\nX_test = test.drop(columns = ['label'] + list(inf_cols.index))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:28:48.246706Z","iopub.execute_input":"2025-06-25T09:28:48.247087Z","iopub.status.idle":"2025-06-25T09:28:55.411118Z","shell.execute_reply.started":"2025-06-25T09:28:48.247062Z","shell.execute_reply":"2025-06-25T09:28:55.410193Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 6: Train on full training data","metadata":{}},{"cell_type":"code","source":"final_model = Ridge()\nfinal_model.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:22:25.657741Z","iopub.execute_input":"2025-06-25T09:22:25.658148Z","iopub.status.idle":"2025-06-25T09:22:43.225716Z","shell.execute_reply.started":"2025-06-25T09:22:25.658120Z","shell.execute_reply":"2025-06-25T09:22:43.224860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:23:44.672754Z","iopub.execute_input":"2025-06-25T09:23:44.673363Z","iopub.status.idle":"2025-06-25T09:23:44.678701Z","shell.execute_reply.started":"2025-06-25T09:23:44.673336Z","shell.execute_reply":"2025-06-25T09:23:44.677799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:23:53.758103Z","iopub.execute_input":"2025-06-25T09:23:53.758423Z","iopub.status.idle":"2025-06-25T09:23:53.763892Z","shell.execute_reply.started":"2025-06-25T09:23:53.758399Z","shell.execute_reply":"2025-06-25T09:23:53.763019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Columns in X but not in X_test\ndiff_in_train = set(X.columns) - set(X_test.columns)\n\n# Columns in X_test but not in X\ndiff_in_test = set(X_test.columns) - set(X.columns)\n\nprint(\"Columns in train but missing in test:\", diff_in_train)\nprint(\"Columns in test but missing in train:\", diff_in_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:28:59.369095Z","iopub.execute_input":"2025-06-25T09:28:59.369937Z","iopub.status.idle":"2025-06-25T09:28:59.375848Z","shell.execute_reply.started":"2025-06-25T09:28:59.369905Z","shell.execute_reply":"2025-06-25T09:28:59.374603Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 7: Predict on test data","metadata":{}},{"cell_type":"code","source":"test_preds = final_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:29:03.694792Z","iopub.execute_input":"2025-06-25T09:29:03.695144Z","iopub.status.idle":"2025-06-25T09:29:04.310861Z","shell.execute_reply.started":"2025-06-25T09:29:03.695119Z","shell.execute_reply":"2025-06-25T09:29:04.310129Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 8: Create submission file","metadata":{}},{"cell_type":"code","source":"# Generate row IDs (starting at 1)\nrow_ids = range(1, len(test_preds) + 1)\n\n# Create submission DataFrame\nsubmission = pd.DataFrame({\n    'ID': row_ids,\n    'prediction': test_preds\n})\n\n# Save to CSV without index\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"✅ Submission file saved: submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T09:33:51.488907Z","iopub.execute_input":"2025-06-25T09:33:51.489352Z","iopub.status.idle":"2025-06-25T09:33:52.758703Z","shell.execute_reply.started":"2025-06-25T09:33:51.489316Z","shell.execute_reply":"2025-06-25T09:33:52.757842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}