{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Step 1: Import Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pandas.plotting import autocorrelation_plot\n\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import TimeSeriesSplit, train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\nfrom sklearn.metrics import r2_score\n\nfrom scipy.stats import pearsonr\n\nimport xgboost as xgb\nfrom xgboost import plot_importance\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:48:19.081936Z","iopub.execute_input":"2025-06-28T10:48:19.082234Z","iopub.status.idle":"2025-06-28T10:48:21.244483Z","shell.execute_reply.started":"2025-06-28T10:48:19.082205Z","shell.execute_reply":"2025-06-28T10:48:21.243320Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: Load the Data","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\ntest_path = '/kaggle/input/drw-crypto-market-prediction/test.parquet'\n\ntrain_df = pd.read_parquet(train_path)\ntest_df = pd.read_parquet(test_path)\n\nprint(\"Train shape:\", train_df.shape)\nprint(\"Test shape:\", test_df.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:48:21.245737Z","iopub.execute_input":"2025-06-28T10:48:21.246319Z","iopub.status.idle":"2025-06-28T10:49:15.183501Z","shell.execute_reply.started":"2025-06-28T10:48:21.246291Z","shell.execute_reply":"2025-06-28T10:49:15.182753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ No conversion needed, just sort by index (already datetime)\ntrain_df = train_df.sort_index()\nprint(train_df.index[:5])  # Confirm it's datetime","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:15.186152Z","iopub.execute_input":"2025-06-28T10:49:15.186419Z","iopub.status.idle":"2025-06-28T10:49:17.670167Z","shell.execute_reply.started":"2025-06-28T10:49:15.186398Z","shell.execute_reply":"2025-06-28T10:49:17.669230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.head())\nprint(train_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:17.673634Z","iopub.execute_input":"2025-06-28T10:49:17.673891Z","iopub.status.idle":"2025-06-28T10:49:17.705523Z","shell.execute_reply.started":"2025-06-28T10:49:17.673872Z","shell.execute_reply":"2025-06-28T10:49:17.704541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: EDA","metadata":{}},{"cell_type":"code","source":"# EDA: Missing values\nprint(train_df.isnull().sum())\nprint(test_df.isnull().sum())\n\n# EDA: Inf values\nprint(\"Inf values in train:\")\nprint(np.isinf(train_df).sum())\n\nprint(\"Inf values in test:\")\nprint(np.isinf(test_df).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:17.706536Z","iopub.execute_input":"2025-06-28T10:49:17.706855Z","iopub.status.idle":"2025-06-28T10:49:25.311479Z","shell.execute_reply.started":"2025-06-28T10:49:17.706832Z","shell.execute_reply":"2025-06-28T10:49:25.310422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df['label'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:25.312761Z","iopub.execute_input":"2025-06-28T10:49:25.313856Z","iopub.status.idle":"2025-06-28T10:49:25.353907Z","shell.execute_reply.started":"2025-06-28T10:49:25.313826Z","shell.execute_reply":"2025-06-28T10:49:25.352937Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What Do I Think About It?\n\n* Centered around 0: The median is ~0.016 and the mean is ~0.036 — very close to 0.\n\n* Highly concentrated: 50% of the data lies between -0.38 and 0.43 — a narrow range.\n\n* Some extreme outliers: Min = -24.4 and Max = 20.7. These are far from the center, but rare (long tails).\n\n* Standard deviation ~1: This suggests that most values are within [-1, 1], consistent with the histogram being peaked around 0.\n\n","metadata":{}},{"cell_type":"code","source":"print(train_df['label'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:25.354970Z","iopub.execute_input":"2025-06-28T10:49:25.355977Z","iopub.status.idle":"2025-06-28T10:49:25.506268Z","shell.execute_reply.started":"2025-06-28T10:49:25.355952Z","shell.execute_reply":"2025-06-28T10:49:25.505381Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Interpretation of value_counts():\n\n* Every value in the 'label' column appears only once.\n\n* The total number of unique values is exactly the same as the number of rows: 525,887 unique values.\n\n* So this is purely continuous data with no duplicates at all.","metadata":{}},{"cell_type":"markdown","source":"### What This Means:\n\n* Your target variable is not categorical, not ordinal, and not rounded — it's a real-valued continuous variable (likely regression).\n\n* This further confirms that the histogram peak near 0 is not due to repeated values like many zeros, but due to density: more values are close to 0, fewer are in the tails.\n\n* This is typical in financial data (like log returns), residuals, or normalized scores.","metadata":{}},{"cell_type":"code","source":"print(train_df['label'].head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:25.510260Z","iopub.execute_input":"2025-06-28T10:49:25.510538Z","iopub.status.idle":"2025-06-28T10:49:25.517219Z","shell.execute_reply.started":"2025-06-28T10:49:25.510517Z","shell.execute_reply":"2025-06-28T10:49:25.516321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Any NaNs?\", train_df.isna().sum().sum(), test_df.isna().sum().sum())\nprint(\"Any Infs?\", np.isinf(train_df.to_numpy()).sum(), np.isinf(test_df.to_numpy()).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:25.517998Z","iopub.execute_input":"2025-06-28T10:49:25.518289Z","iopub.status.idle":"2025-06-28T10:49:33.095432Z","shell.execute_reply.started":"2025-06-28T10:49:25.518261Z","shell.execute_reply":"2025-06-28T10:49:33.094389Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Those are huge numbers of inf values:\n\n* 11+ million infs in train\n\n* 11+ million infs in test\n\nThat’s a very significant portion of the data and likely coming from specific features.","metadata":{}},{"cell_type":"markdown","source":"### Why This Is a Problem\n\n* These values will break most models, or make them behave unpredictably.\n\n* If the infs come from only a few columns, it’s best to drop those columns entirely.\n\n* If they are spread everywhere, you’ll need to decide whether to fill or clip them.","metadata":{}},{"cell_type":"code","source":"# Count inf values per column\ninf_counts = pd.DataFrame({\n    'train_inf': np.isinf(train_df).sum(axis=0),\n    'test_inf': np.isinf(test_df).sum(axis=0)\n})\n\n# Filter only columns with inf values\ninf_cols = inf_counts[(inf_counts['train_inf'] > 0) | (inf_counts['test_inf'] > 0)]\nprint(\"Columns with inf values:\")\nprint(inf_cols.sort_values(by='train_inf', ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:33.096531Z","iopub.execute_input":"2025-06-28T10:49:33.096792Z","iopub.status.idle":"2025-06-28T10:49:37.217235Z","shell.execute_reply.started":"2025-06-28T10:49:33.096772Z","shell.execute_reply":"2025-06-28T10:49:37.216380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## What I Found?\n\nThe following 21 features (X697 through X717) contain only inf values in every row:\n\n* train_inf = 525,887 (every row in train)\n\n* test_inf = 538,150 (every row in test)\n\nThis means:\n\n* These columns are completely unusable — they carry no real signal, just garbage (infinite values).\n\n* They will crash or severely bias your model if not removed.","metadata":{}},{"cell_type":"code","source":"# Drop columns with inf values from both train and test\ncols_to_drop = list(inf_cols.index)\ntrain_df = train_df.drop(columns=cols_to_drop)\ntest_df = test_df.drop(columns=cols_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:37.218526Z","iopub.execute_input":"2025-06-28T10:49:37.218857Z","iopub.status.idle":"2025-06-28T10:49:41.940234Z","shell.execute_reply.started":"2025-06-28T10:49:37.218826Z","shell.execute_reply":"2025-06-28T10:49:41.939181Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Check for Outliers","metadata":{}},{"cell_type":"code","source":"# Count how many values are outside [-10, 10]\noutliers = (train_df['label'].abs() > 10).sum()\nprint(f\"Number of extreme outliers (|label| > 10): {outliers}\")\n\n# View a few of them\nprint(train_df[train_df['label'].abs() > 10].sort_values(by='label'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:41.941385Z","iopub.execute_input":"2025-06-28T10:49:41.941624Z","iopub.status.idle":"2025-06-28T10:49:41.970198Z","shell.execute_reply.started":"2025-06-28T10:49:41.941606Z","shell.execute_reply":"2025-06-28T10:49:41.969424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Outlier Summary:\n\n* There is 126 outliers with |label| > 10, out of 525,887 rows.\n\n* That’s ~0.024% of the data — very rare, but they are extreme (as low as -24.4 and as high as +20.7).\n\n* These outliers might:\n\n    * Skew our loss function (especially MSE or linear regression).\n\n    * Indicate rare but important events (e.g., market shocks).\n\n    * Require special handling like clipping, log transform, or robust loss functions (like Huber).","metadata":{}},{"cell_type":"markdown","source":"## Plot the Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 4))\nsns.histplot(train_df['label'], bins=100, kde=True)\nplt.title(\"Distribution of Target Variable (label)\")\nplt.xlabel(\"label\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:41.971120Z","iopub.execute_input":"2025-06-28T10:49:41.971409Z","iopub.status.idle":"2025-06-28T10:49:44.872000Z","shell.execute_reply.started":"2025-06-28T10:49:41.971390Z","shell.execute_reply":"2025-06-28T10:49:44.870992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plotting the target over time to see trends, cycles, or spikes visually","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\ntrain_df['label'].plot()\nplt.title(\"Label Over Time\")\nplt.xlabel(\"Timestamp\")\nplt.ylabel(\"label\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:44.873158Z","iopub.execute_input":"2025-06-28T10:49:44.873390Z","iopub.status.idle":"2025-06-28T10:49:46.333526Z","shell.execute_reply.started":"2025-06-28T10:49:44.873373Z","shell.execute_reply":"2025-06-28T10:49:46.332457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What those spikes mean:\n\n* The target usually hovers near zero.\n\n* Occasionally, there are sharp spikes (big positive or negative jumps).\n\n* Those spikes could represent rare events, anomalies, or important signals that might heavily influence your model.","metadata":{}},{"cell_type":"markdown","source":"### Smooth the series to see trends better","metadata":{}},{"cell_type":"code","source":"train_df['label'].rolling(window=60).mean().plot(figsize=(15,5))\nplt.title(\"Rolling Mean of Label (60-minute window)\")\nplt.xlabel(\"Timestamp\")\nplt.ylabel(\"Smoothed label\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:46.334470Z","iopub.execute_input":"2025-06-28T10:49:46.334811Z","iopub.status.idle":"2025-06-28T10:49:47.876408Z","shell.execute_reply.started":"2025-06-28T10:49:46.334779Z","shell.execute_reply":"2025-06-28T10:49:47.875416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Autocorrelation plot\n\n* We ran autocorrelation plot to see if the target values relate to past values.\n\n* This tells you if there’s a time dependency (important for time series).","metadata":{}},{"cell_type":"code","source":"lags = list(range(1, 11))\nautocorrs = [train_df['label'].autocorr(lag=lag) for lag in lags]\n\nplt.figure(figsize=(8,4))\nplt.bar(lags, autocorrs)\nplt.xlabel('Lag')\nplt.ylabel('Autocorrelation')\nplt.title('Autocorrelation by Lag (1 to 10)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:47.877396Z","iopub.execute_input":"2025-06-28T10:49:47.877620Z","iopub.status.idle":"2025-06-28T10:49:48.211096Z","shell.execute_reply.started":"2025-06-28T10:49:47.877603Z","shell.execute_reply":"2025-06-28T10:49:48.210220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for lag in range(1, 11):\n    autocorr = train_df['label'].autocorr(lag=lag)\n    print(f\"Autocorrelation at lag {lag}: {autocorr:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:48.211994Z","iopub.execute_input":"2025-06-28T10:49:48.212250Z","iopub.status.idle":"2025-06-28T10:49:48.369819Z","shell.execute_reply.started":"2025-06-28T10:49:48.212231Z","shell.execute_reply":"2025-06-28T10:49:48.368751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What the numbers mean ?\n\n* Lag 1 autocorrelation ~ 0.98 is very high.\n\n* Even at lag 10, the autocorrelation is still around 0.84 — that’s a strong positive correlation.\n\n* This means your target variable is highly dependent on its recent past values — very typical for smooth, slowly changing time series.\n\n* The target is not white noise at all.\n\n* It’s very auto-correlated — past values strongly influence future ones.\n\n* Models that capture temporal dependencies (like ARIMA, LSTM, or feature-engineered lags) might perform well.","metadata":{}},{"cell_type":"markdown","source":"## Check Skewness of Your Target","metadata":{}},{"cell_type":"code","source":"skewness = train_df['label'].skew()\nprint(f\"Skewness of label: {skewness:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:48.370928Z","iopub.execute_input":"2025-06-28T10:49:48.371248Z","iopub.status.idle":"2025-06-28T10:49:48.384031Z","shell.execute_reply.started":"2025-06-28T10:49:48.371224Z","shell.execute_reply":"2025-06-28T10:49:48.383161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Great! A skewness of -0.1135 means:\n\n* The target distribution is slightly negatively skewed (left tail a bit longer).\n\n* But since it’s very close to zero (|skew| < 0.5 is usually considered low), The distribution is fairly symmetric.\n\n* This suggests you probably don’t need to do a transformation just for skewness.","metadata":{}},{"cell_type":"markdown","source":"# Step 4: Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Adding lag features","metadata":{}},{"cell_type":"code","source":"# # 0. Sort and lag BEFORE split\n# train_df = train_df.sort_index()  # just in case\n# for lag in [1, 2, 3]:\n#     for col in ['volume', 'bid_qty', 'ask_qty', 'buy_qty', 'sell_qty']:\n#         train_df[f'{col}_lag{lag}'] = train_df[col].shift(lag)\n#         test_df[f'{col}_lag{lag}'] = test_df[col].shift(lag)\n\n# # 1. Drop rows with NaNs from lagging\n# train_df.dropna(inplace=True)\n\n# # ⚠️ DO NOT drop rows from test_df — it's the competition test set!\n# # Instead, fill or leave NaNs (XGBoost handles them fine)\n# #optional\n# test_df.fillna(0, inplace=True)  # or use forward fill if that's logical\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:48.385198Z","iopub.execute_input":"2025-06-28T10:49:48.385476Z","iopub.status.idle":"2025-06-28T10:49:48.398389Z","shell.execute_reply.started":"2025-06-28T10:49:48.385455Z","shell.execute_reply":"2025-06-28T10:49:48.397343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spliting the data","metadata":{}},{"cell_type":"code","source":"df_train, df_valid = train_test_split(train_df, test_size=0.2, random_state=42)\n\n# 2. Separate features and target\nX_train = df_train.drop(columns=['label'])\ny_train = df_train['label']\n\nX_valid = df_valid.drop(columns=['label'])\ny_valid = df_valid['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:48.399490Z","iopub.execute_input":"2025-06-28T10:49:48.399857Z","iopub.status.idle":"2025-06-28T10:49:59.522266Z","shell.execute_reply.started":"2025-06-28T10:49:48.399831Z","shell.execute_reply":"2025-06-28T10:49:59.521474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Apply clipping on train and valid separately with the train thresholds","metadata":{}},{"cell_type":"markdown","source":"* The extreme spikes (the really tall peaks and deep troughs above or below ±5) will now be cut off or limited at those thresholds.\n\n* So instead of those huge jumps going to, say, +20 or -24, the clipped values will max out at +5 or -5.\n\n* This makes the smoothed curve less wild, easier to analyze, and less sensitive to rare extreme values.\n\n* It helps your model not get overwhelmed by extreme outliers, which can distort learning.\n\n","metadata":{}},{"cell_type":"code","source":"Q1 = y_train.quantile(0.25)\nQ3 = y_train.quantile(0.75)\nIQR = Q3 - Q1\nlow = Q1 - 1.5 * IQR\nhigh = Q3 + 1.5 * IQR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:59.523437Z","iopub.execute_input":"2025-06-28T10:49:59.523665Z","iopub.status.idle":"2025-06-28T10:49:59.546642Z","shell.execute_reply.started":"2025-06-28T10:49:59.523648Z","shell.execute_reply":"2025-06-28T10:49:59.545629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For IQR clipping, you might want to handle edge cases\nif low < high:  # Ensure valid bounds\n    y_train_clipped = y_train.clip(low, high)\n    y_valid_clipped = y_valid.clip(low, high)\nelse:\n    # Handle case where IQR is very small\n    y_train_clipped = y_train.copy()\n    y_valid_clipped = y_valid.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:59.547967Z","iopub.execute_input":"2025-06-28T10:49:59.548583Z","iopub.status.idle":"2025-06-28T10:49:59.564630Z","shell.execute_reply.started":"2025-06-28T10:49:59.548559Z","shell.execute_reply":"2025-06-28T10:49:59.563740Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 6: Train on full training data","metadata":{}},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"# Prepare DMatrix (optional but recommended for XGBoost)\ndtrain = xgb.DMatrix(X_train, label=y_train_clipped)\ndvalid = xgb.DMatrix(X_valid, label=y_valid_clipped)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:49:59.565948Z","iopub.execute_input":"2025-06-28T10:49:59.566359Z","iopub.status.idle":"2025-06-28T10:50:11.718854Z","shell.execute_reply.started":"2025-06-28T10:49:59.566325Z","shell.execute_reply":"2025-06-28T10:50:11.718160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ndel train_df\n\ndel y_train\ndel y_valid\n\ndel X_train\ndel X_valid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:50:11.719471Z","iopub.execute_input":"2025-06-28T10:50:11.719720Z","iopub.status.idle":"2025-06-28T10:50:12.016882Z","shell.execute_reply.started":"2025-06-28T10:50:11.719701Z","shell.execute_reply":"2025-06-28T10:50:12.015588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set XGBoost parameters (regression example)\nparams = {\n    \"objective\": \"reg:squarederror\",\n    \"eval_metric\": \"rmse\",\n    \"tree_method\": \"hist\",  # fast histogram algorithm\n    \"seed\": 42\n}\nevals = [(dtrain, \"train\"), (dvalid, \"valid\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:50:12.017764Z","iopub.execute_input":"2025-06-28T10:50:12.018112Z","iopub.status.idle":"2025-06-28T10:50:12.037177Z","shell.execute_reply.started":"2025-06-28T10:50:12.018080Z","shell.execute_reply":"2025-06-28T10:50:12.036192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Without any new Feature ","metadata":{}},{"cell_type":"code","source":"# Train with early stopping on validation\nmodel = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=1000,\n    evals=evals,\n    early_stopping_rounds=50,\n    verbose_eval=10\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T10:50:12.041923Z","iopub.execute_input":"2025-06-28T10:50:12.042214Z","iopub.status.idle":"2025-06-28T11:15:39.473886Z","shell.execute_reply.started":"2025-06-28T10:50:12.042194Z","shell.execute_reply":"2025-06-28T11:15:39.472631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on validation set\ny_pred = model.predict(dvalid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:15:39.476044Z","iopub.execute_input":"2025-06-28T11:15:39.476447Z","iopub.status.idle":"2025-06-28T11:15:41.049175Z","shell.execute_reply.started":"2025-06-28T11:15:39.476422Z","shell.execute_reply":"2025-06-28T11:15:41.048467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate\nrmse = mean_squared_error(y_valid_clipped, y_pred, squared=False)\nr2 = r2_score(y_valid_clipped, y_pred)\n\nprint(f\"Validation RMSE: {rmse:.4f}\")\nprint(f\"Validation R2: {r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:15:41.049979Z","iopub.execute_input":"2025-06-28T11:15:41.050421Z","iopub.status.idle":"2025-06-28T11:15:41.066593Z","shell.execute_reply.started":"2025-06-28T11:15:41.050396Z","shell.execute_reply":"2025-06-28T11:15:41.065792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"That’s a very strong result, especially for a time series regression task:\n\nRMSE = 0.1453 → low absolute error\n\nR² = 0.9611 → the model explains 96.11% of the variance in the target!\n\nThat means the model is capturing the dynamics of the data really well, at least on the validation set.","metadata":{}},{"cell_type":"code","source":"# Top 20 features by average gain (most useful metric)\nplot_importance(model, importance_type='gain', max_num_features=20)\nplt.title(\"Top 20 Feature Importances\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:15:41.067777Z","iopub.execute_input":"2025-06-28T11:15:41.068343Z","iopub.status.idle":"2025-06-28T11:15:41.445813Z","shell.execute_reply.started":"2025-06-28T11:15:41.068290Z","shell.execute_reply":"2025-06-28T11:15:41.444745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## With new Features","metadata":{}},{"cell_type":"code","source":"X_test = test_df.drop(columns = ['label'])\n\ndtest = xgb.DMatrix(X_test)\n\n# Make predictions on test set\nxgb_test_preds = model.predict(dtest)\n\n# Generate row IDs (starting at 1)\nrow_ids = range(1, len(xgb_test_preds) + 1)\n\n# Create submission DataFrame\nsubmission = pd.DataFrame({\n    'ID': row_ids,\n    'prediction': xgb_test_preds\n})\n\n# Save to CSV\nsubmission.to_csv('submission.csv', index=False)\nprint(\"✅ Submission file saved: submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T11:15:41.447139Z","iopub.execute_input":"2025-06-28T11:15:41.447445Z","iopub.status.idle":"2025-06-28T11:16:13.276906Z","shell.execute_reply.started":"2025-06-28T11:15:41.447423Z","shell.execute_reply":"2025-06-28T11:16:13.275491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}