{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:08:28.099469Z","iopub.execute_input":"2024-12-02T20:08:28.100657Z","iopub.status.idle":"2024-12-02T20:08:29.785418Z","shell.execute_reply.started":"2024-12-02T20:08:28.100612Z","shell.execute_reply":"2024-12-02T20:08:29.784331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet'\ndf = pd.read_parquet(file_path)\ndf.head(10)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:08:29.787401Z","iopub.execute_input":"2024-12-02T20:08:29.787916Z","iopub.status.idle":"2024-12-02T20:08:34.671646Z","shell.execute_reply.started":"2024-12-02T20:08:29.787879Z","shell.execute_reply":"2024-12-02T20:08:34.670425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:34.672914Z","iopub.execute_input":"2024-12-02T20:08:34.673284Z","iopub.status.idle":"2024-12-02T20:08:34.680168Z","shell.execute_reply.started":"2024-12-02T20:08:34.673253Z","shell.execute_reply":"2024-12-02T20:08:34.679028Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values = df.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nmissing_values[missing_values['Missing Values'] > 0]","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:34.682958Z","iopub.execute_input":"2024-12-02T20:08:34.683441Z","iopub.status.idle":"2024-12-02T20:08:34.951717Z","shell.execute_reply.started":"2024-12-02T20:08:34.683378Z","shell.execute_reply":"2024-12-02T20:08:34.950532Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.describe())","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:34.953101Z","iopub.execute_input":"2024-12-02T20:08:34.953452Z","iopub.status.idle":"2024-12-02T20:08:41.328223Z","shell.execute_reply.started":"2024-12-02T20:08:34.953418Z","shell.execute_reply":"2024-12-02T20:08:41.327151Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify columns where all values are null\nnull_columns = df.columns[df.isnull().all()]\n\n# Print the names of the columns that will be dropped\n# print(f\"Columns with all null values: {null_columns.tolist()}\")\n\ndf_cleaned = df.drop(columns=null_columns)\n\nprint(f\"Original shape: {df.shape}\")\nprint(f\"Shape after dropping null columns: {df_cleaned.shape}\")\n\n# print(\"Remaining columns with all null values (if any):\")\n# print(df_cleaned.isnull().all()[df_cleaned.isnull().all()])\n\n# df_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:41.329685Z","iopub.execute_input":"2024-12-02T20:08:41.330215Z","iopub.status.idle":"2024-12-02T20:08:41.636217Z","shell.execute_reply.started":"2024-12-02T20:08:41.330174Z","shell.execute_reply":"2024-12-02T20:08:41.635027Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:41.637409Z","iopub.execute_input":"2024-12-02T20:08:41.637730Z","iopub.status.idle":"2024-12-02T20:08:41.845626Z","shell.execute_reply.started":"2024-12-02T20:08:41.637699Z","shell.execute_reply":"2024-12-02T20:08:41.844563Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fill missing values with 0\ndf_filled = df.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:41.847050Z","iopub.execute_input":"2024-12-02T20:08:41.847447Z","iopub.status.idle":"2024-12-02T20:08:42.662519Z","shell.execute_reply.started":"2024-12-02T20:08:41.847413Z","shell.execute_reply":"2024-12-02T20:08:42.661457Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:08:42.663546Z","iopub.execute_input":"2024-12-02T20:08:42.663883Z","iopub.status.idle":"2024-12-02T20:08:42.671297Z","shell.execute_reply.started":"2024-12-02T20:08:42.663851Z","shell.execute_reply":"2024-12-02T20:08:42.670225Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**> For the below code I refered to our dataset as df_cleaned","metadata":{}},{"cell_type":"markdown","source":"# Visualization","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport sys\nimport time\n\ndata = df_cleaned\nprint(\"Plotting graphs:\")\n\n# Determine the number of features and responders\ncols_to_plot = [col for col in data.columns if 'feature' in col or 'responder' in col]\nn_cols = 3  # Number of columns in the grid\nn_rows = (len(cols_to_plot) + n_cols - 1) // n_cols  # Calculate rows needed\n\n# Create a grid of subplots\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Visualize distributions of features and responders\nfor i, col in enumerate(cols_to_plot):\n    # Overwrite previous message in the same line\n    sys.stdout.write(f'\\rPlotting graph {i + 1}/{len(cols_to_plot)}...')\n    sys.stdout.flush()\n    \n    sns.histplot(data[col], bins=50, kde=True, ax=axes[i])\n    axes[i].set_title(f'Distribution of {col}')\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('Frequency')\n    \n    time.sleep(0.5) \n# Hide any unused subplots\nfor j in range(i + 1, len(axes)):\n    axes[j].axis('off')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()\n\nprint(\"\\nPlotting complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:09:50.079624Z","iopub.execute_input":"2024-12-02T20:09:50.080080Z","iopub.status.idle":"2024-12-02T20:21:59.745039Z","shell.execute_reply.started":"2024-12-02T20:09:50.080042Z","shell.execute_reply":"2024-12-02T20:21:59.743769Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Heatmaps","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:25:07.657727Z","iopub.execute_input":"2024-12-02T20:25:07.658168Z","iopub.status.idle":"2024-12-02T20:25:45.265719Z","shell.execute_reply.started":"2024-12-02T20:25:07.658134Z","shell.execute_reply":"2024-12-02T20:25:45.264696Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> **heatmap between features and first responder**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned.iloc[:, :-8]\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder0')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:25:54.652132Z","iopub.execute_input":"2024-12-02T20:25:54.652512Z","iopub.status.idle":"2024-12-02T20:26:26.015558Z","shell.execute_reply.started":"2024-12-02T20:25:54.652476Z","shell.execute_reply":"2024-12-02T20:26:26.014473Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> **heatmap between features and second responder**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.iloc[:, :-7]\ndata = data.drop(data.columns[-2], axis=1)\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:27:45.688163Z","iopub.execute_input":"2024-12-02T20:27:45.688586Z","iopub.status.idle":"2024-12-02T20:28:16.786855Z","shell.execute_reply.started":"2024-12-02T20:27:45.688549Z","shell.execute_reply":"2024-12-02T20:28:16.785780Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Heatmap for the target variable (responder_6)**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.drop(columns=['responder_0', 'responder_1', 'responder_2', \n                                 'responder_3', 'responder_4', 'responder_5', \n                                'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:29:54.413835Z","iopub.execute_input":"2024-12-02T20:29:54.414237Z","iopub.status.idle":"2024-12-02T20:30:25.310613Z","shell.execute_reply.started":"2024-12-02T20:29:54.414205Z","shell.execute_reply":"2024-12-02T20:30:25.309626Z"},"trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **XGBoost Model**\n\n- **Data Preparation**: Select features and target (`responder_6`).\n- **Train-Test Split**: Split data (80/20) without shuffling to preserve order.\n- **Scaling**: Standardize features using `StandardScaler`.  ....(`altough not recommended`)\n- **Model Training**: Train an XGBoost regressor on the scaled data.\n- **Evaluation**: Calculate metrics like R², MAE, MSE, RMSE, MAPE, and accuracy within 10% of actual values.\n- **Visualization**: Plot actual vs predicted values for the first 100 rows.","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# 1. Define data\ndf_sample = df_cleaned\n\n# 2. Select Features and Target\n# Here 'responder_6' is the target variable (assumed), adjust if needed\ntarget = 'responder_6'\nfeatures = [col for col in df_cleaned.columns if 'feature_' in col]\n\nX = df_cleaned[features]  # Feature matrix\ny = df_cleaned[target]    # Target variable\n\n# 3. Train-Test Split\n# Since this could be time-series data, make sure to split carefully\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.10, shuffle=False)\n\n# 4. Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:30:37.933458Z","iopub.execute_input":"2024-12-02T20:30:37.933842Z","iopub.status.idle":"2024-12-02T20:30:41.867790Z","shell.execute_reply.started":"2024-12-02T20:30:37.933810Z","shell.execute_reply":"2024-12-02T20:30:41.866747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Fine tuning XGBoost hyperparameters**","metadata":{}},{"cell_type":"code","source":"#Finding the best n_estimators and learning_rate values\nfrom sklearn.metrics import make_scorer #Import here \nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import r2_score\n\nparam_grid = {\n    'n_estimators': [40,50,80,100, 200],  # Values for n_estimators to try\n    'learning_rate': [0.01, 0.05, 0.1, 0.2]  # Values for learning_rate to try\n}\nmodel = XGBRegressor(objective='reg:squarederror')\n# Use make_scorer to create a scorer for R-squared\nr2_scorer = make_scorer(r2_score) \n\ngrid_search = GridSearchCV(\n    estimator=model, \n    param_grid=param_grid, \n    scoring=r2_scorer,  # Use R-squared as the scoring metric\n    cv=5,                # Number of cross-validation folds (adjust as needed)\n    verbose=2\n)\ngrid_search.fit(X_train_scaled, y_train)\nbest_params = grid_search.best_params_\nbest_model = grid_search.best_estimator_\n\nprint(\"Best Hyperparameters:\", best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:08:42.795388Z","iopub.status.idle":"2024-12-02T20:08:42.795746Z","shell.execute_reply.started":"2024-12-02T20:08:42.795576Z","shell.execute_reply":"2024-12-02T20:08:42.795594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Finding the best max_depth value\n\nparam_grid = {\n    'max_depth': [3, 4, 5, 6, 7, 8, 9, 10]  # Range of max_depth values to try\n}\n\n# 2. Create the XGBoost model\nmodel = XGBRegressor(objective='reg:squarederror', learning_rate=0.05, n_estimators=50)  \n\n# 3. Create the GridSearchCV object\nr2_scorer = make_scorer(r2_score)  # Create scorer for R-squared\n\ngrid_search = GridSearchCV(\n    estimator=model,\n    param_grid=param_grid,\n    scoring=r2_scorer,       # Use R-squared as the scoring metric\n    cv=5,                   # Number of cross-validation folds (adjust if needed)\n    verbose=2 \n)\n\n# 4. Fit the GridSearchCV to your data\ngrid_search.fit(X_train_scaled, y_train)\n\n# 5. Get the best max_depth and model\nbest_max_depth = grid_search.best_params_['max_depth']\nbest_model = grid_search.best_estimator_\n\nprint(\"Best max_depth:\", best_max_depth)\n\n# 6. Evaluate the best model \ny_pred = best_model.predict(X_test_scaled)\nr2 = r2_score(y_test, y_pred)\nprint(\"R-squared of the best model:\", r2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:08:42.797742Z","iopub.status.idle":"2024-12-02T20:08:42.798140Z","shell.execute_reply.started":"2024-12-02T20:08:42.797923Z","shell.execute_reply":"2024-12-02T20:08:42.797964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Finding the best gamma, reg_alpha, and reg_lambda values\n\nparam_grid = {\n       'gamma': [0.1, 0.5, 1, 2],\n       'reg_alpha': [0.1,0.5, 1,2],\n       'reg_lambda': [ 0.1,0.5, 1, 2]\n   }\n\n# 2. Create the XGBoost model\nmodel = XGBRegressor(objective='reg:squarederror', learning_rate=0.05, n_estimators=50, max_depth=6)  \n\n# 3. Create the GridSearchCV object\nr2_scorer = make_scorer(r2_score)  # Create scorer for R-squared\n\ngrid_search = GridSearchCV(\n    estimator=model,\n    param_grid=param_grid,\n    scoring=r2_scorer,       # Use R-squared as the scoring metric\n    cv=5,                   # Number of cross-validation folds (adjust if needed)\n    verbose=2 \n)\n\n# 4. Fit the GridSearchCV to your data\ngrid_search.fit(X_train_scaled, y_train)\n\n# 5. Get the best parameters and model\nbest_params = grid_search.best_params_\nbest_model = grid_search.best_estimator_\n\nprint(\"Best Hyperparameters:\", best_params)\n\n# 6. Evaluate the best model \ny_pred = best_model.predict(X_test_scaled)\nr2 = r2_score(y_test, y_pred)\nprint(\"R-squared of the best model:\", r2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:08:42.800196Z","iopub.status.idle":"2024-12-02T20:08:42.800559Z","shell.execute_reply.started":"2024-12-02T20:08:42.800380Z","shell.execute_reply":"2024-12-02T20:08:42.800398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Model Training (XGBoost)\nmodel = XGBRegressor(objective='reg:squarederror', n_estimators=50, learning_rate=0.05, max_depth=6,gamma= 2, reg_alpha= 2, reg_lambda= 0.5)  # Reduced estimators for faster testing\nmodel.fit(X_train_scaled, y_train)\n\n# 2. Model Prediction\ny_pred = model.predict(X_test_scaled)\n\n# 3. Calculate Evaluation Metrics\n\n# R-squared\nr2 = r2_score(y_test, y_pred)\n\n# Mean Absolute Error (MAE)\nmae = mean_absolute_error(y_test, y_pred)\n\n# Mean Squared Error (MSE)\nmse = mean_squared_error(y_test, y_pred)\n\n# Root Mean Squared Error (RMSE)\nrmse = np.sqrt(mse)\n\n# Mean Absolute Percentage Error (MAPE)\nmape = np.mean(np.abs((y_test - y_pred) / y_test)) * 100\n\n# Custom Accuracy: Percentage of predictions within 10% of the actual values\naccuracy = np.mean(np.abs((y_pred - y_test) / y_test) < 0.10) * 100\n\n\n# 4. Print all metrics\nprint(f\"R-squared: {r2:.4f}\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"Mean Absolute Percentage Error (MAPE): {mape:.2f}%\")\nprint(f\"Custom Accuracy (within 10% of actual): {accuracy:.2f}%\")\n\n# Optional: Plot the predictions vs actuals\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(10,6))\nplt.plot(y_test.reset_index(drop=True), label='Actual')\nplt.plot(y_pred, label='Predicted')\nplt.title('Actual vs Predicted - Responder_6 (First 100 Rows)')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-02T20:40:46.366731Z","iopub.execute_input":"2024-12-02T20:40:46.367614Z","iopub.status.idle":"2024-12-02T20:41:19.487252Z","shell.execute_reply.started":"2024-12-02T20:40:46.367554Z","shell.execute_reply":"2024-12-02T20:41:19.485470Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate and print weighted R-squared inside your model evaluation section\ny_pred = model.predict(X_test_scaled)  # Get your predictions\nweights = df_cleaned['weight'][y_test.index] # assuming y_test is a pd.Series\n\n# Function to calculate R² score\ndef calculate_r2(y_test, y_pred, weights):\n    numerator = np.sum(weights * (y_test - y_pred) ** 2)\n    denominator = np.sum(weights * (y_test ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score\n\nweighted_r2 = calculate_r2(y_test, y_pred, weights)\n\nprint(f\"Weighted R-squared: {weighted_r2:.4f}\") # Print the weighted R-squared","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T20:46:21.502173Z","iopub.execute_input":"2024-12-02T20:46:21.502682Z","iopub.status.idle":"2024-12-02T20:46:21.667363Z","shell.execute_reply.started":"2024-12-02T20:46:21.502641Z","shell.execute_reply":"2024-12-02T20:46:21.666198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###### ","metadata":{}}]}