{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!pip install -U xgboost","metadata":{"execution":{"iopub.status.busy":"2024-10-26T01:58:21.318491Z","iopub.execute_input":"2024-10-26T01:58:21.318934Z","iopub.status.idle":"2024-10-26T01:58:21.343851Z","shell.execute_reply.started":"2024-10-26T01:58:21.318892Z","shell.execute_reply":"2024-10-26T01:58:21.342599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-10-26T01:58:21.346194Z","iopub.execute_input":"2024-10-26T01:58:21.346679Z","iopub.status.idle":"2024-10-26T01:58:25.164811Z","shell.execute_reply.started":"2024-10-26T01:58:21.346629Z","shell.execute_reply":"2024-10-26T01:58:25.163585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duper_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-10-26T01:59:35.846988Z","iopub.execute_input":"2024-10-26T01:59:35.847494Z","iopub.status.idle":"2024-10-26T01:59:35.852675Z","shell.execute_reply.started":"2024-10-26T01:59:35.847448Z","shell.execute_reply":"2024-10-26T01:59:35.851387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dudf = pd.read_parquet(duper_path)\ndudf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-10-26T02:00:21.178007Z","iopub.execute_input":"2024-10-26T02:00:21.178467Z","iopub.status.idle":"2024-10-26T02:00:21.229380Z","shell.execute_reply.started":"2024-10-26T02:00:21.178423Z","shell.execute_reply":"2024-10-26T02:00:21.227885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet'\ndf = pd.read_parquet(file_path)\ndf.head(10)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-26T01:58:25.166694Z","iopub.execute_input":"2024-10-26T01:58:25.167804Z","iopub.status.idle":"2024-10-26T01:58:28.898405Z","shell.execute_reply.started":"2024-10-26T01:58:25.167746Z","shell.execute_reply":"2024-10-26T01:58:28.897200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:37:48.331306Z","iopub.execute_input":"2024-10-25T17:37:48.332265Z","iopub.status.idle":"2024-10-25T17:37:48.338295Z","shell.execute_reply.started":"2024-10-25T17:37:48.332223Z","shell.execute_reply":"2024-10-25T17:37:48.337340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nmissing_values[missing_values['Missing Values'] > 0]","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:41:19.743542Z","iopub.execute_input":"2024-10-25T17:41:19.744345Z","iopub.status.idle":"2024-10-25T17:41:19.989988Z","shell.execute_reply.started":"2024-10-25T17:41:19.744303Z","shell.execute_reply":"2024-10-25T17:41:19.988953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.describe())","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:42:04.528322Z","iopub.execute_input":"2024-10-25T17:42:04.528731Z","iopub.status.idle":"2024-10-25T17:42:10.469015Z","shell.execute_reply.started":"2024-10-25T17:42:04.528696Z","shell.execute_reply":"2024-10-25T17:42:10.468057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dropping Columns with All Null Values from DataFrame\n\nThis code snippet identifies and removes columns in a DataFrame (`df`) where all values are null.\n","metadata":{}},{"cell_type":"code","source":"# Identify columns where all values are null\nnull_columns = df.columns[df.isnull().all()]\n\n# Print the names of the columns that will be dropped\n# print(f\"Columns with all null values: {null_columns.tolist()}\")\n\ndf_cleaned = df.drop(columns=null_columns)\n\nprint(f\"Original shape: {df.shape}\")\nprint(f\"Shape after dropping null columns: {df_cleaned.shape}\")\n\n# print(\"Remaining columns with all null values (if any):\")\n# print(df_cleaned.isnull().all()[df_cleaned.isnull().all()])\n\n# df_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:43:59.694353Z","iopub.execute_input":"2024-10-25T17:43:59.695512Z","iopub.status.idle":"2024-10-25T17:43:59.990791Z","shell.execute_reply.started":"2024-10-25T17:43:59.695465Z","shell.execute_reply":"2024-10-25T17:43:59.989605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:44:03.318405Z","iopub.execute_input":"2024-10-25T17:44:03.319228Z","iopub.status.idle":"2024-10-25T17:44:03.523380Z","shell.execute_reply.started":"2024-10-25T17:44:03.319182Z","shell.execute_reply":"2024-10-25T17:44:03.521560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Impute the missing values in features and not in responder_06.\n\nNow we are considering it as out target. \nSince it has no nan values we dont drop any values**\n\n*> Just used `knn imputation` I guess using `Forward Fill / Backward Fill` might also work well*\n\nBullshit - The simple imputer just replaces the NaNs with the mean value in the column.","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Define the target column\ntarget_column = 'responder_6'\n\n# Separate features and target\nX = df_cleaned.drop(columns=[target_column])  # Features\ny = df_cleaned[target_column]  # Target variable\n\n# Create a SimpleImputer (faster than KNNImputer)\nimputer = SimpleImputer(strategy='mean') \nX_imputed = imputer.fit_transform(X)\n\n# Create a new DataFrame with imputed values\ndf_imputed = pd.DataFrame(X_imputed, columns=X.columns)\ndf_imputed[target_column] = y.reset_index(drop=True)  # Add the target column back\n\ndf_cleaned = df_imputed","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:47:53.286104Z","iopub.execute_input":"2024-10-25T17:47:53.286572Z","iopub.status.idle":"2024-10-25T17:47:56.005130Z","shell.execute_reply.started":"2024-10-25T17:47:53.286532Z","shell.execute_reply":"2024-10-25T17:47:56.004235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:48:01.279478Z","iopub.execute_input":"2024-10-25T17:48:01.280289Z","iopub.status.idle":"2024-10-25T17:48:01.476925Z","shell.execute_reply.started":"2024-10-25T17:48:01.280240Z","shell.execute_reply":"2024-10-25T17:48:01.475962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**> For the below code I refered to our dataset as df_cleaned","metadata":{}},{"cell_type":"markdown","source":"## Visualizing Distributions of Features and Responders\n\nThis code snippet generates histograms to visualize the distributions of selected features and responders from the cleaned DataFrame (`df_cleaned`).\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport sys\nimport time\n\ndata = df_cleaned\nprint(\"Plotting graphs:\")\n\n# Determine the number of features and responders\ncols_to_plot = [col for col in data.columns if 'feature' in col or 'responder' in col]\nn_cols = 3  # Number of columns in the grid\nn_rows = (len(cols_to_plot) + n_cols - 1) // n_cols  # Calculate rows needed\n\n# Create a grid of subplots\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Visualize distributions of features and responders\nfor i, col in enumerate(cols_to_plot):\n    # Overwrite previous message in the same line\n    sys.stdout.write(f'\\rPlotting graph {i + 1}/{len(cols_to_plot)}...')\n    sys.stdout.flush()\n    \n    sns.histplot(data[col], bins=50, kde=True, ax=axes[i])\n    axes[i].set_title(f'Distribution of {col}')\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('Frequency')\n    \n    time.sleep(0.5) \n# Hide any unused subplots\nfor j in range(i + 1, len(axes)):\n    axes[j].axis('off')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()\n\nprint(\"\\nPlotting complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-25T17:48:16.503336Z","iopub.execute_input":"2024-10-25T17:48:16.504097Z","iopub.status.idle":"2024-10-25T18:00:27.192654Z","shell.execute_reply.started":"2024-10-25T17:48:16.504058Z","shell.execute_reply":"2024-10-25T18:00:27.191567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Heatmaps","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:22:02.651407Z","iopub.execute_input":"2024-10-25T14:22:02.651925Z","iopub.status.idle":"2024-10-25T14:22:43.378799Z","shell.execute_reply.started":"2024-10-25T14:22:02.651877Z","shell.execute_reply":"2024-10-25T14:22:43.377388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **heatmap between features and first responder**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned.iloc[:, :-8]\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder0')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:23:33.657131Z","iopub.execute_input":"2024-10-25T14:23:33.657628Z","iopub.status.idle":"2024-10-25T14:24:07.875016Z","shell.execute_reply.started":"2024-10-25T14:23:33.657585Z","shell.execute_reply":"2024-10-25T14:24:07.873165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **heatmap between features and second responder**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.iloc[:, :-7]\ndata = data.drop(data.columns[-2], axis=1)\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T14:25:08.068052Z","iopub.execute_input":"2024-10-25T14:25:08.068614Z","iopub.status.idle":"2024-10-25T14:25:42.369612Z","shell.execute_reply.started":"2024-10-25T14:25:08.068562Z","shell.execute_reply":"2024-10-25T14:25:42.368245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Heatmap for the target variable (responder_6)**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.drop(columns=['responder_0', 'responder_1', 'responder_2', \n                                 'responder_3', 'responder_4', 'responder_5', \n                                'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:02:37.896938Z","iopub.execute_input":"2024-10-25T18:02:37.898048Z","iopub.status.idle":"2024-10-25T18:03:09.063460Z","shell.execute_reply.started":"2024-10-25T18:02:37.898002Z","shell.execute_reply":"2024-10-25T18:03:09.062478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **XGBoost Model**\n\n- **Data Preparation**: Select features and target (`responder_6`).\n- **Train-Test Split**: Split data (80/20) without shuffling to preserve order.\n- **Scaling**: Standardize features using `StandardScaler`.  ....(`altough not recommended`)\n- **Model Training**: Train an XGBoost regressor on the scaled data.\n- **Evaluation**: Calculate metrics like R², MAE, MSE, RMSE, MAPE, and accuracy within 10% of actual values.\n- **Visualization**: Plot actual vs predicted values for the first 100 rows.","metadata":{}},{"cell_type":"code","source":"# Lets remind ourselves what df_cleaned is:\ndf_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:04:44.847976Z","iopub.execute_input":"2024-10-25T18:04:44.848380Z","iopub.status.idle":"2024-10-25T18:04:44.875557Z","shell.execute_reply.started":"2024-10-25T18:04:44.848332Z","shell.execute_reply":"2024-10-25T18:04:44.874532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample = df_cleaned\n\n# Step 2: Select Features and Target\n# Here 'responder_6' is the target variable (assumed), adjust if needed\ntarget = 'responder_6'\nfeatures = [col for col in df_sample.columns if 'feature_' in col]\n\nX = df_sample[features]  # Feature matrix\ny = df_sample[target]    # Target variable\n\n# Step 3: Train-Test Split\n# Since this could be time-series data, make sure to split carefully\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, shuffle=False)\n\n# Step 4: Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Step 5: Model Training (XGBoost)\nmodel = XGBRegressor(objective='reg:squarederror', n_estimators=50, learning_rate=0.1)  # Reduced estimators for faster testing\nmodel.fit(X_train_scaled, y_train)\n\n# Step 6: Model Prediction\ny_pred = model.predict(X_test_scaled)\n\n# Step 7: Calculate Evaluation Metrics\n\n# R-squared\nr2 = r2_score(y_test, y_pred)\n\n# Mean Absolute Error (MAE)\nmae = mean_absolute_error(y_test, y_pred)\n\n# Mean Squared Error (MSE)\nmse = mean_squared_error(y_test, y_pred)\n\n# Root Mean Squared Error (RMSE)\nrmse = np.sqrt(mse)\n\n# Mean Absolute Percentage Error (MAPE)\nmape = np.mean(np.abs((y_test - y_pred) / y_test)) * 100\n\n# Custom Accuracy: Percentage of predictions within 10% of the actual values\naccuracy = np.mean(np.abs((y_pred - y_test) / y_test) < 0.10) * 100\n\n# Step 8: Print all metrics\nprint(f\"R-squared: {r2:.4f}\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"Mean Absolute Percentage Error (MAPE): {mape:.2f}%\")\nprint(f\"Custom Accuracy (within 10% of actual): {accuracy:.2f}%\")\n\n# Optional: Plot the predictions vs actuals\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(10,6))\nplt.plot(y_test.reset_index(drop=True), label='Actual')\nplt.plot(y_pred, label='Predicted')\nplt.title('Actual vs Predicted - Responder_6 (First 100 Rows)')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:08:58.992039Z","iopub.execute_input":"2024-10-25T18:08:58.993148Z","iopub.status.idle":"2024-10-25T18:09:27.578218Z","shell.execute_reply.started":"2024-10-25T18:08:58.993104Z","shell.execute_reply":"2024-10-25T18:09:27.576968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nprint(xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:16:20.644779Z","iopub.execute_input":"2024-10-25T18:16:20.645219Z","iopub.status.idle":"2024-10-25T18:16:20.650657Z","shell.execute_reply.started":"2024-10-25T18:16:20.645183Z","shell.execute_reply":"2024-10-25T18:16:20.649636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}