{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:37.705511Z","iopub.execute_input":"2024-10-16T08:09:37.706231Z","iopub.status.idle":"2024-10-16T08:09:41.133787Z","shell.execute_reply.started":"2024-10-16T08:09:37.706179Z","shell.execute_reply":"2024-10-16T08:09:41.132600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet'\ndf = pd.read_parquet(file_path)\ndf.head(10)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-16T08:09:41.138530Z","iopub.execute_input":"2024-10-16T08:09:41.139063Z","iopub.status.idle":"2024-10-16T08:09:48.514015Z","shell.execute_reply.started":"2024-10-16T08:09:41.139020Z","shell.execute_reply":"2024-10-16T08:09:48.512896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nmissing_values[missing_values['Missing Values'] > 0]","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:48.515240Z","iopub.execute_input":"2024-10-16T08:09:48.515561Z","iopub.status.idle":"2024-10-16T08:09:48.755519Z","shell.execute_reply.started":"2024-10-16T08:09:48.515526Z","shell.execute_reply":"2024-10-16T08:09:48.754466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.describe())","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:48.757051Z","iopub.execute_input":"2024-10-16T08:09:48.757439Z","iopub.status.idle":"2024-10-16T08:09:54.991280Z","shell.execute_reply.started":"2024-10-16T08:09:48.757399Z","shell.execute_reply":"2024-10-16T08:09:54.989948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dropping Columns with All Null Values from DataFrame\n\nThis code snippet identifies and removes columns in a DataFrame (`df`) where all values are null.\n","metadata":{}},{"cell_type":"code","source":"# Identify columns where all values are null\nnull_columns = df.columns[df.isnull().all()]\n\n# Print the names of the columns that will be dropped\n# print(f\"Columns with all null values: {null_columns.tolist()}\")\n\ndf_cleaned = df.drop(columns=null_columns)\n\nprint(f\"Original shape: {df.shape}\")\nprint(f\"Shape after dropping null columns: {df_cleaned.shape}\")\n\n# print(\"Remaining columns with all null values (if any):\")\n# print(df_cleaned.isnull().all()[df_cleaned.isnull().all()])\n\n# df_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:54.995176Z","iopub.execute_input":"2024-10-16T08:09:54.996136Z","iopub.status.idle":"2024-10-16T08:09:55.315777Z","shell.execute_reply.started":"2024-10-16T08:09:54.996084Z","shell.execute_reply":"2024-10-16T08:09:55.314699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:55.317196Z","iopub.execute_input":"2024-10-16T08:09:55.317602Z","iopub.status.idle":"2024-10-16T08:09:55.520019Z","shell.execute_reply.started":"2024-10-16T08:09:55.317565Z","shell.execute_reply":"2024-10-16T08:09:55.518937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Impute the missing values in features and not in responder_06 for Now we are considering it as out target, Since it has no nan values we dont drop any values**\n\n*> Just used `knn imputation` I guess using `Forward Fill / Backward Fill` might also work well*","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Define the target column\ntarget_column = 'responder_6'\n\n# Separate features and target\nX = df_cleaned.drop(columns=[target_column])  # Features\ny = df_cleaned[target_column]  # Target variable\n\n# Create a SimpleImputer (faster than KNNImputer)\nimputer = SimpleImputer(strategy='mean') \nX_imputed = imputer.fit_transform(X)\n\n# Create a new DataFrame with imputed values\ndf_imputed = pd.DataFrame(X_imputed, columns=X.columns)\ndf_imputed[target_column] = y.reset_index(drop=True)  # Add the target column back\n\ndf_cleaned = df_imputed","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:55.521494Z","iopub.execute_input":"2024-10-16T08:09:55.521980Z","iopub.status.idle":"2024-10-16T08:09:58.223463Z","shell.execute_reply.started":"2024-10-16T08:09:55.521927Z","shell.execute_reply":"2024-10-16T08:09:58.222272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_cleaned.isnull().sum().reset_index()\nmissing_values.columns = ['Column', 'Missing Values']\n\n# Display columns with missing values only\nprint(\"total number of columns having any row as null:\", len(missing_values[missing_values['Missing Values'] > 0]))","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:58.225141Z","iopub.execute_input":"2024-10-16T08:09:58.225577Z","iopub.status.idle":"2024-10-16T08:09:58.432819Z","shell.execute_reply.started":"2024-10-16T08:09:58.225528Z","shell.execute_reply":"2024-10-16T08:09:58.431558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**> For the below code I refered to our dataset as df_cleane**","metadata":{}},{"cell_type":"markdown","source":"## Visualizing Distributions of Features and Responders\n\nThis code snippet generates histograms to visualize the distributions of selected features and responders from the cleaned DataFrame (`df_cleaned`).\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport sys\nimport time\n\ndata = df_cleaned\nprint(\"Plotting graphs:\")\n\n# Determine the number of features and responders\ncols_to_plot = [col for col in data.columns if 'feature' in col or 'responder' in col]\nn_cols = 3  # Number of columns in the grid\nn_rows = (len(cols_to_plot) + n_cols - 1) // n_cols  # Calculate rows needed\n\n# Create a grid of subplots\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(20, 5 * n_rows))\naxes = axes.flatten()  # Flatten the axes array for easy indexing\n\n# Visualize distributions of features and responders\nfor i, col in enumerate(cols_to_plot):\n    # Overwrite previous message in the same line\n    sys.stdout.write(f'\\rPlotting graph {i + 1}/{len(cols_to_plot)}...')\n    sys.stdout.flush()\n    \n    sns.histplot(data[col], bins=50, kde=True, ax=axes[i])\n    axes[i].set_title(f'Distribution of {col}')\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('Frequency')\n    \n    time.sleep(0.5) \n# Hide any unused subplots\nfor j in range(i + 1, len(axes)):\n    axes[j].axis('off')\n\n# Adjust layout\nplt.tight_layout()\nplt.show()\n\nprint(\"\\nPlotting complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:09:58.434281Z","iopub.execute_input":"2024-10-16T08:09:58.434725Z","iopub.status.idle":"2024-10-16T08:23:48.292015Z","shell.execute_reply.started":"2024-10-16T08:09:58.434658Z","shell.execute_reply":"2024-10-16T08:23:48.290285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Heatmaps\n* > all the heatmaps are here\n\n* ****heatmap between features and all responders****","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:23:48.293781Z","iopub.execute_input":"2024-10-16T08:23:48.294213Z","iopub.status.idle":"2024-10-16T08:24:26.342277Z","shell.execute_reply.started":"2024-10-16T08:23:48.294165Z","shell.execute_reply":"2024-10-16T08:24:26.341202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **heatmap between features and first responder**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndata = df_cleaned.iloc[:, :-8]\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder0')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:24:26.343735Z","iopub.execute_input":"2024-10-16T08:24:26.344195Z","iopub.status.idle":"2024-10-16T08:24:58.267251Z","shell.execute_reply.started":"2024-10-16T08:24:26.344147Z","shell.execute_reply":"2024-10-16T08:24:58.266144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **heatmap between features and second responder**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.iloc[:, :-7]\ndata = data.drop(data.columns[-2], axis=1)\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:24:58.268714Z","iopub.execute_input":"2024-10-16T08:24:58.269186Z","iopub.status.idle":"2024-10-16T08:25:30.089808Z","shell.execute_reply.started":"2024-10-16T08:24:58.269138Z","shell.execute_reply":"2024-10-16T08:25:30.088568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Heatmap for the target variable (responder_6)**","metadata":{}},{"cell_type":"code","source":"data = df_cleaned.iloc[:, :-2].drop(columns=['responder_0', 'responder_1', 'responder_2', \n                                 'responder_3', 'responder_4', 'responder_5'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(25, 20))\n\n# Generate the heatmap for the entire dataset\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for responder1')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-16T08:25:30.091460Z","iopub.execute_input":"2024-10-16T08:25:30.091981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **XGBoost Model**\n\n- **Data Preparation**: Select features and target (`responder_6`).\n- **Train-Test Split**: Split data (80/20) without shuffling to preserve order.\n- **Scaling**: Standardize features using `StandardScaler`.  ....(`altough not recommended`)\n- **Model Training**: Train an XGBoost regressor on the scaled data.\n- **Evaluation**: Calculate metrics like R², MAE, MSE, RMSE, MAPE, and accuracy within 10% of actual values.\n- **Visualization**: Plot actual vs predicted values for the first 100 rows.","metadata":{}},{"cell_type":"code","source":"df_sample = df_cleaned\n\n# Step 2: Select Features and Target\n# Here 'responder_6' is the target variable (assumed), adjust if needed\ntarget = 'responder_6'\nfeatures = [col for col in df_sample.columns if 'feature_' in col]\n\nX = df_sample[features]  # Feature matrix\ny = df_sample[target]    # Target variable\n\n# Step 3: Train-Test Split\n# Since this could be time-series data, make sure to split carefully\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, shuffle=False)\n\n# Step 4: Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Step 5: Model Training (XGBoost)\nmodel = XGBRegressor(objective='reg:squarederror', n_estimators=50, learning_rate=0.1)  # Reduced estimators for faster testing\nmodel.fit(X_train_scaled, y_train)\n\n# Step 6: Model Prediction\ny_pred = model.predict(X_test_scaled)\n\n# Step 7: Calculate Evaluation Metrics\n\n# R-squared\nr2 = r2_score(y_test, y_pred)\n\n# Mean Absolute Error (MAE)\nmae = mean_absolute_error(y_test, y_pred)\n\n# Mean Squared Error (MSE)\nmse = mean_squared_error(y_test, y_pred)\n\n# Root Mean Squared Error (RMSE)\nrmse = np.sqrt(mse)\n\n# Mean Absolute Percentage Error (MAPE)\nmape = np.mean(np.abs((y_test - y_pred) / y_test)) * 100\n\n# Custom Accuracy: Percentage of predictions within 10% of the actual values\naccuracy = np.mean(np.abs((y_pred - y_test) / y_test) < 0.10) * 100\n\n# Step 8: Print all metrics\nprint(f\"R-squared: {r2:.4f}\")\nprint(f\"Mean Absolute Error (MAE): {mae:.4f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.4f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.4f}\")\nprint(f\"Mean Absolute Percentage Error (MAPE): {mape:.2f}%\")\nprint(f\"Custom Accuracy (within 10% of actual): {accuracy:.2f}%\")\n\n# Optional: Plot the predictions vs actuals\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(10,6))\nplt.plot(y_test.reset_index(drop=True), label='Actual')\nplt.plot(y_pred, label='Predicted')\nplt.title('Actual vs Predicted - Responder_6 (First 100 Rows)')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select the specific row (e.g., row 89) as a sample input\nsample_input = df.loc[[89], features]  # Use double brackets to maintain DataFrame structure\n\n# Step 10: Scale the Sample Input\nsample_input_scaled = scaler.transform(sample_input)\n\n# Step 11: Make Predictions\nsample_predictions = model.predict(sample_input_scaled)\n\n# Step 12: Display Predictions\nprint(\"Sample Input Data:\")\nprint(sample_input)\nprint(\"\\nPredicted Value for Sample Input:\")\nprint(sample_predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **LSTM for Time Series Analysis**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import MinMaxScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score\n\ndata = df_cleaned\n# Step 1: Prepare the data\ntarget_column = 'responder_6'\nfeatures = data.drop(columns=[target_column]).values  # Feature matrix\ntarget = data[target_column].values  # Target variable\n\n# Scale the features\nscaler = MinMaxScaler()\nfeatures_scaled = scaler.fit_transform(features)\n\n# Create sequences for LSTM\ndef create_sequences(data, target, time_steps=1):\n    X, y = [], []\n    for i in range(len(data) - time_steps):\n        X.append(data[i:(i + time_steps)])\n        y.append(target[i + time_steps])\n    return np.array(X), np.array(y)\n\n# Set the number of time steps\ntime_steps = 5\nX, y = create_sequences(features_scaled, target, time_steps)\n\n# Step 2: Train-Test Split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, shuffle=False)\n\n# Step 3: Reshape input to be [samples, time steps, features]\nX_train = X_train.reshape((X_train.shape[0], X_train.shape[1], X_train.shape[2]))\nX_test = X_test.reshape((X_test.shape[0], X_test.shape[1], X_test.shape[2]))\n\n# Step 4: Build the LSTM model\nmodel = Sequential()\nmodel.add(LSTM(50, activation='relu', return_sequences=True, input_shape=(X_train.shape[1], X_train.shape[2])))\nmodel.add(Dropout(0.2))\nmodel.add(LSTM(50, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1))  # Output layer for regression\n\n# Step 5: Compile and fit the model\nmodel.compile(optimizer='adam', loss='mean_squared_error')\nmodel.fit(X_train, y_train, epochs=50, batch_size=32)\n\n# Step 6: Model prediction\ny_pred = model.predict(X_test)\n\n# Step 7: Evaluate the model\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f'Mean Squared Error: {mse:.4f}')\nprint(f'R-squared: {r2:.4f}')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}