{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-19T14:49:45.870287Z","iopub.execute_input":"2024-10-19T14:49:45.871217Z","iopub.status.idle":"2024-10-19T14:49:48.232103Z","shell.execute_reply.started":"2024-10-19T14:49:45.871147Z","shell.execute_reply":"2024-10-19T14:49:48.231130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:49:48.233893Z","iopub.execute_input":"2024-10-19T14:49:48.234470Z","iopub.status.idle":"2024-10-19T14:49:54.074274Z","shell.execute_reply.started":"2024-10-19T14:49:48.234421Z","shell.execute_reply":"2024-10-19T14:49:54.073225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Some basic dataProcessing**","metadata":{}},{"cell_type":"code","source":"#Removing columns that are null entirely\nnullColumns = []\n\nnullColumns = df.columns[pd.isnull(df).all()].tolist()\ndf = df.drop(nullColumns, axis=1)\ndf.shape\n\nnanValues = df.isnull().sum()\ndf.dropna(inplace=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:49:54.075467Z","iopub.execute_input":"2024-10-19T14:49:54.075816Z","iopub.status.idle":"2024-10-19T14:49:55.022312Z","shell.execute_reply.started":"2024-10-19T14:49:54.075777Z","shell.execute_reply":"2024-10-19T14:49:55.021291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Drop all responder columns except responder_6\nfeature_columns = df.columns.difference(['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                                         'responder_4', 'responder_5', 'responder_7', 'responder_8', \n                                         'responder_6'])\n\n# Compute the correlation of features with responder_6\ncorr_with_responder6 = df[feature_columns].corrwith(df['responder_6'])\n\n# Visualize the correlation with a bar plot\nplt.figure(figsize=(10, 12))\nsns.barplot(y=corr_with_responder6.index, x=corr_with_responder6.values, palette='coolwarm')\nplt.title('Correlation of Features with responder_6')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:49:55.025488Z","iopub.execute_input":"2024-10-19T14:49:55.025958Z","iopub.status.idle":"2024-10-19T14:49:58.236018Z","shell.execute_reply.started":"2024-10-19T14:49:55.025910Z","shell.execute_reply":"2024-10-19T14:49:58.234926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set a threshold for low correlation (e.g., abs(correlation) < 0.05)\nlow_correlation_features = corr_with_responder6[abs(corr_with_responder6) < 0.05].index\n\n# Drop low-correlation features from the DataFrame\ndf_cleaned = df.drop(columns=low_correlation_features)\n\nprint(f\"Features dropped due to low correlation with responder_6: {list(low_correlation_features)}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:49:58.237410Z","iopub.execute_input":"2024-10-19T14:49:58.237764Z","iopub.status.idle":"2024-10-19T14:49:58.274291Z","shell.execute_reply.started":"2024-10-19T14:49:58.237724Z","shell.execute_reply":"2024-10-19T14:49:58.273235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Drop all responder columns except responder_6\ndata = df.drop(columns=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                        'responder_4', 'responder_5', 'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(20, 16))\n\n# Generate the heatmap for the dataset correlations\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns (Including responder_6)')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:50:39.668459Z","iopub.execute_input":"2024-10-19T14:50:39.669244Z","iopub.status.idle":"2024-10-19T14:51:04.198731Z","shell.execute_reply.started":"2024-10-19T14:50:39.669180Z","shell.execute_reply":"2024-10-19T14:51:04.197673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Drop all responder columns except responder_6\ndata = df_cleaned.drop(columns=['responder_0', 'responder_1', 'responder_2', 'responder_3', \n                        'responder_4', 'responder_5', 'responder_7', 'responder_8'])\n\n# Set up the figure size for a large heatmap\nplt.figure(figsize=(8, 8))\n\n# Generate the heatmap for the dataset correlations\nsns.heatmap(data.corr(), annot=False, cmap='coolwarm', linewidths=0.5)\n\n# Add title\nplt.title('Heatmap of Correlations for All Columns (Including responder_6)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:51:04.200819Z","iopub.execute_input":"2024-10-19T14:51:04.201296Z","iopub.status.idle":"2024-10-19T14:51:04.600460Z","shell.execute_reply.started":"2024-10-19T14:51:04.201241Z","shell.execute_reply":"2024-10-19T14:51:04.599469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cleaned.drop(columns=['responder_0',\t'responder_1',\t'responder_2',\t'responder_3',\t'responder_4',\t'responder_5',  'responder_7',\t'responder_8']).head()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:51:04.601802Z","iopub.execute_input":"2024-10-19T14:51:04.602174Z","iopub.status.idle":"2024-10-19T14:51:04.623150Z","shell.execute_reply.started":"2024-10-19T14:51:04.602134Z","shell.execute_reply":"2024-10-19T14:51:04.621911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Assuming df_cleaned is already preprocessed and loaded\n\n# Define the target variable (responder_6) and features\nX = df_cleaned.drop(columns=['responder_6'])  # Features\ny = df_cleaned['responder_6']  # Target variable\n\n# Split the data into training and testing sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Initialize the Random Forest Regressor\nrf_model = RandomForestRegressor(n_estimators=100, random_state=42)\n\n# Train the model\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = rf_model.predict(X_test)\n\n# Evaluate the model performance\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error: {mse}\")\nprint(f\"R-squared: {r2}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T14:51:04.625646Z","iopub.execute_input":"2024-10-19T14:51:04.626050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Assuming df_cleaned is already preprocessed and loaded\n\n# Define the target variable (responder_6) and features\nX = df_cleaned.drop(columns=['responder_6'])  # Features\ny = df_cleaned['responder_6']  # Target variable\n\n# Split the data into training and testing sets (80% train, 20% test)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Standardize the features\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Build the neural network model\nmodel = Sequential()\nmodel.add(Dense(64, activation='relu', input_shape=(X_train.shape[1],)))  # First hidden layer\nmodel.add(Dense(32, activation='relu'))  # Second hidden layer\nmodel.add(Dense(1))  # Output layer for regression\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mean_squared_error')\n\n# Train the model\nhistory = model.fit(X_train, y_train, validation_split=0.2, epochs=100, batch_size=32, verbose=1)\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Evaluate the model performance\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error: {mse}\")\nprint(f\"R-squared: {r2}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}