{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T18:16:42.540558Z","iopub.execute_input":"2024-10-30T18:16:42.541084Z","iopub.status.idle":"2024-10-30T18:16:44.339383Z","shell.execute_reply.started":"2024-10-30T18:16:42.541035Z","shell.execute_reply":"2024-10-30T18:16:44.337742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install required libraries (if not already installed)\n!pip install pandas numpy matplotlib seaborn scikit-learn xgboost","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:16:44.341313Z","iopub.execute_input":"2024-10-30T18:16:44.342508Z","iopub.status.idle":"2024-10-30T18:17:01.670311Z","shell.execute_reply.started":"2024-10-30T18:16:44.342456Z","shell.execute_reply":"2024-10-30T18:17:01.668936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import r2_score\nfrom xgboost import XGBRegressor\nimport glob\n\n# Load and Combine Train Parquet Partitions\n# train_files = glob.glob(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/*/*.parquet\")\n# df_list = [pd.read_parquet(file) for file in train_files]\n# train_df = pd.concat(df_list, ignore_index=True)\n# print(\"Combined Train Data Shape:\", train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:35:14.982344Z","iopub.execute_input":"2024-10-30T18:35:14.982900Z","iopub.status.idle":"2024-10-30T18:35:16.481682Z","shell.execute_reply.started":"2024-10-30T18:35:14.982823Z","shell.execute_reply":"2024-10-30T18:35:16.480207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Specify paths to only two of the partition files\ntrain_file_1 = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet\"\ntrain_file_2 = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1/part-0.parquet\"\n\n# Load only these two files\ntrain_df_1 = pd.read_parquet(train_file_1)\ntrain_df_2 = pd.read_parquet(train_file_2)\n\n# Concatenate the two DataFrames\ntrain_df = pd.concat([train_df_1, train_df_2], ignore_index=True)\n\nprint(\"Optimized Combined Train Data Shape:\", train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:29:37.659449Z","iopub.execute_input":"2024-10-30T18:29:37.660021Z","iopub.status.idle":"2024-10-30T18:29:49.480272Z","shell.execute_reply.started":"2024-10-30T18:29:37.659968Z","shell.execute_reply":"2024-10-30T18:29:49.478804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:30:34.156051Z","iopub.execute_input":"2024-10-30T18:30:34.156542Z","iopub.status.idle":"2024-10-30T18:30:34.167537Z","shell.execute_reply.started":"2024-10-30T18:30:34.156496Z","shell.execute_reply":"2024-10-30T18:30:34.166010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Initial Data Inspection\nprint(\"Data Shape:\", train_df.shape)\nprint(\"Data Columns:\", train_df.columns)\nprint(train_df.info())\nprint(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:32:10.412306Z","iopub.execute_input":"2024-10-30T18:32:10.412822Z","iopub.status.idle":"2024-10-30T18:32:10.477790Z","shell.execute_reply.started":"2024-10-30T18:32:10.412771Z","shell.execute_reply":"2024-10-30T18:32:10.476305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Check for Missing Values\nmissing_values = train_df.isnull().sum()\nprint(\"Missing Values in Each Column:\\n\", missing_values[missing_values > 0])\n\n# 3. Basic Statistical Summaries\nprint(\"Statistical Summary:\\n\", train_df.describe())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:32:30.334209Z","iopub.execute_input":"2024-10-30T18:32:30.334759Z","iopub.status.idle":"2024-10-30T18:32:48.602637Z","shell.execute_reply.started":"2024-10-30T18:32:30.334709Z","shell.execute_reply":"2024-10-30T18:32:48.601311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Correlation Analysis\ncorrelation_matrix = train_df.corr()\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, cmap=\"coolwarm\", vmin=-1, vmax=1)\nplt.title(\"Correlation Matrix\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:35:29.986330Z","iopub.execute_input":"2024-10-30T18:35:29.987443Z","iopub.status.idle":"2024-10-30T18:35:31.210783Z","shell.execute_reply.started":"2024-10-30T18:35:29.987386Z","shell.execute_reply":"2024-10-30T18:35:31.209305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 5. Distribution of `weight` and `feature_00`\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nsns.histplot(train_df['weight'], kde=True, color='blue', bins=30)\nplt.title(\"Distribution of 'weight'\")\nplt.subplot(1, 2, 2)\nsns.histplot(train_df['feature_00'], kde=True, color='green', bins=30)\nplt.title(\"Distribution of 'feature_00'\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:35:44.947970Z","iopub.execute_input":"2024-10-30T18:35:44.949601Z","iopub.status.idle":"2024-10-30T18:36:44.334994Z","shell.execute_reply.started":"2024-10-30T18:35:44.949538Z","shell.execute_reply":"2024-10-30T18:36:44.333246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 6. Exploring Relationship between Selected Features and Responders\nsns.pairplot(train_df[['feature_00', 'feature_01', 'feature_02', 'responder_0']])\nplt.suptitle(\"Pairplot for Selected Features and Responder_0\", y=1.02)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:36:44.337579Z","iopub.execute_input":"2024-10-30T18:36:44.338031Z","iopub.status.idle":"2024-10-30T18:40:51.960443Z","shell.execute_reply.started":"2024-10-30T18:36:44.337985Z","shell.execute_reply":"2024-10-30T18:40:51.958883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the responder and feature CSVs\nfeatures_df = pd.read_csv(\"/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv\")\nresponders_df = pd.read_csv(\"/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv\")\n\n# Verify column names before merging\nprint(\"Columns in train_df:\", train_df.columns)\nprint(\"Columns in responders_df:\", responders_df.columns)\nprint(\"Columns in features_df:\", features_df.columns)\n\n# Merge Responders and Features with Train Data\n# Adjust column names if necessary based on the print statements\ntry:\n    train_df = train_df.merge(responders_df, on='date_id', how='left')\n    train_df = train_df.merge(features_df, on='feature_id', how='left')\n    print(\"Merged Data Shape:\", train_df.shape)\nexcept KeyError as e:\n    print(f\"Column not found during merge: {e}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:47:12.299932Z","iopub.execute_input":"2024-10-30T18:47:12.301434Z","iopub.status.idle":"2024-10-30T18:47:12.330355Z","shell.execute_reply.started":"2024-10-30T18:47:12.301374Z","shell.execute_reply":"2024-10-30T18:47:12.328735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initial Exploration\nprint(\"Dataset Columns:\", train_df.columns)\nprint(\"Dataset Types:\\n\", train_df.dtypes)\nprint(\"Missing Values:\\n\", train_df.isnull().sum())\n\n# Target Variable Distribution (Responder_6)\nplt.figure(figsize=(10, 6))\nsns.histplot(train_df['responder_6'], kde=True, bins=30)\nplt.title('Responder_6 Distribution')\nplt.xlabel('Responder_6')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:47:47.249420Z","iopub.execute_input":"2024-10-30T18:47:47.250004Z","iopub.status.idle":"2024-10-30T18:48:40.870953Z","shell.execute_reply.started":"2024-10-30T18:47:47.249948Z","shell.execute_reply":"2024-10-30T18:48:40.869412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation Analysis\ncorr_matrix = train_df.corr()\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr_matrix, cmap='coolwarm', annot=False, fmt=\".2f\")\nplt.title(\"Correlation Heatmap\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:48:40.873480Z","iopub.execute_input":"2024-10-30T18:48:40.873927Z","iopub.status.idle":"2024-10-30T18:50:25.694646Z","shell.execute_reply.started":"2024-10-30T18:48:40.873880Z","shell.execute_reply":"2024-10-30T18:50:25.693240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Selection by Correlation\ncorrelation_target = corr_matrix['responder_6'].sort_values(ascending=False)\ntop_features = correlation_target.index[1:11]  # Top 10 correlated features, excluding responder_6 itself\nprint(\"Top Correlated Features with responder_6:\\n\", correlation_target.head(10))","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:25.696399Z","iopub.execute_input":"2024-10-30T18:50:25.696816Z","iopub.status.idle":"2024-10-30T18:50:25.709459Z","shell.execute_reply.started":"2024-10-30T18:50:25.696774Z","shell.execute_reply":"2024-10-30T18:50:25.706689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Distributions\ntrain_df[top_features].hist(figsize=(15, 10))\nplt.suptitle(\"Histograms of Top Correlated Features\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:25.713637Z","iopub.execute_input":"2024-10-30T18:50:25.714115Z","iopub.status.idle":"2024-10-30T18:50:29.654288Z","shell.execute_reply.started":"2024-10-30T18:50:25.714071Z","shell.execute_reply":"2024-10-30T18:50:29.652774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocessing\nX = train_df[top_features]  # Selecting only top features for initial model\ny = train_df['responder_6']\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:29.655835Z","iopub.execute_input":"2024-10-30T18:50:29.656282Z","iopub.status.idle":"2024-10-30T18:50:31.004274Z","shell.execute_reply.started":"2024-10-30T18:50:29.656236Z","shell.execute_reply":"2024-10-30T18:50:31.002620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Time-Series Cross Validation\ntscv = TimeSeriesSplit(n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:31.006342Z","iopub.execute_input":"2024-10-30T18:50:31.007244Z","iopub.status.idle":"2024-10-30T18:50:31.012633Z","shell.execute_reply.started":"2024-10-30T18:50:31.007160Z","shell.execute_reply":"2024-10-30T18:50:31.011293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Training and Validation\nxgb_params = {\n    'objective': 'reg:squarederror',\n    'n_estimators': 100,\n    'learning_rate': 0.1,\n    'max_depth': 6,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'random_state': 42\n}\n\nxgb_model = XGBRegressor(**xgb_params)\nr2_scores = []","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:31.014098Z","iopub.execute_input":"2024-10-30T18:50:31.014603Z","iopub.status.idle":"2024-10-30T18:50:31.030902Z","shell.execute_reply.started":"2024-10-30T18:50:31.014557Z","shell.execute_reply":"2024-10-30T18:50:31.029278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for train_index, test_index in tscv.split(X_scaled):\n    X_train, X_test = X_scaled[train_index], X_scaled[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n    \n    xgb_model.fit(X_train, y_train)\n    y_pred = xgb_model.predict(X_test)\n    r2 = r2_score(y_test, y_pred)\n    r2_scores.append(r2)\n    print(f\"R-squared Score: {r2:.4f}\")\n\nprint(\"Average R-squared Score:\", np.mean(r2_scores))","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:50:31.033075Z","iopub.execute_input":"2024-10-30T18:50:31.033592Z","iopub.status.idle":"2024-10-30T18:52:14.583839Z","shell.execute_reply.started":"2024-10-30T18:50:31.033545Z","shell.execute_reply":"2024-10-30T18:52:14.582684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Importance Plot\nplt.figure(figsize=(10, 6))\nfeature_importances = xgb_model.feature_importances_\nsns.barplot(x=top_features, y=feature_importances)\nplt.xticks(rotation=45)\nplt.title(\"Feature Importances from XGBoost Model\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T18:52:14.588387Z","iopub.execute_input":"2024-10-30T18:52:14.588863Z","iopub.status.idle":"2024-10-30T18:52:14.901885Z","shell.execute_reply.started":"2024-10-30T18:52:14.588810Z","shell.execute_reply":"2024-10-30T18:52:14.900468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}