{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd  # For handling tabular data\nimport numpy as np  # For numerical computations\nimport matplotlib.pyplot as plt  # For basic visualizations\nimport seaborn as sns  # For advanced visualizations\nimport os  # For file and directory operations\n\n# Set visualization style\nsns.set_theme(style=\"whitegrid\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:31:42.181405Z","iopub.execute_input":"2024-11-30T08:31:42.181815Z","iopub.status.idle":"2024-11-30T08:31:43.427395Z","shell.execute_reply.started":"2024-11-30T08:31:42.181779Z","shell.execute_reply":"2024-11-30T08:31:43.426198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Path to the training data folder\ntrain_folder = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/'\n\n# Function to load and process each partition incrementally\ndef load_and_sample_partitions(folder_path, sample_frac=0.1):\n    \"\"\"\n    Load and process dataset partitions incrementally.\n    Args:\n    - folder_path: Path to the partitioned dataset folder.\n    - sample_frac: Fraction of rows to sample from each partition.\n    \n    Returns:\n    - Combined sampled DataFrame.\n    \"\"\"\n    combined_df = []\n    for i in range(10):  # Loop through partitions 0 to 9\n        partition_path = os.path.join(folder_path, f'partition_id={i}')\n        print(f\"Loading partition {i}...\")\n        partition_df = pd.read_parquet(partition_path)\n        \n        # Sample a fraction of rows to reduce memory usage\n        sampled_df = partition_df.sample(frac=sample_frac, random_state=42)\n        combined_df.append(sampled_df)\n        \n        print(f\"Partition {i} loaded and sampled. Shape: {sampled_df.shape}\")\n    \n    # Combine all sampled DataFrames into one\n    return pd.concat(combined_df, ignore_index=True)\n\n# Load and sample the partitions\ntrain_data_sampled = load_and_sample_partitions(train_folder, sample_frac=0.1)\n\n# Display the shape of the sampled DataFrame\nprint(f\"Sampled DataFrame Shape: {train_data_sampled.shape}\")\n\n# View the first few rows\ntrain_data_sampled.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:31:43.429564Z","iopub.execute_input":"2024-11-30T08:31:43.430259Z","iopub.status.idle":"2024-11-30T08:33:11.765909Z","shell.execute_reply.started":"2024-11-30T08:31:43.430205Z","shell.execute_reply":"2024-11-30T08:33:11.764530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmissing_values = train_data_sampled.isnull().sum()\n\n# Display missing values for each column\nprint(\"Missing Values:\\n\", missing_values[missing_values > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:33:11.768155Z","iopub.execute_input":"2024-11-30T08:33:11.768671Z","iopub.status.idle":"2024-11-30T08:33:12.320428Z","shell.execute_reply.started":"2024-11-30T08:33:11.768619Z","shell.execute_reply":"2024-11-30T08:33:12.319353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the shape and first few rows of the combined dataset\nprint(f\"Shape of train_data_sampled: {train_data_sampled.shape}\")\nprint(train_data_sampled.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:33:12.322722Z","iopub.execute_input":"2024-11-30T08:33:12.323061Z","iopub.status.idle":"2024-11-30T08:33:12.338256Z","shell.execute_reply.started":"2024-11-30T08:33:12.323029Z","shell.execute_reply":"2024-11-30T08:33:12.337001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute mean for all missing values\ntrain_data_imputed = train_data_sampled.fillna(train_data_sampled.mean())\n\n# Verify if there are any remaining missing values\nmissing_after_imputation = train_data_imputed.isnull().sum().sum()\nprint(f\"Total missing values after imputation: {missing_after_imputation}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:33:12.339794Z","iopub.execute_input":"2024-11-30T08:33:12.340140Z","iopub.status.idle":"2024-11-30T08:33:19.269378Z","shell.execute_reply.started":"2024-11-30T08:33:12.340089Z","shell.execute_reply":"2024-11-30T08:33:19.268045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_imputed.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:34:13.076712Z","iopub.execute_input":"2024-11-30T08:34:13.077900Z","iopub.status.idle":"2024-11-30T08:34:13.851368Z","shell.execute_reply.started":"2024-11-30T08:34:13.077838Z","shell.execute_reply":"2024-11-30T08:34:13.850174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"print(train_data_imputed.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:36:44.017941Z","iopub.execute_input":"2024-11-30T08:36:44.019169Z","iopub.status.idle":"2024-11-30T08:37:03.277082Z","shell.execute_reply.started":"2024-11-30T08:36:44.019082Z","shell.execute_reply":"2024-11-30T08:37:03.275827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze correlations between features and responders to identify useful predictors\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 8))\ncorr = train_data_imputed.corr()\nsns.heatmap(corr, cmap=\"coolwarm\", center=0, annot=False, fmt=\".2f\")\nplt.title(\"Feature Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:40:41.856120Z","iopub.execute_input":"2024-11-30T08:40:41.856636Z","iopub.status.idle":"2024-11-30T08:42:33.735917Z","shell.execute_reply.started":"2024-11-30T08:40:41.856579Z","shell.execute_reply":"2024-11-30T08:42:33.734483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Focus on correlations with responder_6 (target variable)\nresponder_corr = corr[\"responder_6\"].sort_values(ascending=False)\nprint(responder_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:44:52.921796Z","iopub.execute_input":"2024-11-30T08:44:52.922285Z","iopub.status.idle":"2024-11-30T08:44:52.933394Z","shell.execute_reply.started":"2024-11-30T08:44:52.922246Z","shell.execute_reply":"2024-11-30T08:44:52.932168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", None)  # Show all rows\nprint(responder_corr)\npd.reset_option(\"display.max_rows\")  # Reset after viewing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:56:21.683248Z","iopub.execute_input":"2024-11-30T08:56:21.683773Z","iopub.status.idle":"2024-11-30T08:56:21.692485Z","shell.execute_reply.started":"2024-11-30T08:56:21.683730Z","shell.execute_reply":"2024-11-30T08:56:21.691365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Filtering the correlations\nthreshold = 0.1  \nsignificant_corr = responder_corr[responder_corr.abs() > threshold]\nprint(significant_corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:56:57.865743Z","iopub.execute_input":"2024-11-30T08:56:57.867103Z","iopub.status.idle":"2024-11-30T08:56:57.875176Z","shell.execute_reply.started":"2024-11-30T08:56:57.867056Z","shell.execute_reply":"2024-11-30T08:56:57.873697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Visualizing the correlations\nimport matplotlib.pyplot as plt\n\ntop_corr = significant_corr.sort_values(ascending=False)\nplt.figure(figsize=(10, 8))\ntop_corr.plot(kind='barh', color='skyblue')\nplt.title(\"Feature Correlations with responder_6\")\nplt.xlabel(\"Correlation Coefficient\")\nplt.ylabel(\"Features\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T08:57:57.704054Z","iopub.execute_input":"2024-11-30T08:57:57.704454Z","iopub.status.idle":"2024-11-30T08:57:57.990860Z","shell.execute_reply.started":"2024-11-30T08:57:57.704421Z","shell.execute_reply":"2024-11-30T08:57:57.989697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identifying the top predictors\ntop_positive_predictors = significant_corr[significant_corr > 0].sort_values(ascending=False).head(10)\ntop_negative_predictors = significant_corr[significant_corr < 0].sort_values().head(10)\nprint(\"Top Positive Predictors:\\n\", top_positive_predictors)\nprint(\"Top Negative Predictors:\\n\", top_negative_predictors)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T09:00:05.861982Z","iopub.execute_input":"2024-11-30T09:00:05.862417Z","iopub.status.idle":"2024-11-30T09:00:05.872310Z","shell.execute_reply.started":"2024-11-30T09:00:05.862380Z","shell.execute_reply":"2024-11-30T09:00:05.870906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize to better understand the relationships between responder_6 and its top predictors:\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\ntop_predictors = [\"responder_3\", \"responder_8\", \"responder_7\", \"responder_4\", \"responder_5\", \"responder_0\"]\nfor predictor in top_predictors:\n    plt.figure(figsize=(6, 4))\n    sns.scatterplot(x=train_data_sampled[predictor], y=train_data_sampled['responder_6'])\n    plt.title(f'Scatter Plot: responder_6 vs {predictor}')\n    plt.xlabel(predictor)\n    plt.ylabel('responder_6')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T09:01:32.346031Z","iopub.execute_input":"2024-11-30T09:01:32.346494Z","iopub.status.idle":"2024-11-30T09:02:33.270966Z","shell.execute_reply.started":"2024-11-30T09:01:32.346455Z","shell.execute_reply":"2024-11-30T09:02:33.269679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for predictor in top_predictors + [\"responder_6\"]:\n    plt.figure(figsize=(6, 4))\n    sns.histplot(train_data_sampled[predictor], kde=True, bins=50)\n    plt.title(f'Distribution of {predictor}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T09:02:33.273169Z","iopub.execute_input":"2024-11-30T09:02:33.273539Z","iopub.status.idle":"2024-11-30T09:05:04.108111Z","shell.execute_reply.started":"2024-11-30T09:02:33.273500Z","shell.execute_reply":"2024-11-30T09:05:04.106384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}