{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:10.731480Z","iopub.execute_input":"2024-12-27T09:26:10.731835Z","iopub.status.idle":"2024-12-27T09:26:10.756734Z","shell.execute_reply.started":"2024-12-27T09:26:10.731806Z","shell.execute_reply":"2024-12-27T09:26:10.755524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\nimport warnings\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:10.758558Z","iopub.execute_input":"2024-12-27T09:26:10.759133Z","iopub.status.idle":"2024-12-27T09:26:10.764294Z","shell.execute_reply.started":"2024-12-27T09:26:10.759093Z","shell.execute_reply":"2024-12-27T09:26:10.763103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 数据加载与清洗\n\n首先，我们加载了 `train.parquet` 数据文件，并选择了其中的三个特征：`feature_24`、`feature_25` 和 `responder_6`，以进行后续分析。为了确保数据的质量，我们替换了无穷值（`np.inf` 和 `-np.inf`）为缺失值（`NaN`），并删除了含有缺失值的行。\n\n通过这一步骤，我们确保了数据的整洁性，并为进一步的分析奠定了基础。\n","metadata":{}},{"cell_type":"code","source":"# 屏蔽 FutureWarning\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n# 文件路径\nfile_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet'\n\n# 加载数据\nprint(\"加载数据中...\")\ntrain = pd.read_parquet(file_path, columns=['feature_24', 'feature_25', 'responder_6']).head(500000)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:10.766789Z","iopub.execute_input":"2024-12-27T09:26:10.767179Z","iopub.status.idle":"2024-12-27T09:26:11.713958Z","shell.execute_reply.started":"2024-12-27T09:26:10.767151Z","shell.execute_reply":"2024-12-27T09:26:11.712907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 替换无穷值为 NaN\ntrain.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# 数据清洗：删除缺失值\nprint(\"数据清洗中...\")\ndata_cleaned = train.dropna()\ndel train  # 释放内存\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:11.715737Z","iopub.execute_input":"2024-12-27T09:26:11.716087Z","iopub.status.idle":"2024-12-27T09:26:11.951380Z","shell.execute_reply.started":"2024-12-27T09:26:11.716059Z","shell.execute_reply":"2024-12-27T09:26:11.950101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 描述统计\n\n为了更好地了解数据的基本情况，我们对 `feature_24`、`feature_25` 和 `responder_6` 进行了描述统计。以下是数据的主要统计信息：\n\n- **feature_24**：均值为 -0.6146，标准差为 0.7414，数据分布大致呈现负偏态。\n- **feature_25**：均值为 0.4112，标准差为 1.0986，数据分布较为广泛。\n- **responder_6**：均值为 -0.0048，标准差为 0.8186，数据表现出一定的离散性。\n\n这些统计信息帮助我们对数据有了初步的认识，尤其是在后续可视化和建模时非常有用。\n","metadata":{}},{"cell_type":"code","source":"# 描述统计\nprint(\"描述统计:\")\nprint(data_cleaned.describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:11.952590Z","iopub.execute_input":"2024-12-27T09:26:11.952962Z","iopub.status.idle":"2024-12-27T09:26:12.017090Z","shell.execute_reply.started":"2024-12-27T09:26:11.952935Z","shell.execute_reply":"2024-12-27T09:26:12.015848Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 数据分布可视化\n\n我们通过直方图和核密度估计（KDE）来展示 `feature_25` 和 `responder_6` 的数据分布。通过观察这些分布，我们可以更清楚地了解每个特征的值域以及它们的分布形态。\n\n- **feature_25** 的分布相对较为均匀，且呈现出一定的对称性。\n- **responder_6** 的分布相对集中，大部分数据位于接近 0 的区域。\n\n这些图表有助于我们了解数据的分布情况，从而为模型训练做出相应调整。\n","metadata":{}},{"cell_type":"code","source":"# 1. 绘制 feature_25 和 responder_6 分布\nplt.figure(figsize=(10, 6))\nsns.histplot(data_cleaned['feature_25'], bins=50, kde=True, color=\"skyblue\")\nplt.title(\"Distribution of feature_25\")\nplt.xlabel(\"feature_25\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:12.018350Z","iopub.execute_input":"2024-12-27T09:26:12.018675Z","iopub.status.idle":"2024-12-27T09:26:14.505158Z","shell.execute_reply.started":"2024-12-27T09:26:12.018619Z","shell.execute_reply":"2024-12-27T09:26:14.504061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(data_cleaned['responder_6'], bins=50, kde=True, color=\"orange\")\nplt.title(\"Distribution of responder_6\")\nplt.xlabel(\"responder_6\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:14.506180Z","iopub.execute_input":"2024-12-27T09:26:14.506514Z","iopub.status.idle":"2024-12-27T09:26:17.066498Z","shell.execute_reply.started":"2024-12-27T09:26:14.506488Z","shell.execute_reply":"2024-12-27T09:26:17.065469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 散点图分析\n\n我们绘制了 `feature_25` 和 `responder_6` 的散点图，以观察它们之间的关系。从散点图中可以看出，二者之间并没有明显的线性关系，但仍然可能存在一定的非线性关联。\n\n这种分析方式可以帮助我们探索特征间的潜在关系，并为后续的建模提供有用的线索。\n","metadata":{}},{"cell_type":"code","source":"\n# 2. 绘制 feature_25 和 responder_6 的散点图\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='feature_25', y='responder_6', data=data_cleaned, alpha=0.3)\nplt.title(\"Scatter Plot of feature_25 vs responder_6\")\nplt.xlabel(\"feature_25\")\nplt.ylabel(\"responder_6\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:17.067459Z","iopub.execute_input":"2024-12-27T09:26:17.067744Z","iopub.status.idle":"2024-12-27T09:26:18.524349Z","shell.execute_reply.started":"2024-12-27T09:26:17.067718Z","shell.execute_reply":"2024-12-27T09:26:18.522906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 计算相关性\ncorrelation = data_cleaned['feature_25'].corr(data_cleaned['responder_6'])\nprint(f\"feature_25 和 responder_6 的相关性: {correlation:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:18.528156Z","iopub.execute_input":"2024-12-27T09:26:18.528622Z","iopub.status.idle":"2024-12-27T09:26:18.545776Z","shell.execute_reply.started":"2024-12-27T09:26:18.528582Z","shell.execute_reply":"2024-12-27T09:26:18.544626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 箱线图分析\n\n我们对 `feature_25` 进行了分箱处理，并使用箱线图展示了不同箱体内 `feature_25` 和 `responder_6` 的关系。箱线图清晰地展示了不同区间内数据的分布情况，能够有效地揭示潜在的异常值和数据分布的差异。\n\n从图中可以看到，`feature_25` 的不同分箱区间与 `responder_6` 的分布存在一定差异，说明分箱对预测有可能带来一定的影响。\n","metadata":{}},{"cell_type":"code","source":"# 3. 分箱分析（箱线图）\ndata_cleaned['feature_25_bin'] = pd.qcut(data_cleaned['feature_25'], 10, duplicates='drop')\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='feature_25_bin', y='responder_6', data=data_cleaned)\nplt.title(\"Boxplot of feature_25 Bins vs responder_6\")\nplt.xlabel(\"feature_25 (Binned)\")\nplt.ylabel(\"responder_6\")\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:18.547414Z","iopub.execute_input":"2024-12-27T09:26:18.547811Z","iopub.status.idle":"2024-12-27T09:26:19.001655Z","shell.execute_reply.started":"2024-12-27T09:26:18.547771Z","shell.execute_reply":"2024-12-27T09:26:19.000555Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 交互热力图分析\n\n为了进一步探究 `feature_24` 和 `feature_25` 对 `responder_6` 的影响，我们绘制了这两个特征的交互热力图。热力图通过计算每个特征组合的 `responder_6` 均值来展示它们之间的关系。\n\n从热力图中可以看出，`feature_24` 和 `feature_25` 的组合对 `responder_6` 的均值有一定影响，某些组合区域表现出较高的均值。\n","metadata":{}},{"cell_type":"code","source":"# 4. 绘制交互热力图\n# 对 feature_24 和 feature_25 进行分箱\ndata_cleaned['feature_24_bin'] = pd.qcut(data_cleaned['feature_24'], 10, duplicates='drop')\ndata_cleaned['feature_25_bin'] = pd.qcut(data_cleaned['feature_25'], 10, duplicates='drop')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:19.003237Z","iopub.execute_input":"2024-12-27T09:26:19.003650Z","iopub.status.idle":"2024-12-27T09:26:19.059734Z","shell.execute_reply.started":"2024-12-27T09:26:19.003608Z","shell.execute_reply":"2024-12-27T09:26:19.058126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 使用 groupby 对 feature_24_bin 和 feature_25_bin 进行聚合，计算 responder_6 的均值\nheatmap_data = data_cleaned.groupby(['feature_24_bin', 'feature_25_bin'])['responder_6'].mean().unstack()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:19.060932Z","iopub.execute_input":"2024-12-27T09:26:19.061353Z","iopub.status.idle":"2024-12-27T09:26:19.087933Z","shell.execute_reply.started":"2024-12-27T09:26:19.061317Z","shell.execute_reply":"2024-12-27T09:26:19.086940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 填充空值为 0 或其他合适的值\nheatmap_data = heatmap_data.fillna(0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:19.088946Z","iopub.execute_input":"2024-12-27T09:26:19.089320Z","iopub.status.idle":"2024-12-27T09:26:19.093705Z","shell.execute_reply.started":"2024-12-27T09:26:19.089279Z","shell.execute_reply":"2024-12-27T09:26:19.092681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# 绘制热力图\nplt.figure(figsize=(12, 8))\nsns.heatmap(heatmap_data, cmap='coolwarm', cbar_kws={'label': 'Mean responder_6'})\nplt.title(\"Heatmap of feature_24 vs feature_25 (Responder_6 Mean)\")\nplt.xlabel(\"feature_25 (Binned)\")\nplt.ylabel(\"feature_24 (Binned)\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:19.094639Z","iopub.execute_input":"2024-12-27T09:26:19.094939Z","iopub.status.idle":"2024-12-27T09:26:19.591275Z","shell.execute_reply.started":"2024-12-27T09:26:19.094914Z","shell.execute_reply":"2024-12-27T09:26:19.590087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# 5. 绘制交互分箱柱状图\ndata_cleaned['feature_24_bin'] = pd.qcut(data_cleaned['feature_24'], 10, duplicates='drop')\ndata_cleaned['feature_25_bin'] = pd.qcut(data_cleaned['feature_25'], 10, duplicates='drop')\ngrouped = data_cleaned.groupby(['feature_24_bin', 'feature_25_bin'])['responder_6'].mean().reset_index()\n\nplt.figure(figsize=(12, 8))\nsns.barplot(x='feature_24_bin', y='responder_6', hue='feature_25_bin', data=grouped, palette='Spectral')\nplt.title(\"Responder_6 by Binned Feature_24 and Feature_25\")\nplt.xlabel(\"Feature_24 (Binned)\")\nplt.ylabel(\"Responder_6\")\nplt.xticks(rotation=45)\nplt.legend(title=\"Feature_25 (Binned)\", bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:19.592413Z","iopub.execute_input":"2024-12-27T09:26:19.592713Z","iopub.status.idle":"2024-12-27T09:26:20.362353Z","shell.execute_reply.started":"2024-12-27T09:26:19.592688Z","shell.execute_reply":"2024-12-27T09:26:20.361453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# 保存清理后的数据\ndata_cleaned.to_csv(\"cleaned_data_combined.csv\", index=False)\nprint(\"清理后的数据已保存为 cleaned_data_combined.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-27T09:26:20.363309Z","iopub.execute_input":"2024-12-27T09:26:20.363618Z","iopub.status.idle":"2024-12-27T09:26:23.790678Z","shell.execute_reply.started":"2024-12-27T09:26:20.363594Z","shell.execute_reply":"2024-12-27T09:26:23.789618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}