{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:05:54.468027Z","iopub.execute_input":"2024-12-28T05:05:54.468366Z","iopub.status.idle":"2024-12-28T05:05:54.912737Z","shell.execute_reply.started":"2024-12-28T05:05:54.468321Z","shell.execute_reply":"2024-12-28T05:05:54.911459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\nimport gc\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:05:54.913963Z","iopub.execute_input":"2024-12-28T05:05:54.914462Z","iopub.status.idle":"2024-12-28T05:05:56.279184Z","shell.execute_reply.started":"2024-12-28T05:05:54.914408Z","shell.execute_reply":"2024-12-28T05:05:56.278058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\npath = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nsamples = [] \n\n# Load a data from each file:\nr = range(10)\nfor i in r:\n    file_path = f\"{path}/train.parquet/partition_id={i}/part-0.parquet\"\n    part = pd.read_parquet(file_path)\n    part = part[['date_id','time_id','symbol_id','weight','responder_6']]\n    samples.append(part)\n    \n#sample_df = pd.concat(samples, ignore_index=True) # Concatenate all samples into one DataFrame if needed\n\n#sample_df.round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:05:56.284777Z","iopub.execute_input":"2024-12-28T05:05:56.285043Z","iopub.status.idle":"2024-12-28T05:07:28.814105Z","shell.execute_reply.started":"2024-12-28T05:05:56.285018Z","shell.execute_reply":"2024-12-28T05:07:28.812556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = pd.concat(samples, ignore_index=True) # Concatenate all samples into one DataFrame if needed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:28.816800Z","iopub.execute_input":"2024-12-28T05:07:28.817127Z","iopub.status.idle":"2024-12-28T05:07:29.143809Z","shell.execute_reply.started":"2024-12-28T05:07:28.817101Z","shell.execute_reply":"2024-12-28T05:07:29.142683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:29.144990Z","iopub.execute_input":"2024-12-28T05:07:29.145357Z","iopub.status.idle":"2024-12-28T05:07:29.174559Z","shell.execute_reply.started":"2024-12-28T05:07:29.145328Z","shell.execute_reply":"2024-12-28T05:07:29.173026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path1 = f\"{path}/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv\"\npart1 = pd.read_parquet(file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:29.175675Z","iopub.execute_input":"2024-12-28T05:07:29.175976Z","iopub.status.idle":"2024-12-28T05:07:32.742817Z","shell.execute_reply.started":"2024-12-28T05:07:29.175947Z","shell.execute_reply":"2024-12-28T05:07:32.741758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['feature_23'] = part1['feature_23']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:32.743902Z","iopub.execute_input":"2024-12-28T05:07:32.744202Z","iopub.status.idle":"2024-12-28T05:07:34.451541Z","shell.execute_reply.started":"2024-12-28T05:07:32.744173Z","shell.execute_reply":"2024-12-28T05:07:34.450263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:34.452653Z","iopub.execute_input":"2024-12-28T05:07:34.452952Z","iopub.status.idle":"2024-12-28T05:07:34.467035Z","shell.execute_reply.started":"2024-12-28T05:07:34.452923Z","shell.execute_reply":"2024-12-28T05:07:34.465964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned = sample_df.dropna(subset=['feature_23'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:34.467999Z","iopub.execute_input":"2024-12-28T05:07:34.468291Z","iopub.status.idle":"2024-12-28T05:07:34.971608Z","shell.execute_reply.started":"2024-12-28T05:07:34.468264Z","shell.execute_reply":"2024-12-28T05:07:34.970454Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 删除sample_df中feature_23为NAN的行","metadata":{}},{"cell_type":"code","source":"df_cleaned","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:34.972651Z","iopub.execute_input":"2024-12-28T05:07:34.972946Z","iopub.status.idle":"2024-12-28T05:07:34.986817Z","shell.execute_reply.started":"2024-12-28T05:07:34.972919Z","shell.execute_reply":"2024-12-28T05:07:34.985628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:34.987868Z","iopub.execute_input":"2024-12-28T05:07:34.988261Z","iopub.status.idle":"2024-12-28T05:07:35.278264Z","shell.execute_reply.started":"2024-12-28T05:07:34.988222Z","shell.execute_reply":"2024-12-28T05:07:35.277247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:35.281645Z","iopub.execute_input":"2024-12-28T05:07:35.281948Z","iopub.status.idle":"2024-12-28T05:07:35.327675Z","shell.execute_reply.started":"2024-12-28T05:07:35.281921Z","shell.execute_reply":"2024-12-28T05:07:35.326697Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 由上述可见：sample_df中symbol_id 的值为5, 20, 21, 22, 25, 26, 29, 36, 23, 27, 28, 35, 37, 4, 24, 31, 6, 18, 32 的行feature_23的值全为NAN","metadata":{}},{"cell_type":"code","source":"#数据的统计描述\nprint(df_cleaned[['weight','responder_6']].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:35.329074Z","iopub.execute_input":"2024-12-28T05:07:35.329454Z","iopub.status.idle":"2024-12-28T05:07:35.809065Z","shell.execute_reply.started":"2024-12-28T05:07:35.329424Z","shell.execute_reply":"2024-12-28T05:07:35.808040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# feature_23的值主要集中分布在[-2,2]之间，0左右由为密集","metadata":{}},{"cell_type":"code","source":"# 分析feature_23的分布\nsns.histplot(df_cleaned['feature_23'], kde=True)  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:07:35.810309Z","iopub.execute_input":"2024-12-28T05:07:35.810641Z","iopub.status.idle":"2024-12-28T05:08:05.399836Z","shell.execute_reply.started":"2024-12-28T05:07:35.810613Z","shell.execute_reply":"2024-12-28T05:08:05.398686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 分析responder_6的分布\nsns.histplot(df_cleaned['responder_6'], kde=True)  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:12:40.557288Z","iopub.execute_input":"2024-12-28T05:12:40.558118Z","iopub.status.idle":"2024-12-28T05:13:10.394796Z","shell.execute_reply.started":"2024-12-28T05:12:40.558058Z","shell.execute_reply":"2024-12-28T05:13:10.393693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 相关性系数分析---0.0005168035099590274 相关性较弱","metadata":{}},{"cell_type":"code","source":"#sample_df中所有数据与feature_23的相关性\ncorrelation_matrix = df_cleaned.corr()\nprint(correlation_matrix)\n\n# 可视化相关性矩阵\n#sns.heatmap()：这是Seaborn库中用于创建热图的函数。\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:05.401109Z","iopub.execute_input":"2024-12-28T05:08:05.401534Z","iopub.status.idle":"2024-12-28T05:08:06.572618Z","shell.execute_reply.started":"2024-12-28T05:08:05.401490Z","shell.execute_reply":"2024-12-28T05:08:06.571431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#看responder_6和feature_23之间的相关系数\ncorrelation = df_cleaned['feature_23'].corr(df_cleaned['responder_6'])\nprint(f\"Correlation between feature_23 and responder_6: {correlation}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:06.573765Z","iopub.execute_input":"2024-12-28T05:08:06.574183Z","iopub.status.idle":"2024-12-28T05:08:06.733010Z","shell.execute_reply.started":"2024-12-28T05:08:06.574141Z","shell.execute_reply":"2024-12-28T05:08:06.732000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 箱形图和散点图意义不是很大（不对数据做处理）","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(x='feature_23', y='responder_6', data=df_cleaned)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:06.734003Z","iopub.execute_input":"2024-12-28T05:08:06.734409Z","iopub.status.idle":"2024-12-28T05:08:19.897653Z","shell.execute_reply.started":"2024-12-28T05:08:06.734352Z","shell.execute_reply":"2024-12-28T05:08:19.896137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#绘制箱形图\n#sns.boxplot(x='feature_23', y='responder_6', data=df_cleaned)\n#plt.title('Grouped Box Plot using Seaborn')\n#plt.xlabel('feature_23')\n#plt.ylabel('responder_6')\n#plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:19.898845Z","iopub.execute_input":"2024-12-28T05:08:19.899164Z","iopub.status.idle":"2024-12-28T05:08:19.903687Z","shell.execute_reply.started":"2024-12-28T05:08:19.899136Z","shell.execute_reply":"2024-12-28T05:08:19.902264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=df_cleaned.groupby('symbol_id',as_index=False)[['feature_23']].mean()\ndf1=df_cleaned.groupby('symbol_id',as_index=False)[['responder_6']].mean()\ndf['responder_6']=df1['responder_6']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:19.904850Z","iopub.execute_input":"2024-12-28T05:08:19.905167Z","iopub.status.idle":"2024-12-28T05:08:20.141974Z","shell.execute_reply.started":"2024-12-28T05:08:19.905121Z","shell.execute_reply":"2024-12-28T05:08:20.141103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:08:20.142891Z","iopub.execute_input":"2024-12-28T05:08:20.143240Z","iopub.status.idle":"2024-12-28T05:08:20.154222Z","shell.execute_reply.started":"2024-12-28T05:08:20.143205Z","shell.execute_reply":"2024-12-28T05:08:20.153185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 分类处理后，responder_6和feature_23的相关性上升","metadata":{}},{"cell_type":"code","source":"correlation_matrix_m = df.corr()\nprint(correlation_matrix_m)\n\n# 可视化相关性矩阵\n#sns.heatmap()：这是Seaborn库中用于创建热图的函数。\nsns.heatmap(correlation_matrix_m, annot=True, cmap='coolwarm')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:21:54.949946Z","iopub.execute_input":"2024-12-28T05:21:54.950228Z","iopub.status.idle":"2024-12-28T05:21:55.232826Z","shell.execute_reply.started":"2024-12-28T05:21:54.950202Z","shell.execute_reply":"2024-12-28T05:21:55.231832Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 构造线性回归","metadata":{}},{"cell_type":"code","source":"import statsmodels.api as sm\n\n# 添加常数项，用于计算截距\ndf['constant'] = 1\n#x为自变量y为因变量\nX = df[['feature_23', 'constant']]\ny = df['responder_6']\nX = sm.add_constant(X)\n#拟合线性回归模型\nmodel = sm.OLS(y, X).fit()\nprint(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:26:57.292198Z","iopub.execute_input":"2024-12-28T05:26:57.292620Z","iopub.status.idle":"2024-12-28T05:26:58.268803Z","shell.execute_reply.started":"2024-12-28T05:26:57.292586Z","shell.execute_reply":"2024-12-28T05:26:58.267785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def group(num_feature):\n    if 0.1 <= num_feature < 0.15:\n        return 0  # Group 1\n    elif 0.15 <= num_feature < 0.2:\n        return 1  # Group 2 \n    elif 0.2 <= num_feature < 0.25:\n        return 2 # Group 3 \n    elif 0.25 <= num_feature < 0.3:\n        return 3  # Group 4 \n    else:\n        return -1  # 如果有其他取值，可以单独标记为异常组\n\n# 应用分组编码，确保 df 和 'feature_23' 列已正确定义和存在\ndf['feature_23_group'] = df['feature_23'].apply(group)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:46:53.646914Z","iopub.execute_input":"2024-12-28T05:46:53.647344Z","iopub.status.idle":"2024-12-28T05:46:53.654485Z","shell.execute_reply.started":"2024-12-28T05:46:53.647315Z","shell.execute_reply":"2024-12-28T05:46:53.653144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = df[['feature_23_group', 'responder_6']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:46:55.745171Z","iopub.execute_input":"2024-12-28T05:46:55.745561Z","iopub.status.idle":"2024-12-28T05:46:55.751642Z","shell.execute_reply.started":"2024-12-28T05:46:55.745531Z","shell.execute_reply":"2024-12-28T05:46:55.750227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:46:56.223199Z","iopub.execute_input":"2024-12-28T05:46:56.223601Z","iopub.status.idle":"2024-12-28T05:46:56.233978Z","shell.execute_reply.started":"2024-12-28T05:46:56.223566Z","shell.execute_reply":"2024-12-28T05:46:56.232935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 组1，组3明显负影响，组2正影响","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\nsns.boxplot(x=\"feature_23_group\",y=\"responder_6\",data=train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:49:19.039042Z","iopub.execute_input":"2024-12-28T05:49:19.039429Z","iopub.status.idle":"2024-12-28T05:49:19.501117Z","shell.execute_reply.started":"2024-12-28T05:49:19.039396Z","shell.execute_reply":"2024-12-28T05:49:19.500039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(x='feature_23_group', y='responder_6', data=train)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T05:50:18.317836Z","iopub.execute_input":"2024-12-28T05:50:18.318279Z","iopub.status.idle":"2024-12-28T05:50:18.616328Z","shell.execute_reply.started":"2024-12-28T05:50:18.318243Z","shell.execute_reply":"2024-12-28T05:50:18.615156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}