{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:54:24.993042Z","iopub.execute_input":"2024-12-24T12:54:24.993319Z","iopub.status.idle":"2024-12-24T12:54:25.449067Z","shell.execute_reply.started":"2024-12-24T12:54:24.993291Z","shell.execute_reply":"2024-12-24T12:54:25.448003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\nimport gc\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:54:25.450255Z","iopub.execute_input":"2024-12-24T12:54:25.450837Z","iopub.status.idle":"2024-12-24T12:54:27.379839Z","shell.execute_reply.started":"2024-12-24T12:54:25.450796Z","shell.execute_reply":"2024-12-24T12:54:27.378810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\npath = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nsamples = [] \n\n# Load a data from each file:\nr = range(10)\nfor i in r:\n    file_path = f\"{path}/train.parquet/partition_id={i}/part-0.parquet\"\n    part = pd.read_parquet(file_path)\n    part = part[['date_id','time_id','symbol_id','weight','responder_6']]\n    samples.append(part)\n    \n#sample_df = pd.concat(samples, ignore_index=True) # Concatenate all samples into one DataFrame if needed\n\n#sample_df.round(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:54:27.381095Z","iopub.execute_input":"2024-12-24T12:54:27.381576Z","iopub.status.idle":"2024-12-24T12:56:15.635352Z","shell.execute_reply.started":"2024-12-24T12:54:27.381549Z","shell.execute_reply":"2024-12-24T12:56:15.633697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = pd.concat(samples, ignore_index=True) # Concatenate all samples into one DataFrame if needed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:56:15.637281Z","iopub.execute_input":"2024-12-24T12:56:15.637886Z","iopub.status.idle":"2024-12-24T12:56:15.952884Z","shell.execute_reply.started":"2024-12-24T12:56:15.637797Z","shell.execute_reply":"2024-12-24T12:56:15.951925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:56:15.953898Z","iopub.execute_input":"2024-12-24T12:56:15.954156Z","iopub.status.idle":"2024-12-24T12:56:15.981234Z","shell.execute_reply.started":"2024-12-24T12:56:15.954134Z","shell.execute_reply":"2024-12-24T12:56:15.980296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_path1 = f\"{path}/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv\"\npart1 = pd.read_parquet(file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:56:15.983705Z","iopub.execute_input":"2024-12-24T12:56:15.983978Z","iopub.status.idle":"2024-12-24T12:56:19.572780Z","shell.execute_reply.started":"2024-12-24T12:56:15.983955Z","shell.execute_reply":"2024-12-24T12:56:19.571754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['feature_23'] = part1['feature_23']\nsample_df['feature_22'] = part1['feature_22']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:56:19.574083Z","iopub.execute_input":"2024-12-24T12:56:19.574386Z","iopub.status.idle":"2024-12-24T12:56:22.288135Z","shell.execute_reply.started":"2024-12-24T12:56:19.574363Z","shell.execute_reply":"2024-12-24T12:56:22.286999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T12:56:22.289247Z","iopub.execute_input":"2024-12-24T12:56:22.289634Z","iopub.status.idle":"2024-12-24T12:56:22.305828Z","shell.execute_reply.started":"2024-12-24T12:56:22.289567Z","shell.execute_reply":"2024-12-24T12:56:22.304729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 由上述可见：sample_df中symbol_id 的值为5, 20, 21, 22, 25, 26, 29, 36, 23, 27, 28, 35, 37, 4, 24, 31, 6, 18, 32 的行feature_22.feature_23的值均全为NAN（feature_23在12.21_2中体现）","metadata":{}},{"cell_type":"code","source":"#df_cleaned = sample_df.dropna(subset=['feature_23'])\ndf_cleaned = sample_df.dropna(subset=['feature_22'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:31.468002Z","iopub.execute_input":"2024-12-24T13:00:31.468375Z","iopub.status.idle":"2024-12-24T13:00:31.909362Z","shell.execute_reply.started":"2024-12-24T13:00:31.468348Z","shell.execute_reply":"2024-12-24T13:00:31.908212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:31.910860Z","iopub.execute_input":"2024-12-24T13:00:31.911230Z","iopub.status.idle":"2024-12-24T13:00:31.926730Z","shell.execute_reply.started":"2024-12-24T13:00:31.911194Z","shell.execute_reply":"2024-12-24T13:00:31.925346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:31.928764Z","iopub.execute_input":"2024-12-24T13:00:31.929269Z","iopub.status.idle":"2024-12-24T13:00:32.222048Z","shell.execute_reply.started":"2024-12-24T13:00:31.929214Z","shell.execute_reply":"2024-12-24T13:00:32.220751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:32.223587Z","iopub.execute_input":"2024-12-24T13:00:32.224002Z","iopub.status.idle":"2024-12-24T13:00:32.269568Z","shell.execute_reply.started":"2024-12-24T13:00:32.223963Z","shell.execute_reply":"2024-12-24T13:00:32.268314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned = sample_df.dropna(subset=['feature_23'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:51.224096Z","iopub.execute_input":"2024-12-24T13:00:51.224438Z","iopub.status.idle":"2024-12-24T13:00:51.706180Z","shell.execute_reply.started":"2024-12-24T13:00:51.224414Z","shell.execute_reply":"2024-12-24T13:00:51.705191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:51.707480Z","iopub.execute_input":"2024-12-24T13:00:51.707806Z","iopub.status.idle":"2024-12-24T13:00:51.990380Z","shell.execute_reply.started":"2024-12-24T13:00:51.707780Z","shell.execute_reply":"2024-12-24T13:00:51.989139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:00:51.992193Z","iopub.execute_input":"2024-12-24T13:00:51.992636Z","iopub.status.idle":"2024-12-24T13:00:52.036006Z","shell.execute_reply.started":"2024-12-24T13:00:51.992589Z","shell.execute_reply":"2024-12-24T13:00:52.034963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_cleaned[['weight','responder_6','feature_23','feature_22']].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:07:48.133400Z","iopub.execute_input":"2024-12-24T13:07:48.133840Z","iopub.status.idle":"2024-12-24T13:07:49.121117Z","shell.execute_reply.started":"2024-12-24T13:07:48.133807Z","shell.execute_reply":"2024-12-24T13:07:49.119668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# fearure_22的数据峰值在-1左右，responder_6的数据关于0几乎对称；feature_22和feature_23似乎有些线性相关","metadata":{}},{"cell_type":"code","source":"# 可视化\nsns.pairplot(df_cleaned[['feature_22', 'feature_23', 'responder_6']])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:07:52.388986Z","iopub.execute_input":"2024-12-24T13:07:52.389416Z","iopub.status.idle":"2024-12-24T13:09:23.834431Z","shell.execute_reply.started":"2024-12-24T13:07:52.389386Z","shell.execute_reply":"2024-12-24T13:09:23.832844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 可视化 feature_22 和 feature_23 与 responder_6 的关系\nsns.lmplot(x='feature_22', y='responder_6', data=df_cleaned, ci=None, scatter_kws={'alpha':0.5})\nplt.title('Relationship between feature_22 and responder_6')\nplt.show()\n\nsns.lmplot(x='feature_23', y='responder_6', data=df_cleaned, ci=None, scatter_kws={'alpha':0.5})\nplt.title('Relationship between feature_23 and responder_6')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:19:59.005973Z","iopub.execute_input":"2024-12-24T13:19:59.006453Z","iopub.status.idle":"2024-12-24T13:20:35.829015Z","shell.execute_reply.started":"2024-12-24T13:19:59.006407Z","shell.execute_reply":"2024-12-24T13:20:35.827569Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  回归系数：\n         const（常数项）：-0.0063，具有统计显著性（P值<0.05）。\n         feature_22：-0.0009，P值为0.125，不具有统计显著性。\n         feature_23：0.0003，P值为0.718，同样不具有统计显著性。","metadata":{}},{"cell_type":"code","source":"import statsmodels.api as sm\n\n# 准备自变量和因变量\nX = df_cleaned[['feature_22', 'feature_23']]\ny = df_cleaned['responder_6']\n\n# 添加常数项以拟合截距\nX = sm.add_constant(X)\n\n# 拟合多元线性回归模型\nmodel = sm.OLS(y, X).fit()\n\n# 打印模型摘要\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:21:31.948057Z","iopub.execute_input":"2024-12-24T13:21:31.948499Z","iopub.status.idle":"2024-12-24T13:21:35.329581Z","shell.execute_reply.started":"2024-12-24T13:21:31.948469Z","shell.execute_reply":"2024-12-24T13:21:35.326793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = df_cleaned['feature_22'] + df_cleaned['feature_23']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:23:25.464585Z","iopub.execute_input":"2024-12-24T13:23:25.465521Z","iopub.status.idle":"2024-12-24T13:23:25.486922Z","shell.execute_reply.started":"2024-12-24T13:23:25.465474Z","shell.execute_reply":"2024-12-24T13:23:25.485165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:23:28.090276Z","iopub.execute_input":"2024-12-24T13:23:28.090798Z","iopub.status.idle":"2024-12-24T13:23:28.598190Z","shell.execute_reply.started":"2024-12-24T13:23:28.090762Z","shell.execute_reply":"2024-12-24T13:23:28.596780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = df_cleaned['feature_22'] / df_cleaned['feature_23']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:26:10.373409Z","iopub.execute_input":"2024-12-24T13:26:10.373956Z","iopub.status.idle":"2024-12-24T13:26:10.391283Z","shell.execute_reply.started":"2024-12-24T13:26:10.373920Z","shell.execute_reply":"2024-12-24T13:26:10.389698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:26:16.641585Z","iopub.execute_input":"2024-12-24T13:26:16.642014Z","iopub.status.idle":"2024-12-24T13:26:17.202196Z","shell.execute_reply.started":"2024-12-24T13:26:16.641982Z","shell.execute_reply":"2024-12-24T13:26:17.200942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = df_cleaned['feature_22'] - 3*df_cleaned['feature_23']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:26:50.106769Z","iopub.execute_input":"2024-12-24T13:26:50.107185Z","iopub.status.idle":"2024-12-24T13:26:50.132295Z","shell.execute_reply.started":"2024-12-24T13:26:50.107158Z","shell.execute_reply":"2024-12-24T13:26:50.130822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:26:52.161415Z","iopub.execute_input":"2024-12-24T13:26:52.161855Z","iopub.status.idle":"2024-12-24T13:26:52.717599Z","shell.execute_reply.started":"2024-12-24T13:26:52.161823Z","shell.execute_reply":"2024-12-24T13:26:52.716368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = 3*df_cleaned['feature_22'] - df_cleaned['feature_23']\n# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:28:22.692681Z","iopub.execute_input":"2024-12-24T13:28:22.693180Z","iopub.status.idle":"2024-12-24T13:28:23.260957Z","shell.execute_reply.started":"2024-12-24T13:28:22.693140Z","shell.execute_reply":"2024-12-24T13:28:23.259651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 下面这个组合方法，相关性有所提高","metadata":{}},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = df_cleaned['feature_22']*df_cleaned['feature_22'] - df_cleaned['feature_23']\n# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:29:08.424024Z","iopub.execute_input":"2024-12-24T13:29:08.424522Z","iopub.status.idle":"2024-12-24T13:29:08.943123Z","shell.execute_reply.started":"2024-12-24T13:29:08.424482Z","shell.execute_reply":"2024-12-24T13:29:08.941710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建组合特征\ndf_cleaned['feature_combined'] = df_cleaned['feature_22']*df_cleaned['feature_22'] - df_cleaned['feature_23']\n# 探索性数据分析：计算新特征与responder_6之间的相关性\ncorrelation_matrix = df_cleaned[['feature_22', 'feature_23', 'feature_combined', 'responder_6']].corr()\ncorrelation_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:32:45.189568Z","iopub.execute_input":"2024-12-24T13:32:45.190078Z","iopub.status.idle":"2024-12-24T13:32:45.720540Z","shell.execute_reply.started":"2024-12-24T13:32:45.190040Z","shell.execute_reply":"2024-12-24T13:32:45.719095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 可视化：新特征与responder_6之间的关系\n# 使用pairplot可视化所有相关变量之间的关系（包括新特征）\nsns.pairplot(df_cleaned[['feature_combined', 'responder_6']])\nplt.suptitle('Pairplot of Features and Responder_6', y=1.02)  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-24T13:34:11.605066Z","iopub.execute_input":"2024-12-24T13:34:11.605516Z","iopub.status.idle":"2024-12-24T13:34:44.394416Z","shell.execute_reply.started":"2024-12-24T13:34:11.605474Z","shell.execute_reply":"2024-12-24T13:34:44.392576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}