{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":13849184,"sourceType":"datasetVersion","datasetId":8821586},{"sourceId":13909684,"sourceType":"datasetVersion","datasetId":8862728},{"sourceId":266277621,"sourceType":"kernelVersion"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n!pip install /kaggle/input/janestreet2025-code/janestreet-0.1-py3-none-any.whl --force-reinstall --no-deps","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-28T14:46:20.669187Z","iopub.execute_input":"2025-11-28T14:46:20.670843Z","iopub.status.idle":"2025-11-28T14:46:22.837209Z","shell.execute_reply.started":"2025-11-28T14:46:20.670784Z","shell.execute_reply":"2025-11-28T14:46:22.835497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:48:35.104484Z","iopub.execute_input":"2025-11-29T15:48:35.105186Z","iopub.status.idle":"2025-11-29T15:48:35.205044Z","shell.execute_reply.started":"2025-11-29T15:48:35.105155Z","shell.execute_reply":"2025-11-29T15:48:35.203980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\"\n\nfiles = [\n    f\"{base}/partition_id={i}/part-0.parquet\"\n    for i in range(8)  # partition 0–6\n]\n\ndf = pl.scan_parquet(base)\ndf = df.filter(\n            pl.col(\"partition_id\").is_in([1, 2, 3, 4, 5, 6, 7])\n        )\ndf = df.drop(\"partition_id\")\ndf = df.fill_null(0.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:23.766351Z","iopub.execute_input":"2025-11-29T15:57:23.766682Z","iopub.status.idle":"2025-11-29T15:57:23.773308Z","shell.execute_reply.started":"2025-11-29T15:57:23.766654Z","shell.execute_reply":"2025-11-29T15:57:23.772208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.limit(5).collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T13:55:14.777419Z","iopub.execute_input":"2025-11-29T13:55:14.781580Z","iopub.status.idle":"2025-11-29T13:55:59.428042Z","shell.execute_reply.started":"2025-11-29T13:55:14.781498Z","shell.execute_reply":"2025-11-29T13:55:59.426316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_counts = df.group_by(\"date_id\").agg(pl.col(\"time_id\").n_unique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T13:57:49.163933Z","iopub.execute_input":"2025-11-29T13:57:49.164350Z","iopub.status.idle":"2025-11-29T13:57:49.171211Z","shell.execute_reply.started":"2025-11-29T13:57:49.164320Z","shell.execute_reply":"2025-11-29T13:57:49.169719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"time_counts = time_counts.collect().sort('date_id')\ntime_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T13:58:59.381713Z","iopub.execute_input":"2025-11-29T13:58:59.383248Z","iopub.status.idle":"2025-11-29T13:59:00.018659Z","shell.execute_reply.started":"2025-11-29T13:58:59.383214Z","shell.execute_reply":"2025-11-29T13:59:00.017655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndates = time_counts[\"date_id\"].to_numpy()\ncounts = time_counts[\"time_id\"].to_numpy()\nplt.figure(figsize=(14, 7))\n\n# 绘制散点图/折线图\nplt.plot(dates, counts, marker='o', linestyle='-', markersize=4, color='b', label='Timesteps per Day')\n\n# 标记关键的稳定值 (968)\nplt.axhline(y=968, color='r', linestyle='--', label='Stabilization Target (968)') \n\n# 设置标题和标签\nplt.title('Time Steps Stability Over Trading Days (Date ID)', fontsize=16)\nplt.xlabel('Date ID', fontsize=14)\nplt.ylabel('Number of Time Steps (TimestepsPerDay)', fontsize=14)\n\n# 显示图例和网格\nplt.grid(True, linestyle='--', alpha=0.6)\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T14:01:30.440795Z","iopub.execute_input":"2025-11-29T14:01:30.441294Z","iopub.status.idle":"2025-11-29T14:01:31.011550Z","shell.execute_reply.started":"2025-11-29T14:01:30.441223Z","shell.execute_reply":"2025-11-29T14:01:31.010313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stable_dates = time_counts.filter(\n    pl.col(\"time_id\") == 968\n)\nstable_dates[0:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T14:06:35.389122Z","iopub.execute_input":"2025-11-29T14:06:35.389487Z","iopub.status.idle":"2025-11-29T14:06:35.400149Z","shell.execute_reply.started":"2025-11-29T14:06:35.389464Z","shell.execute_reply":"2025-11-29T14:06:35.398695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_after677 = df.filter(pl.col(\"date_id\") >= 677)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:29.711078Z","iopub.execute_input":"2025-11-29T15:57:29.711385Z","iopub.status.idle":"2025-11-29T15:57:29.716923Z","shell.execute_reply.started":"2025-11-29T15:57:29.711361Z","shell.execute_reply":"2025-11-29T15:57:29.715887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f61_corr_with_time = df_after677.select(\n    corr_with_date = pl.corr(\"feature_61\", \"date_id\")\n).collect()\nprint(f61_corr_with_time)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:37.549186Z","iopub.execute_input":"2025-11-29T15:57:37.550065Z","iopub.status.idle":"2025-11-29T15:57:38.079986Z","shell.execute_reply.started":"2025-11-29T15:57:37.550028Z","shell.execute_reply":"2025-11-29T15:57:38.078834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_features = [\n    \"feature_61\", \"feature_02\", \"feature_03\", \"feature_00\", \"feature_34\", \n    \"feature_35\", \"feature_32\", \"feature_27\", \"feature_28\", \"feature_20\", \n    \"feature_62\"\n]\n\navg_exprs = [\n    pl.col(col).mean().alias(f\"avg_{col}\")\n    for col in target_features\n]\n\ndf_daily_avg = df_after677.group_by(\"date_id\").agg(\n    avg_exprs\n)\ncorr_exprs = [\n    pl.corr(f\"avg_{col}\", \"date_id\").alias(col) # 使用原始 feature name 作为列名\n    for col in target_features\n]\ncorrelation_result = df_daily_avg.select(\n    corr_exprs\n).collect()\n\nfeature_names = correlation_result.columns\nresult_transposed = correlation_result.transpose(\n    column_names=[\"Correlation Value\"], \n    include_header=True, \n    header_name=\"Feature\"\n).with_columns(\n    pl.Series(feature_names).alias(\"Feature\")\n).select(\n    \"Feature\",\n    \"Correlation Value\"\n)\n\nresult_sorted = result_transposed.with_columns(\n    pl.col(\"Correlation Value\").abs().alias(\"Abs_Corr\")\n).sort(\"Abs_Corr\", descending=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:43.139218Z","iopub.execute_input":"2025-11-29T15:57:43.139503Z","iopub.status.idle":"2025-11-29T15:57:46.861837Z","shell.execute_reply.started":"2025-11-29T15:57:43.139481Z","shell.execute_reply":"2025-11-29T15:57:46.860779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(result_sorted.drop(\"Abs_Corr\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:46.863176Z","iopub.execute_input":"2025-11-29T15:57:46.863451Z","iopub.status.idle":"2025-11-29T15:57:46.869409Z","shell.execute_reply.started":"2025-11-29T15:57:46.863420Z","shell.execute_reply":"2025-11-29T15:57:46.868382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_cols = [f\"feature_{i:02d}\" for i in range(79)]\ncorr_date_df = df_after677.select(\n    [pl.corr(col, \"date_id\").alias(col) for col in feature_cols]\n).collect()\n\ncorr_time_df = df_after677.select(\n    [pl.corr(col, \"time_id\").alias(col) for col in feature_cols]\n).collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:57:50.976388Z","iopub.execute_input":"2025-11-29T15:57:50.976717Z","iopub.status.idle":"2025-11-29T15:59:46.746069Z","shell.execute_reply.started":"2025-11-29T15:57:50.976683Z","shell.execute_reply":"2025-11-29T15:59:46.744616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_data = {\n    \"Feature\": feature_cols,\n    \"Corr_with_Date\": corr_date_df.transpose().to_series().to_list(),\n    \"Corr_with_Time\": corr_time_df.transpose().to_series().to_list()\n}\n\nresult_df = pl.DataFrame(corr_data)\n\n# 5. 添加绝对值列以便排序（我们需要看相关性强弱，不论正负）\nresult_df = result_df.with_columns(\n    pl.col(\"Corr_with_Date\").abs().alias(\"Abs_Corr_Date\"),\n    pl.col(\"Corr_with_Time\").abs().alias(\"Abs_Corr_Time\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:59:46.748232Z","iopub.execute_input":"2025-11-29T15:59:46.748603Z","iopub.status.idle":"2025-11-29T15:59:46.758375Z","shell.execute_reply.started":"2025-11-29T15:59:46.748555Z","shell.execute_reply":"2025-11-29T15:59:46.757333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 榜单 A: 与日期(长期趋势)相关性最高的特征 ---\nprint(\"\\n🔥 Top 10 与 日期 (Date_ID) 相关性最高的特征：\")\ntop_date = result_df.sort(\"Abs_Corr_Date\", descending=True).head(10)\nprint(top_date.select([\"Feature\", \"Corr_with_Date\"]))\n\n# --- 榜单 B: 与日内时间(日内模式)相关性最高的特征 ---\nprint(\"\\n⏰ Top 10 与 日内时间 (Time_ID) 相关性最高的特征：\")\ntop_time = result_df.sort(\"Abs_Corr_Time\", descending=True).head(10)\nprint(top_time.select([\"Feature\", \"Corr_with_Time\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:59:46.759453Z","iopub.execute_input":"2025-11-29T15:59:46.759841Z","iopub.status.idle":"2025-11-29T15:59:46.793683Z","shell.execute_reply.started":"2025-11-29T15:59:46.759816Z","shell.execute_reply":"2025-11-29T15:59:46.792402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n\n# 1. 定义特征列表\nCOLS_FEATURES_CORR = [\n    'feature_06', 'feature_04', 'feature_07', 'feature_36',\n    'feature_60', 'feature_45', 'feature_56', 'feature_05',\n    'feature_51', 'feature_19', 'feature_66', 'feature_59',\n    'feature_54', 'feature_70', 'feature_71', 'feature_72',\n]\n\n# 2. 定义你想计算的目标列表\ntarget_cols = ['responder_6', 'responder_7', 'responder_8']\n\n# 3. 循环计算并打印结果\nfor target in target_cols:\n    print(f\"\\n=== 各特征与 {target} 的相关性 (Pearson) ===\")\n    \n    try:\n        # 计算当前 target 与列表特征的相关性\n        # 注意：如果 df_after677 已经是 DataFrame 而不是 LazyFrame，请去掉 .collect()\n        df_corr = df_after677.select(\n            [pl.corr(col, target).alias(col) for col in COLS_FEATURES_CORR]\n        ).collect()\n\n        # 提取结果并打印\n        results = df_corr.row(0, named=True)\n        for feat, corr_val in results.items():\n            print(f\"{feat:<15}: {corr_val:.6f}\")\n            \n    except Exception as e:\n        print(f\"计算 {target} 时出错: {e}\")\n        # 可能是因为数据中没有 responder_7 或 responder_8","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T16:11:12.751275Z","iopub.execute_input":"2025-11-29T16:11:12.751647Z","iopub.status.idle":"2025-11-29T16:11:23.156693Z","shell.execute_reply.started":"2025-11-29T16:11:12.751622Z","shell.execute_reply":"2025-11-29T16:11:23.155706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}