{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245336,"sourceType":"datasetVersion","datasetId":6336300}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:18.742526Z","iopub.execute_input":"2024-12-29T05:46:18.742920Z","iopub.status.idle":"2024-12-29T05:46:20.144444Z","shell.execute_reply.started":"2024-12-29T05:46:18.742890Z","shell.execute_reply":"2024-12-29T05:46:20.143348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:20.145727Z","iopub.execute_input":"2024-12-29T05:46:20.146143Z","iopub.status.idle":"2024-12-29T05:46:20.616249Z","shell.execute_reply.started":"2024-12-29T05:46:20.146114Z","shell.execute_reply":"2024-12-29T05:46:20.615138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 加载数据","metadata":{}},{"cell_type":"code","source":"train = pl.scan_parquet(\"/kaggle/input/20241219-data/training.parquet\").collect().to_pandas()\nvalid = pl.scan_parquet(\"/kaggle/input/20241219-data/validation.parquet\").collect().to_pandas()\ntrain.shape, valid.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:20.618370Z","iopub.execute_input":"2024-12-29T05:46:20.618865Z","iopub.status.idle":"2024-12-29T05:46:38.575959Z","shell.execute_reply.started":"2024-12-29T05:46:20.618832Z","shell.execute_reply":"2024-12-29T05:46:38.574345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:38.577872Z","iopub.execute_input":"2024-12-29T05:46:38.578294Z","iopub.status.idle":"2024-12-29T05:46:40.043654Z","shell.execute_reply.started":"2024-12-29T05:46:38.578253Z","shell.execute_reply":"2024-12-29T05:46:40.040282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['feature_13']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:40.046490Z","iopub.execute_input":"2024-12-29T05:46:40.047600Z","iopub.status.idle":"2024-12-29T05:46:40.069805Z","shell.execute_reply.started":"2024-12-29T05:46:40.047520Z","shell.execute_reply":"2024-12-29T05:46:40.066875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_1=train[train['feature_13']<20]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:40.072797Z","iopub.execute_input":"2024-12-29T05:46:40.073164Z","iopub.status.idle":"2024-12-29T05:46:42.155252Z","shell.execute_reply.started":"2024-12-29T05:46:40.073125Z","shell.execute_reply":"2024-12-29T05:46:42.154401Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 查看feature_13","metadata":{}},{"cell_type":"code","source":"min_value = train['feature_13'].min()\nmax_value = train['feature_13'].max()\nprint(f\"Minimum value: {min_value}\")\nprint(f\"Maximum value: {max_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:42.155985Z","iopub.execute_input":"2024-12-29T05:46:42.156265Z","iopub.status.idle":"2024-12-29T05:46:42.187701Z","shell.execute_reply.started":"2024-12-29T05:46:42.156241Z","shell.execute_reply":"2024-12-29T05:46:42.186537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nplt.hist(train_1['feature_13'], bins=30, color='skyblue', alpha=0.7)\nplt.title('Histogram of Feature 13', fontsize=14)\nplt.xlabel('Feature 13', fontsize=12)\nplt.ylabel('Frequency', fontsize=12)\nplt.grid(axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:42.190445Z","iopub.execute_input":"2024-12-29T05:46:42.190801Z","iopub.status.idle":"2024-12-29T05:46:42.650779Z","shell.execute_reply.started":"2024-12-29T05:46:42.190769Z","shell.execute_reply":"2024-12-29T05:46:42.649492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.boxplot(train_1['feature_13'], vert=False, patch_artist=True)\nplt.title('Boxplot of Feature 13', fontsize=14)\nplt.xlabel('Feature 13', fontsize=12)\nplt.grid(axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:42.652201Z","iopub.execute_input":"2024-12-29T05:46:42.652508Z","iopub.status.idle":"2024-12-29T05:46:43.800835Z","shell.execute_reply.started":"2024-12-29T05:46:42.652482Z","shell.execute_reply":"2024-12-29T05:46:43.799388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 对数据进行转换，使其分布更加均匀","metadata":{}},{"cell_type":"code","source":"# 对数转换\ntrain['feature_13_log'] = np.log(train['feature_13'] + 1)  # 加1避免对0取对数","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:43.802137Z","iopub.execute_input":"2024-12-29T05:46:43.802486Z","iopub.status.idle":"2024-12-29T05:46:43.863202Z","shell.execute_reply.started":"2024-12-29T05:46:43.802460Z","shell.execute_reply":"2024-12-29T05:46:43.862043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 可视化对数变换后的数据","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))  # 调整图表尺寸\nplt.hist(train['feature_13_log'], bins=30, color='#4F81BD', alpha=0.7)  # 调整颜色和透明度\nplt.title('Histogram of Feature 13 after Log Transformation', fontsize=16)  # 优化标题\nplt.xlabel('Feature 13 (Log Scale)', fontsize=14)  # 优化X轴标签\nplt.ylabel('Frequency', fontsize=14)  # 优化Y轴标签\nplt.grid(axis='y', linestyle='--', alpha=0.7)  # 调整网格线\nplt.xlim(-2, 5)  # 调整X轴范围，根据数据分布情况\nplt.ylim(0, 2500000)  # 调整Y轴范围，根据数据分布情况\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:43.864322Z","iopub.execute_input":"2024-12-29T05:46:43.864726Z","iopub.status.idle":"2024-12-29T05:46:44.288765Z","shell.execute_reply.started":"2024-12-29T05:46:43.864690Z","shell.execute_reply":"2024-12-29T05:46:44.287372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Mean of Feature 13 after Log Transformation: {train['feature_13_log'].mean()}\")\nprint(f\"Median of Feature 13 after Log Transformation: {train['feature_13_log'].median()}\")\nprint(f\"Variance of Feature 13 after Log Transformation: {train['feature_13_log'].var()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:44.289866Z","iopub.execute_input":"2024-12-29T05:46:44.290271Z","iopub.status.idle":"2024-12-29T05:46:44.518294Z","shell.execute_reply.started":"2024-12-29T05:46:44.290222Z","shell.execute_reply":"2024-12-29T05:46:44.515006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 画图查看未变化前与responder_6关系","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.scatter(train['feature_13'], train['responder_6_lag_1'], alpha=0.5)\nplt.title('Scatter Plot of Feature 13 vs responder_6_lag_1', fontsize=14)\nplt.xlabel('Feature 13', fontsize=12)\nplt.ylabel('responder_6_lag_1', fontsize=12)\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:46:44.520333Z","iopub.execute_input":"2024-12-29T05:46:44.520732Z","iopub.status.idle":"2024-12-29T05:47:02.872212Z","shell.execute_reply.started":"2024-12-29T05:46:44.520701Z","shell.execute_reply":"2024-12-29T05:47:02.870409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt  \nimport numpy as np  \n\nplt.figure(figsize=(20, 5))  \n\n# 如果数据量太大，可以选择每隔几个点绘制一次  \nstep = 10  \nplt.plot(train['feature_13'][::step], train['responder_6_lag_1'][::step],  \n         marker='o', linestyle='-', color='b', markersize=4, alpha=0.5)  \n\n# 添加均值线  \nmean_y = np.mean(train['responder_6_lag_1'])  \nplt.axhline(mean_y, color='r', linestyle='--', label='Mean Line')  \n\n# 设定标题和标签  \nplt.title('Feature 13 vs Responder 6 Lag 1', fontsize=14)  \nplt.xlabel('Feature 13', fontsize=12)  \nplt.ylabel('Responder 6 Lag 1', fontsize=12)  \n\n# 显示图例  \nplt.legend()  \n\n# 设定坐标轴范围（根据数据分布）  \nplt.xlim(0, 80)   \nplt.ylim(-1.5, 5) \n\n# 显示栅格  \nplt.grid(True)  \n\n# 显示图表  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:47:02.874149Z","iopub.execute_input":"2024-12-29T05:47:02.874689Z","iopub.status.idle":"2024-12-29T05:47:10.064221Z","shell.execute_reply.started":"2024-12-29T05:47:02.874614Z","shell.execute_reply":"2024-12-29T05:47:10.062936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 查看对数变化后与它的关系","metadata":{}},{"cell_type":"markdown","source":"### 散点图","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.scatter(train['feature_13_log'], train['responder_6_lag_1'], alpha=0.5)\nplt.title('Scatter Plot of Feature 13 (Log) vs responder_6_lag_1', fontsize=14)\nplt.xlabel('Feature 13 (Log Scale)', fontsize=12)\nplt.ylabel('responder_6_lag_1', fontsize=12)\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:47:16.996390Z","iopub.execute_input":"2024-12-29T05:47:16.996789Z","iopub.status.idle":"2024-12-29T05:47:34.527337Z","shell.execute_reply.started":"2024-12-29T05:47:16.996756Z","shell.execute_reply":"2024-12-29T05:47:34.526198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hexbin(train['feature_13_log'], train['responder_6_lag_1'], gridsize=50, cmap='Blues', alpha=0.7)\nplt.colorbar(label='log10(N) of points')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:47:34.528750Z","iopub.execute_input":"2024-12-29T05:47:34.529179Z","iopub.status.idle":"2024-12-29T05:47:35.690510Z","shell.execute_reply.started":"2024-12-29T05:47:34.529132Z","shell.execute_reply":"2024-12-29T05:47:35.689335Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 与其他feature做交互","metadata":{}},{"cell_type":"code","source":"train['13+5']=train['feature_13']+train['feature_05']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:58:06.766394Z","iopub.execute_input":"2024-12-29T05:58:06.766762Z","iopub.status.idle":"2024-12-29T05:58:06.782653Z","shell.execute_reply.started":"2024-12-29T05:58:06.766734Z","shell.execute_reply":"2024-12-29T05:58:06.781530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['13+5']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:58:08.547163Z","iopub.execute_input":"2024-12-29T05:58:08.547536Z","iopub.status.idle":"2024-12-29T05:58:08.556112Z","shell.execute_reply.started":"2024-12-29T05:58:08.547506Z","shell.execute_reply":"2024-12-29T05:58:08.555072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\ntrain['13+5'].hist(bins=50, color='blue', alpha=0.7)\nplt.title('Histogram of Feature 13+5')\nplt.xlabel('Value')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:58:12.062014Z","iopub.execute_input":"2024-12-29T05:58:12.062472Z","iopub.status.idle":"2024-12-29T05:58:12.516929Z","shell.execute_reply.started":"2024-12-29T05:58:12.062435Z","shell.execute_reply":"2024-12-29T05:58:12.515734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# 散点图\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='responder_6_lag_1', y='13+5', data=train)\nplt.title('Scatter plot of responder_6_lag_1 vs 13+5')\nplt.xlabel('responder_6_lag_1')\nplt.ylabel('13+5')\nplt.grid(True)\nplt.show()\n\n# 计算相关系数\ncorrelation = train['responder_6_lag_1'].corr(train['13+5'])\nprint(f\"The correlation coefficient between 'responder_6_lag_1' and '13+5' is: {correlation}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T07:14:00.553571Z","iopub.execute_input":"2024-12-29T07:14:00.553990Z","iopub.status.idle":"2024-12-29T07:14:13.989620Z","shell.execute_reply.started":"2024-12-29T07:14:00.553958Z","shell.execute_reply":"2024-12-29T07:14:13.988709Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"没有线性关系","metadata":{}},{"cell_type":"code","source":"train['13*6']=train['feature_13']*train['feature_06']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:07:00.345299Z","iopub.execute_input":"2024-12-29T06:07:00.345729Z","iopub.status.idle":"2024-12-29T06:07:00.366065Z","shell.execute_reply.started":"2024-12-29T06:07:00.345693Z","shell.execute_reply":"2024-12-29T06:07:00.364920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['13*6']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:07:02.099771Z","iopub.execute_input":"2024-12-29T06:07:02.100140Z","iopub.status.idle":"2024-12-29T06:07:02.108706Z","shell.execute_reply.started":"2024-12-29T06:07:02.100106Z","shell.execute_reply":"2024-12-29T06:07:02.107529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\ntrain['13*6'].hist(bins=60, color='blue', alpha=0.7)\nplt.title('Histogram of Feature 13*6')\nplt.xlabel('Value')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:08:05.271343Z","iopub.execute_input":"2024-12-29T06:08:05.271812Z","iopub.status.idle":"2024-12-29T06:08:05.687903Z","shell.execute_reply.started":"2024-12-29T06:08:05.271774Z","shell.execute_reply":"2024-12-29T06:08:05.686488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 对 'feature_13' 和 'feature_06' 进行对数变换\ntrain['log_13'] = np.log1p(train['feature_13'])\ntrain['log_06'] = np.log1p(train['feature_06'])\n\n# 将对数变换后的特征相乘\ntrain['13_log+6_log'] = train['log_13'] + train['log_06']\n\n# 查看变换后的数据\nprint(train[['feature_13', 'feature_06', 'log_13', 'log_06', '13_log+6_log']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:14:46.285487Z","iopub.execute_input":"2024-12-29T06:14:46.285924Z","iopub.status.idle":"2024-12-29T06:14:46.606820Z","shell.execute_reply.started":"2024-12-29T06:14:46.285889Z","shell.execute_reply":"2024-12-29T06:14:46.605522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\ntrain['13_log+6_log'].hist(bins=50, color='orange', alpha=0.8)\nplt.title('Histogram of Feature 13_log+6_log')\nplt.xlabel('Value')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T06:16:10.955952Z","iopub.execute_input":"2024-12-29T06:16:10.956316Z","iopub.status.idle":"2024-12-29T06:16:11.390306Z","shell.execute_reply.started":"2024-12-29T06:16:10.956289Z","shell.execute_reply":"2024-12-29T06:16:11.388677Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"分布形状：直方图显示 13_log+6_log 特征的分布大致呈正态分布，中心集中在0附近，这是对数变换后常见的分布形态。\n\n集中趋势：大多数数据点集中在-2.5到2.5之间，这表明这个特征的值大多数落在这个范围内。\n\n偏态：分布看起来相对对称，没有明显的左偏或右偏，这可能意味着对数变换有效地减少了原始数据的偏态。\n\n频率：y轴表示频率，即每个区间内数据点的数量。从图中可以看出，频率最高的区间是0附近，随着值远离0，频率逐渐降低。\n\n异常值：在-12.5和5.0附近，频率非常低，这可能表明这些区域的数据点是异常值或极端值。\n\n数据范围：x轴的范围从-12.5到5.0，这显示了 13_log+6_log 特征的值域。\n\n对数变换的效果：对数变换通常用于处理具有长尾分布或偏态分布的数据。从直方图来看，对数变换可能已经有效地稳定了数据的方差，并使其更接近正态分布。\n\n模型输入：如果这个特征将被用于机器学习模型，其分布形态是一个重要的考虑因素。正态分布的特征通常更容易被大多数模型处理。","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# 散点图\nplt.figure(figsize=(10, 6))\nsns.scatterplot(x='responder_6_lag_1', y='13_log+6_log', data=train)\nplt.title('Scatter plot of responder_6_lag_1 vs 13_log+6_log')\nplt.xlabel('responder_6_lag_1')\nplt.ylabel('13_log+6_log')\nplt.grid(True)\nplt.show()\n\n# 计算相关系数\ncorrelation = train['responder_6_lag_1'].corr(train['13_log+6_log'])\nprint(f\"The correlation coefficient between 'responder_6_lag_1' and '13_log+6_log' is: {correlation}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T07:16:46.112128Z","iopub.execute_input":"2024-12-29T07:16:46.112528Z","iopub.status.idle":"2024-12-29T07:16:58.469158Z","shell.execute_reply.started":"2024-12-29T07:16:46.112495Z","shell.execute_reply":"2024-12-29T07:16:58.468016Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"分布形状：数据点的分布呈现出一个大致的椭圆形，这可能表明两个特征之间存在某种程度的相关性。\n\n集中趋势：大多数数据点集中在图的中心区域，这表明这两个特征的大多数值都接近它们的平均值或中位数。\n\n离群点：在x轴的两端，尤其是正值区域，有一些数据点远离主要的集中区域，这些可能是离群点。\n\n对称性：数据点的分布似乎在y轴上是对称的，这可能意味着 13_log+6_log 特征在 responder_6_lag_1 特征的正负值上具有相似的分布。\n\n线性关系：如果数据点沿着一条直线分布，那么两个特征之间可能存在线性关系。然而，从这个图中，数据点的分布并不完全沿着一条直线，这表明线性关系可能不是两个特征之间唯一的关系。\n\n对数变换的影响：由于两个特征都进行了对数变换，这可能意味着原始数据的分布是偏态的，对数变换有助于稳定方差并使数据更接近正态分布。\n\n数据范围：x轴和y轴的范围显示了两个特征的值域，responder_6_lag_1 的值域大约在 -5 到 5 之间，而 13_log+6_log 的值域大约在 -12.5 到 5 之间。","metadata":{}},{"cell_type":"code","source":"lags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 1 minute of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n\n    # Replace this section with your own predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0).alias('responder_6'),\n    )\n\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:47:35.710731Z","iopub.execute_input":"2024-12-29T05:47:35.711149Z","iopub.status.idle":"2024-12-29T05:47:35.722231Z","shell.execute_reply.started":"2024-12-29T05:47:35.711104Z","shell.execute_reply":"2024-12-29T05:47:35.721128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T05:47:35.723372Z","iopub.execute_input":"2024-12-29T05:47:35.723800Z","iopub.status.idle":"2024-12-29T05:47:35.971864Z","shell.execute_reply.started":"2024-12-29T05:47:35.723753Z","shell.execute_reply":"2024-12-29T05:47:35.970785Z"}},"outputs":[],"execution_count":null}]}