{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245336,"sourceType":"datasetVersion","datasetId":6336300}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:23.531031Z","iopub.execute_input":"2024-12-28T14:23:23.531493Z","iopub.status.idle":"2024-12-28T14:23:25.198685Z","shell.execute_reply.started":"2024-12-28T14:23:23.531458Z","shell.execute_reply":"2024-12-28T14:23:25.197809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:25.199957Z","iopub.execute_input":"2024-12-28T14:23:25.200262Z","iopub.status.idle":"2024-12-28T14:23:25.204346Z","shell.execute_reply.started":"2024-12-28T14:23:25.200237Z","shell.execute_reply":"2024-12-28T14:23:25.203351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:25.206253Z","iopub.execute_input":"2024-12-28T14:23:25.206547Z","iopub.status.idle":"2024-12-28T14:23:25.224860Z","shell.execute_reply.started":"2024-12-28T14:23:25.206523Z","shell.execute_reply":"2024-12-28T14:23:25.223726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:25.226382Z","iopub.execute_input":"2024-12-28T14:23:25.226780Z","iopub.status.idle":"2024-12-28T14:23:25.249037Z","shell.execute_reply.started":"2024-12-28T14:23:25.226702Z","shell.execute_reply":"2024-12-28T14:23:25.248172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.dataset_download(\"bruceqdu/20241219-data\")\n\nprint(\"Path to dataset files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:25.249924Z","iopub.execute_input":"2024-12-28T14:23:25.250271Z","iopub.status.idle":"2024-12-28T14:23:26.616485Z","shell.execute_reply.started":"2024-12-28T14:23:25.250244Z","shell.execute_reply":"2024-12-28T14:23:26.615509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.scan_parquet(\n    f\"/kaggle/input/20241219-data/training.parquet\"\n).collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:26.617503Z","iopub.execute_input":"2024-12-28T14:23:26.617877Z","iopub.status.idle":"2024-12-28T14:23:36.466281Z","shell.execute_reply.started":"2024-12-28T14:23:26.617841Z","shell.execute_reply":"2024-12-28T14:23:36.465177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:23:36.467261Z","iopub.execute_input":"2024-12-28T14:23:36.467607Z","iopub.status.idle":"2024-12-28T14:23:36.492930Z","shell.execute_reply.started":"2024-12-28T14:23:36.467571Z","shell.execute_reply":"2024-12-28T14:23:36.491845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:18.678618Z","iopub.execute_input":"2024-12-28T14:32:18.678972Z","iopub.status.idle":"2024-12-28T14:32:21.479297Z","shell.execute_reply.started":"2024-12-28T14:32:18.678935Z","shell.execute_reply":"2024-12-28T14:32:21.478182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:25.415164Z","iopub.execute_input":"2024-12-28T14:32:25.415493Z","iopub.status.idle":"2024-12-28T14:32:26.453151Z","shell.execute_reply.started":"2024-12-28T14:32:25.415468Z","shell.execute_reply":"2024-12-28T14:32:26.452143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val = pl.scan_parquet(\n    f\"/kaggle/input/20241219-data/validation.parquet\"\n).collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:42.445552Z","iopub.execute_input":"2024-12-28T14:32:42.445876Z","iopub.status.idle":"2024-12-28T14:32:42.842434Z","shell.execute_reply.started":"2024-12-28T14:32:42.445849Z","shell.execute_reply":"2024-12-28T14:32:42.841367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:45.164741Z","iopub.execute_input":"2024-12-28T14:32:45.165153Z","iopub.status.idle":"2024-12-28T14:32:45.182170Z","shell.execute_reply.started":"2024-12-28T14:32:45.165118Z","shell.execute_reply":"2024-12-28T14:32:45.181168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val = val.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:47.730504Z","iopub.execute_input":"2024-12-28T14:32:47.730834Z","iopub.status.idle":"2024-12-28T14:32:48.203544Z","shell.execute_reply.started":"2024-12-28T14:32:47.730807Z","shell.execute_reply":"2024-12-28T14:32:48.202576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:32:50.913080Z","iopub.execute_input":"2024-12-28T14:32:50.913405Z","iopub.status.idle":"2024-12-28T14:32:51.063223Z","shell.execute_reply.started":"2024-12-28T14:32:50.913380Z","shell.execute_reply":"2024-12-28T14:32:51.062300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['symbol_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:33:34.401995Z","iopub.execute_input":"2024-12-28T14:33:34.402395Z","iopub.status.idle":"2024-12-28T14:33:34.451888Z","shell.execute_reply.started":"2024-12-28T14:33:34.402363Z","shell.execute_reply":"2024-12-28T14:33:34.450873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 特征变换","metadata":{}},{"cell_type":"markdown","source":"# 绘制每个 symbol_id 对应的 feature_19 平均值随 date_id 变化的折线图","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\nimport pandas as pd\ngrouped = train.groupby(['symbol_id', 'date_id'])['feature_19'].mean().reset_index()\nfig, axes = plt.subplots(nrows=39, ncols=1, figsize=(15, 100))\nfor i, symbol in enumerate(grouped['symbol_id'].unique()):\n    symbol_data = grouped[grouped['symbol_id'] == symbol]\n    if i >= len(axes.flat):\n        break  \n    ax = axes.flat[i] \n    ax.plot(symbol_data['date_id'], symbol_data['feature_19'], marker='o', linestyle='-')\n    ax.set_xlabel('Date')\n    ax.set_ylabel('Feature_19 Average')\n    ax.yaxis.set_major_formatter(ticker.FormatStrFormatter('%.2f'))\nplt.tight_layout()\nplt.savefig('/kaggle/working/feature_19.png', dpi=600)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T14:51:49.416716Z","iopub.execute_input":"2024-12-28T14:51:49.417187Z","iopub.status.idle":"2024-12-28T14:52:36.091134Z","shell.execute_reply.started":"2024-12-28T14:51:49.417150Z","shell.execute_reply":"2024-12-28T14:52:36.089925Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 比较不同 symbol_id 下 feature_19 的分布：","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 10))\nsns.boxplot(x='symbol_id', y='feature_19', data=train)\nplt.xlabel('Symbol ID')\nplt.ylabel('Feature 19')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:11:17.444858Z","iopub.execute_input":"2024-12-28T15:11:17.445267Z","iopub.status.idle":"2024-12-28T15:11:19.209678Z","shell.execute_reply.started":"2024-12-28T15:11:17.445237Z","shell.execute_reply":"2024-12-28T15:11:19.208538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:14:06.957286Z","iopub.execute_input":"2024-12-28T15:14:06.957665Z","iopub.status.idle":"2024-12-28T15:14:06.963451Z","shell.execute_reply.started":"2024-12-28T15:14:06.957635Z","shell.execute_reply":"2024-12-28T15:14:06.962454Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 特征19与特征20的交互（两者平均值）","metadata":{}},{"cell_type":"code","source":"train['feature19,20_average'] = (train['feature_19'] + train['feature_20']) / 2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:17:59.396015Z","iopub.execute_input":"2024-12-28T15:17:59.396429Z","iopub.status.idle":"2024-12-28T15:17:59.429812Z","shell.execute_reply.started":"2024-12-28T15:17:59.396397Z","shell.execute_reply":"2024-12-28T15:17:59.428852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:18:37.309937Z","iopub.execute_input":"2024-12-28T15:18:37.310345Z","iopub.status.idle":"2024-12-28T15:18:38.879645Z","shell.execute_reply.started":"2024-12-28T15:18:37.310314Z","shell.execute_reply":"2024-12-28T15:18:38.878640Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 每个 symbol_id 对应的 feature_19 平均值随 date_id 变化的折线图","metadata":{}},{"cell_type":"code","source":"grouped = train.groupby(['symbol_id', 'date_id'])['feature19,20_average'].mean().reset_index()\nfig, axes = plt.subplots(nrows=39, ncols=1, figsize=(15, 100))\nfor i, symbol in enumerate(grouped['symbol_id'].unique()):\n    symbol_data = grouped[grouped['symbol_id'] == symbol]\n    if i >= len(axes.flat):\n        break  \n    ax = axes.flat[i] \n    ax.plot(symbol_data['date_id'], symbol_data['feature19,20_average'], marker='o', linestyle='-')\n    ax.set_xlabel('Date')\n    ax.set_ylabel('feature19,20_average')\n    ax.yaxis.set_major_formatter(ticker.FormatStrFormatter('%.2f'))\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-28T15:27:45.738905Z","iopub.execute_input":"2024-12-28T15:27:45.739298Z","iopub.status.idle":"2024-12-28T15:27:52.981270Z","shell.execute_reply.started":"2024-12-28T15:27:45.739266Z","shell.execute_reply":"2024-12-28T15:27:52.980204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}