{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245345,"sourceType":"datasetVersion","datasetId":6336305}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:45:58.247790Z","iopub.execute_input":"2024-12-29T13:45:58.248066Z","iopub.status.idle":"2024-12-29T13:46:00.440718Z","shell.execute_reply.started":"2024-12-29T13:45:58.248033Z","shell.execute_reply":"2024-12-29T13:46:00.439411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport numpy as np\nimport os, gc\nfrom tqdm.auto import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:00.442126Z","iopub.execute_input":"2024-12-29T13:46:00.442575Z","iopub.status.idle":"2024-12-29T13:46:00.923289Z","shell.execute_reply.started":"2024-12-29T13:46:00.442544Z","shell.execute_reply":"2024-12-29T13:46:00.922153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:00.924328Z","iopub.execute_input":"2024-12-29T13:46:00.924731Z","iopub.status.idle":"2024-12-29T13:46:00.929499Z","shell.execute_reply.started":"2024-12-29T13:46:00.924689Z","shell.execute_reply":"2024-12-29T13:46:00.928557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_paths = [f\"{ROOT_DIR}/train.parquet/partition_id={i}/part-0.parquet\" for i in range(0,10)]\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:00.931967Z","iopub.execute_input":"2024-12-29T13:46:00.932262Z","iopub.status.idle":"2024-12-29T13:46:00.953195Z","shell.execute_reply.started":"2024-12-29T13:46:00.932237Z","shell.execute_reply":"2024-12-29T13:46:00.951821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 统计feature_00的缺失值","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:01:46.782314Z","iopub.execute_input":"2024-12-28T14:01:46.782654Z","iopub.status.idle":"2024-12-28T14:01:46.862988Z","shell.execute_reply.started":"2024-12-28T14:01:46.782631Z","shell.execute_reply":"2024-12-28T14:01:46.862219Z"}}},{"cell_type":"code","source":"missing_counts = {}\n\nfor file_path in file_paths:\n    # 只读取文件的架构（不加载数据）\n    df = pl.read_parquet(file_path, columns=[\"feature_00\"]).select(pl.col(\"feature_00\").is_null().sum())\n    \n    # 获取缺失值数量\n    missing_count = df.row(0)[0]  # 获取 'feature_00' 列的缺失值数量\n    missing_counts[file_path] = missing_count\n    \n# 打印每个文件中 'feature_00' 的缺失值数量\nfor file, count in missing_counts.items():\n    print(f\"{file}: {count} missing values in 'feature_00'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:00.955440Z","iopub.execute_input":"2024-12-29T13:46:00.955918Z","iopub.status.idle":"2024-12-29T13:46:02.367862Z","shell.execute_reply.started":"2024-12-29T13:46:00.955882Z","shell.execute_reply":"2024-12-29T13:46:02.366779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_01 = pd.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id=0/part-0.parquet\")\n\ntrain_02=pd.read_parquet(f\"{ROOT_DIR}/train.parquet/partition_id=1/part-0.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:02.368937Z","iopub.execute_input":"2024-12-29T13:46:02.369253Z","iopub.status.idle":"2024-12-29T13:46:10.852550Z","shell.execute_reply.started":"2024-12-29T13:46:02.369225Z","shell.execute_reply":"2024-12-29T13:46:10.851471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_01.shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:10.853496Z","iopub.execute_input":"2024-12-29T13:46:10.853765Z","iopub.status.idle":"2024-12-29T13:46:10.860821Z","shell.execute_reply.started":"2024-12-29T13:46:10.853740Z","shell.execute_reply":"2024-12-29T13:46:10.859832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_02.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:10.861778Z","iopub.execute_input":"2024-12-29T13:46:10.862136Z","iopub.status.idle":"2024-12-29T13:46:10.884173Z","shell.execute_reply.started":"2024-12-29T13:46:10.862102Z","shell.execute_reply":"2024-12-29T13:46:10.883158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_01","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:10.885224Z","iopub.execute_input":"2024-12-29T13:46:10.885604Z","iopub.status.idle":"2024-12-29T13:46:10.946969Z","shell.execute_reply.started":"2024-12-29T13:46:10.885566Z","shell.execute_reply":"2024-12-29T13:46:10.945950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_02","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:10.947937Z","iopub.execute_input":"2024-12-29T13:46:10.948331Z","iopub.status.idle":"2024-12-29T13:46:11.029066Z","shell.execute_reply.started":"2024-12-29T13:46:10.948293Z","shell.execute_reply":"2024-12-29T13:46:11.027963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.scan_parquet(#读取文件的架构并没有读取文件内容\n    f\"/kaggle/input/dateset1/training.parquet\"\n).collect()","metadata":{"execution":{"iopub.status.busy":"2024-12-29T13:46:11.030031Z","iopub.execute_input":"2024-12-29T13:46:11.030288Z","iopub.status.idle":"2024-12-29T13:46:22.679066Z","shell.execute_reply.started":"2024-12-29T13:46:11.030267Z","shell.execute_reply":"2024-12-29T13:46:22.677040Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:22.680395Z","iopub.execute_input":"2024-12-29T13:46:22.680927Z","iopub.status.idle":"2024-12-29T13:46:28.044100Z","shell.execute_reply.started":"2024-12-29T13:46:22.680887Z","shell.execute_reply":"2024-12-29T13:46:28.042998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df=train[train[\"date_id\"]>1600]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:46:28.046713Z","iopub.execute_input":"2024-12-29T13:46:28.047046Z","iopub.status.idle":"2024-12-29T13:46:28.815289Z","shell.execute_reply.started":"2024-12-29T13:46:28.047015Z","shell.execute_reply":"2024-12-29T13:46:28.814380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 时序图","metadata":{}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nxx= sample_df[(sample_df.symbol_id==1)] ['id']\nyy=sample_df[ (sample_df.symbol_id==1)]['responder_6']\n\nplt.figure(figsize=(16, 5))\nplt.plot(xx,yy, color = 'black', linewidth =0.05)\nplt.suptitle('Returns, responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Returns\", fontsize=12)\n\nplt.axhline(0, color='red', linestyle='-', linewidth=1.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:52:01.856437Z","iopub.execute_input":"2024-12-29T13:52:01.857378Z","iopub.status.idle":"2024-12-29T13:52:02.487423Z","shell.execute_reply.started":"2024-12-29T13:52:01.857325Z","shell.execute_reply":"2024-12-29T13:52:02.486144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xx= sample_df[(sample_df.symbol_id==2)] ['id']\nyy=sample_df[ (sample_df.symbol_id==2)]['responder_6']\n\nplt.figure(figsize=(16, 5))\nplt.plot(xx,yy, color = 'black', linewidth =0.05)\nplt.suptitle('Returns, responder_6', weight='bold', fontsize=16)\nplt.xlabel(\"Time\", fontsize=12)\nplt.ylabel(\"Returns\", fontsize=12)\n\nplt.axhline(0, color='red', linestyle='-', linewidth=1.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T13:55:18.221629Z","iopub.execute_input":"2024-12-29T13:55:18.222049Z","iopub.status.idle":"2024-12-29T13:55:18.744344Z","shell.execute_reply.started":"2024-12-29T13:55:18.222018Z","shell.execute_reply":"2024-12-29T13:55:18.743039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"feature_00具有一定的周期性","metadata":{}},{"cell_type":"code","source":"features = pd.read_csv(f\"{ROOT_DIR}/features.csv\")\nfeatures","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T14:01:55.989348Z","iopub.execute_input":"2024-12-29T14:01:55.989714Z","iopub.status.idle":"2024-12-29T14:01:56.034637Z","shell.execute_reply.started":"2024-12-29T14:01:55.989681Z","shell.execute_reply":"2024-12-29T14:01:56.033621Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 数据矩阵","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(18, 6))\nplt.imshow(features.iloc[:, 1:].T.values, cmap=\"gray_r\")\nplt.xlabel(\"feature_00 - feature_78\")\nplt.ylabel(\"tag_0 - tag_16\")\nplt.yticks(np.arange(17))\nplt.xticks(np.arange(79))\nplt.grid(color = 'lightgrey')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T14:02:10.584265Z","iopub.execute_input":"2024-12-29T14:02:10.584602Z","iopub.status.idle":"2024-12-29T14:02:11.231000Z","shell.execute_reply.started":"2024-12-29T14:02:10.584574Z","shell.execute_reply":"2024-12-29T14:02:11.229891Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"feature_00具有的tag与其他一些特征有重合，可能特征之间具有相关性","metadata":{}},{"cell_type":"markdown","source":"### 热力图","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize=(11, 11))\nmatrix = features[[ f\"tag_{no}\" for no in range(0,17,1) ] ].T.corr()\nsns.heatmap(matrix, square=True, cmap=\"coolwarm\", alpha =0.9, vmin=-1, vmax=1, center= 0, linewidths=0.5, linecolor='white')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T14:07:44.684563Z","iopub.execute_input":"2024-12-29T14:07:44.684981Z","iopub.status.idle":"2024-12-29T14:07:46.633921Z","shell.execute_reply.started":"2024-12-29T14:07:44.684946Z","shell.execute_reply":"2024-12-29T14:07:46.632747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"通过热力图可以看出feature_00与feature_01、feature_02等特征的相关性较高，且图中呈现出了强相关区域。","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}