{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245336,"sourceType":"datasetVersion","datasetId":6336300}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom tqdm.auto import tqdm\nfrom matplotlib import pyplot as plt\nimport pickle\nfrom sklearn.metrics import r2_score\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-28T08:30:48.639068Z","iopub.execute_input":"2024-12-28T08:30:48.639580Z","iopub.status.idle":"2024-12-28T08:30:48.646697Z","shell.execute_reply.started":"2024-12-28T08:30:48.639546Z","shell.execute_reply":"2024-12-28T08:30:48.645431Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)] \\\n        + [f\"responder_{idx}_lag_1\" for idx in range(9)]\n    categorical_cols = []","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:30:51.208195Z","iopub.execute_input":"2024-12-28T08:30:51.208582Z","iopub.status.idle":"2024-12-28T08:30:51.214373Z","shell.execute_reply.started":"2024-12-28T08:30:51.208550Z","shell.execute_reply":"2024-12-28T08:30:51.213010Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pl.scan_parquet(\"/kaggle/input/20241219-data/training.parquet\").collect().to_pandas()\nvalid = pl.scan_parquet(\"/kaggle/input/20241219-data/validation.parquet\").collect().to_pandas()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:30:53.571261Z","iopub.execute_input":"2024-12-28T08:30:53.571654Z","iopub.status.idle":"2024-12-28T08:31:03.900393Z","shell.execute_reply.started":"2024-12-28T08:30:53.571619Z","shell.execute_reply":"2024-12-28T08:31:03.899116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:11:13.412916Z","iopub.execute_input":"2024-12-28T08:11:13.414042Z","iopub.status.idle":"2024-12-28T08:11:13.525526Z","shell.execute_reply.started":"2024-12-28T08:11:13.413961Z","shell.execute_reply":"2024-12-28T08:11:13.524219Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 计算皮尔逊相关系数\ncorrelation = train[['feature_05', 'responder_6']].corr()\nprint(correlation)\n# 绘制热力图\nplt.figure(figsize=(8, 6))\nsns.heatmap(correlation, annot=True, cmap='coolwarm', center=0, fmt='.2f', linewidths=2)\nplt.title('Correlation Heatmap of feature_05 and responder_6')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:32:36.632453Z","iopub.execute_input":"2024-12-28T08:32:36.632746Z","iopub.status.idle":"2024-12-28T08:32:37.050783Z","shell.execute_reply.started":"2024-12-28T08:32:36.632720Z","shell.execute_reply":"2024-12-28T08:32:37.049655Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**没线性关系**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.linear_model import LinearRegression\n# X 是 feature_05, y 是 responder_6\nX = train[['feature_05']].values\ny = train['responder_6'].values\npoly = PolynomialFeatures(degree=2)\nX_poly = poly.fit_transform(X)\nmodel = LinearRegression()\nmodel.fit(X_poly, y)\ny_poly_pred = model.predict(X_poly)\nplt.figure(figsize=(10, 6))\nplt.plot(X, y_poly_pred, color='red', label=\"Polynomial regression (degree=2)\", linewidth=2)\nplt.title('Polynomial Regression Plot (Degree=2)')\nplt.xlabel('Feature 05')\nplt.ylabel('Responder 6')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:31:50.356414Z","iopub.execute_input":"2024-12-28T08:31:50.356836Z","iopub.status.idle":"2024-12-28T08:32:04.350310Z","shell.execute_reply.started":"2024-12-28T08:31:50.356799Z","shell.execute_reply":"2024-12-28T08:32:04.348940Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"******非线性的二次关系**","metadata":{}},{"cell_type":"code","source":"#total_rows = len(train)\n#train = train.iloc[int(total_rows * (9 / 10)):]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['feature_05_binned'] = pd.cut(train['feature_05'], bins=10)\n\n# 绘制箱线图：x 为分箱后的 feature_05，y 为 responder_6\nplt.figure(figsize=(12, 8))\nsns.boxplot(x='feature_05_binned', y='responder_6', data=train)\nplt.title('Box Plot of responder_6 vs Binned feature_05')\nplt.xlabel('Binned feature_05')\nplt.ylabel('responder_6')\nplt.xticks(rotation=45) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:32:04.351841Z","iopub.execute_input":"2024-12-28T08:32:04.352231Z","iopub.status.idle":"2024-12-28T08:32:06.360504Z","shell.execute_reply.started":"2024-12-28T08:32:04.352190Z","shell.execute_reply":"2024-12-28T08:32:06.359327Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n**feature_05对responder_6的影响在不同区间差异较大。在低值区间，feature_05 越低，responder_6 的响应越高；在中值区间，feature_05 的增大导致 responder_6 的响应逐渐下降，且数据分布变得更为分散；在高值区间，feature_05 增大时，responder_6 的响应有所增加，但响应波动较大**\n","metadata":{}},{"cell_type":"code","source":"train['feature_05_binned'] = pd.cut(train['feature_05'], bins=10)\n\n# 绘制小提琴图：x 为分箱后的 feature_05，y 为 responder_6\nplt.figure(figsize=(12, 8))\nsns.violinplot(x='feature_05_binned', y='responder_6', data=train, inner=\"quart\", palette=\"muted\")\nplt.title('Violin Plot of responder_6 vs Binned feature_05')\nplt.xlabel('Binned feature_05')\nplt.ylabel('responder_6')\nplt.xticks(rotation=45)  # 旋转 x 轴标签，便于查看\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T08:32:21.851122Z","iopub.execute_input":"2024-12-28T08:32:21.851525Z","iopub.status.idle":"2024-12-28T08:32:36.631222Z","shell.execute_reply.started":"2024-12-28T08:32:21.851492Z","shell.execute_reply":"2024-12-28T08:32:36.630157Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}