{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":11686826,"sourceType":"datasetVersion","datasetId":7335172}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 项目：DRW Crypto Market Prediction\n# 编写者：周赋瑶\n\n# 1. 导入核心工具包\nimport pandas as pd  # 数据处理主力库\nimport numpy as np   # 数值计算支持\nimport matplotlib.pyplot as plt  # 可视化\nimport seaborn as sns  # 高级可视化辅助\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nfrom sklearn.feature_selection import SelectKBest, f_regression\nimport optuna  # 超参数优化库\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:49:49.153315Z","iopub.execute_input":"2025-06-21T13:49:49.153699Z","iopub.status.idle":"2025-06-21T13:49:52.866217Z","shell.execute_reply.started":"2025-06-21T13:49:49.153672Z","shell.execute_reply":"2025-06-21T13:49:52.865069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 2. 读取训练集\ndata_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\ndf = pd.read_parquet(data_path)  # 读取 parquet 文件格式\nprint(\"数据维度:\", df.shape)\nprint(\"前几列名称:\", df.columns[:10].tolist())\nprint(df.dtypes.head())  # 显示前几列数据类型","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:49:52.867774Z","iopub.execute_input":"2025-06-21T13:49:52.868334Z","iopub.status.idle":"2025-06-21T13:50:15.322470Z","shell.execute_reply.started":"2025-06-21T13:49:52.868299Z","shell.execute_reply":"2025-06-21T13:50:15.321183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 3. 缺失值 / 重复值 检查\nprint(\"\\n 缺失值统计（仅显示存在缺失的列）:\")\nprint(df.isna().sum()[df.isna().sum() > 0])\nprint(\" 重复行数量:\", df.duplicated().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:50:15.323947Z","iopub.execute_input":"2025-06-21T13:50:15.324302Z","iopub.status.idle":"2025-06-21T13:50:55.427354Z","shell.execute_reply.started":"2025-06-21T13:50:15.324276Z","shell.execute_reply":"2025-06-21T13:50:55.425886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 4. 异常值处理（inf / -inf）\ninf_cols = df.columns[np.isinf(df).any()]\ndf[inf_cols] = df[inf_cols].replace([np.inf, -np.inf], 0)  # 替换为 0\nprint(\"替换 inf 的列数:\", len(inf_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:50:55.430513Z","iopub.execute_input":"2025-06-21T13:50:55.431029Z","iopub.status.idle":"2025-06-21T13:50:57.660378Z","shell.execute_reply.started":"2025-06-21T13:50:55.430999Z","shell.execute_reply":"2025-06-21T13:50:57.658974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 5. 可视化分析\n\nimport matplotlib.pyplot as plt\nfrom matplotlib import font_manager\nfrom statsmodels.tsa.stattools import adfuller\n\n# 加载并注册 SimHei 字体\nsimhei_path = \"/kaggle/input/simhei/SIMHEI.TTF\"\nfont_prop = font_manager.FontProperties(fname=simhei_path)\n\n# 设置 matplotlib 正确显示负号\nplt.rcParams['axes.unicode_minus'] = False\n\n# 画目标收益的时序图\nplt.figure(figsize=(10, 3))\nplt.plot(df['label'], color='blue', label='目标收益')\n\n# 中文标题使用 fontproperties 指定字体\nplt.title(\"Label 时序走势\", fontproperties=font_prop)\nplt.legend(prop=font_prop)  # 图例也要指定字体\nplt.tight_layout()\nplt.show()\n\n# 滚动均值和标准差\nroll_mean = df['label'].rolling(window=500).mean()\nroll_std = df['label'].rolling(window=500).std()\n\nplt.figure(figsize=(10, 3))\nplt.plot(df['label'], label='原始', alpha=0.5)\nplt.plot(roll_mean, label='滑动均值', color='green')\nplt.plot(roll_std, label='滑动标准差', color='orange')\nplt.title(\"Label 滚动统计分析\",fontproperties=font_prop)\nplt.legend(prop=font_prop)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:50:57.661677Z","iopub.execute_input":"2025-06-21T13:50:57.661972Z","iopub.status.idle":"2025-06-21T13:51:00.328797Z","shell.execute_reply.started":"2025-06-21T13:50:57.661948Z","shell.execute_reply":"2025-06-21T13:51:00.327575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 6. 特征工程（Lag, Rolling, 市场行为）\ndf['label_lag_1'] = df['label'].shift(1)\ndf['label_lag_2'] = df['label'].shift(2)\ndf['label_lag_3'] = df['label'].shift(3)\ndf['label_lag_5'] = df['label'].shift(5)\n\ndf['label_roll_mean_3'] = df['label'].rolling(3).mean()\ndf['label_roll_mean_5'] = df['label'].rolling(5).mean()\ndf['label_roll_std_3'] = df['label'].rolling(3).std()\ndf['label_roll_std_5'] = df['label'].rolling(5).std()\n\n# 微结构特征\ndf['volume_diff'] = df['volume'].diff()\ndf['buy_sell_diff'] = df['buy_qty'] - df['sell_qty']\ndf['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-6)\ndf['buy_volume_ratio'] = df['buy_qty'] / (df['volume'] + 1e-6)\ndf['sell_volume_ratio'] = df['sell_qty'] / (df['volume'] + 1e-6)\n\n# 删除因 shift 和 rolling 产生的缺失值\ndf.dropna(inplace=True)\ndf.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:51:00.329793Z","iopub.execute_input":"2025-06-21T13:51:00.330299Z","iopub.status.idle":"2025-06-21T13:51:03.362112Z","shell.execute_reply.started":"2025-06-21T13:51:00.330252Z","shell.execute_reply":"2025-06-21T13:51:03.360870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 7. 特征选择（保留最有用的100个特征）\ny = df['label']  # 目标变量\nX = df.drop(columns=['label'])  # 所有其他变量作为输入特征\n\nselector = SelectKBest(score_func=f_regression, k=100)  # 选择前100个与 y 相关性最高的特征\nX_selected = selector.fit_transform(X, y)\nselected_columns = X.columns[selector.get_support()]  # 提取被选中的列名\nX = X[selected_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:51:03.363353Z","iopub.execute_input":"2025-06-21T13:51:03.363757Z","iopub.status.idle":"2025-06-21T13:51:09.921400Z","shell.execute_reply.started":"2025-06-21T13:51:03.363721Z","shell.execute_reply":"2025-06-21T13:51:09.920138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 8. 拆分训练/验证集\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:51:09.922582Z","iopub.execute_input":"2025-06-21T13:51:09.922954Z","iopub.status.idle":"2025-06-21T13:51:10.088863Z","shell.execute_reply.started":"2025-06-21T13:51:09.922920Z","shell.execute_reply":"2025-06-21T13:51:10.087756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 9. 定义模型评估函数\ndef evaluate_model(name, y_true, y_pred):\n    print(f\"\\n📊 模型评估: {name}\")\n    print(\"MAE:\", mean_absolute_error(y_true, y_pred))\n    print(\"RMSE:\", np.sqrt(mean_squared_error(y_true, y_pred)))\n    print(\"R2 Score:\", r2_score(y_true, y_pred))\n    print(\"-\" * 40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:51:42.729542Z","iopub.execute_input":"2025-06-21T13:51:42.729905Z","iopub.status.idle":"2025-06-21T13:51:42.736795Z","shell.execute_reply.started":"2025-06-21T13:51:42.729882Z","shell.execute_reply":"2025-06-21T13:51:42.735409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 10. 训练多个模型对比表现\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\n\n# 线性回归\nlr = LinearRegression().fit(X_train, y_train)\nevaluate_model(\"线性回归\", y_val, lr.predict(X_val))\n\n# 决策树\ntree = DecisionTreeRegressor(max_depth=10, random_state=0).fit(X_train, y_train)\nevaluate_model(\"决策树\", y_val, tree.predict(X_val))\n\n# XGBoost\nxgb = XGBRegressor(n_estimators=100, max_depth=6, learning_rate=0.1, random_state=0).fit(X_train, y_train)\nevaluate_model(\"XGBoost\", y_val, xgb.predict(X_val))\n\n# LightGBM\nlgb = LGBMRegressor(n_estimators=100, learning_rate=0.1, max_depth=6, random_state=0).fit(X_train, y_train)\nevaluate_model(\"LightGBM\", y_val, lgb.predict(X_val))\n\n# CatBoost 默认参数\ncat = CatBoostRegressor(iterations=100, learning_rate=0.1, depth=6, verbose=0, random_seed=0).fit(X_train, y_train)\nevaluate_model(\"CatBoost\", y_val, cat.predict(X_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:51:45.637418Z","iopub.execute_input":"2025-06-21T13:51:45.638284Z","iopub.status.idle":"2025-06-21T13:53:56.623881Z","shell.execute_reply.started":"2025-06-21T13:51:45.638247Z","shell.execute_reply":"2025-06-21T13:53:56.622548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 11. 使用 Optuna 调优 CatBoost\ndef objective(trial):\n    params = {\n        'iterations': trial.suggest_int('iterations', 150, 300),\n        'depth': trial.suggest_int('depth', 5, 10),\n        'learning_rate': trial.suggest_float('learning_rate', 0.05, 0.3),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1, 9),\n        'random_strength': trial.suggest_float('random_strength', 0.0, 1.0),\n        'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n        'verbose': 0,\n        'random_seed': 0\n    }\n    model = CatBoostRegressor(**params)\n    model.fit(X_train, y_train)\n    preds = model.predict(X_val)\n    return mean_squared_error(y_val, preds, squared=False)\n\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=30)\n\n# 输出最优参数\nprint(\"最佳参数:\", study.best_params)\nprint(\"最小 RMSE:\", study.best_value)\n\n# 用最佳参数重新训练 CatBoost\nbest_cat = CatBoostRegressor(**study.best_params, verbose=0).fit(X_train, y_train)\nevaluate_model(\"CatBoost + Optuna\", y_val, best_cat.predict(X_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T13:54:32.209661Z","iopub.execute_input":"2025-06-21T13:54:32.211049Z","iopub.status.idle":"2025-06-21T14:37:18.580617Z","shell.execute_reply.started":"2025-06-21T13:54:32.210995Z","shell.execute_reply":"2025-06-21T14:37:18.579289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 12. 预测测试集并生成提交文件\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\ntest.reset_index(drop=True, inplace=True)\n\n# 应用与训练一致的特征构造逻辑（排除不能用 label 做滞后的部分）\ntest['volume_diff'] = test['volume'].diff()\ntest['buy_sell_diff'] = test['buy_qty'] - test['sell_qty']\ntest['buy_sell_ratio'] = test['buy_qty'] / (test['sell_qty'] + 1e-6)\ntest['buy_volume_ratio'] = test['buy_qty'] / (test['volume'] + 1e-6)\ntest['sell_volume_ratio'] = test['sell_qty'] / (test['volume'] + 1e-6)\n\n# 按照特征选择器输出的顺序选择列\nX_kaggle = test.reindex(columns=selected_columns).fillna(0)\nkaggle_preds = best_cat.predict(X_kaggle)\n\n# 提交文件\nsubmission = pd.DataFrame({\n    \"row_id\": test[\"row_id\"] if \"row_id\" in test.columns else test.index,\n    \"target\": kaggle_preds\n})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-21T14:38:56.889199Z","iopub.execute_input":"2025-06-21T14:38:56.889704Z","iopub.status.idle":"2025-06-21T14:39:29.131070Z","shell.execute_reply.started":"2025-06-21T14:38:56.889675Z","shell.execute_reply":"2025-06-21T14:39:29.130061Z"}},"outputs":[],"execution_count":null}]}