{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"},{"sourceId":11686826,"sourceType":"datasetVersion","datasetId":7335172}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 项目：DRW Crypto Market Prediction\n# 编写者：周赋瑶\n\n# 1. 导入核心工具包\nimport pandas as pd  # 数据处理主力库\nimport numpy as np   # 数值计算支持\nimport matplotlib.pyplot as plt  # 可视化\nimport seaborn as sns  # 高级可视化辅助\nfrom sklearn.model_selection import train_test_split  # 从scikit-learn导入数据集分割工具，用于划分训练集和测试集\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score  # 导入回归评估指标：均方误差、平均绝对误差、R²分数\nfrom sklearn.feature_selection import SelectKBest, f_regression  # 导入特征选择工具：SelectKBest和f_regression\nimport optuna  # 超参数优化库\nimport warnings  # 导入Python警告模块，用于控制警告信息的处理\nwarnings.filterwarnings(\"ignore\")  # 设置忽略所有警告信息，避免干扰输出结果","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:20.933539Z","iopub.execute_input":"2025-06-28T13:17:20.934259Z","iopub.status.idle":"2025-06-28T13:17:20.938815Z","shell.execute_reply.started":"2025-06-28T13:17:20.934232Z","shell.execute_reply":"2025-06-28T13:17:20.938050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 2. 读取训练集\n# 指定训练数据文件路径\ndata_path = '/kaggle/input/drw-crypto-market-prediction/train.parquet'\ndf = pd.read_parquet(data_path)  # 读取 parquet 文件格式\n\n# 输出数据集维度信息\nprint(\"数据维度:\", df.shape) \n# 输出前10列的名称\n# columns属性返回所有列名，[:10]切片取前10个，tolist()转换为列表格式\nprint(\"前几列名称:\", df.columns[:10].tolist())\n\n# dtypes返回每列的数据类型，head()限制输出行数避免控制台信息过长\nprint(df.dtypes.head())  # 显示DataFrame前5列的数据类型信息","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:20.939963Z","iopub.execute_input":"2025-06-28T13:17:20.940200Z","iopub.status.idle":"2025-06-28T13:17:26.118403Z","shell.execute_reply.started":"2025-06-28T13:17:20.940180Z","shell.execute_reply":"2025-06-28T13:17:26.117446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 3. 缺失值 / 重复值 检查\nprint(\"\\n 缺失值统计（仅显示存在缺失的列）:\")\nprint(df.isna().sum()[df.isna().sum() > 0])\nprint(\" 重复行数量:\", df.duplicated().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:26.119843Z","iopub.execute_input":"2025-06-28T13:17:26.120479Z","iopub.status.idle":"2025-06-28T13:17:49.036087Z","shell.execute_reply.started":"2025-06-28T13:17:26.120451Z","shell.execute_reply":"2025-06-28T13:17:49.035337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 4. 异常值处理（inf / -inf）\ninf_cols = df.columns[np.isinf(df).any()]\ndf[inf_cols] = df[inf_cols].replace([np.inf, -np.inf], 0)  # 替换为 0\nprint(\"替换 inf 的列数:\", len(inf_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:49.036963Z","iopub.execute_input":"2025-06-28T13:17:49.037280Z","iopub.status.idle":"2025-06-28T13:17:50.597054Z","shell.execute_reply.started":"2025-06-28T13:17:49.037255Z","shell.execute_reply":"2025-06-28T13:17:50.596202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 5. 可视化分析\n\nimport matplotlib.pyplot as plt\nfrom matplotlib import font_manager\nfrom statsmodels.tsa.stattools import adfuller\n\n# 加载并注册 SimHei 字体\nsimhei_path = \"/kaggle/input/simhei/SIMHEI.TTF\"\nfont_prop = font_manager.FontProperties(fname=simhei_path)\n\n# 设置 matplotlib 正确显示负号\nplt.rcParams['axes.unicode_minus'] = False\n\n# 画目标收益的时序图\nplt.figure(figsize=(10, 3))\nplt.plot(df['label'], color='blue', label='目标收益')\n\n# 中文标题使用 fontproperties 指定字体\nplt.title(\"Label 时序走势\", fontproperties=font_prop)\nplt.legend(prop=font_prop)  # 图例也要指定字体\nplt.tight_layout()\nplt.show()\n\n# 滚动均值和标准差\nroll_mean = df['label'].rolling(window=500).mean()\nroll_std = df['label'].rolling(window=500).std()\n\nplt.figure(figsize=(10, 3))\nplt.plot(df['label'], label='原始', alpha=0.5)\nplt.plot(roll_mean, label='滑动均值', color='green')\nplt.plot(roll_std, label='滑动标准差', color='orange')\nplt.title(\"Label 滚动统计分析\",fontproperties=font_prop)\nplt.legend(prop=font_prop)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:50.599379Z","iopub.execute_input":"2025-06-28T13:17:50.600189Z","iopub.status.idle":"2025-06-28T13:17:52.284646Z","shell.execute_reply.started":"2025-06-28T13:17:50.600158Z","shell.execute_reply":"2025-06-28T13:17:52.283926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 6. 特征工程（Lag, Rolling, 市场行为）\ndf['label_lag_1'] = df['label'].shift(1)\ndf['label_lag_2'] = df['label'].shift(2)\ndf['label_lag_3'] = df['label'].shift(3)\ndf['label_lag_5'] = df['label'].shift(5)\n\ndf['label_roll_mean_3'] = df['label'].rolling(3).mean()\ndf['label_roll_mean_5'] = df['label'].rolling(5).mean()\ndf['label_roll_std_3'] = df['label'].rolling(3).std()\ndf['label_roll_std_5'] = df['label'].rolling(5).std()\n\n# 微结构特征\ndf['volume_diff'] = df['volume'].diff()\ndf['buy_sell_diff'] = df['buy_qty'] - df['sell_qty']\ndf['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-6)\ndf['buy_volume_ratio'] = df['buy_qty'] / (df['volume'] + 1e-6)\ndf['sell_volume_ratio'] = df['sell_qty'] / (df['volume'] + 1e-6)\n\n# 删除因 shift 和 rolling 产生的缺失值\ndf.dropna(inplace=True)\ndf.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:52.285367Z","iopub.execute_input":"2025-06-28T13:17:52.285553Z","iopub.status.idle":"2025-06-28T13:17:55.755569Z","shell.execute_reply.started":"2025-06-28T13:17:52.285539Z","shell.execute_reply":"2025-06-28T13:17:55.754761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 7. 特征选择（保留最有用的100个特征）\ny = df['label']  # 目标变量\nX = df.drop(columns=['label'])  # 所有其他变量作为输入特征\n\nselector = SelectKBest(score_func=f_regression, k=100)  # 选择前100个与 y 相关性最高的特征\nX_selected = selector.fit_transform(X, y)\nselected_columns = X.columns[selector.get_support()]  # 提取被选中的列名\nX = X[selected_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:17:55.756424Z","iopub.execute_input":"2025-06-28T13:17:55.756688Z","iopub.status.idle":"2025-06-28T13:18:07.723214Z","shell.execute_reply.started":"2025-06-28T13:17:55.756669Z","shell.execute_reply":"2025-06-28T13:18:07.722600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 8. 拆分训练/验证集\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:18:07.724093Z","iopub.execute_input":"2025-06-28T13:18:07.724614Z","iopub.status.idle":"2025-06-28T13:18:07.880161Z","shell.execute_reply.started":"2025-06-28T13:18:07.724588Z","shell.execute_reply":"2025-06-28T13:18:07.879590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 9. 定义模型评估函数\ndef evaluate_model(name, y_true, y_pred):\n    print(f\"\\n📊 模型评估: {name}\")\n    print(\"MAE:\", mean_absolute_error(y_true, y_pred))\n    print(\"RMSE:\", np.sqrt(mean_squared_error(y_true, y_pred)))\n    print(\"R2 Score:\", r2_score(y_true, y_pred))\n    print(\"-\" * 40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:18:07.880921Z","iopub.execute_input":"2025-06-28T13:18:07.881192Z","iopub.status.idle":"2025-06-28T13:18:07.885899Z","shell.execute_reply.started":"2025-06-28T13:18:07.881167Z","shell.execute_reply":"2025-06-28T13:18:07.885027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 10. 训练多个模型对比表现\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\n\n# 线性回归\nlr = LinearRegression().fit(X_train, y_train)\nevaluate_model(\"线性回归\", y_val, lr.predict(X_val))\n\n# 决策树\ntree = DecisionTreeRegressor(max_depth=10, random_state=0).fit(X_train, y_train)\nevaluate_model(\"决策树\", y_val, tree.predict(X_val))\n\n# XGBoost\nxgb = XGBRegressor(n_estimators=100, max_depth=6, learning_rate=0.1, random_state=0).fit(X_train, y_train)\nevaluate_model(\"XGBoost\", y_val, xgb.predict(X_val))\n\n# LightGBM\nlgb = LGBMRegressor(n_estimators=100, learning_rate=0.1, max_depth=6, random_state=0).fit(X_train, y_train)\nevaluate_model(\"LightGBM\", y_val, lgb.predict(X_val))\n\n# CatBoost 默认参数\ncat = CatBoostRegressor(iterations=100, learning_rate=0.1, depth=6, verbose=0, random_seed=0).fit(X_train, y_train)\nevaluate_model(\"CatBoost\", y_val, cat.predict(X_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:18:07.886838Z","iopub.execute_input":"2025-06-28T13:18:07.887038Z","iopub.status.idle":"2025-06-28T13:19:29.305939Z","shell.execute_reply.started":"2025-06-28T13:18:07.887022Z","shell.execute_reply":"2025-06-28T13:19:29.305113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport optuna\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_error\nimport time\nimport joblib\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\n\n#  定义Optuna优化目标函数\ndef xgb_objective(trial):\n    params = {\n        # 核心参数：使用多棵树和合理的学习率\n        'n_estimators': trial.suggest_categorical('n_estimators', [100, 200, 300]),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3, log=True),\n        \n        # 树结构参数\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_float('min_child_weight', 1e-5, 10),\n        'gamma': trial.suggest_float('gamma', 0, 1),\n        \n        # 正则化参数\n        'reg_alpha': trial.suggest_float('reg_alpha', 0, 10),\n        'reg_lambda': trial.suggest_float('reg_lambda', 0.5, 5),\n        \n        # 采样参数\n        'subsample': trial.suggest_float('subsample', 0.6, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),\n        \n        # GPU优化\n        'tree_method': 'gpu_hist',\n        'gpu_id': 0,\n        'random_state': 42,\n        'verbosity': 0,\n        'eval_metric': 'rmse'\n    }\n    \n    # 自动早停\n    early_stop = trial.suggest_int('early_stopping_rounds', 10, 50)\n    \n    model = XGBRegressor(**params)\n    \n    start_time = time.time()\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        early_stopping_rounds=early_stop,\n        verbose=False\n    )\n    train_time = time.time() - start_time\n    \n    # 使用最佳迭代预测\n    preds = model.predict(X_val)\n    rmse = np.sqrt(mean_squared_error(y_val, preds))\n    \n    trial.set_user_attr('train_time', train_time)\n    trial.set_user_attr('best_iteration', model.best_iteration)\n    return rmse\n\n# 4. 创建Optuna研究并优化\nstudy = optuna.create_study(\n    direction='minimize',\n    sampler=optuna.samplers.TPESampler(\n        n_startup_trials=10,\n        seed=42\n    ),\n    pruner=optuna.pruners.MedianPruner(\n        n_startup_trials=5,\n        n_warmup_steps=10\n    )\n)\n\n# 优化参数 - 增加试验次数以获得更好的结果\nstudy.optimize(\n    xgb_objective, \n    n_trials=30,\n    timeout=7200,\n    show_progress_bar=True\n)\n\n# 输出最优参数\nprint(\"=\"*50)\nprint(\"最佳参数:\", study.best_params)\nprint(\"最小 RMSE:\", study.best_value)\nprint(f\"训练时间: {study.best_trial.user_attrs['train_time']:.2f}秒\")\nprint(f\"最佳迭代次数: {study.best_trial.user_attrs['best_iteration']}\")\n\n\n# 5. 用最佳参数训练最终模型\nbest_params = study.best_params.copy()\n# 移除早停参数，因为它不是模型本身的参数\nbest_params.pop('early_stopping_rounds', None)\n\n# 使用最佳迭代次数作为n_estimators\nif 'best_iteration' in study.best_trial.user_attrs:\n    best_iteration = study.best_trial.user_attrs['best_iteration']\n    best_params['n_estimators'] = best_iteration\n\nfinal_model = XGBRegressor(\n    **best_params,\n    tree_method='gpu_hist',\n    random_state=42\n)\n\n# 合并训练集和验证集进行最终训练\nX_full = np.vstack((X_train, X_val))\ny_full = np.concatenate((y_train, y_val))\n\nfinal_model.fit(X_full, y_full)\n\n# 评估最终模型\nval_preds = final_model.predict(X_val)\nevaluate_model(\"最终模型验证集\", y_val, val_preds)\n\n# 保存模型\nfinal_model.save_model('optimized_xgb_model.json')\njoblib.dump(final_model, 'optimized_xgb_model.pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:31:46.215336Z","iopub.execute_input":"2025-06-28T13:31:46.215903Z","iopub.status.idle":"2025-06-28T13:34:39.792644Z","shell.execute_reply.started":"2025-06-28T13:31:46.215880Z","shell.execute_reply":"2025-06-28T13:34:39.792037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 编写者：周赋瑶\n# 12. 预测测试集并生成提交文件\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\ntest.reset_index(drop=True, inplace=True)\n\n# 应用与训练一致的特征构造逻辑（排除不能用 label 做滞后的部分）\ntest['volume_diff'] = test['volume'].diff()\ntest['buy_sell_diff'] = test['buy_qty'] - test['sell_qty']\ntest['buy_sell_ratio'] = test['buy_qty'] / (test['sell_qty'] + 1e-6)\ntest['buy_volume_ratio'] = test['buy_qty'] / (test['volume'] + 1e-6)\ntest['sell_volume_ratio'] = test['sell_qty'] / (test['volume'] + 1e-6)\n\n# 按照特征选择器输出的顺序选择列\nX_kaggle = test.reindex(columns=selected_columns).fillna(0)\nkaggle_preds = final_model.predict(X_kaggle)\n\n# 提交文件\nsubmission = pd.DataFrame({\n    \"ID\": (test[\"row_id\"] if \"row_id\" in test.columns else test.index)+1,\n    \"prediction\": kaggle_preds\n})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-28T13:36:07.416524Z","iopub.execute_input":"2025-06-28T13:36:07.416817Z","iopub.status.idle":"2025-06-28T13:36:15.625575Z","shell.execute_reply.started":"2025-06-28T13:36:07.416797Z","shell.execute_reply":"2025-06-28T13:36:15.624963Z"}},"outputs":[],"execution_count":null}]}