{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:56:19.694881Z","iopub.execute_input":"2025-06-16T04:56:19.699508Z","iopub.status.idle":"2025-06-16T04:56:19.728912Z","shell.execute_reply.started":"2025-06-16T04:56:19.699379Z","shell.execute_reply":"2025-06-16T04:56:19.725245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom scipy.stats import pearsonr\nimport gc # Garbage Collector for memory management\n\n# 设置绘图风格\nplt.style.use('seaborn-v0_8-whitegrid')\nprint(\"✅ 库导入成功\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:56:19.738234Z","iopub.execute_input":"2025-06-16T04:56:19.740821Z","iopub.status.idle":"2025-06-16T04:56:19.782527Z","shell.execute_reply.started":"2025-06-16T04:56:19.740493Z","shell.execute_reply":"2025-06-16T04:56:19.775196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. 数据加载与初步探查 (Data Loading & Initial Inspection)\n# ===================================================================\n# 数据路径\nDATA_PATH = '/kaggle/input/drw-crypto-market-prediction/'\n\n# 使用 pandas 读取高效的 parquet 文件\ntry:\n    train_df = pd.read_parquet(f'{DATA_PATH}train.parquet')\n    test_df = pd.read_parquet(f'{DATA_PATH}test.parquet')\n    sample_submission_df = pd.read_csv(f'{DATA_PATH}sample_submission.csv')\n    \n    print(\"✅ 数据加载成功\")\n    print(f\"训练集形状: {train_df.shape}\")\n    print(f\"测试集形状: {test_df.shape}\")\nexcept FileNotFoundError:\n    print(\"❌ 数据文件未找到。请确保比赛数据已正确添加到Notebook的 /kaggle/input/ 目录下。\")\n    exit()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:56:19.788819Z","iopub.execute_input":"2025-06-16T04:56:19.790712Z","iopub.status.idle":"2025-06-16T04:56:58.803427Z","shell.execute_reply.started":"2025-06-16T04:56:19.790536Z","shell.execute_reply":"2025-06-16T04:56:58.802442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 初步检查 ---\nprint(\"\\n--- 训练集信息 ---\")\n\n# 新增：在使用数据前，先将索引重置为普通列\n# 这是因为 'timestamp' 在 parquet 文件中被存储为索引\ntrain_df = train_df.reset_index()\ntest_df = test_df.reset_index()\n\nprint(\"✅ 已将索引 'timestamp' 转换为普通列。\")\n\n# 使用 display 可以更好地在Notebook中展示DataFrame\ndisplay(train_df.head())\n\n# 将 Unix 时间戳转换为 datetime 对象，方便后续处理\n# unit='ms' 表示时间戳是毫秒级的\ntrain_df['timestamp'] = pd.to_datetime(train_df['timestamp'], unit='ms')\n\n# 测试集的时间戳是id，我们将其重命名以避免混淆\ntest_df.rename(columns={'timestamp': 'id'}, inplace=True)\n\nprint(\"\\n✅ 时间戳处理和重命名完成。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:56:58.805551Z","iopub.execute_input":"2025-06-16T04:56:58.805825Z","iopub.status.idle":"2025-06-16T04:57:06.644946Z","shell.execute_reply.started":"2025-06-16T04:56:58.805804Z","shell.execute_reply":"2025-06-16T04:57:06.644187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查缺失值\nprint(\"\\n--- 缺失值检查 ---\")\n# 计算缺失值比例\nmissing_values = train_df.isnull().sum() / len(train_df)\nprint(missing_values[missing_values > 0].sort_values(ascending=False))\n\n# 处理缺失值：使用前向填充 (ffill)\n# 这对于时间序列数据是合理的，因为它假设当前缺失的值与上一个时间点的值相同\ntrain_df.fillna(method='ffill', inplace=True)\ntest_df.fillna(method='ffill', inplace=True)\nprint(\"\\n✅ 使用前向填充(ffill)处理了缺失值。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:57:06.645776Z","iopub.execute_input":"2025-06-16T04:57:06.646041Z","iopub.status.idle":"2025-06-16T04:57:13.723105Z","shell.execute_reply.started":"2025-06-16T04:57:06.646021Z","shell.execute_reply":"2025-06-16T04:57:13.722339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================================================================\n# 3. 探索性数据分析 (Exploratory Data Analysis - EDA)\n# ===================================================================\nprint(\"\\n--- 开始探索性数据分析 (EDA) ---\")\n\n# --- 3.1. 目标变量 'label' 的分布 ---\nplt.figure(figsize=(14, 6))\nsns.histplot(train_df['label'], bins=100, kde=True, color='blue')\nplt.title('目标变量 (label) 的分布', fontsize=16)\nplt.xlabel('Label 值', fontsize=12)\nplt.ylabel('频数', fontsize=12)\nplt.show()\nprint(f\"目标变量统计信息:\\n{train_df['label'].describe()}\")\n\n\n# --- 3.2. 特征与目标变量的相关性 (高效优化版) ---\nprint(\"\\n--- 高效计算特征与目标的相关性 ---\")\n\n# 找出所有的特征列（排除非数值列和目标列）\nfeature_cols = [col for col in train_df.columns if col not in ['timestamp', 'label']]\n\n# 使用 .corrwith() 高效计算所有特征与 'label' 的相关性\n# 这比 .corr() 快几个数量级\ncorrelations = train_df[feature_cols].corrwith(train_df['label']).sort_values()\n\nprint(\"\\n--- 与 'label' 负相关性最高的10个特征 ---\")\ndisplay(correlations.head(10))\nprint(\"\\n--- 与 'label' 正相关性最高的10个特征 ---\")\n# 注意：最后一个是label与自身的相关性(值为1)，所以我们看 tail(11) 然后去掉最后一个\ndisplay(correlations.tail(11).iloc[:-1]) \n\n# 释放内存\ndel correlations\ngc.collect()\nprint(\"✅ 相关性计算完成。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:57:13.723981Z","iopub.execute_input":"2025-06-16T04:57:13.724213Z","iopub.status.idle":"2025-06-16T04:57:27.534622Z","shell.execute_reply.started":"2025-06-16T04:57:13.724188Z","shell.execute_reply":"2025-06-16T04:57:27.533832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================================================================\n# 3. 探索性数据分析 (EDA) & 动态特征选择\n# ===================================================================\nprint(\"\\n--- 开始探索性数据分析 (EDA) ---\")\n\n# --- 3.1. 目标变量 'label' 的分布 ---\nplt.figure(figsize=(14, 6))\nsns.histplot(train_df['label'], bins=100, kde=True, color='blue')\nplt.title('目标变量 (label) 的分布', fontsize=16)\nplt.xlabel('Label 值', fontsize=12)\nplt.ylabel('频数', fontsize=12)\nplt.show()\nprint(f\"目标变量统计信息:\\n{train_df['label'].describe()}\")\n\n\n# --- 3.2. 特征与目标变量的相关性 (高效优化版) & 动态选择 ---\nprint(\"\\n--- 高效计算特征与目标的相关性 ---\")\n\nfeature_cols = [col for col in train_df.columns if col.startswith('f_')] # 更精确地选择特征列\ncorrelations = train_df[feature_cols].corrwith(train_df['label']).sort_values(ascending=False)\n\n# 动态地从相关性计算结果中选择最重要的特征\n# 我们合并正相关和负相关最强的特征\nN_TOP_FEATURES = 3 # 选择前N个正相关和前N个负相关的特征\ntop_positive_corr_features = correlations.head(N_TOP_FEATURES).index.tolist()\ntop_negative_corr_features = correlations.tail(N_TOP_FEATURES).index.tolist()\n\n# 这将是我们在特征工程中使用的关键特征列表\nkey_features_for_eng = top_positive_corr_features + top_negative_corr_features\n\nprint(\"\\n--- 与 'label' 正相关性最高的特征 ---\")\ndisplay(correlations.head(N_TOP_FEATURES))\nprint(\"\\n--- 与 'label' 负相关性最高的特征 ---\")\ndisplay(correlations.tail(N_TOP_FEATURES))\nprint(f\"\\n✅ 动态选择的关键特征为: {key_features_for_eng}\")\n\n# 释放内存\ndel correlations\ngc.collect()\n\n\n# ===================================================================\n# 4. 特征工程 (Feature Engineering) - 动态版\n# ===================================================================\nprint(\"\\n--- 开始特征工程 (动态版) ---\")\n\ndef create_features_dynamic(df, key_features):\n    \"\"\"为数据集创建新的特征 (使用动态传入的特征列表)\"\"\"\n    df_out = df.copy()\n\n    # --- 基础特征 ---\n    df_out['bid_ask_spread'] = df_out['ask_qty'] - df_out['bid_qty']\n    df_out['buy_sell_ratio'] = df_out['buy_qty'] / (df_out['sell_qty'] + 1e-6)\n    \n    print(f\"将对以下关键特征进行滞后和滚动计算: {key_features}\")\n\n    # --- 滞后和滚动特征 ---\n    for col in key_features:\n        # 添加一个检查，确保列真的存在，增加代码稳健性\n        if col in df_out.columns:\n            df_out[f'{col}_lag1'] = df_out[col].shift(1)\n            df_out[f'{col}_lag3'] = df_out[col].shift(3)\n            df_out[f'{col}_rolling_mean_10'] = df_out[col].rolling(window=10, min_periods=1).mean()\n            df_out[f'{col}_rolling_std_10'] = df_out[col].rolling(window=10, min_periods=1).std()\n\n    df_out.fillna(method='ffill', inplace=True)\n    df_out.fillna(method='bfill', inplace=True)\n\n    return df_out\n\n# 对训练集和测试集应用特征工程，传入我们动态生成的key_features_for_eng\ntrain_featured_df = create_features_dynamic(train_df, key_features_for_eng)\ntest_featured_df = create_features_dynamic(test_df, key_features_for_eng)\n\nprint(\"✅ 特征工程完成\")\nprint(f\"新的训练集形状: {train_featured_df.shape}\")\nprint(f\"新的测试集形状: {test_featured_df.shape}\")\n\n\n# 第4部分代码末尾的推荐写法：\ndel train_df, test_df\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:57:27.535503Z","iopub.execute_input":"2025-06-16T04:57:27.535740Z","iopub.status.idle":"2025-06-16T04:57:49.792397Z","shell.execute_reply.started":"2025-06-16T04:57:27.535720Z","shell.execute_reply":"2025-06-16T04:57:49.791316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================================================================\n# 5. 模型训练与验证 (Model Training & Validation) - 极限速度优化版\n# ===================================================================\nprint(\"\\n--- 开始模型训练与验证 (极限速度优化版) ---\")\n\n# --- 5.1. 核心优化：使用部分近期数据进行训练 ---\n# 如果完整数据导致崩溃，只使用最新的50%数据是一个非常有效的策略\n# 这不仅能解决内存问题，也符合时间序列中近期数据更重要的直觉\nTRAIN_SAMPLE_RATIO = 0.5 \nstart_row = int(len(train_featured_df) * (1 - TRAIN_SAMPLE_RATIO))\nprint(f\"⚠️ 为提高效率，将仅使用最新的 {TRAIN_SAMPLE_RATIO*100}% 数据进行训练。\")\ntrain_subset_df = train_featured_df.iloc[start_row:].copy()\n\n# 使用完完整训练集后，立即删除\ndel train_featured_df\ngc.collect()\n\nprint(f\"用于训练的数据形状: {train_subset_df.shape}\")\n\n\n# --- 5.2. 定义特征列和目标列 (在子集上操作) ---\nfeatures = [col for col in train_subset_df.columns if col not in ['timestamp', 'label']]\nX = train_subset_df[features]\ny = train_subset_df['label']\n\n# 释放内存\ndel train_subset_df\ngc.collect()\n\n# --- 5.3. 时序验证集划分 (在子集上操作) ---\n# 我们仍然在子集的80%上训练，20%上验证\nsplit_index = int(len(X) * 0.8)\nX_train, X_val = X.iloc[:split_index], X.iloc[split_index:]\ny_train, y_val = y.iloc[:split_index], y.iloc[split_index:]\n\nprint(f\"新的训练集大小: {len(X_train)}\")\nprint(f\"新的验证集大小: {len(X_val)}\")\n\n\n# --- 5.4. 智能设备选择与模型训练 ---\ntry:\n    import cupy\n    cupy.zeros(1)\n    device = 'gpu'\n    print(\"✅ GPU环境检测成功，将使用 GPU 进行训练。\")\nexcept (ImportError, Exception):\n    device = 'cpu'\n    print(\"⚠️ 未检测到可用的GPU环境，将自动切换到 CPU 进行训练。\")\n\n# LightGBM 参数设置 - 专为低内存和高速度优化\nlgb_params_fast = {\n    'objective': 'regression_l1',\n    'metric': 'mae',\n    'n_estimators': 1000,          # 减少树的数量\n    'learning_rate': 0.05,         # 稍微增大学习率以配合较少的树\n    'feature_fraction': 0.7,       # 减少特征采样\n    'bagging_fraction': 0.7,\n    'bagging_freq': 1,\n    'lambda_l1': 0.5,\n    'lambda_l2': 0.5,\n    'num_leaves': 31,              # 核心优化：显著减少叶子数 (从64到31)\n    'verbose': -1,\n    'n_jobs': -1, \n    'seed': 42,\n    'boosting_type': 'gbdt',\n    'device': device,\n}\n\nmodel = lgb.LGBMRegressor(**lgb_params_fast)\n\nprint(\"\\n开始模型训练...\")\nmodel.fit(X_train, y_train,\n          eval_set=[(X_val, y_val)],\n          eval_metric='mae',\n          callbacks=[lgb.early_stopping(30, verbose=True)]) # 更激进的早停\n\n# --- 5.5. 评估与分析 ---\n# 这部分不耗时，予以保留\nprint(\"\\n在验证集上进行评估...\")\nval_preds = model.predict(X_val)\npearson_corr, _ = pearsonr(y_val, val_preds)\nprint(f\"\\n✅ 验证集上的皮尔逊相关系数 (Pearson Correlation): {pearson_corr:.6f}\")\n\nprint(\"\\n绘制特征重要性图...\")\nlgb.plot_importance(model, max_num_features=20, figsize=(10, 10))\nplt.title('Top 20 Feature Importances', fontsize=16)\nplt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:57:49.793706Z","iopub.execute_input":"2025-06-16T04:57:49.794092Z","iopub.status.idle":"2025-06-16T04:58:40.165787Z","shell.execute_reply.started":"2025-06-16T04:57:49.794058Z","shell.execute_reply":"2025-06-16T04:58:40.164837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================================================================================\n# # 6. 预测与提交 (最终决定版 - 严格遵循官方格式)\n# ===================================================================================\n\n# --- 6.1. 准备测试数据 ---\n# 从经过特征工程的 test_featured_df 中，提取出与模型训练时完全相同的特征列。\n# 'features' 变量是在模型训练前定义的，我们必须用它来确保特征一致性。\nprint(\"--- 准备测试数据集以进行预测...\")\nX_test = test_featured_df[features]\n\n# --- 6.2. 使用模型进行预测 ---\nprint(\"\\n--- 使用已训练好的模型进行预测...\")\ntest_predictions = model.predict(X_test)\n\n# --- 6.3. 创建符合官方格式的提交文件 ---\nprint(\"\\n--- 创建完全符合 sample_submission.csv 格式的提交文件...\")\n\n# 官方要求 'id' 列是整数行号，从 0 开始。test_featured_df.index 正好就是这个序列。\n# 官方要求预测列名为 'prediction'。\nsubmission_df = pd.DataFrame({\n    'id': test_featured_df.index,\n    'prediction': test_predictions\n})\n\n# --- 6.4. 内存释放与保存 ---\nprint(\"释放内存...\")\ndel test_featured_df, X_test, test_predictions\ngc.collect()\n\n# --- 6.5. 保存到 CSV 文件 ---\n# 关键: index=False 是必须的，它能防止 pandas 将 DataFrame 的索引 (0, 1, 2...) 作为第一列写入文件。\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\n✅ 'submission.csv' 文件已成功生成！\")\nprint(\"文件预览:\")\ndisplay(submission_df.head())\nprint(f\"提交文件总行数: {len(submission_df)}\") # 应该显示 538150\nprint(\"id 列数据类型:\", submission_df['id'].dtype) # 应该是 int\nprint(\"prediction 列数据类型:\", submission_df['prediction'].dtype) # 应该是 float","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T04:58:40.166772Z","iopub.execute_input":"2025-06-16T04:58:40.167424Z","iopub.status.idle":"2025-06-16T04:58:49.794856Z","shell.execute_reply.started":"2025-06-16T04:58:40.167393Z","shell.execute_reply":"2025-06-16T04:58:49.793299Z"}},"outputs":[],"execution_count":null}]}