{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:18:53.747327Z","iopub.execute_input":"2025-06-15T14:18:53.748293Z","iopub.status.idle":"2025-06-15T14:18:53.757544Z","shell.execute_reply.started":"2025-06-15T14:18:53.748254Z","shell.execute_reply":"2025-06-15T14:18:53.756704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# 检查数据集目录\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# 加载数据\ntrain_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_data = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:18:53.759275Z","iopub.execute_input":"2025-06-15T14:18:53.759589Z","iopub.status.idle":"2025-06-15T14:19:31.255429Z","shell.execute_reply.started":"2025-06-15T14:18:53.759570Z","shell.execute_reply":"2025-06-15T14:19:31.253858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 查看数据的基本信息\nprint(\"训练数据的基本信息：\")\nprint(train_data.info())\nprint(\"\\n测试数据的基本信息：\")\nprint(test_data.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:31.257831Z","iopub.execute_input":"2025-06-15T14:19:31.258075Z","iopub.status.idle":"2025-06-15T14:19:31.339465Z","shell.execute_reply.started":"2025-06-15T14:19:31.258056Z","shell.execute_reply":"2025-06-15T14:19:31.338498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查缺失值\nprint(\"训练数据的缺失值情况：\")\nprint(train_data.isnull().sum())\nprint(\"\\n测试数据的缺失值情况：\")\nprint(test_data.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:31.340233Z","iopub.execute_input":"2025-06-15T14:19:31.340463Z","iopub.status.idle":"2025-06-15T14:19:34.735297Z","shell.execute_reply.started":"2025-06-15T14:19:31.340429Z","shell.execute_reply":"2025-06-15T14:19:34.734181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 检查列名\nprint(\"\\n训练数据的列名：\")\nprint(train_data.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:34.738009Z","iopub.execute_input":"2025-06-15T14:19:34.738258Z","iopub.status.idle":"2025-06-15T14:19:34.743998Z","shell.execute_reply.started":"2025-06-15T14:19:34.738240Z","shell.execute_reply":"2025-06-15T14:19:34.742873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 查看数据结构\nprint(\"\\n训练数据的前五行：\")\nprint(train_data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:34.744861Z","iopub.execute_input":"2025-06-15T14:19:34.745233Z","iopub.status.idle":"2025-06-15T14:19:34.775525Z","shell.execute_reply.started":"2025-06-15T14:19:34.745200Z","shell.execute_reply":"2025-06-15T14:19:34.774598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 简单的数据可视化\n\n# 目标变量的分布\nplt.figure(figsize=(10, 6))\nsns.histplot(train_data['label'], bins=50, kde=True)\nplt.title('目标变量（label）的分布')\nplt.xlabel('目标值')\nplt.ylabel('频率')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:34.776588Z","iopub.execute_input":"2025-06-15T14:19:34.776910Z","iopub.status.idle":"2025-06-15T14:19:37.600240Z","shell.execute_reply.started":"2025-06-15T14:19:34.776882Z","shell.execute_reply":"2025-06-15T14:19:37.599283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 特征X1的分布（修正后的列名）\nplt.figure(figsize=(10, 6))\nsns.histplot(train_data['X1'], bins=50, kde=True)\nplt.title('特征X1的分布')\nplt.xlabel('X1值')\nplt.ylabel('频率')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:37.601277Z","iopub.execute_input":"2025-06-15T14:19:37.601613Z","iopub.status.idle":"2025-06-15T14:19:40.378092Z","shell.execute_reply.started":"2025-06-15T14:19:37.601577Z","shell.execute_reply":"2025-06-15T14:19:40.377042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 时间戳与目标变量的关系（抽样数据）\n# 假设时间戳列名为“timestamp”，如果不是，请替换为正确的列名\nsample_data = train_data.sample(n=1000, random_state=42)\nplt.figure(figsize=(12, 6))\nplt.scatter(sample_data.index, sample_data['label'], alpha=0.5)  # 使用索引作为时间戳\nplt.title('时间戳与目标变量的关系（抽样数据）')\nplt.xlabel('时间戳（索引）')\nplt.ylabel('目标值')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:40.379056Z","iopub.execute_input":"2025-06-15T14:19:40.379337Z","iopub.status.idle":"2025-06-15T14:19:40.708288Z","shell.execute_reply.started":"2025-06-15T14:19:40.379315Z","shell.execute_reply":"2025-06-15T14:19:40.707500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 特征相关性分析（选择部分特征）\nselected_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume', 'X1', 'label']\ncorr_matrix = train_data[selected_features].corr()\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', square=True)\nplt.title('部分特征的相关性矩阵')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:40.709341Z","iopub.execute_input":"2025-06-15T14:19:40.709935Z","iopub.status.idle":"2025-06-15T14:19:41.145219Z","shell.execute_reply.started":"2025-06-15T14:19:40.709913Z","shell.execute_reply":"2025-06-15T14:19:41.144330Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 导入必要的库\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.146199Z","iopub.execute_input":"2025-06-15T14:19:41.146484Z","iopub.status.idle":"2025-06-15T14:19:41.152034Z","shell.execute_reply.started":"2025-06-15T14:19:41.146435Z","shell.execute_reply":"2025-06-15T14:19:41.150961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 特征和目标变量的选择\nfeatures = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume', 'X1']\nX = train_data[features]\ny = train_data['label']\n\n# 划分训练集和验证集\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# 特征标准化\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.153040Z","iopub.execute_input":"2025-06-15T14:19:41.153338Z","iopub.status.idle":"2025-06-15T14:19:41.702031Z","shell.execute_reply.started":"2025-06-15T14:19:41.153313Z","shell.execute_reply":"2025-06-15T14:19:41.701125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 初始化线性回归模型\nlr_model = LinearRegression()\n\n# 训练模型\nlr_model.fit(X_train_scaled, y_train)\n\n# 在训练集上进行预测\ny_train_pred = lr_model.predict(X_train_scaled)\n\n# 计算训练集上的均方误差（MSE）\ntrain_mse = mean_squared_error(y_train, y_train_pred)\nprint(f'训练集上的均方误差（MSE）：{train_mse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.703011Z","iopub.execute_input":"2025-06-15T14:19:41.703267Z","iopub.status.idle":"2025-06-15T14:19:41.837121Z","shell.execute_reply.started":"2025-06-15T14:19:41.703248Z","shell.execute_reply":"2025-06-15T14:19:41.836355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 在验证集上进行预测\ny_val_pred = lr_model.predict(X_val_scaled)\n\n# 计算验证集上的均方误差（MSE）\nval_mse = mean_squared_error(y_val, y_val_pred)\nprint(f'验证集上的均方误差（MSE）：{val_mse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.839885Z","iopub.execute_input":"2025-06-15T14:19:41.840157Z","iopub.status.idle":"2025-06-15T14:19:41.848064Z","shell.execute_reply.started":"2025-06-15T14:19:41.840136Z","shell.execute_reply":"2025-06-15T14:19:41.847393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 特征重要性分析（线性回归的系数）\nfeature_importance_df = pd.DataFrame({\n    '特征': features,\n    '系数': lr_model.coef_\n}).sort_values(by='系数', ascending=False)\n\nprint(\"\\n特征系数：\")\nprint(feature_importance_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.848607Z","iopub.execute_input":"2025-06-15T14:19:41.848808Z","iopub.status.idle":"2025-06-15T14:19:41.867340Z","shell.execute_reply.started":"2025-06-15T14:19:41.848789Z","shell.execute_reply":"2025-06-15T14:19:41.866545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 对测试集进行预测并提交\nX_test = test_data[features]\nX_test_scaled = scaler.transform(X_test)\ntest_pred = lr_model.predict(X_test_scaled)\n\n# 创建提交文件\nsubmission_df = pd.DataFrame({\n    'ID': test_data.index,  # 使用索引作为ID\n    'label': test_pred\n})\n\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"\\n提交文件已保存为 'submission.csv'\")\n\n# 评估模型的拟合情况\nprint(\"\\n模型评估：\")\nprint(f\"训练集 MSE: {train_mse:.4f}\")\nprint(f\"验证集 MSE: {val_mse:.4f}\")\n\nif val_mse > train_mse * 1.5:\n    print(\"\\n警告：模型可能存在过拟合现象。\")\nelif val_mse < train_mse * 0.8:\n    print(\"\\n警告：模型可能存在欠拟合现象。\")\nelse:\n    print(\"\\n模型拟合情况良好。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:41.868358Z","iopub.execute_input":"2025-06-15T14:19:41.868667Z","iopub.status.idle":"2025-06-15T14:19:43.170124Z","shell.execute_reply.started":"2025-06-15T14:19:41.868645Z","shell.execute_reply":"2025-06-15T14:19:43.169162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建提交文件\nsubmission_df = pd.DataFrame({\n    'ID': test_data.index,  # 使用索引作为ID\n    'prediction': test_pred  # 将列名改为 prediction\n})\n\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"\\n提交文件已保存为 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T14:19:43.171241Z","iopub.execute_input":"2025-06-15T14:19:43.171578Z","iopub.status.idle":"2025-06-15T14:19:44.435981Z","shell.execute_reply.started":"2025-06-15T14:19:43.171547Z","shell.execute_reply":"2025-06-15T14:19:44.435134Z"}},"outputs":[],"execution_count":null}]}