{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-14T21:59:31.199348Z","iopub.execute_input":"2025-07-14T21:59:31.200094Z","iopub.status.idle":"2025-07-14T21:59:33.707062Z","shell.execute_reply.started":"2025-07-14T21:59:31.200063Z","shell.execute_reply":"2025-07-14T21:59:33.706335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pickle\nimport polars as pl\nimport numpy as np\nimport pandas as pd\nimport gc\nimport warnings\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\nfrom pytorch_lightning.loggers import WandbLogger\nimport lightgbm as lgb\nfrom pandas import read_parquet\n\nfrom datetime import datetime\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom torch.utils.data import Dataset, DataLoader\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom scipy.stats import pearsonr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T21:59:50.332044Z","iopub.execute_input":"2025-07-14T21:59:50.332832Z","iopub.status.idle":"2025-07-14T22:00:16.568601Z","shell.execute_reply.started":"2025-07-14T21:59:50.332809Z","shell.execute_reply":"2025-07-14T22:00:16.567659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_model(model, path):\n    with open(path, \"wb\") as f:\n        pickle.dump(model, f)\n\n\ndef load_model(path):\n    with open(path, \"rb\") as f:\n        model = pickle.load(f)\n    return model\n\ndef _pearsonr(y_true, y_pred):\n    return pearsonr(y_true, y_pred)[0]\n\nclass VotingModel(BaseEstimator, RegressorMixin):\n    \"\"\"\n    A voting ensemble model that averages predictions from multiple estimators.\n\n    Parameters:\n    - estimators: List of estimators to include in the voting ensemble\n\n    Methods:\n    - fit(X, y=None): No training is performed as it's just an aggregator.\n    - predict(X): Returns the average prediction from all included estimators.\n    - predict_proba(X): Returns the average class probabilities from all included estimators.\n    \"\"\"\n\n    def __init__(self, estimators):\n        \"\"\"\n        Initializes the VotingModel with a list of estimators.\n\n        Parameters:\n        - estimators: List of estimators to include in the voting ensemble\n        \"\"\"\n        super().__init__()\n        self.estimators = estimators\n\n    def fit(self, X, y=None):\n        \"\"\"Fits the voting model (no operation).\"\"\"\n        return self\n\n    def predict(self, X):\n        \"\"\"Returns the average prediction from all included estimators.\"\"\"\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\n    def predict_proba(self, X):\n        \"\"\"Returns the average class probabilities from all included estimators.\"\"\"\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \ndef reduce_mem_usage(df):\n    \"\"\"\n    Optimizes the memory usage of a DataFrame by downcasting numeric columns to smaller data types.\n\n    Parameters:\n    - df: DataFrame to be optimized\n\n    Returns:\n    - df: Optimized DataFrame\n    \"\"\"\n\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(\"Memory usage of dataframe is {:.2f} MB\".format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type) == \"category\":\n            continue\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == \"int\":\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if (\n                    c_min > np.finfo(np.float16).min\n                    and c_max < np.finfo(np.float16).max\n                ):\n                    df[col] = df[col].astype(np.float16)\n                elif (\n                    c_min > np.finfo(np.float32).min\n                    and c_max < np.finfo(np.float32).max\n                ):\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(\"Memory usage after optimization is: {:.2f} MB\".format(end_mem))\n    print(\"Decreased by {:.1f}%\".format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:00:20.410582Z","iopub.execute_input":"2025-07-14T22:00:20.411320Z","iopub.status.idle":"2025-07-14T22:00:20.424271Z","shell.execute_reply.started":"2025-07-14T22:00:20.411293Z","shell.execute_reply":"2025-07-14T22:00:20.423408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # === NEW MICROSTRUCTURE FEATURES ===\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:00:23.458681Z","iopub.execute_input":"2025-07-14T22:00:23.458976Z","iopub.status.idle":"2025-07-14T22:00:23.473393Z","shell.execute_reply.started":"2025-07-14T22:00:23.458955Z","shell.execute_reply":"2025-07-14T22:00:23.472740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\npath = r'/kaggle/input/drw-crypto-market-prediction/'\ntrain = reduce_mem_usage(read_parquet(path+r'train.parquet'))\ntest = reduce_mem_usage(read_parquet(path+r'test.parquet'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:00:26.712409Z","iopub.execute_input":"2025-07-14T22:00:26.713133Z","iopub.status.idle":"2025-07-14T22:01:33.816452Z","shell.execute_reply.started":"2025-07-14T22:00:26.713108Z","shell.execute_reply":"2025-07-14T22:01:33.815784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrain = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:01:59.725127Z","iopub.execute_input":"2025-07-14T22:01:59.725390Z","iopub.status.idle":"2025-07-14T22:02:16.611973Z","shell.execute_reply.started":"2025-07-14T22:01:59.725372Z","shell.execute_reply":"2025-07-14T22:02:16.611315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_feaures = [f\"X{num}\" for num in range(1,890)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:02:27.499935Z","iopub.execute_input":"2025-07-14T22:02:27.500496Z","iopub.status.idle":"2025-07-14T22:02:27.504117Z","shell.execute_reply.started":"2025-07-14T22:02:27.500473Z","shell.execute_reply":"2025-07-14T22:02:27.503479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport torch\n\ndef set_seed(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)  # if using multi-GPU\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(42)  # 设置随机种子\n\n# 假设 train 是你的 DataFrame，选择除最后两列以外的列名\nx_features = random.sample(train.columns[:-2].tolist(), 100)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:02:29.909060Z","iopub.execute_input":"2025-07-14T22:02:29.909750Z","iopub.status.idle":"2025-07-14T22:02:29.922506Z","shell.execute_reply.started":"2025-07-14T22:02:29.909728Z","shell.execute_reply":"2025-07-14T22:02:29.921826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = [\n            \"X344\",\"X598\",\"X385\",\"X603\",\n        \"X674\",\"X415\",\"X345\",\"X137\",\"X174\",\"X302\",\n        \"X178\",\"X532\",\"X168\",\"X612\",\n    'bid_ask_interaction',\n    'bid_buy_interaction',\n    'bid_sell_interaction',\n    'ask_buy_interaction',\n    'ask_sell_interaction',\n    'volume_weighted_sell',\n    'buy_sell_ratio',\n    'selling_pressure',\n    'log_volume',\n    'effective_spread_proxy',\n    'bid_ask_imbalance',\n    'order_flow_imbalance',\n    'liquidity_ratio',\n    'net_order_flow',\n    'normalized_net_flow',\n    'buying_pressure',\n    'volume_weighted_buy',\n    'total_depth',\n    'depth_imbalance',\n    'relative_spread',\n    'log_depth',\n    'kyle_lambda',\n    'flow_toxicity',\n    'aggressive_flow_ratio',\n    'volume_depth_ratio',\n    'activity_intensity',\n    'log_buy_qty',\n    'log_sell_qty',\n    'log_bid_qty',\n    'log_ask_qty',\n    'realized_spread_proxy',\n    'price_impact_proxy',\n    'quote_volatility_proxy',\n    'flow_depth_interaction',\n    'imbalance_volume_interaction',\n    'depth_volume_interaction',\n    'buy_sell_spread',\n    'bid_ask_spread',\n    'trade_informativeness',\n    'execution_shortfall_proxy',\n    'adverse_selection_proxy',\n    'fill_probability',\n    'execution_rate',\n    'market_efficiency',\n    'sqrt_volume',\n    'sqrt_depth',\n    'volume_squared',\n    'imbalance_squared',\n    'bid_ratio',\n    'ask_ratio',\n    'buy_ratio',\n    'sell_ratio',\n    'liquidity_consumption',\n    'market_stress',\n    'depth_depletion',\n    'net_buying_ratio',\n    'directional_volume',\n    'signed_volume'\n]\n\n\n# feature_names = x_feaures\n# feature_names = feature_names + new_features\nlabel_name = 'label'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:02:31.822199Z","iopub.execute_input":"2025-07-14T22:02:31.822828Z","iopub.status.idle":"2025-07-14T22:02:31.832088Z","shell.execute_reply.started":"2025-07-14T22:02:31.822797Z","shell.execute_reply":"2025-07-14T22:02:31.831276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_fold_slices(n_samples):\n    # 按比例划分样本索引的切片\n    fold_slices = [\n        {\"name\": \"fold_1\", \"start\": 0, \"end\": int(0.167 * n_samples)},\n        {\"name\": \"fold_2\", \"start\": int(0.167 * n_samples), \"end\": int(0.333 * n_samples)},\n        {\"name\": \"fold_3\", \"start\": int(0.333 * n_samples), \"end\": int(0.5 * n_samples)},\n        {\"name\": \"fold_4\", \"start\": int(0.5 * n_samples), \"end\": int(0.667 * n_samples)},\n        {\"name\": \"fold_5\", \"start\": int(0.667 * n_samples), \"end\": int(0.833 * n_samples)},\n        {\"name\": \"fold_6\", \"start\": int(0.833 * n_samples), \"end\": n_samples},\n    ]\n    return fold_slices\n\n# 给DataFrame添加Fold列示例\nfolds = get_fold_slices(len(train))\ntrain['Fold'] = -1\nfor i, fs in enumerate(folds, start=1):\n    train.iloc[fs[\"start\"]:fs[\"end\"], train.columns.get_loc(\"Fold\")] = i\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:02:34.418542Z","iopub.execute_input":"2025-07-14T22:02:34.419101Z","iopub.status.idle":"2025-07-14T22:02:34.428168Z","shell.execute_reply.started":"2025-07-14T22:02:34.419079Z","shell.execute_reply":"2025-07-14T22:02:34.427635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef create_time_weights(n: int, decay: float = 0.95) -> np.ndarray:\n    \"\"\"\n    Create exponentially decaying weights based on time order.\n    More recent samples (later in sequence) have higher weights.\n    \n    Args:\n        n (int): Number of samples\n        decay (float): Decay factor between 0 and 1 (e.g., 0.95 means 5% decay per unit)\n\n    Returns:\n        np.ndarray: Array of weights summing to n\n    \"\"\"\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)  # ✅ 使用 decay 参数\n    return weights * n / weights.sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:03:09.189071Z","iopub.execute_input":"2025-07-14T22:03:09.189656Z","iopub.status.idle":"2025-07-14T22:03:09.194008Z","shell.execute_reply.started":"2025-07-14T22:03:09.189633Z","shell.execute_reply":"2025-07-14T22:03:09.193113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['weight'] = create_time_weights(len(train), decay=0.95)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:03:19.030344Z","iopub.execute_input":"2025-07-14T22:03:19.030654Z","iopub.status.idle":"2025-07-14T22:03:19.053108Z","shell.execute_reply.started":"2025-07-14T22:03:19.030632Z","shell.execute_reply":"2025-07-14T22:03:19.052490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pearsonr_coeff(preds, data):\n    y_true = data.get_label()\n    # weights = data.get_weight()\n    valid_score = _pearsonr(y_true, preds)\n    return 'pearsonr_coeff_score',valid_score,True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:03:21.882962Z","iopub.execute_input":"2025-07-14T22:03:21.883706Z","iopub.status.idle":"2025-07-14T22:03:21.887222Z","shell.execute_reply.started":"2025-07-14T22:03:21.883681Z","shell.execute_reply":"2025-07-14T22:03:21.886521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # 训练模型\ndef TrainModel(train_data, valid_data, lgb_params):\n    print(\"Training Model...\")\n    model = lgb.train(lgb_params,\n                        train_data,\n                        num_boost_round=150,\n                        valid_sets=[valid_data],\n                        feval=pearsonr_coeff,\n                        callbacks=[\n                        # lgb.callback.early_stopping(stopping_rounds=300),\n                        lgb.callback.log_evaluation(period=50)]\n                        )\n\n    valid_pred = model.predict(valid_data.get_data())\n    valid_score = _pearsonr(valid_data.get_label(),valid_pred)\n    print(\"Valid Score:\", valid_score)\n    return model,valid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:03:23.321913Z","iopub.execute_input":"2025-07-14T22:03:23.322697Z","iopub.status.idle":"2025-07-14T22:03:23.327295Z","shell.execute_reply.started":"2025-07-14T22:03:23.322674Z","shell.execute_reply":"2025-07-14T22:03:23.326649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = []\nvalid_scores = []\nfor fold in range(1,6):\n    X_train = train[(train['Fold']!=fold)][ feature_names ]\n    w_train = train[(train['Fold']!=fold)][ 'weight' ]\n    X_valid = train[train['Fold']==6][ feature_names ]\n    w_valid = train[train['Fold']==6][ 'weight' ]\n    y_train = train[(train['Fold']!=fold)][ label_name ]\n    y_valid = train[train['Fold']==6][ label_name ]\n\n\n    train_data = lgb.Dataset(X_train, label=y_train, weight=w_train,free_raw_data=False).construct()\n    valid_data = lgb.Dataset(X_valid, label=y_valid, weight=w_valid,reference=train_data, free_raw_data=False).construct()\n    print(f'train time {X_train.index.min()},{X_train.index.max()}')\n    print(f'valid time {X_valid.index.min()},{X_valid.index.max()}')\n\n    lgb_params = {\n            \"boosting_type\": \"gbdt\",\n            \"objective\": \"regression\",       # 回归任务\n            \"metric\": \"mae\",                 # 使用 MAE 作为评估指标\n            \"colsample_bytree\": 0.55,\n            \"learning_rate\": 0.021,\n            \"min_child_samples\": 32,\n            \"min_child_weight\": 0.15,\n            'max_depth':-1,\n            \"n_jobs\": -1,\n            \"num_leaves\":64,\n            \"random_state\": 42,\n            \"reg_alpha\": 80,\n            \"reg_lambda\": 100,\n            \"subsample\": 0.85,\n            \"verbosity\": 1,  \n            \"device\": \"gpu\",                 # 使用 GPU 加速\n            # \"max_bin\":1024\n            }\n\n    model,valid_score = TrainModel(train_data,valid_data,lgb_params)\n\n    models.append(model)\n    valid_scores.append(valid_score)\nprint(f'Average score is {np.mean(valid_scores)}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:03:26.950319Z","iopub.execute_input":"2025-07-14T22:03:26.950601Z","iopub.status.idle":"2025-07-14T22:04:50.357551Z","shell.execute_reply.started":"2025-07-14T22:03:26.950584Z","shell.execute_reply":"2025-07-14T22:04:50.356684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm = VotingModel(models)\nsubmission = pd.read_csv(path+r'sample_submission.csv')\nsubmission['prediction'] = lgbm.predict(test[feature_names])\nsubmission.to_csv(r'submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-14T22:06:49.705356Z","iopub.execute_input":"2025-07-14T22:06:49.705718Z","iopub.status.idle":"2025-07-14T22:07:14.751669Z","shell.execute_reply.started":"2025-07-14T22:06:49.705694Z","shell.execute_reply":"2025-07-14T22:07:14.750786Z"}},"outputs":[],"execution_count":null}]}