{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pickle\nimport polars as pl\nimport numpy as np\nimport pandas as pd\nimport gc\nimport warnings\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pytorch_lightning import (LightningDataModule, LightningModule, Trainer)\nfrom pytorch_lightning.callbacks import EarlyStopping, ModelCheckpoint, Timer\nfrom pytorch_lightning.loggers import WandbLogger\nimport lightgbm as lgb\nfrom pandas import read_parquet\n\nfrom datetime import datetime\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom torch.utils.data import Dataset, DataLoader\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom scipy.stats import pearsonr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:17:38.257488Z","iopub.execute_input":"2025-06-04T10:17:38.258090Z","iopub.status.idle":"2025-06-04T10:17:38.263509Z","shell.execute_reply.started":"2025-06-04T10:17:38.258066Z","shell.execute_reply":"2025-06-04T10:17:38.262654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_model(model, path):\n    with open(path, \"wb\") as f:\n        pickle.dump(model, f)\n\n\ndef load_model(path):\n    with open(path, \"rb\") as f:\n        model = pickle.load(f)\n    return model\n\ndef _pearsonr(y_true, y_pred):\n    return pearsonr(y_true, y_pred)[0]\n\nclass VotingModel(BaseEstimator, RegressorMixin):\n    \"\"\"\n    A voting ensemble model that averages predictions from multiple estimators.\n\n    Parameters:\n    - estimators: List of estimators to include in the voting ensemble\n\n    Methods:\n    - fit(X, y=None): No training is performed as it's just an aggregator.\n    - predict(X): Returns the average prediction from all included estimators.\n    - predict_proba(X): Returns the average class probabilities from all included estimators.\n    \"\"\"\n\n    def __init__(self, estimators):\n        \"\"\"\n        Initializes the VotingModel with a list of estimators.\n\n        Parameters:\n        - estimators: List of estimators to include in the voting ensemble\n        \"\"\"\n        super().__init__()\n        self.estimators = estimators\n\n    def fit(self, X, y=None):\n        \"\"\"Fits the voting model (no operation).\"\"\"\n        return self\n\n    def predict(self, X):\n        \"\"\"Returns the average prediction from all included estimators.\"\"\"\n        y_preds = [estimator.predict(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n\n    def predict_proba(self, X):\n        \"\"\"Returns the average class probabilities from all included estimators.\"\"\"\n        y_preds = [estimator.predict_proba(X) for estimator in self.estimators]\n        return np.mean(y_preds, axis=0)\n    \ndef reduce_mem_usage(df):\n    \"\"\"\n    Optimizes the memory usage of a DataFrame by downcasting numeric columns to smaller data types.\n\n    Parameters:\n    - df: DataFrame to be optimized\n\n    Returns:\n    - df: Optimized DataFrame\n    \"\"\"\n\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(\"Memory usage of dataframe is {:.2f} MB\".format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n        if str(col_type) == \"category\":\n            continue\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == \"int\":\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if (\n                    c_min > np.finfo(np.float16).min\n                    and c_max < np.finfo(np.float16).max\n                ):\n                    df[col] = df[col].astype(np.float16)\n                elif (\n                    c_min > np.finfo(np.float32).min\n                    and c_max < np.finfo(np.float32).max\n                ):\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            continue\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(\"Memory usage after optimization is: {:.2f} MB\".format(end_mem))\n    print(\"Decreased by {:.1f}%\".format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:17:38.264729Z","iopub.execute_input":"2025-06-04T10:17:38.264935Z","iopub.status.idle":"2025-06-04T10:17:38.289208Z","shell.execute_reply.started":"2025-06-04T10:17:38.264921Z","shell.execute_reply":"2025-06-04T10:17:38.288516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Create interaction features for df\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    df['buy_sell_interaction'] = df['buy_qty'] * df['sell_qty']\n\n    # Calculate spread indicators for df\n    df['spread_indicator'] = (df['ask_qty'] - df['bid_qty']) / (df['ask_qty'] + df['bid_qty'] + 1e-8)\n\n    # Volume-weighted features for df\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['volume_weighted_bid'] = df['bid_qty'] * df['volume']\n    df['volume_weighted_ask'] = df['ask_qty'] * df['volume']\n\n    # NEW FEATURES - Add ratio features\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n\n    # NEW FEATURES - Add order flow imbalance\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n\n    # NEW FEATURES - Add market pressure indicators\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n\n    # ADDITIONAL NEW MARKET FEATURES - Liquidity measures\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-8)\n    df['relative_spread'] = (df['ask_qty'] - df['bid_qty']) / (df['volume'] + 1e-8)\n\n    # ADDITIONAL NEW MARKET FEATURES - Trade intensity\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-8)\n    df['avg_trade_size'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    df['net_trade_flow'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n\n    # ADDITIONAL NEW MARKET FEATURES - Market depth and activity\n    df['depth_ratio'] = df['total_liquidity'] / (df['volume'] + 1e-8)\n    df['volume_participation'] = (df['buy_qty'] + df['sell_qty']) / (df['total_liquidity'] + 1e-8)\n    df['market_activity'] = df['volume'] * df['total_liquidity']\n\n    # ADDITIONAL NEW MARKET FEATURES - Execution quality indicators\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    df['realized_volatility_proxy'] = np.abs(df['order_flow_imbalance']) * df['volume']\n\n    # ADDITIONAL NEW MARKET FEATURES - Normalized volumes\n    df['normalized_buy_volume'] = df['buy_qty'] / (df['bid_qty'] + 1e-8)\n    df['normalized_sell_volume'] = df['sell_qty'] / (df['ask_qty'] + 1e-8)\n\n    # ADDITIONAL NEW MARKET FEATURES - Complex interactions\n    df['liquidity_adjusted_imbalance'] = df['order_flow_imbalance'] * df['depth_ratio']\n    df['pressure_spread_interaction'] = df['buying_pressure'] * df['spread_indicator']\n\n    # Replace any inf or -inf values with NaN, then fill NaN with 0\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    return df ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:17:38.290023Z","iopub.execute_input":"2025-06-04T10:17:38.290305Z","iopub.status.idle":"2025-06-04T10:17:38.309828Z","shell.execute_reply.started":"2025-06-04T10:17:38.290279Z","shell.execute_reply":"2025-06-04T10:17:38.309221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\npath = r'/kaggle/input/drw-crypto-market-prediction/'\ntrain = reduce_mem_usage(read_parquet(path+r'train.parquet'))\ntest = reduce_mem_usage(read_parquet(path+r'test.parquet'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:17:38.311440Z","iopub.execute_input":"2025-06-04T10:17:38.311745Z","iopub.status.idle":"2025-06-04T10:18:01.851318Z","shell.execute_reply.started":"2025-06-04T10:17:38.311694Z","shell.execute_reply":"2025-06-04T10:18:01.850488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntrain = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:01.852361Z","iopub.execute_input":"2025-06-04T10:18:01.852699Z","iopub.status.idle":"2025-06-04T10:18:18.778358Z","shell.execute_reply.started":"2025-06-04T10:18:01.852674Z","shell.execute_reply":"2025-06-04T10:18:18.777577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_feaures = [f\"X{num}\" for num in range(1,890)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.780256Z","iopub.execute_input":"2025-06-04T10:18:18.780474Z","iopub.status.idle":"2025-06-04T10:18:18.784604Z","shell.execute_reply.started":"2025-06-04T10:18:18.780458Z","shell.execute_reply":"2025-06-04T10:18:18.783807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nrandom.seed(42)\nx_feaures = random.sample(train.columns[:-2].tolist(),100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.785290Z","iopub.execute_input":"2025-06-04T10:18:18.785504Z","iopub.status.idle":"2025-06-04T10:18:18.801980Z","shell.execute_reply.started":"2025-06-04T10:18:18.785489Z","shell.execute_reply":"2025-06-04T10:18:18.801319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_names = [\n                 'X863',\n                 'X856',\n                 'X344',\n                 'X598',\n                 'X862',\n                 'X385',\n                 'X852',\n                 'X603',\n                 'X860',\n                 'X674',\n                 'X415',\n                 'X345',\n                 'X137',\n                 'X855',\n                 'X174',\n                 'X302',\n                 'X178',\n                 'X532',\n                 'X168',\n                 'X612',\n                 'bid_qty',\n                 'ask_qty',\n                 'buy_qty',\n                 'sell_qty',\n                 'volume',\n                 'bid_ask_interaction',\n                 'bid_buy_interaction',\n                 'bid_sell_interaction',\n                 'ask_buy_interaction',\n                 'ask_sell_interaction'\n                        ]\n# feature_names = x_feaures\n# feature_names = feature_names + new_features\nlabel_name = 'label'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.803308Z","iopub.execute_input":"2025-06-04T10:18:18.803581Z","iopub.status.idle":"2025-06-04T10:18:18.816190Z","shell.execute_reply.started":"2025-06-04T10:18:18.803564Z","shell.execute_reply":"2025-06-04T10:18:18.815586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.loc['2023-03-01 00:00:00':'2023-05-01 00:00:00','Fold'] =  1\ntrain.loc['2023-05-01 00:00:00':'2023-07-01 00:00:00','Fold'] =  2\ntrain.loc['2023-07-01 00:00:00':'2023-09-01 00:00:00','Fold'] =  3\ntrain.loc['2023-09-01 00:00:00':'2023-11-01 00:00:00','Fold'] =  4\ntrain.loc['2023-11-01 00:00:00':'2024-01-01 00:00:00','Fold'] =  5\ntrain.loc['2024-01-01 00:00:00':'2024-03-01 00:00:00','Fold'] =  6","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.816936Z","iopub.execute_input":"2025-06-04T10:18:18.817240Z","iopub.status.idle":"2025-06-04T10:18:18.887306Z","shell.execute_reply.started":"2025-06-04T10:18:18.817220Z","shell.execute_reply":"2025-06-04T10:18:18.886559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.888086Z","iopub.execute_input":"2025-06-04T10:18:18.888312Z","iopub.status.idle":"2025-06-04T10:18:18.892519Z","shell.execute_reply.started":"2025-06-04T10:18:18.888287Z","shell.execute_reply":"2025-06-04T10:18:18.891799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['weight'] = create_time_weights(len(train), decay_factor=0.95)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.894782Z","iopub.execute_input":"2025-06-04T10:18:18.895306Z","iopub.status.idle":"2025-06-04T10:18:18.920665Z","shell.execute_reply.started":"2025-06-04T10:18:18.895290Z","shell.execute_reply":"2025-06-04T10:18:18.919852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pearsonr_coeff(preds, data):\n    y_true = data.get_label()\n    # weights = data.get_weight()\n    valid_score = _pearsonr(y_true, preds)\n    return 'pearsonr_coeff_score',valid_score,True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.921475Z","iopub.execute_input":"2025-06-04T10:18:18.921757Z","iopub.status.idle":"2025-06-04T10:18:18.926510Z","shell.execute_reply.started":"2025-06-04T10:18:18.921734Z","shell.execute_reply":"2025-06-04T10:18:18.925731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" # 训练模型\ndef TrainModel(train_data, valid_data, lgb_params):\n    print(\"Training Model...\")\n    model = lgb.train(lgb_params,\n                        train_data,\n                        num_boost_round=150,\n                        valid_sets=[valid_data],\n                        feval=pearsonr_coeff,\n                        callbacks=[\n                        # lgb.callback.early_stopping(stopping_rounds=300),\n                        lgb.callback.log_evaluation(period=50)]\n                        )\n\n    valid_pred = model.predict(valid_data.get_data())\n    valid_score = _pearsonr(valid_data.get_label(),valid_pred)\n    print(\"Valid Score:\", valid_score)\n    return model,valid_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.927300Z","iopub.execute_input":"2025-06-04T10:18:18.927603Z","iopub.status.idle":"2025-06-04T10:18:18.941021Z","shell.execute_reply.started":"2025-06-04T10:18:18.927578Z","shell.execute_reply":"2025-06-04T10:18:18.940312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nmodels = []\nvalid_scores = []\nfor fold in range(1,6):\n    X_train = train[(train['Fold']!=fold)][ feature_names ]\n    w_train = train[(train['Fold']!=fold)][ 'weight' ]\n    X_valid = train[train['Fold']==6][ feature_names ]\n    w_valid = train[train['Fold']==6][ 'weight' ]\n    y_train = train[(train['Fold']!=fold)][ label_name ]\n    y_valid = train[train['Fold']==6][ label_name ]\n\n\n    train_data = lgb.Dataset(X_train, label=y_train, weight=w_train,free_raw_data=False).construct()\n    valid_data = lgb.Dataset(X_valid, label=y_valid, weight=w_valid,reference=train_data, free_raw_data=False).construct()\n    print(f'train time {X_train.index.min()},{X_train.index.max()}')\n    print(f'valid time {X_valid.index.min()},{X_valid.index.max()}')\n\n    lgb_params = {\n            \"boosting_type\": \"gbdt\",\n            \"objective\": \"regression\",       # 回归任务\n            \"metric\": \"mae\",                 # 使用 MAE 作为评估指标\n            \"colsample_bytree\": 0.55,\n            \"learning_rate\": 0.021,\n            \"min_child_samples\": 32,\n            \"min_child_weight\": 0.15,\n            'max_depth':-1,\n            \"n_jobs\": -1,\n            \"num_leaves\":64,\n            \"random_state\": 42,\n            \"reg_alpha\": 80,\n            \"reg_lambda\": 100,\n            \"subsample\": 0.85,\n            \"verbosity\": 1,  \n            \"device\": \"gpu\",                 # 使用 GPU 加速\n            # \"max_bin\":1024\n            }\n\n    model,valid_score = TrainModel(train_data,valid_data,lgb_params)\n\n    models.append(model)\n    valid_scores.append(valid_score)\nprint(f'Average score is {np.mean(valid_scores)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T10:18:18.941857Z","iopub.execute_input":"2025-06-04T10:18:18.942474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm = VotingModel(models)\nsubmission = pd.read_csv(path+r'sample_submission.csv')\nsubmission['prediction'] = lgbm.predict(test[feature_names])\nsubmission.to_csv(r'submission.csv',index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}