{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DeepGBM Implementation for Crypto Market Prediction\n# Fixed version addressing all dimension and compatibility issues\n\nimport random\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, QuantileTransformer\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.decomposition import PCA\nfrom tqdm import tqdm\nimport gc\nfrom typing import List, Tuple, Dict\nimport math\nimport pickle\nimport os\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau, CosineAnnealingWarmRestarts, OneCycleLR\nimport torch.nn.functional as F\n\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom sklearn.tree import DecisionTreeRegressor\n\nfrom scipy.stats import pearsonr, spearmanr\nfrom scipy.special import erfinv\nfrom scipy.stats import kurtosis, skew\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Feature Configuration\n# =========================\nMARKET_FEATURES = [\"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"]\n\nPROPRIETARY_FEATURES = [\n    \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n    \"X425\", \"X132\", \"X691\", \"X593\", \"X377\", \"X285\", \"X126\", \"X419\", \"X604\",\n    \"X84\", \"X138\", \"X413\", \"X291\", \"X40\", \"X123\", \"X81\", \"X853\", \"X854\",\n    \"X777\", \"X219\", \"X776\", \"X180\", \"X781\", \"X445\", \"X444\", \"X384\", \"X466\",\n    \"X95\", \"X583\", \"X272\", \"X137\", \"X533\", \"X758\", \"X279\", \"X297\",\n    \"X21\", \"X20\", \"X28\", \"X29\", \"X19\", \"X27\", \"X22\", \"X198\", \"X89\", \"X90\",\n    \"X98\", \"X96\", \"X97\", \"X383\", \"X427\", \"X451\", \"X283\",\n    \"X753\", \"X497\", \"X748\", \"X820\", \"X566\", \"X535\", \"X394\", \"X618\",\n    \"X429\", \"X381\", \"X387\", \"X890\", \"X752\", \"X375\", \"X68\", \"X152\",\n    \"X110\", \"X850\", \"X851\", \"X481\", \"X321\", \"X363\", \"X405\", \"X492\",\n    \"X888\", \"X421\", \"X333\", \"X817\", \"X586\", \"X292\", \"X344\", \"X532\"\n]\n\nSELECTED_FEATURES = MARKET_FEATURES + PROPRIETARY_FEATURES + [\"label\"]\n\n# =========================\n# Feature Engineering\n# =========================\ndef add_advanced_market_features(df):\n    \"\"\"Add comprehensive market microstructure features for crypto markets\"\"\"\n    \n    # Core microstructure features\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Advanced volume analysis\n    df['log_volume'] = np.log1p(df['volume'])\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['volume_intensity'] = df['volume'] / (df['volume'].rolling(window=10, min_periods=1).mean() + 1e-10)\n    \n    # Liquidity metrics\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['depth_exhaustion'] = df['volume'] / (df['total_depth'] + 1e-10)\n    \n    # Price pressure and toxicity indicators\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['amihud_illiquidity'] = np.abs(df['order_flow_imbalance']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * np.log1p(df['volume'])\n    \n    # Market stress and volatility proxies\n    df['volume_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['depth_volume_interaction'] = df['total_depth'] * np.log1p(df['volume'])\n    df['flow_intensity'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['volatility_proxy'] = np.abs(df['net_order_flow']) / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Order book shape metrics\n    df['book_pressure'] = (df['bid_qty'] - df['ask_qty']) / (df['volume'] + 1e-10)\n    df['book_skew'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10) * np.sign(df['net_order_flow'])\n    df['depth_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-10)\n    df['inverse_depth_ratio'] = df['ask_qty'] / (df['bid_qty'] + 1e-10)\n    \n    # Execution quality metrics\n    df['execution_probability'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['fill_rate'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['trade_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Information asymmetry measures\n    df['pin_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['information_share'] = df['net_order_flow'] / (df['volume'] + 1e-10) * np.abs(df['depth_imbalance'])\n    df['adverse_selection'] = np.abs(df['net_order_flow']) / (df['bid_qty'] + df['ask_qty'] + 1e-10) * df['volume']\n    \n    # Complex interaction features\n    df['volume_adjusted_imbalance'] = df['order_flow_imbalance'] * np.log1p(df['volume'])\n    df['depth_adjusted_flow'] = df['net_order_flow'] / (np.log1p(df['total_depth']) + 1)\n    df['liquidity_weighted_pressure'] = df['market_stress'] * df['liquidity_ratio']\n    df['toxic_volume_ratio'] = df['flow_toxicity'] / (df['volume'] + 1e-10)\n    \n    # Non-linear transformations\n    df['volume_squared'] = df['volume'] ** 2\n    df['volume_cubed'] = df['volume'] ** 3\n    df['log_total_activity'] = np.log1p(df['buy_qty'] + df['sell_qty'] + df['volume'])\n    df['sqrt_market_activity'] = np.sqrt(df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'])\n    \n    # Ratios and normalized metrics\n    df['buy_total_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['sell_total_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['bid_total_ratio'] = df['bid_qty'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['ask_total_ratio'] = df['ask_qty'] / (df['buy_qty'] + df['sell_qty'] + df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Market efficiency indicators\n    df['price_discovery'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * np.sign(df['order_flow_imbalance'])\n    df['market_quality'] = df['total_depth'] / (df['volume'] + 1e-10) * (1 - np.abs(df['order_flow_imbalance']))\n    df['liquidity_provision'] = df['total_depth'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n\ndef compute_rolling_features(df, windows=[5, 10, 20]):\n    \"\"\"Compute rolling statistics for key features\"\"\"\n    key_features = ['volume', 'net_order_flow', 'total_depth', 'order_flow_imbalance']\n    \n    for feature in key_features:\n        if feature in df.columns:\n            for window in windows:\n                df[f'{feature}_ma_{window}'] = df[feature].rolling(window=window, min_periods=1).mean()\n                df[f'{feature}_std_{window}'] = df[feature].rolling(window=window, min_periods=1).std().fillna(0)\n                df[f'{feature}_skew_{window}'] = df[feature].rolling(window=window, min_periods=3).skew().fillna(0)\n                df[f'{feature}_rel_ma_{window}'] = df[feature] / (df[f'{feature}_ma_{window}'] + 1e-10)\n    \n    return df\n\n# =========================\n# Data Transformation Functions\n# =========================\ndef apply_advanced_transformations(X, method='quantile'):\n    \"\"\"Apply advanced transformations to features\"\"\"\n    if method == 'quantile':\n        transformer = QuantileTransformer(output_distribution='normal', random_state=42)\n    elif method == 'robust':\n        transformer = RobustScaler()\n    elif method == 'standard':\n        transformer = StandardScaler()\n    else:\n        raise ValueError(f\"Unknown transformation method: {method}\")\n    \n    return transformer.fit_transform(X), transformer\n\ndef compute_comprehensive_feature_importance(X, y, feature_names, n_samples=50000, n_permutations=10):\n    \"\"\"Compute feature importance using multiple methods with large sample size\"\"\"\n    print(f\"Computing feature importance with {n_samples} samples...\")\n    \n    if len(X) <= n_samples:\n        X_subset = X\n        y_subset = y\n    else:\n        indices = np.random.choice(len(X), n_samples, replace=False)\n        X_subset = X[indices]\n        y_subset = y.iloc[indices] if hasattr(y, 'iloc') else y[indices]\n    \n    mi_scores = mutual_info_regression(X_subset, y_subset, random_state=42, n_neighbors=5)\n    \n    baseline_corr = pearsonr(y_subset, np.mean(X_subset, axis=1))[0]\n    \n    perm_importance = np.zeros(X_subset.shape[1])\n    for i in range(X_subset.shape[1]):\n        perm_scores = []\n        for _ in range(n_permutations):\n            X_perm = X_subset.copy()\n            np.random.shuffle(X_perm[:, i])\n            perm_corr = pearsonr(y_subset, np.mean(X_perm, axis=1))[0]\n            perm_scores.append(baseline_corr - perm_corr)\n        perm_importance[i] = np.mean(perm_scores)\n    \n    combined_scores = (mi_scores / mi_scores.max() + \n                      perm_importance / (perm_importance.max() + 1e-10)) / 2\n    \n    importance_df = pd.DataFrame({\n        'feature': feature_names,\n        'mi_score': mi_scores,\n        'perm_importance': perm_importance,\n        'combined_score': combined_scores\n    }).sort_values('combined_score', ascending=False)\n    \n    return importance_df\n\n# =========================\n# DeepGBM Architecture Components\n# =========================\ndef set_seed(seed=42):\n    \"\"\"Set random seeds for reproducibility\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nclass FeatureTransformer(nn.Module):\n    \"\"\"Feature transformation layer for DeepGBM\"\"\"\n    def __init__(self, input_dim, output_dim, dropout=0.1):\n        super().__init__()\n        self.linear = nn.Linear(input_dim, output_dim)\n        self.norm = nn.LayerNorm(output_dim)\n        self.activation = nn.GELU()\n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x):\n        return self.dropout(self.activation(self.norm(self.linear(x))))\n\nclass CatNN(nn.Module):\n    \"\"\"Categorical-aware Neural Network component of DeepGBM\"\"\"\n    def __init__(self, config):\n        super().__init__()\n        \n        self.input_dim = config['input_dim']\n        self.embedding_dim = config['embedding_dim']\n        self.hidden_dims = config['hidden_dims']\n        \n        # Feature embedding layers\n        self.feature_embedder = nn.Sequential(\n            FeatureTransformer(self.input_dim, self.embedding_dim, config['dropout']),\n            FeatureTransformer(self.embedding_dim, self.embedding_dim, config['dropout'])\n        )\n        \n        # Deep layers\n        layers = []\n        prev_dim = self.embedding_dim\n        \n        for hidden_dim in self.hidden_dims:\n            layers.append(FeatureTransformer(prev_dim, hidden_dim, config['dropout']))\n            prev_dim = hidden_dim\n        \n        self.deep_layers = nn.Sequential(*layers)\n        \n        # Store the final hidden dimension\n        self.final_hidden_dim = prev_dim\n        \n        # Output projection - outputs 1 dimension for regression\n        self.output_projection = nn.Linear(prev_dim, 1)\n        \n        # Attention mechanism for feature interactions\n        self.feature_attention = nn.MultiheadAttention(\n            embed_dim=self.embedding_dim,\n            num_heads=config['num_attention_heads'],\n            dropout=config['attention_dropout'],\n            batch_first=True\n        )\n        \n        # Gating mechanism\n        self.gate = nn.Sequential(\n            nn.Linear(self.embedding_dim, self.embedding_dim // 2),\n            nn.ReLU(),\n            nn.Linear(self.embedding_dim // 2, self.embedding_dim),\n            nn.Sigmoid()\n        )\n        \n        self._initialize_weights()\n        \n    def _initialize_weights(self):\n        for module in self.modules():\n            if isinstance(module, nn.Linear):\n                nn.init.xavier_uniform_(module.weight)\n                if module.bias is not None:\n                    nn.init.constant_(module.bias, 0)\n                    \n    def forward(self, x):\n        # Initial feature embedding\n        embedded = self.feature_embedder(x)\n        \n        # Self-attention for feature interactions\n        attended, _ = self.feature_attention(\n            embedded.unsqueeze(1), \n            embedded.unsqueeze(1), \n            embedded.unsqueeze(1)\n        )\n        attended = attended.squeeze(1)\n        \n        # Gating mechanism\n        gate_values = self.gate(embedded)\n        features = gate_values * attended + (1 - gate_values) * embedded\n        \n        # Deep transformation\n        deep_features = self.deep_layers(features)\n        \n        # Output\n        output = self.output_projection(deep_features)\n        \n        return output.squeeze(-1), deep_features\n\nclass DeepGBM(nn.Module):\n    \"\"\"Deep Gradient Boosting Machine combining NN and GBDT\"\"\"\n    def __init__(self, config):\n        super().__init__()\n        \n        self.config = config\n        self.use_gbdt_features = config.get('use_gbdt_features', True)\n        \n        # CatNN component\n        catnn_config = {\n            'input_dim': config['input_dim'],\n            'embedding_dim': config['catnn_embedding_dim'],\n            'hidden_dims': config['catnn_hidden_dims'],\n            'dropout': config['dropout'],\n            'num_attention_heads': config.get('num_attention_heads', 4),\n            'attention_dropout': config.get('attention_dropout', 0.1)\n        }\n        self.catnn = CatNN(catnn_config)\n        \n        # Calculate fusion input dimension correctly\n        catnn_final_dim = config['catnn_hidden_dims'][-1] if config['catnn_hidden_dims'] else config['catnn_embedding_dim']\n        \n        if self.use_gbdt_features:\n            fusion_input_dim = catnn_final_dim + config.get('gbdt_output_dim', 100)\n        else:\n            fusion_input_dim = catnn_final_dim\n            \n        self.fusion_layers = nn.Sequential(\n            nn.Linear(fusion_input_dim, config['fusion_hidden_dim']),\n            nn.LayerNorm(config['fusion_hidden_dim']),\n            nn.GELU(),\n            nn.Dropout(config['fusion_dropout']),\n            nn.Linear(config['fusion_hidden_dim'], config['fusion_hidden_dim'] // 2),\n            nn.LayerNorm(config['fusion_hidden_dim'] // 2),\n            nn.GELU(),\n            nn.Dropout(config['fusion_dropout']),\n            nn.Linear(config['fusion_hidden_dim'] // 2, 1)\n        )\n        \n        # Residual connection\n        self.residual_weight = nn.Parameter(torch.tensor(0.1))\n        \n        self._initialize_weights()\n        \n    def _initialize_weights(self):\n        for module in self.fusion_layers.modules():\n            if isinstance(module, nn.Linear):\n                nn.init.xavier_uniform_(module.weight)\n                if module.bias is not None:\n                    nn.init.constant_(module.bias, 0)\n                    \n    def forward(self, x, gbdt_features=None):\n        # Get CatNN outputs\n        catnn_output, catnn_features = self.catnn(x)\n        \n        # Combine with GBDT features if available\n        if self.use_gbdt_features and gbdt_features is not None:\n            combined_features = torch.cat([catnn_features, gbdt_features], dim=-1)\n        else:\n            combined_features = catnn_features\n        \n        # Fusion\n        fusion_output = self.fusion_layers(combined_features).squeeze(-1)\n        \n        # Residual connection\n        final_output = fusion_output + self.residual_weight * catnn_output\n        \n        return final_output\n\nclass GBDTFeatureExtractor:\n    \"\"\"Extract features from GBDT models for DeepGBM\"\"\"\n    def __init__(self, config):\n        self.config = config\n        self.models = []\n        self.feature_dim = config.get('gbdt_output_dim', 100)\n        \n    def train_gbdt_models(self, X_train, y_train, X_val, y_val):\n        \"\"\"Train multiple GBDT models and extract features\"\"\"\n        \n        # XGBoost model\n        xgb_params = {\n            'objective': 'reg:squarederror',\n            'eval_metric': 'rmse',\n            'max_depth': self.config.get('xgb_max_depth', 6),\n            'learning_rate': self.config.get('xgb_learning_rate', 0.05),\n            'n_estimators': self.config.get('xgb_n_estimators', 200),\n            'subsample': self.config.get('xgb_subsample', 0.8),\n            'colsample_bytree': self.config.get('xgb_colsample', 0.8),\n            'reg_alpha': self.config.get('xgb_reg_alpha', 0.1),\n            'reg_lambda': self.config.get('xgb_reg_lambda', 0.1),\n            'random_state': 42,\n            'n_jobs': -1\n        }\n        \n        print(\"Training XGBoost model...\")\n        xgb_model = xgb.XGBRegressor(**xgb_params)\n        xgb_model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            early_stopping_rounds=50,\n            verbose=False\n        )\n        self.models.append(('xgboost', xgb_model))\n        \n        # LightGBM model\n        lgb_params = {\n            'objective': 'regression',\n            'metric': 'rmse',\n            'max_depth': self.config.get('lgb_max_depth', 6),\n            'learning_rate': self.config.get('lgb_learning_rate', 0.05),\n            'n_estimators': self.config.get('lgb_n_estimators', 200),\n            'num_leaves': self.config.get('lgb_num_leaves', 31),\n            'subsample': self.config.get('lgb_subsample', 0.8),\n            'colsample_bytree': self.config.get('lgb_colsample', 0.8),\n            'reg_alpha': self.config.get('lgb_reg_alpha', 0.1),\n            'reg_lambda': self.config.get('lgb_reg_lambda', 0.1),\n            'random_state': 42,\n            'n_jobs': -1,\n            'verbose': -1\n        }\n        \n        print(\"Training LightGBM model...\")\n        lgb_model = lgb.LGBMRegressor(**lgb_params)\n        lgb_model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            callbacks=[lgb.early_stopping(50), lgb.log_evaluation(0)]\n        )\n        self.models.append(('lightgbm', lgb_model))\n        \n    def extract_features(self, X):\n        \"\"\"Extract features from trained GBDT models\"\"\"\n        features = []\n        target_dim = self.feature_dim\n        \n        for name, model in self.models:\n            if name == 'xgboost':\n                # Get predictions from different tree iterations\n                n_trees = min(50, model.n_estimators)\n                tree_preds = []\n                for i in range(0, n_trees, 2):\n                    pred = model.predict(X, iteration_range=(i, i+1))\n                    tree_preds.append(pred)\n                features.append(np.column_stack(tree_preds))\n                \n            elif name == 'lightgbm':\n                # Get predictions from individual trees for LightGBM\n                n_trees_to_use = min(50, model.n_estimators_)\n                tree_preds = []\n                \n                # Get predictions at different iterations\n                for i in range(0, n_trees_to_use, 2):\n                    pred = model.predict(X, num_iteration=i+1)\n                    if i > 0:\n                        pred_prev = model.predict(X, num_iteration=i)\n                        pred = pred - pred_prev\n                    tree_preds.append(pred)\n                \n                tree_features = np.column_stack(tree_preds)\n                features.append(tree_features)\n        \n        # Concatenate all features\n        if features:\n            all_features = np.hstack(features)\n            \n            # Ensure we have the right dimension\n            if all_features.shape[1] > target_dim:\n                all_features = all_features[:, :target_dim]\n            elif all_features.shape[1] < target_dim:\n                # Pad with zeros if needed\n                padding = np.zeros((all_features.shape[0], target_dim - all_features.shape[1]))\n                all_features = np.hstack([all_features, padding])\n            \n            return all_features.astype(np.float32)\n        else:\n            # Return zeros if feature extraction failed\n            return np.zeros((X.shape[0], target_dim), dtype=np.float32)\n\n# =========================\n# Training Functions for DeepGBM\n# =========================\ndef train_deepgbm(model, gbdt_extractor, train_loader, val_loader, X_train_raw, y_train_raw, \n                  X_val_raw, y_val_raw, config, device):\n    \"\"\"Train DeepGBM with joint optimization\"\"\"\n    \n    # First, train GBDT models\n    print(\"\\nTraining GBDT components...\")\n    gbdt_extractor.train_gbdt_models(X_train_raw, y_train_raw, X_val_raw, y_val_raw)\n    \n    # Extract GBDT features\n    print(\"Extracting GBDT features...\")\n    train_gbdt_features = gbdt_extractor.extract_features(X_train_raw)\n    val_gbdt_features = gbdt_extractor.extract_features(X_val_raw)\n    \n    print(f\"GBDT feature shape: {train_gbdt_features.shape}\")\n    \n    # Convert to tensors\n    train_gbdt_tensor = torch.tensor(train_gbdt_features, dtype=torch.float32).to(device)\n    val_gbdt_tensor = torch.tensor(val_gbdt_features, dtype=torch.float32).to(device)\n    \n    # Loss and optimizer\n    criterion = nn.HuberLoss(delta=config['huber_delta'])\n    \n    # Optimizer with different learning rates\n    param_groups = [\n        {'params': model.catnn.parameters(), 'lr': config['learning_rate']},\n        {'params': model.fusion_layers.parameters(), 'lr': config['learning_rate'] * 0.5},\n        {'params': [model.residual_weight], 'lr': config['learning_rate'] * 0.01}\n    ]\n    \n    optimizer = optim.AdamW(param_groups, weight_decay=config['weight_decay'], eps=1e-8)\n    \n    # Learning rate scheduler\n    scheduler = ReduceLROnPlateau(\n        optimizer, \n        mode='max', \n        factor=0.5, \n        patience=3, \n        verbose=True\n    )\n    \n    # Training history\n    history = {'train_loss': [], 'val_loss': [], 'val_pearson': []}\n    best_pearson = -np.inf\n    patience_counter = 0\n    \n    for epoch in range(config['num_epochs']):\n        # Training phase\n        model.train()\n        train_loss = 0.0\n        train_batches = 0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{config['num_epochs']}\")\n        \n        for batch_idx, (inputs, targets) in enumerate(progress_bar):\n            inputs, targets = inputs.to(device), targets.to(device)\n            targets = targets.squeeze()  # Ensure targets are 1D\n            \n            # Get corresponding GBDT features\n            start_idx = batch_idx * train_loader.batch_size\n            end_idx = min(start_idx + train_loader.batch_size, len(train_gbdt_tensor))\n            gbdt_batch = train_gbdt_tensor[start_idx:end_idx]\n            \n            # Ensure batch sizes match\n            if len(inputs) != len(gbdt_batch):\n                min_len = min(len(inputs), len(gbdt_batch))\n                inputs = inputs[:min_len]\n                targets = targets[:min_len]\n                gbdt_batch = gbdt_batch[:min_len]\n            \n            # Apply augmentation to neural network inputs\n            if np.random.random() < 0.8:\n                noise_factor = config.get('noise_factor', 0.01)\n                noise = torch.randn_like(inputs) * noise_factor\n                inputs = inputs + noise\n            \n            optimizer.zero_grad()\n            \n            # Forward pass\n            outputs = model(inputs, gbdt_batch)\n            loss = criterion(outputs, targets)\n            \n            # Add L2 regularization on outputs\n            if config.get('output_penalty', 0) > 0:\n                output_penalty = config['output_penalty'] * torch.mean(outputs ** 2)\n                loss = loss + output_penalty\n            \n            # Backward pass\n            loss.backward()\n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=config['grad_clip'])\n            optimizer.step()\n            \n            train_loss += loss.item()\n            train_batches += 1\n            \n            progress_bar.set_postfix({'loss': f'{loss.item():.4f}'})\n        \n        avg_train_loss = train_loss / train_batches\n        history['train_loss'].append(avg_train_loss)\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        val_batches = 0\n        \n        with torch.no_grad():\n            for batch_idx, (inputs, targets) in enumerate(val_loader):\n                inputs, targets = inputs.to(device), targets.to(device)\n                targets = targets.squeeze()  # Ensure targets are 1D\n                \n                # Get corresponding GBDT features\n                start_idx = batch_idx * val_loader.batch_size\n                end_idx = min(start_idx + val_loader.batch_size, len(val_gbdt_tensor))\n                gbdt_batch = val_gbdt_tensor[start_idx:end_idx]\n                \n                # Ensure batch sizes match\n                if len(inputs) != len(gbdt_batch):\n                    min_len = min(len(inputs), len(gbdt_batch))\n                    inputs = inputs[:min_len]\n                    targets = targets[:min_len]\n                    gbdt_batch = gbdt_batch[:min_len]\n                \n                outputs = model(inputs, gbdt_batch)\n                loss = criterion(outputs, targets)\n                \n                val_loss += loss.item()\n                val_preds.extend(outputs.cpu().numpy())\n                val_targets.extend(targets.cpu().numpy())\n                val_batches += 1\n        \n        avg_val_loss = val_loss / val_batches\n        \n        # Calculate correlations\n        if len(val_preds) > 0 and len(val_targets) > 0:\n            val_pearson = pearsonr(val_targets, val_preds)[0]\n            val_spearman = spearmanr(val_targets, val_preds)[0]\n        else:\n            val_pearson = 0.0\n            val_spearman = 0.0\n        \n        history['val_loss'].append(avg_val_loss)\n        history['val_pearson'].append(val_pearson)\n        \n        # Update scheduler\n        scheduler.step(val_pearson)\n        \n        print(f\"\\nEpoch {epoch+1}: Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n        print(f\"Val Pearson: {val_pearson:.4f}, Val Spearman: {val_spearman:.4f}\")\n        print(f\"LR: {optimizer.param_groups[0]['lr']:.6f}\")\n        \n        # Save best model\n        if val_pearson > best_pearson:\n            best_pearson = val_pearson\n            torch.save({\n                'model_state_dict': model.state_dict(),\n                'gbdt_models': gbdt_extractor.models,\n                'config': config\n            }, f\"best_deepgbm_{config['model_id']}.pt\")\n            print(f\"✅ New best model saved with Pearson: {val_pearson:.4f}\")\n            patience_counter = 0\n        else:\n            patience_counter += 1\n            if patience_counter >= config['patience']:\n                print(f\"Early stopping triggered after {epoch+1} epochs\")\n                break\n    \n    # Load best model\n    checkpoint = torch.load(f\"best_deepgbm_{config['model_id']}.pt\", weights_only=False)\n    model.load_state_dict(checkpoint['model_state_dict'])\n    gbdt_extractor.models = checkpoint['gbdt_models']\n    \n    return model, gbdt_extractor, history, best_pearson\n\n# =========================\n# Main Execution with DeepGBM\n# =========================\ndef main():\n    print(\"=== DeepGBM for Crypto Market Prediction ===\")\n    \n    # Load training data\n    print(\"\\nLoading training data...\")\n    train = pl.scan_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\n    train = train.select(SELECTED_FEATURES).collect().to_pandas()\n    \n    print(f\"Initial training data shape: {train.shape}\")\n    \n    # Use recent data for better relevance\n    train_size = int(0.85 * len(train))\n    train = train.iloc[-train_size:].reset_index(drop=True)\n    print(f\"Using last {train_size} samples for training\")\n    \n    # Add market microstructure features\n    print(\"\\nAdding advanced market microstructure features...\")\n    train = add_advanced_market_features(train)\n    train = compute_rolling_features(train)\n    \n    # Get all feature names\n    all_features = [col for col in train.columns if col != 'label']\n    print(f\"Total number of features after engineering: {len(all_features)}\")\n    \n    # Split data\n    X_train, X_val = train_test_split(train, test_size=0.2, shuffle=False, random_state=42)\n    \n    y_train = X_train.pop(\"label\")\n    y_val = X_val.pop(\"label\")\n    \n    # Feature importance analysis\n    print(\"\\nComputing comprehensive feature importance...\")\n    importance_df = compute_comprehensive_feature_importance(\n        X_train[all_features].values,\n        y_train,\n        all_features,\n        n_samples=min(100000, len(X_train))\n    )\n    \n    # Select features with importance threshold\n    feature_threshold = 0.01\n    selected_features = importance_df[importance_df['combined_score'] > feature_threshold]['feature'].tolist()\n    print(f\"Features with importance > {feature_threshold}: {len(selected_features)}\")\n    \n    # Prepare data with selected features\n    X_train_selected = X_train[selected_features].values\n    X_val_selected = X_val[selected_features].values\n    \n    # Store raw data for GBDT\n    X_train_raw = X_train_selected.copy()\n    X_val_raw = X_val_selected.copy()\n    y_train_raw = y_train.values\n    y_val_raw = y_val.values\n    \n    # Apply transformations for neural network\n    print(\"\\nApplying data transformations...\")\n    X_train_transformed, transformer = apply_advanced_transformations(X_train_selected, method='quantile')\n    X_val_transformed = transformer.transform(X_val_selected)\n    \n    print(f\"Final training shape: {X_train_transformed.shape}\")\n    print(f\"Final validation shape: {X_val_transformed.shape}\")\n    \n    # DeepGBM configuration\n    config = {\n        'model_id': 'deepgbm_v1',\n        'input_dim': X_train_transformed.shape[1],\n        'catnn_embedding_dim': 256,\n        'catnn_hidden_dims': [512, 256, 128],\n        'gbdt_output_dim': 100,\n        'fusion_hidden_dim': 256,\n        'fusion_dropout': 0.3,\n        'dropout': 0.2,\n        'num_attention_heads': 4,\n        'attention_dropout': 0.1,\n        'use_gbdt_features': True,\n        'num_epochs': 50,\n        'batch_size': 512,\n        'learning_rate': 0.001,\n        'weight_decay': 0.01,\n        'huber_delta': 1.0,\n        'noise_factor': 0.01,\n        'grad_clip': 1.0,\n        'patience': 10,\n        'output_penalty': 0.001,\n        # GBDT parameters\n        'xgb_max_depth': 6,\n        'xgb_learning_rate': 0.05,\n        'xgb_n_estimators': 300,\n        'xgb_subsample': 0.8,\n        'xgb_colsample': 0.8,\n        'lgb_max_depth': 6,\n        'lgb_learning_rate': 0.05,\n        'lgb_n_estimators': 300,\n        'lgb_num_leaves': 31,\n        'lgb_subsample': 0.8,\n        'lgb_colsample': 0.8\n    }\n    \n    # Create data loaders\n    train_dataset = TensorDataset(\n        torch.tensor(X_train_transformed, dtype=torch.float32),\n        torch.tensor(y_train.values, dtype=torch.float32).unsqueeze(1)\n    )\n    val_dataset = TensorDataset(\n        torch.tensor(X_val_transformed, dtype=torch.float32),\n        torch.tensor(y_val.values, dtype=torch.float32).unsqueeze(1)\n    )\n    \n    train_loader = DataLoader(train_dataset, batch_size=config['batch_size'], \n                            shuffle=True, num_workers=0)\n    val_loader = DataLoader(val_dataset, batch_size=config['batch_size'], \n                          shuffle=False, num_workers=0)\n    \n    # Train ensemble of DeepGBM models\n    print(\"\\n=== Training Ensemble of DeepGBM Models ===\")\n    \n    ensemble_models = []\n    ensemble_configs = [\n        {'catnn_hidden_dims': [512, 256, 128], 'gbdt_output_dim': 100},\n        {'catnn_hidden_dims': [384, 192, 96], 'gbdt_output_dim': 80},\n        {'catnn_hidden_dims': [640, 320, 160], 'gbdt_output_dim': 120}\n    ]\n    \n    for i, model_variation in enumerate(ensemble_configs[:2]):  # Use 2 models for ensemble\n        print(f\"\\n--- Training DeepGBM Model {i+1} ---\")\n        \n        # Update configuration\n        model_config = config.copy()\n        model_config.update(model_variation)\n        model_config['model_id'] = f'deepgbm_v{i+1}'\n        model_config['seed'] = 42 + i * 1000\n        \n        set_seed(model_config['seed'])\n        \n        # Create model and GBDT extractor\n        model = DeepGBM(model_config).to(device)\n        gbdt_extractor = GBDTFeatureExtractor(model_config)\n        \n        print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n        \n        # Train model\n        model, gbdt_extractor, history, best_pearson = train_deepgbm(\n            model, gbdt_extractor, train_loader, val_loader,\n            X_train_raw, y_train_raw, X_val_raw, y_val_raw,\n            model_config, device\n        )\n        \n        print(f\"Model {i+1} best validation Pearson: {best_pearson:.4f}\")\n        \n        # Store model for ensemble\n        ensemble_models.append((model, gbdt_extractor, transformer, selected_features))\n    \n    # Test prediction\n    print(\"\\n=== Making Test Predictions ===\")\n    \n    # Load test data\n    test = pl.scan_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\n    test_features = [f for f in SELECTED_FEATURES if f != \"label\"]\n    test = test.select(test_features).collect().to_pandas()\n    \n    # Add features to test data\n    test = add_advanced_market_features(test)\n    test = compute_rolling_features(test)\n    \n    # Make ensemble predictions\n    all_predictions = []\n    \n    for model, gbdt_extractor, transformer, features in ensemble_models:\n        # Prepare test data\n        X_test_selected = test[features].values\n        X_test_raw = X_test_selected.copy()\n        X_test_transformed = transformer.transform(X_test_selected)\n        \n        # Extract GBDT features\n        test_gbdt_features = gbdt_extractor.extract_features(X_test_raw)\n        test_gbdt_tensor = torch.tensor(test_gbdt_features, dtype=torch.float32).to(device)\n        \n        # Make predictions\n        model.eval()\n        test_dataset = TensorDataset(torch.tensor(X_test_transformed, dtype=torch.float32))\n        test_loader = DataLoader(test_dataset, batch_size=2048, shuffle=False)\n        \n        predictions = []\n        with torch.no_grad():\n            for batch_idx, (inputs,) in enumerate(test_loader):\n                inputs = inputs.to(device)\n                \n                # Get corresponding GBDT features\n                start_idx = batch_idx * test_loader.batch_size\n                end_idx = min(start_idx + test_loader.batch_size, len(test_gbdt_tensor))\n                gbdt_batch = test_gbdt_tensor[start_idx:end_idx]\n                \n                # Ensure batch sizes match\n                if len(inputs) != len(gbdt_batch):\n                    min_len = min(len(inputs), len(gbdt_batch))\n                    inputs = inputs[:min_len]\n                    gbdt_batch = gbdt_batch[:min_len]\n                \n                outputs = model(inputs, gbdt_batch)\n                predictions.extend(outputs.cpu().numpy())\n        \n        all_predictions.append(np.array(predictions))\n    \n    # Ensemble predictions\n    final_predictions = np.mean(all_predictions, axis=0)\n    \n    # Post-processing\n    pred_mean = y_train.mean()\n    pred_std = y_train.std()\n    final_predictions = np.clip(final_predictions, \n                               pred_mean - 4 * pred_std, \n                               pred_mean + 4 * pred_std)\n    \n    # Create submission\n    submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\n    submission[\"prediction\"] = final_predictions\n    submission.to_csv(\"submission_deepgbm.csv\", index=False)\n    \n    # Display results\n    print(\"\\n=== Final Results ===\")\n    print(f\"Ensemble size: {len(ensemble_models)} DeepGBM models\")\n    print(f\"Features used: {len(selected_features)}\")\n    print(f\"\\nPrediction statistics:\")\n    print(f\"Mean: {final_predictions.mean():.6f}\")\n    print(f\"Std: {final_predictions.std():.6f}\")\n    print(f\"Min: {final_predictions.min():.6f}\")\n    print(f\"Max: {final_predictions.max():.6f}\")\n    print(f\"Skewness: {skew(final_predictions):.4f}\")\n    print(f\"Kurtosis: {kurtosis(final_predictions):.4f}\")\n    \n    print(\"\\nTop 15 most important features:\")\n    print(importance_df.head(15)[['feature', 'combined_score']])\n    \n    # Clean up model files\n    for i in range(len(ensemble_models)):\n        model_file = f\"best_deepgbm_deepgbm_v{i+1}.pt\"\n        if os.path.exists(model_file):\n            os.remove(model_file)\n    \n    print(\"\\nSubmission saved as 'submission_deepgbm.csv'\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}