{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom pathlib import Path\nfrom sklearn.model_selection import KFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer, RobustScaler\nfrom sklearn.feature_selection import mutual_info_regression, SelectKBest\nfrom sklearn.impute import SimpleImputer\nfrom scipy.stats import pearsonr, spearmanr, skew, kurtosis, entropy\nfrom scipy.special import expit\nfrom tqdm import tqdm\nimport random\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\nimport gc\n\n# Deep Learning imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nimport torch.nn.functional as F\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau, CosineAnnealingWarmRestarts\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Configuration\n# =========================\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Features for GANDALF\n    GANDALF_FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        \"X425\", \"X132\", \"X691\", \"X593\", \"X377\", \"X285\", \"X126\", \"X419\", \"X604\",\n        \"X84\", \"X138\", \"X413\", \"X291\", \"X40\", \"X123\", \"X81\", \"X853\", \"X854\",\n        \"X777\", \"X219\", \"X776\", \"X180\", \"X781\", \"X445\", \"X444\", \"X384\", \"X466\",\n        \"X95\", \"X583\", \"X272\", \"X137\", \"X533\", \"X758\", \"X279\", \"X297\",\n        \"X21\", \"X20\", \"X28\", \"X29\", \"X19\", \"X27\", \"X22\", \"X198\", \"X89\", \"X90\",\n        \"X98\", \"X96\", \"X97\", \"X383\", \"X427\", \"X451\", \"X283\",\n        \"X753\", \"X497\", \"X748\", \"X820\", \"X566\", \"X535\", \"X394\", \"X618\",\n        \"X429\", \"X381\", \"X387\", \"X890\", \"X752\", \"X375\", \"X68\", \"X152\",\n        \"X110\", \"X850\", \"X851\", \"X481\", \"X321\", \"X363\", \"X405\", \"X492\",\n        \"X888\", \"X421\", \"X333\", \"X817\", \"X586\", \"X292\", \"X344\", \"X532\",\n        \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n    ]\n\n    LABEL_COLUMN = \"label\"\n    RANDOM_STATE = 42\n\ndef set_seed(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\ndef free_memory():\n    gc.collect()\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n\n# =========================\n# Advanced Feature Engineering\n# =========================\ndef add_advanced_features(df):\n    \"\"\"Add comprehensive feature engineering with interactions and derived variables\"\"\"\n    eps = 1e-10\n    \n    # Store original columns to avoid fragmentation warning\n    new_features = {}\n    \n    # 1. Original features with safety\n    new_features['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    new_features['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    new_features['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    new_features['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    new_features['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n    new_features['buy_sell_interaction'] = df['buy_qty'] * df['sell_qty']\n\n    # 2. Volume-based features\n    new_features['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    new_features['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    new_features['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + eps)\n    new_features['selling_pressure'] = df['sell_qty'] / (df['volume'] + eps)\n    new_features['buying_pressure'] = df['buy_qty'] / (df['volume'] + eps)\n    new_features['log_volume'] = np.log1p(df['volume'])\n    new_features['sqrt_volume'] = np.sqrt(df['volume'])\n    new_features['volume_squared'] = df['volume'] ** 2\n\n    # 3. Order book features\n    new_features['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + eps)\n    new_features['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + eps)\n    new_features['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + eps)\n    new_features['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + eps)\n    new_features['total_depth'] = df['bid_qty'] + df['ask_qty']\n    new_features['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (new_features['total_depth'] + eps)\n    new_features['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (new_features['total_depth'] + eps)\n    new_features['log_depth'] = np.log1p(new_features['total_depth'])\n    \n    # 4. Microstructure features\n    new_features['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    new_features['normalized_net_flow'] = new_features['net_order_flow'] / (df['volume'] + eps)\n    new_features['kyle_lambda'] = np.abs(new_features['net_order_flow']) / (df['volume'] + eps)\n    new_features['signed_kyle_lambda'] = new_features['net_order_flow'] / (df['volume'] + eps)\n    new_features['flow_toxicity'] = np.abs(new_features['order_flow_imbalance']) * df['volume']\n    new_features['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (new_features['total_depth'] + eps)\n    \n    # 5. Ratios and relationships\n    new_features['volume_depth_ratio'] = df['volume'] / (new_features['total_depth'] + eps)\n    new_features['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + eps)\n    new_features['log_buy_qty'] = np.log1p(df['buy_qty'])\n    new_features['log_sell_qty'] = np.log1p(df['sell_qty'])\n    new_features['log_bid_qty'] = np.log1p(df['bid_qty'])\n    new_features['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # 6. Price impact proxies\n    new_features['price_impact_proxy'] = new_features['net_order_flow'] / (new_features['total_depth'] + eps)\n    new_features['market_stress'] = df['volume'] / (new_features['total_depth'] + eps) * np.abs(new_features['order_flow_imbalance'])\n    new_features['illiquidity_measure'] = np.abs(new_features['net_order_flow']) / (df['volume'] * new_features['total_depth'] + eps)\n    \n    # 7. Advanced market microstructure features\n    new_features['amihud_illiquidity'] = np.abs(new_features['net_order_flow']) / (df['volume'] ** 1.5 + eps)\n    new_features['hasbrouck_lambda'] = new_features['net_order_flow'] / np.sqrt(df['volume'] + eps)\n    new_features['execution_cost_proxy'] = np.abs(new_features['order_flow_imbalance']) * np.sqrt(df['volume'])\n    \n    # 8. Liquidity consumption features\n    new_features['bid_consumption_rate'] = df['buy_qty'] / (df['bid_qty'] + eps)\n    new_features['ask_consumption_rate'] = df['sell_qty'] / (df['ask_qty'] + eps)\n    new_features['total_consumption_rate'] = (df['buy_qty'] + df['sell_qty']) / (new_features['total_depth'] + eps)\n    \n    # 9. Non-linear transformations\n    new_features['bid_ask_imbalance_squared'] = new_features['bid_ask_imbalance'] ** 2\n    new_features['order_flow_imbalance_squared'] = new_features['order_flow_imbalance'] ** 2\n    new_features['bid_ask_imbalance_cubed'] = new_features['bid_ask_imbalance'] ** 3\n    new_features['order_flow_imbalance_cubed'] = new_features['order_flow_imbalance'] ** 3\n    \n    # 10. Interaction features between X features\n    important_x_features = ['X863', 'X856', 'X598', 'X862', 'X385', 'X852', 'X603', 'X860', \n                           'X674', 'X415', 'X345', 'X855', 'X174', 'X302']\n    \n    for i, feat1 in enumerate(important_x_features[:7]):\n        if feat1 in df.columns:\n            new_features[f'{feat1}_squared'] = df[feat1] ** 2\n            new_features[f'{feat1}_x_volume'] = df[feat1] * df['volume']\n            new_features[f'{feat1}_x_bid_ask_imb'] = df[feat1] * new_features['bid_ask_imbalance']\n            new_features[f'{feat1}_x_order_flow_imb'] = df[feat1] * new_features['order_flow_imbalance']\n            \n            for feat2 in important_x_features[i+1:i+3]:\n                if feat2 in df.columns:\n                    new_features[f'{feat1}_x_{feat2}'] = df[feat1] * df[feat2]\n    \n    # 11. Market regime indicators\n    new_features['high_volume_regime'] = (df['volume'] > df['volume'].quantile(0.75)).astype(float)\n    new_features['imbalanced_market'] = (np.abs(new_features['bid_ask_imbalance']) > 0.5).astype(float)\n    new_features['toxic_flow'] = (np.abs(new_features['order_flow_imbalance']) > 0.5).astype(float)\n    \n    # 12. Composite scores\n    new_features['liquidity_score'] = (\n        new_features['total_depth'] / (new_features['total_depth'].mean() + eps) * 0.4 +\n        df['volume'] / (df['volume'].mean() + eps) * 0.3 +\n        (1 - np.abs(new_features['bid_ask_imbalance'])) * 0.3\n    )\n    \n    new_features['market_quality_score'] = (\n        new_features['liquidity_score'] * 0.5 +\n        (1 - np.abs(new_features['order_flow_imbalance'])) * 0.3 +\n        new_features['total_depth'] / (df['volume'] + eps) * 0.2\n    )\n    \n    # 13. Advanced transformations\n    new_features['tanh_bid_ask_imb'] = np.tanh(new_features['bid_ask_imbalance'])\n    new_features['tanh_order_flow_imb'] = np.tanh(new_features['order_flow_imbalance'])\n    new_features['sigmoid_volume_ratio'] = expit(df['volume'] / (df['volume'].mean() + eps))\n    \n    # 14. Volatility proxies\n    new_features['flow_volatility'] = np.abs(new_features['net_order_flow']) * np.sqrt(df['volume'])\n    new_features['depth_volatility'] = np.abs(new_features['depth_imbalance']) * new_features['total_depth']\n    \n    # 15. Information asymmetry proxies\n    new_features['pin_proxy'] = np.abs(new_features['order_flow_imbalance']) / (new_features['activity_intensity'] + eps)\n    new_features['information_share'] = new_features['net_order_flow'] ** 2 / (df['volume'] * new_features['total_depth'] + eps)\n    \n    # Add all new features to dataframe at once\n    new_features_df = pd.DataFrame(new_features, index=df.index)\n    df = pd.concat([df, new_features_df], axis=1)\n    \n    # Handle infinities and NaNs\n    numeric_cols = df.select_dtypes(include=[np.number]).columns\n    for col in numeric_cols:\n        if col != Config.LABEL_COLUMN:\n            lower = df[col].quantile(0.001)\n            upper = df[col].quantile(0.999)\n            df[col] = df[col].clip(lower, upper)\n    \n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\n# =========================\n# Enhanced GANDALF Components\n# =========================\nclass AttentionGate(nn.Module):\n    \"\"\"Attention mechanism for feature importance\"\"\"\n    def __init__(self, input_dim, hidden_dim=64):\n        super().__init__()\n        self.attention = nn.Sequential(\n            nn.Linear(input_dim, hidden_dim),\n            nn.Tanh(),\n            nn.Linear(hidden_dim, 1),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        weights = self.attention(x)\n        return x * weights\n\nclass EnhancedDifferentiableDecisionTree(nn.Module):\n    \"\"\"Enhanced soft decision tree with attention\"\"\"\n    def __init__(self, input_dim, depth, temperature=1.0, use_attention=True):\n        super().__init__()\n        self.depth = depth\n        self.n_leaves = 2 ** depth\n        self.use_attention = use_attention\n        \n        if use_attention:\n            self.attention_gate = AttentionGate(input_dim)\n        \n        self.internal_nodes = nn.ModuleList()\n        for i in range(2 ** depth - 1):\n            layer = nn.Linear(input_dim, 1)\n            nn.init.xavier_uniform_(layer.weight, gain=0.5)\n            nn.init.zeros_(layer.bias)\n            self.internal_nodes.append(layer)\n        \n        self.leaf_values = nn.Parameter(torch.zeros(self.n_leaves))\n        nn.init.uniform_(self.leaf_values, -0.01, 0.01)\n        \n        self.log_temperature = nn.Parameter(torch.log(torch.tensor(temperature)))\n        \n    def forward(self, x):\n        if self.use_attention:\n            x = self.attention_gate(x)\n        \n        batch_size = x.size(0)\n        device = x.device\n        \n        temp = torch.exp(self.log_temperature).clamp(min=0.1, max=10.0)\n        \n        path_probs = torch.ones(batch_size, 1, device=device)\n        \n        for level in range(self.depth):\n            n_nodes = 2 ** level\n            next_path_probs = []\n            \n            for node in range(n_nodes):\n                node_idx = 2 ** level - 1 + node\n                \n                if node_idx < len(self.internal_nodes):\n                    logit = self.internal_nodes[node_idx](x).squeeze(-1)\n                    logit = torch.clamp(logit, -10, 10)\n                    \n                    split_prob = torch.sigmoid(logit / temp)\n                    split_prob = torch.clamp(split_prob, 1e-7, 1 - 1e-7)\n                    \n                    if node < path_probs.size(1):\n                        current_prob = path_probs[:, node]\n                        left_prob = current_prob * (1 - split_prob)\n                        right_prob = current_prob * split_prob\n                        \n                        next_path_probs.append(left_prob.unsqueeze(1))\n                        next_path_probs.append(right_prob.unsqueeze(1))\n            \n            if next_path_probs:\n                path_probs = torch.cat(next_path_probs, dim=1)\n        \n        if path_probs.size(1) != self.n_leaves:\n            if path_probs.size(1) < self.n_leaves:\n                padding = torch.zeros(batch_size, self.n_leaves - path_probs.size(1), device=device)\n                path_probs = torch.cat([path_probs, padding], dim=1)\n            else:\n                path_probs = path_probs[:, :self.n_leaves]\n        \n        path_probs = F.normalize(path_probs, p=1, dim=1)\n        output = torch.sum(path_probs * self.leaf_values.unsqueeze(0), dim=1)\n        \n        return output\n\nclass EnhancedGatingNetwork(nn.Module):\n    \"\"\"Enhanced gating with fixed attention dimensions\"\"\"\n    def __init__(self, input_dim, n_trees, hidden_dim=128, n_heads=4, dropout=0.3):\n        super().__init__()\n        \n        # Ensure embed_dim is divisible by num_heads\n        self.attention_dim = ((input_dim + n_heads - 1) // n_heads) * n_heads\n        \n        # Project to attention dimension if needed\n        if input_dim != self.attention_dim:\n            self.input_projection = nn.Linear(input_dim, self.attention_dim)\n        else:\n            self.input_projection = None\n        \n        # Self-attention layer\n        self.self_attention = nn.MultiheadAttention(\n            embed_dim=self.attention_dim,\n            num_heads=n_heads,\n            dropout=dropout,\n            batch_first=True\n        )\n        \n        # Gating network\n        self.network = nn.Sequential(\n            nn.Linear(self.attention_dim, hidden_dim),\n            nn.LayerNorm(hidden_dim),\n            nn.GELU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, hidden_dim // 2),\n            nn.LayerNorm(hidden_dim // 2),\n            nn.GELU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim // 2, n_trees)\n        )\n        \n        for m in self.network.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.xavier_uniform_(m.weight, gain=0.5)\n                nn.init.zeros_(m.bias)\n        \n    def forward(self, x):\n        # Project to attention dimension if needed\n        if self.input_projection is not None:\n            x_att = self.input_projection(x)\n        else:\n            x_att = x\n        \n        # Self-attention\n        x_att = x_att.unsqueeze(1)\n        x_att, _ = self.self_attention(x_att, x_att, x_att)\n        x_att = x_att.squeeze(1)\n        \n        # Gating\n        gates = self.network(x_att)\n        gates = gates / 2.0\n        return F.softmax(gates, dim=-1)\n\nclass EnhancedGANDALF(nn.Module):\n    \"\"\"Enhanced GANDALF with fixed architecture\"\"\"\n    def __init__(self, config):\n        super().__init__()\n        \n        self.n_trees = config['n_trees']\n        self.tree_depth = config['tree_depth']\n        self.input_dim = config['input_dim']\n        self.use_attention = config.get('use_attention', True)\n        \n        # Feature embedder\n        self.feature_embedder = self._build_embedder(config)\n        self.embed_dim = config['embed_dims'][-1]\n        \n        # Skip connection\n        if self.embed_dim != self.input_dim:\n            self.skip_projection = nn.Linear(self.input_dim, self.embed_dim)\n        else:\n            self.skip_projection = None\n        \n        # Decision trees ensemble\n        self.trees = nn.ModuleList()\n        for i in range(self.n_trees):\n            tree_depth = self.tree_depth + (i % 3 - 1)\n            tree_depth = max(2, min(tree_depth, 7))\n            use_attention = self.use_attention and (i % 2 == 0)\n            \n            self.trees.append(\n                EnhancedDifferentiableDecisionTree(\n                    self.embed_dim,\n                    tree_depth,\n                    temperature=config['tree_temperature'],\n                    use_attention=use_attention\n                )\n            )\n        \n        # Enhanced gating network\n        self.gating_network = EnhancedGatingNetwork(\n            self.input_dim,\n            self.n_trees,\n            config['gate_hidden_dim'],\n            config.get('n_heads', 4),\n            config['gate_dropout']\n        )\n        \n        # Neural network head\n        if config.get('use_nn_head', True):\n            self.nn_head = self._build_nn_head(config)\n            self.combination_weight = nn.Parameter(torch.tensor(0.5))\n        else:\n            self.nn_head = None\n        \n        # Final normalization\n        self.final_norm = nn.LayerNorm(1)\n        \n        # Initialize weights\n        self.apply(self._init_weights)\n    \n    def _build_embedder(self, config):\n        \"\"\"Build feature embedder\"\"\"\n        layers = []\n        prev_dim = self.input_dim\n        \n        for i, dim in enumerate(config['embed_dims']):\n            layers.append(nn.Linear(prev_dim, dim))\n            layers.append(nn.LayerNorm(dim))\n            layers.append(nn.GELU())\n            \n            if i < len(config['embed_dims']) - 1:\n                layers.append(nn.Dropout(config['feature_dropout']))\n            \n            prev_dim = dim\n        \n        return nn.Sequential(*layers)\n    \n    def _build_nn_head(self, config):\n        \"\"\"Build neural network head\"\"\"\n        layers = []\n        prev_dim = self.input_dim\n        \n        for dim in config['head_dims']:\n            layers.append(nn.Linear(prev_dim, dim))\n            layers.append(nn.LayerNorm(dim))\n            layers.append(nn.GELU())\n            layers.append(nn.Dropout(config['head_dropout']))\n            prev_dim = dim\n        \n        layers.append(nn.Linear(prev_dim, 1))\n        \n        return nn.Sequential(*layers)\n    \n    def _init_weights(self, module):\n        if isinstance(module, nn.Linear):\n            nn.init.xavier_uniform_(module.weight, gain=0.5)\n            if module.bias is not None:\n                nn.init.zeros_(module.bias)\n    \n    def forward(self, x):\n        x = torch.clamp(x, -10, 10)\n        \n        # Feature embedding\n        embedded = self.feature_embedder(x)\n        \n        # Skip connection\n        if self.skip_projection is not None:\n            embedded = embedded + self.skip_projection(x)\n        elif self.embed_dim == self.input_dim:\n            embedded = embedded + x\n        \n        embedded = F.layer_norm(embedded, embedded.shape[1:])\n        \n        # Get tree outputs\n        tree_outputs = []\n        for tree in self.trees:\n            output = tree(embedded)\n            output = torch.clamp(output, -10, 10)\n            tree_outputs.append(output)\n        \n        tree_outputs = torch.stack(tree_outputs, dim=1)\n        \n        # Gating\n        gates = self.gating_network(x)\n        \n        # Weighted combination\n        forest_output = torch.sum(gates * tree_outputs, dim=1, keepdim=True)\n        \n        # Combine with neural head\n        if self.nn_head is not None:\n            nn_output = self.nn_head(x)\n            weight = torch.sigmoid(self.combination_weight)\n            final_output = weight * forest_output + (1 - weight) * nn_output\n        else:\n            final_output = forest_output\n        \n        # Final normalization\n        final_output = self.final_norm(final_output)\n        final_output = torch.clamp(final_output, -10, 10)\n        \n        return final_output\n\n# =========================\n# Training Function\n# =========================\ndef train_enhanced_gandalf_model(model, train_loader, val_loader, config, device):\n    \"\"\"Train enhanced GANDALF\"\"\"\n    \n    criterion = nn.SmoothL1Loss()\n    \n    # Optimizer\n    optimizer = optim.AdamW(\n        model.parameters(),\n        lr=config['learning_rate'],\n        weight_decay=config['weight_decay'],\n        eps=1e-8\n    )\n    \n    # Scheduler\n    scheduler = ReduceLROnPlateau(optimizer, mode='max', factor=0.5, patience=5, min_lr=1e-6)\n    \n    # Training variables\n    best_val_pearson = -np.inf\n    best_model_state = None\n    patience_counter = 0\n    patience = config.get('patience', 15)\n    num_epochs = config.get('num_epochs', 50)\n    \n    for epoch in range(num_epochs):\n        # Training\n        model.train()\n        train_loss = 0.0\n        train_batches = 0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{num_epochs}\")\n        for inputs, targets in progress_bar:\n            inputs, targets = inputs.to(device), targets.to(device)\n            \n            # Add noise\n            if config.get('noise_factor', 0) > 0:\n                noise = torch.randn_like(inputs) * config['noise_factor']\n                inputs = inputs + noise\n            \n            outputs = model(inputs)\n            loss = criterion(outputs, targets)\n            \n            optimizer.zero_grad()\n            loss.backward()\n            \n            # Gradient clipping\n            torch.nn.utils.clip_grad_norm_(model.parameters(), config.get('grad_clip', 1.0))\n            \n            optimizer.step()\n            \n            train_loss += loss.item()\n            train_batches += 1\n            \n            progress_bar.set_postfix({'loss': f'{loss.item():.4f}'})\n        \n        # Validation\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs, targets = inputs.to(device), targets.to(device)\n                \n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                \n                val_loss += loss.item()\n                val_preds.extend(outputs.cpu().numpy().flatten())\n                val_targets.extend(targets.cpu().numpy().flatten())\n        \n        # Calculate metrics\n        avg_train_loss = train_loss / train_batches\n        avg_val_loss = val_loss / len(val_loader)\n        \n        val_preds = np.array(val_preds)\n        val_targets = np.array(val_targets)\n        \n        # Remove NaN values\n        valid_mask = ~(np.isnan(val_preds) | np.isnan(val_targets))\n        if valid_mask.sum() > 10:\n            val_pearson = pearsonr(val_targets[valid_mask], val_preds[valid_mask])[0]\n            val_spearman = spearmanr(val_targets[valid_mask], val_preds[valid_mask])[0]\n        else:\n            val_pearson = -1.0\n            val_spearman = -1.0\n        \n        print(f\"\\nEpoch {epoch+1}: Train Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n        print(f\"Val Pearson: {val_pearson:.4f}, Val Spearman: {val_spearman:.4f}\")\n        \n        # Update scheduler\n        if not np.isnan(val_pearson):\n            scheduler.step(val_pearson)\n        \n        # Save best model\n        if not np.isnan(val_pearson) and val_pearson > best_val_pearson:\n            best_val_pearson = val_pearson\n            best_model_state = model.state_dict().copy()\n            patience_counter = 0\n            print(f\"✅ New best model! Pearson: {best_val_pearson:.4f}\")\n        else:\n            patience_counter += 1\n            if patience_counter >= patience:\n                print(f\"Early stopping triggered after {epoch+1} epochs\")\n                break\n    \n    # Load best model\n    if best_model_state is not None:\n        model.load_state_dict(best_model_state)\n    \n    return model, best_val_pearson\n\ndef train_enhanced_gandalf(train_df, test_df):\n    \"\"\"Main enhanced GANDALF training function\"\"\"\n    print(\"\\n=== Training Enhanced GANDALF Model ===\")\n    \n    set_seed(42)\n    \n    # Get all features\n    gandalf_features = Config.GANDALF_FEATURES.copy()\n    \n    # Get all available engineered features\n    all_features = [col for col in train_df.columns if col not in [Config.LABEL_COLUMN, 'timestamp']]\n    all_gandalf_features = list(set(gandalf_features + all_features))\n    all_gandalf_features = [f for f in all_gandalf_features if f in train_df.columns]\n    \n    print(f\"Using {len(all_gandalf_features)} features for Enhanced GANDALF\")\n    \n    # Use recent data\n    train_size = int(0.9 * len(train_df))\n    train_data = train_df.iloc[-train_size:].reset_index(drop=True)\n    \n    # Split for validation\n    split_idx = int(0.85 * len(train_data))\n    train_split = train_data[:split_idx].copy()\n    val_split = train_data[split_idx:].copy()\n    \n    y_train = train_split[Config.LABEL_COLUMN].values\n    y_val = val_split[Config.LABEL_COLUMN].values\n    \n    X_train = train_split[all_gandalf_features].values\n    X_val = val_split[all_gandalf_features].values\n    \n    # Handle missing values\n    imputer = SimpleImputer(strategy='median')\n    X_train = imputer.fit_transform(X_train)\n    X_val = imputer.transform(X_val)\n    \n    # Feature selection\n    print(\"\\nSelecting features...\")\n    n_features = min(150, len(all_gandalf_features))\n    selector = SelectKBest(score_func=mutual_info_regression, k=n_features)\n    X_train_selected = selector.fit_transform(X_train, y_train)\n    X_val_selected = selector.transform(X_val)\n    \n    selected_features = [all_gandalf_features[i] for i in selector.get_support(indices=True)]\n    print(f\"Selected {len(selected_features)} features\")\n    \n    # Scaling\n    print(\"\\nScaling data...\")\n    scaler = RobustScaler(quantile_range=(5, 95))\n    X_train_scaled = scaler.fit_transform(X_train_selected)\n    X_val_scaled = scaler.transform(X_val_selected)\n    \n    X_train_scaled = np.clip(X_train_scaled, -5, 5)\n    X_val_scaled = np.clip(X_val_scaled, -5, 5)\n    \n    # Create data loaders\n    train_dataset = TensorDataset(\n        torch.tensor(X_train_scaled, dtype=torch.float32),\n        torch.tensor(y_train, dtype=torch.float32).unsqueeze(1)\n    )\n    val_dataset = TensorDataset(\n        torch.tensor(X_val_scaled, dtype=torch.float32),\n        torch.tensor(y_val, dtype=torch.float32).unsqueeze(1)\n    )\n    \n    # Fixed configurations with proper dimensions\n    configs = [\n        # Config 1: Simple model\n        {\n            'config_name': 'simple_stable',\n            'n_trees': 12,\n            'tree_depth': 4,\n            'tree_temperature': 1.0,\n            'embed_dims': [128, 96, 64],\n            'feature_dropout': 0.1,\n            'gate_hidden_dim': 64,\n            'gate_dropout': 0.1,\n            'n_heads': 2,  # Reduced heads\n            'use_attention': True,\n            'use_nn_head': True,\n            'head_dims': [128, 64],\n            'head_dropout': 0.2,\n            'learning_rate': 0.001,\n            'weight_decay': 0.01,\n            'batch_size': 512,\n            'noise_factor': 0.005,\n            'grad_clip': 1.0,\n            'num_epochs': 30,\n            'patience': 10,\n            'input_dim': X_train_scaled.shape[1]\n        },\n        # Config 2: Medium model\n        {\n            'config_name': 'medium_ensemble',\n            'n_trees': 16,\n            'tree_depth': 5,\n            'tree_temperature': 1.2,\n            'embed_dims': [192, 128, 96],\n            'feature_dropout': 0.15,\n            'gate_hidden_dim': 96,\n            'gate_dropout': 0.15,\n            'n_heads': 3,  # Adjusted for divisibility\n            'use_attention': False,  # Disable attention for stability\n            'use_nn_head': True,\n            'head_dims': [192, 96],\n            'head_dropout': 0.2,\n            'learning_rate': 0.0008,\n            'weight_decay': 0.02,\n            'batch_size': 384,\n            'noise_factor': 0.005,\n            'grad_clip': 1.0,\n            'num_epochs': 25,\n            'patience': 8,\n            'input_dim': X_train_scaled.shape[1]\n        }\n    ]\n    \n    # Train ensemble\n    ensemble_models = []\n    ensemble_scores = []\n    \n    for config in configs:\n        print(f\"\\n=== Training Enhanced GANDALF {config['config_name']} ===\")\n        \n        train_loader = DataLoader(train_dataset, batch_size=config['batch_size'], shuffle=True)\n        val_loader = DataLoader(val_dataset, batch_size=1024, shuffle=False)\n        \n        try:\n            model = EnhancedGANDALF(config).to(device)\n            print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n            \n            # Train\n            model, best_score = train_enhanced_gandalf_model(\n                model, train_loader, val_loader, config, device\n            )\n            \n            if best_score > -0.5:  # Accept models with reasonable scores\n                ensemble_models.append(model)\n                ensemble_scores.append(max(0.01, best_score))  # Ensure positive weight\n                print(f\"Model {config['config_name']} best validation Pearson: {best_score:.4f}\")\n            else:\n                print(f\"Model {config['config_name']} score too low: {best_score:.4f}\")\n                \n        except Exception as e:\n            print(f\"Error training {config['config_name']}: {str(e)}\")\n            import traceback\n            traceback.print_exc()\n            continue\n        \n        # Clean up memory\n        free_memory()\n    \n    # Check if we have any models\n    if len(ensemble_models) == 0:\n        print(\"No models were successfully trained! Creating a simple baseline model.\")\n        # Create a simple baseline model\n        simple_config = configs[0].copy()\n        simple_config['n_trees'] = 5\n        simple_config['tree_depth'] = 3\n        simple_config['use_attention'] = False\n        simple_config['num_epochs'] = 10\n        \n        train_loader = DataLoader(train_dataset, batch_size=512, shuffle=True)\n        val_loader = DataLoader(val_dataset, batch_size=1024, shuffle=False)\n        \n        model = EnhancedGANDALF(simple_config).to(device)\n        model, best_score = train_enhanced_gandalf_model(\n            model, train_loader, val_loader, simple_config, device\n        )\n        ensemble_models.append(model)\n        ensemble_scores.append(1.0)\n    \n    # Make test predictions\n    print(\"\\n=== Making Enhanced GANDALF Test Predictions ===\")\n    \n    # Prepare test data\n    X_test = test_df[all_gandalf_features].values\n    X_test = imputer.transform(X_test)\n    X_test_selected = selector.transform(X_test)\n    X_test_scaled = scaler.transform(X_test_selected)\n    X_test_scaled = np.clip(X_test_scaled, -5, 5)\n    \n    # Make predictions with each model\n    all_predictions = []\n    \n    for model in ensemble_models:\n        model.eval()\n        test_dataset = TensorDataset(torch.tensor(X_test_scaled, dtype=torch.float32))\n        test_loader = DataLoader(test_dataset, batch_size=2048, shuffle=False)\n        \n        predictions = []\n        with torch.no_grad():\n            for (inputs,) in test_loader:\n                inputs = inputs.to(device)\n                outputs = model(inputs)\n                predictions.extend(outputs.cpu().numpy().flatten())\n        \n        all_predictions.append(np.array(predictions))\n    \n    # Weighted ensemble\n    weights = np.array(ensemble_scores)\n    weights = weights / weights.sum()\n    \n    final_predictions = np.zeros_like(all_predictions[0])\n    for pred, weight in zip(all_predictions, weights):\n        final_predictions += weight * pred\n    \n    # Post-processing\n    pred_mean = train_df[Config.LABEL_COLUMN].mean()\n    pred_std = train_df[Config.LABEL_COLUMN].std()\n    \n    # Clip predictions\n    final_predictions = np.clip(\n        final_predictions,\n        pred_mean - 3 * pred_std,\n        pred_mean + 3 * pred_std\n    )\n    \n    # Final NaN check\n    final_predictions = np.nan_to_num(final_predictions, nan=pred_mean)\n    \n    print(f\"\\nEnhanced GANDALF ensemble weights: {weights}\")\n    print(f\"Prediction stats - Mean: {final_predictions.mean():.6f}, Std: {final_predictions.std():.6f}\")\n    \n    return final_predictions\n\ndef load_data():\n    \"\"\"Load data with enhanced features\"\"\"\n    all_features = Config.GANDALF_FEATURES\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=all_features + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=all_features)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n\n    # Add enhanced features\n    train_df = add_advanced_features(train_df)\n    test_df = add_advanced_features(test_df)\n\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\n# =========================\n# Main Execution\n# =========================\nif __name__ == \"__main__\":\n    # Set memory management\n    if torch.cuda.is_available():\n        torch.cuda.set_per_process_memory_fraction(0.9)\n    \n    # Load data with enhanced features\n    train_df, test_df, submission_df = load_data()\n    \n    # Train enhanced GANDALF\n    gandalf_predictions = train_enhanced_gandalf(train_df, test_df)\n    \n    # Save enhanced submission\n    enhanced_submission = submission_df.copy()\n    enhanced_submission[\"prediction\"] = gandalf_predictions\n    enhanced_submission.to_csv(\"submission_enhanced_gandalf_fixed.csv\", index=False)\n    print(f\"\\nSaved: submission_enhanced_gandalf_fixed.csv\")\n    \n    # Show sample predictions\n    print(\"\\nSample predictions (first 10 rows):\")\n    print(enhanced_submission[['ID', 'prediction']].head(10))\n    \n    # Show final statistics\n    print(\"\\n✅ Enhanced GANDALF training completed successfully!\")\n    print(f\"Total features used: {len(train_df.columns) - 1}\")\n    print(f\"Prediction range: [{gandalf_predictions.min():.4f}, {gandalf_predictions.max():.4f}]\")\n    print(f\"Prediction mean: {gandalf_predictions.mean():.6f}\")\n    print(f\"Prediction std: {gandalf_predictions.std():.6f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}