{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.feature_selection import SelectKBest, mutual_info_regression\nfrom scipy.stats import pearsonr, rankdata\nfrom scipy.optimize import minimize, differential_evolution\nimport gc\n\n# Deep learning imports for GANDALF\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\nprint(\"Starting XGBoost + GANDALF Pipeline - Advanced Ensemble Approach...\")\nprint(\"=\" * 80)\n\n# Configuration\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    n_folds = 5\n    random_state = 42\n    use_gpu = torch.cuda.is_available()\n    device = torch.device('cuda' if use_gpu else 'cpu')\n    \n    # Feature settings\n    max_x_features = 80\n    n_interaction_features = 50\n    n_proprietary_features = 30\n    \n    # Feature selection\n    use_feature_selection = True\n    feature_selection_threshold = 0.01\n    \n    # GANDALF settings\n    gandalf_hidden_dims = [256, 128, 64]\n    gandalf_dropout = 0.3\n    gandalf_batch_size = 1024\n    gandalf_epochs = 50\n    gandalf_lr = 0.001\n    gandalf_patience = 10\n    \n    # Discriminator settings\n    disc_hidden_dims = [128, 64, 32]\n    disc_lr = 0.0005\n    disc_epochs = 30\n\n# Memory optimization\ndef reduce_mem_usage(df, name=\"\"):\n    print(f\"Optimizing memory for {name}...\")\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage: {start_mem:.2f} MB -> {end_mem:.2f} MB ({100*(start_mem-end_mem)/start_mem:.1f}% reduction)')\n    return df\n\n# ====================== GANDALF Architecture ======================\n\nclass GatedUnit(nn.Module):\n    \"\"\"Gated Linear Unit for feature selection\"\"\"\n    def __init__(self, input_dim, output_dim):\n        super(GatedUnit, self).__init__()\n        self.fc = nn.Linear(input_dim, output_dim)\n        self.gate = nn.Linear(input_dim, output_dim)\n        \n    def forward(self, x):\n        return self.fc(x) * torch.sigmoid(self.gate(x))\n\nclass AttentionLayer(nn.Module):\n    \"\"\"Multi-head attention for feature interactions\"\"\"\n    def __init__(self, input_dim, num_heads=4):\n        super(AttentionLayer, self).__init__()\n        self.num_heads = num_heads\n        self.head_dim = input_dim // num_heads\n        \n        self.query = nn.Linear(input_dim, input_dim)\n        self.key = nn.Linear(input_dim, input_dim)\n        self.value = nn.Linear(input_dim, input_dim)\n        self.fc_out = nn.Linear(input_dim, input_dim)\n        \n    def forward(self, x):\n        batch_size = x.size(0)\n        \n        Q = self.query(x).view(batch_size, -1, self.num_heads, self.head_dim).transpose(1, 2)\n        K = self.key(x).view(batch_size, -1, self.num_heads, self.head_dim).transpose(1, 2)\n        V = self.value(x).view(batch_size, -1, self.num_heads, self.head_dim).transpose(1, 2)\n        \n        attention_scores = torch.matmul(Q, K.transpose(-2, -1)) / np.sqrt(self.head_dim)\n        attention_weights = F.softmax(attention_scores, dim=-1)\n        \n        attention_output = torch.matmul(attention_weights, V)\n        attention_output = attention_output.transpose(1, 2).contiguous().view(batch_size, -1)\n        \n        return self.fc_out(attention_output)\n\nclass GANDALF(nn.Module):\n    \"\"\"GANDALF: Gated Adaptive Network for Deep Automated Learning of Features\"\"\"\n    def __init__(self, input_dim, hidden_dims, dropout=0.3):\n        super(GANDALF, self).__init__()\n        \n        # Feature extraction layers with gating\n        self.gated_layers = nn.ModuleList()\n        dims = [input_dim] + hidden_dims\n        \n        for i in range(len(dims) - 1):\n            self.gated_layers.append(GatedUnit(dims[i], dims[i+1]))\n        \n        # Attention mechanism for feature interactions\n        self.attention = AttentionLayer(hidden_dims[-1])\n        \n        # Batch normalization layers\n        self.batch_norms = nn.ModuleList([nn.BatchNorm1d(dim) for dim in hidden_dims])\n        \n        # Dropout for regularization\n        self.dropout = nn.Dropout(dropout)\n        \n        # Output layer\n        self.output_layer = nn.Linear(hidden_dims[-1], 1)\n        \n        # Residual connections\n        self.residual_weight = nn.Parameter(torch.tensor(0.1))\n        \n    def forward(self, x):\n        # Store original input for residual connection\n        original = x\n        \n        # Pass through gated layers\n        for i, (gated_layer, bn) in enumerate(zip(self.gated_layers, self.batch_norms)):\n            x = gated_layer(x)\n            x = bn(x)\n            x = F.relu(x)\n            x = self.dropout(x)\n        \n        # Apply attention mechanism\n        x = self.attention(x)\n        \n        # Output with residual connection\n        output = self.output_layer(x)\n        \n        # Add weighted residual from original features\n        if original.size(1) == 1:\n            residual = original\n        else:\n            # Simple linear projection for residual\n            residual = original.mean(dim=1, keepdim=True)\n        \n        output = output + self.residual_weight * residual\n        \n        return output.squeeze()\n\nclass Discriminator(nn.Module):\n    \"\"\"Discriminator network for anti-overfitting\"\"\"\n    def __init__(self, input_dim, hidden_dims):\n        super(Discriminator, self).__init__()\n        \n        layers = []\n        dims = [input_dim] + hidden_dims + [1]\n        \n        for i in range(len(dims) - 1):\n            layers.append(nn.Linear(dims[i], dims[i+1]))\n            if i < len(dims) - 2:\n                layers.append(nn.ReLU())\n                layers.append(nn.Dropout(0.2))\n        \n        self.model = nn.Sequential(*layers)\n        \n    def forward(self, x):\n        return torch.sigmoid(self.model(x))\n\nclass GANDALFTrainer:\n    \"\"\"Trainer class for GANDALF with anti-overfitting discriminator\"\"\"\n    def __init__(self, input_dim, config):\n        self.config = config\n        self.device = config.device\n        \n        # Initialize GANDALF model\n        self.gandalf = GANDALF(\n            input_dim=input_dim,\n            hidden_dims=config.gandalf_hidden_dims,\n            dropout=config.gandalf_dropout\n        ).to(self.device)\n        \n        # Initialize discriminator\n        self.discriminator = Discriminator(\n            input_dim=input_dim + 1,  # features + prediction\n            hidden_dims=config.disc_hidden_dims\n        ).to(self.device)\n        \n        # Optimizers\n        self.gandalf_optimizer = optim.Adam(self.gandalf.parameters(), lr=config.gandalf_lr)\n        self.disc_optimizer = optim.Adam(self.discriminator.parameters(), lr=config.disc_lr)\n        \n        # Loss functions\n        self.regression_loss = nn.MSELoss()\n        self.adversarial_loss = nn.BCELoss()\n        \n        # Learning rate schedulers\n        self.gandalf_scheduler = ReduceLROnPlateau(\n            self.gandalf_optimizer, mode='min', patience=5, factor=0.5\n        )\n        self.disc_scheduler = ReduceLROnPlateau(\n            self.disc_optimizer, mode='min', patience=5, factor=0.5\n        )\n        \n    def train_discriminator(self, real_data, fake_data):\n        \"\"\"Train discriminator to distinguish between real and overfitted predictions\"\"\"\n        self.discriminator.train()\n        \n        # Real data (good predictions)\n        real_labels = torch.ones(real_data.size(0), 1).to(self.device)\n        real_pred = self.discriminator(real_data)\n        real_loss = self.adversarial_loss(real_pred, real_labels)\n        \n        # Fake data (overfitted predictions)\n        fake_labels = torch.zeros(fake_data.size(0), 1).to(self.device)\n        fake_pred = self.discriminator(fake_data)\n        fake_loss = self.adversarial_loss(fake_pred, fake_labels)\n        \n        # Total discriminator loss\n        disc_loss = (real_loss + fake_loss) / 2\n        \n        self.disc_optimizer.zero_grad()\n        disc_loss.backward()\n        self.disc_optimizer.step()\n        \n        return disc_loss.item()\n    \n    def train_gandalf(self, X, y, X_val, y_val):\n        \"\"\"Train GANDALF with adversarial regularization\"\"\"\n        self.gandalf.train()\n        \n        # Create data loaders\n        train_dataset = TensorDataset(\n            torch.FloatTensor(X).to(self.device),\n            torch.FloatTensor(y).to(self.device)\n        )\n        train_loader = DataLoader(\n            train_dataset, \n            batch_size=self.config.gandalf_batch_size, \n            shuffle=True\n        )\n        \n        val_dataset = TensorDataset(\n            torch.FloatTensor(X_val).to(self.device),\n            torch.FloatTensor(y_val).to(self.device)\n        )\n        val_loader = DataLoader(\n            val_dataset, \n            batch_size=self.config.gandalf_batch_size\n        )\n        \n        best_val_loss = float('inf')\n        patience_counter = 0\n        \n        for epoch in range(self.config.gandalf_epochs):\n            # Training phase\n            train_losses = []\n            for batch_X, batch_y in train_loader:\n                # Forward pass\n                predictions = self.gandalf(batch_X)\n                \n                # Regression loss\n                reg_loss = self.regression_loss(predictions, batch_y)\n                \n                # Adversarial loss - fool discriminator\n                disc_input = torch.cat([batch_X, predictions.unsqueeze(1)], dim=1)\n                disc_pred = self.discriminator(disc_input)\n                adv_loss = self.adversarial_loss(\n                    disc_pred, \n                    torch.ones_like(disc_pred)  # Want discriminator to think it's real\n                )\n                \n                # Total loss\n                total_loss = reg_loss + 0.1 * adv_loss\n                \n                # Backward pass\n                self.gandalf_optimizer.zero_grad()\n                total_loss.backward()\n                torch.nn.utils.clip_grad_norm_(self.gandalf.parameters(), 1.0)\n                self.gandalf_optimizer.step()\n                \n                train_losses.append(total_loss.item())\n            \n            # Validation phase\n            self.gandalf.eval()\n            val_losses = []\n            val_predictions = []\n            val_targets = []\n            \n            with torch.no_grad():\n                for batch_X, batch_y in val_loader:\n                    predictions = self.gandalf(batch_X)\n                    val_loss = self.regression_loss(predictions, batch_y)\n                    val_losses.append(val_loss.item())\n                    val_predictions.extend(predictions.cpu().numpy())\n                    val_targets.extend(batch_y.cpu().numpy())\n            \n            # Calculate validation correlation\n            val_corr = pearsonr(val_predictions, val_targets)[0]\n            avg_val_loss = np.mean(val_losses)\n            \n            # Update learning rate\n            self.gandalf_scheduler.step(avg_val_loss)\n            \n            # Early stopping\n            if avg_val_loss < best_val_loss:\n                best_val_loss = avg_val_loss\n                patience_counter = 0\n                # Save best model state\n                self.best_gandalf_state = self.gandalf.state_dict()\n            else:\n                patience_counter += 1\n                if patience_counter >= self.config.gandalf_patience:\n                    print(f\"Early stopping at epoch {epoch+1}\")\n                    break\n            \n            if (epoch + 1) % 10 == 0:\n                print(f\"Epoch {epoch+1}/{self.config.gandalf_epochs} - \"\n                      f\"Train Loss: {np.mean(train_losses):.4f}, \"\n                      f\"Val Loss: {avg_val_loss:.4f}, \"\n                      f\"Val Corr: {val_corr:.4f}\")\n            \n            self.gandalf.train()\n        \n        # Load best model\n        self.gandalf.load_state_dict(self.best_gandalf_state)\n    \n    def train_anti_overfit_discriminator(self, X_train, y_train, X_val, y_val):\n        \"\"\"Train discriminator to detect overfitting patterns\"\"\"\n        print(\"\\nTraining anti-overfitting discriminator...\")\n        \n        # Generate predictions on training and validation sets\n        self.gandalf.eval()\n        with torch.no_grad():\n            train_preds = self.gandalf(torch.FloatTensor(X_train).to(self.device))\n            val_preds = self.gandalf(torch.FloatTensor(X_val).to(self.device))\n        \n        # Calculate prediction errors\n        train_errors = np.abs(train_preds.cpu().numpy() - y_train)\n        val_errors = np.abs(val_preds.cpu().numpy() - y_val)\n        \n        # Create discriminator training data\n        # High error samples are \"fake\" (overfitted)\n        error_threshold = np.percentile(train_errors, 75)\n        \n        real_mask = train_errors < error_threshold\n        fake_mask = train_errors >= error_threshold\n        \n        real_features = np.concatenate([\n            X_train[real_mask],\n            train_preds.cpu().numpy()[real_mask].reshape(-1, 1)\n        ], axis=1)\n        \n        fake_features = np.concatenate([\n            X_train[fake_mask],\n            train_preds.cpu().numpy()[fake_mask].reshape(-1, 1)\n        ], axis=1)\n        \n        # Train discriminator\n        for epoch in range(self.config.disc_epochs):\n            # Sample batch\n            batch_size = min(len(real_features), len(fake_features), 512)\n            real_batch = real_features[np.random.choice(len(real_features), batch_size)]\n            fake_batch = fake_features[np.random.choice(len(fake_features), batch_size)]\n            \n            real_tensor = torch.FloatTensor(real_batch).to(self.device)\n            fake_tensor = torch.FloatTensor(fake_batch).to(self.device)\n            \n            disc_loss = self.train_discriminator(real_tensor, fake_tensor)\n            \n            if (epoch + 1) % 10 == 0:\n                print(f\"Discriminator Epoch {epoch+1}/{self.config.disc_epochs} - Loss: {disc_loss:.4f}\")\n    \n    def predict(self, X):\n        \"\"\"Make predictions with GANDALF\"\"\"\n        self.gandalf.eval()\n        with torch.no_grad():\n            X_tensor = torch.FloatTensor(X).to(self.device)\n            predictions = self.gandalf(X_tensor)\n        return predictions.cpu().numpy()\n    \n    def get_overfit_scores(self, X):\n        \"\"\"Get overfitting scores from discriminator\"\"\"\n        self.gandalf.eval()\n        self.discriminator.eval()\n        with torch.no_grad():\n            X_tensor = torch.FloatTensor(X).to(self.device)\n            predictions = self.gandalf(X_tensor)\n            disc_input = torch.cat([X_tensor, predictions.unsqueeze(1)], dim=1)\n            overfit_scores = self.discriminator(disc_input)\n        return overfit_scores.cpu().numpy().squeeze()\n\n# ====================== Feature Engineering ======================\n\ndef create_proprietary_x_variables(df, n_features=30):\n    \"\"\"Create proprietary X variables with focus on quality over quantity\"\"\"\n    print(f\"Creating {n_features} proprietary X variables...\")\n    \n    # Get existing X features\n    x_features = [col for col in df.columns if col.startswith('X') and col[1:].isdigit()]\n    \n    # Important X features from original pipeline\n    important_x = [\"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\",\n                   \"X344\", \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X137\", \"X174\", \"X178\"]\n    \n    # Use available important features\n    base_features = [f for f in important_x if f in df.columns][:10]\n    \n    # Add some high-variance features\n    if len(base_features) < 10:\n        x_variances = df[x_features].var()\n        high_var_features = x_variances.nlargest(10).index.tolist()\n        for feat in high_var_features:\n            if feat not in base_features:\n                base_features.append(feat)\n                if len(base_features) >= 10:\n                    break\n    \n    prop_idx = 1\n    \n    # 1. Statistical combinations (8 features)\n    if len(base_features) >= 5:\n        df[f'X_prop_{prop_idx}'] = df[base_features[:5]].mean(axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:5]].std(axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:5]].max(axis=1) - df[base_features[:5]].min(axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:5]].median(axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:7]].quantile(0.25, axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:7]].quantile(0.75, axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = (df[base_features[:5]] > df[base_features[:5]].mean(axis=1).values[:, None]).sum(axis=1)\n        prop_idx += 1\n        df[f'X_prop_{prop_idx}'] = df[base_features[:5]].idxmax(axis=1).str.extract('(\\d+)')[0].astype(float)\n        prop_idx += 1\n    \n    # 2. Non-linear transformations (7 features)\n    for i in range(7):\n        feat = base_features[i % len(base_features)]\n        if i < 2:\n            df[f'X_prop_{prop_idx}'] = np.sign(df[feat]) * np.sqrt(np.abs(df[feat]))\n        elif i < 4:\n            df[f'X_prop_{prop_idx}'] = np.tanh(df[feat] / df[feat].std())\n        elif i < 6:\n            df[f'X_prop_{prop_idx}'] = 1 / (1 + np.exp(-df[feat] / df[feat].std()))  # Sigmoid\n        else:\n            df[f'X_prop_{prop_idx}'] = rankdata(df[feat]) / len(df)  # Rank transform\n        prop_idx += 1\n    \n    # 3. Market interaction features (10 features)\n    if 'volume' in df.columns:\n        for i in range(3):\n            df[f'X_prop_{prop_idx}'] = df[base_features[i]] * np.log1p(df['volume'])\n            prop_idx += 1\n    \n    if 'order_flow_imbalance' in df.columns:\n        for i in range(3):\n            df[f'X_prop_{prop_idx}'] = df[base_features[i+3]] * df['order_flow_imbalance']\n            prop_idx += 1\n    \n    if 'kyle_lambda' in df.columns:\n        for i in range(2):\n            df[f'X_prop_{prop_idx}'] = df[base_features[i+6]] * np.sign(df['kyle_lambda']) * np.log1p(np.abs(df['kyle_lambda']))\n            prop_idx += 1\n    \n    if 'vpin' in df.columns:\n        for i in range(2):\n            df[f'X_prop_{prop_idx}'] = df[base_features[i+8]] * df['vpin']\n            prop_idx += 1\n    \n    # 4. Interaction ratios (5 features)\n    for i in range(5):\n        feat1 = base_features[i % len(base_features)]\n        feat2 = base_features[(i + 1) % len(base_features)]\n        df[f'X_prop_{prop_idx}'] = df[feat1] / (np.abs(df[feat2]) + 1e-8)\n        prop_idx += 1\n    \n    # Handle infinities and NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    print(f\"Created {prop_idx-1} proprietary X variables\")\n    return df\n\ndef create_interaction_features(df, selected_features, n_interactions=50):\n    \"\"\"Create high-quality interaction features\"\"\"\n    print(f\"Creating {n_interactions} interaction features...\")\n    \n    interaction_features = []\n    feature_names = []\n    \n    # Prioritize features\n    important_x = [\"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\"]\n    market_features = ['order_flow_imbalance', 'kyle_lambda', 'vpin', 'liquidity_imbalance',\n                      'bid_ask_spread', 'buying_pressure', 'volume', 'log_volume']\n    \n    # Get available priority features\n    priority_x = [f for f in important_x if f in selected_features and f in df.columns][:10]\n    priority_market = [f for f in market_features if f in selected_features and f in df.columns][:8]\n    \n    interaction_count = 0\n    \n    # 1. X features with market microstructure (20 interactions)\n    for i, x_feat in enumerate(priority_x[:10]):\n        if interaction_count >= 20:\n            break\n        for j, market_feat in enumerate(priority_market[:4]):\n            if interaction_count >= 20:\n                break\n            \n            # Multiplication\n            interaction_features.append(df[x_feat] * df[market_feat])\n            feature_names.append(f'{x_feat}_x_{market_feat}')\n            interaction_count += 1\n            \n            # Log interaction\n            if interaction_count < 20 and market_feat in ['volume', 'kyle_lambda']:\n                interaction_features.append(df[x_feat] * np.log1p(np.abs(df[market_feat])))\n                feature_names.append(f'{x_feat}_x_log_{market_feat}')\n                interaction_count += 1\n    \n    # 2. Market feature interactions (15 interactions)\n    market_pairs = [\n        ('order_flow_imbalance', 'kyle_lambda'),\n        ('vpin', 'liquidity_imbalance'),\n        ('buying_pressure', 'selling_pressure'),\n        ('bid_ask_spread', 'total_liquidity'),\n        ('volume', 'liquidity_ratio')\n    ]\n    \n    for feat1, feat2 in market_pairs:\n        if interaction_count >= 35:\n            break\n        if feat1 in df.columns and feat2 in df.columns:\n            # Product\n            interaction_features.append(df[feat1] * df[feat2])\n            feature_names.append(f'{feat1}_x_{feat2}')\n            interaction_count += 1\n            \n            # Ratio\n            if interaction_count < 35:\n                interaction_features.append(df[feat1] / (np.abs(df[feat2]) + 1e-8))\n                feature_names.append(f'{feat1}_div_{feat2}')\n                interaction_count += 1\n            \n            # Difference\n            if interaction_count < 35:\n                interaction_features.append(df[feat1] - df[feat2])\n                feature_names.append(f'{feat1}_minus_{feat2}')\n                interaction_count += 1\n    \n    # 3. Non-linear interactions (10 interactions)\n    for i in range(5):\n        if interaction_count >= 45:\n            break\n        if i < len(priority_x):\n            feat = priority_x[i]\n            \n            # Squared\n            interaction_features.append(df[feat] ** 2)\n            feature_names.append(f'{feat}_squared')\n            interaction_count += 1\n            \n            # Square root of absolute\n            if interaction_count < 45:\n                interaction_features.append(np.sqrt(np.abs(df[feat])))\n                feature_names.append(f'{feat}_sqrt_abs')\n                interaction_count += 1\n    \n    # 4. Three-way interactions (5 interactions)\n    if len(priority_x) >= 3 and len(priority_market) >= 1:\n        for i in range(5):\n            if interaction_count >= n_interactions:\n                break\n            x1 = priority_x[i % len(priority_x)]\n            x2 = priority_x[(i+1) % len(priority_x)]\n            m1 = priority_market[i % len(priority_market)]\n            \n            interaction_features.append(df[x1] * df[x2] * df[m1])\n            feature_names.append(f'{x1}_{x2}_{m1}')\n            interaction_count += 1\n    \n    # Create DataFrame\n    interaction_df = pd.DataFrame(\n        np.column_stack(interaction_features[:interaction_count]),\n        columns=feature_names[:interaction_count],\n        index=df.index\n    )\n    \n    # Handle infinities and NaN\n    interaction_df = interaction_df.replace([np.inf, -np.inf], np.nan)\n    interaction_df = interaction_df.fillna(0)\n    \n    print(f\"Created {interaction_count} interaction features\")\n    return interaction_df\n\ndef add_features(df):\n    \"\"\"Create comprehensive features for market microstructure\"\"\"\n    print(\"Engineering features...\")\n    \n    # Basic interactions\n    df['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    df['bid_ask_ratio'] = df['bid_qty'] / (df['ask_qty'] + 1e-8)\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-8)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-8)\n    \n    # Pressure indicators\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-8)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-8)\n    df['net_pressure'] = df['buying_pressure'] - df['selling_pressure']\n    \n    # Liquidity features\n    df['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    df['liquidity_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_liquidity'] + 1e-8)\n    df['liquidity_ratio'] = df['total_liquidity'] / (df['volume'] + 1e-8)\n    \n    # Volume transformations\n    df['log_volume'] = np.log1p(df['volume'])\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    \n    # Market microstructure\n    df['kyle_lambda'] = df['order_flow_imbalance'] / (df['sqrt_volume'] + 1e-8)\n    df['vpin'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-8)\n    \n    # Additional useful features\n    df['effective_spread'] = 2 * np.abs(df['order_flow_imbalance']) * df['bid_ask_spread']\n    df['realized_spread'] = df['bid_ask_spread'] * df['vpin']\n    df['price_impact'] = df['kyle_lambda'] * df['volume']\n    df['trade_intensity'] = df['volume'] / (df['total_liquidity'] + 1e-8)\n    \n    # Handle infinities and NaN\n    df = df.replace([np.inf, -np.inf], np.nan)\n    df = df.fillna(0)\n    \n    return df\n\ndef select_features_by_importance(X_train, y_train, feature_names, threshold=0.01):\n    \"\"\"Select features based on importance scores\"\"\"\n    print(\"Calculating feature importance...\")\n    \n    # Train a quick model to get feature importance\n    params = {\n        'n_estimators': 100,\n        'max_depth': 6,\n        'learning_rate': 0.1,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'random_state': 42,\n        'n_jobs': -1,\n        'verbosity': 0\n    }\n    \n    model = XGBRegressor(**params)\n    model.fit(X_train, y_train)\n    \n    # Get feature importance\n    importance = model.feature_importances_\n    \n    # Create importance DataFrame\n    importance_df = pd.DataFrame({\n        'feature': feature_names,\n        'importance': importance\n    }).sort_values('importance', ascending=False)\n    \n    # Select features above threshold\n    selected_features = importance_df[importance_df['importance'] > threshold]['feature'].tolist()\n    \n    # Always include critical features\n    critical_features = ['order_flow_imbalance', 'kyle_lambda', 'vpin', 'volume', \n                        'bid_ask_spread', 'liquidity_imbalance', 'buying_pressure']\n    \n    for feat in critical_features:\n        if feat in feature_names and feat not in selected_features:\n            selected_features.append(feat)\n    \n    print(f\"Selected {len(selected_features)} features with importance > {threshold}\")\n    \n    return selected_features, importance_df\n\n# ====================== XGBoost Anti-overfitting ======================\n\nclass AntiOverfitXGB:\n    def __init__(self, base_params=None):\n        self.base_params = base_params or {\n            'tree_method': 'hist',\n            'n_estimators': 500,\n            'learning_rate': 0.01,\n            'max_depth': 6,\n            'random_state': 42,\n            'n_jobs': -1,\n            'verbosity': 0\n        }\n        \n    def train_overfit_model(self, X, y, overfit_direction='high'):\n        \"\"\"Train a model designed to overfit in a specific direction\"\"\"\n        params = self.base_params.copy()\n        \n        if overfit_direction == 'high':\n            params.update({\n                'max_depth': 12,\n                'min_child_weight': 1,\n                'subsample': 0.9,\n                'colsample_bytree': 0.9,\n                'reg_alpha': 0.001,\n                'reg_lambda': 0.001,\n                'learning_rate': 0.05\n            })\n        else:\n            params.update({\n                'max_depth': 3,\n                'min_child_weight': 50,\n                'subsample': 0.5,\n                'colsample_bytree': 0.5,\n                'reg_alpha': 10,\n                'reg_lambda': 10,\n                'learning_rate': 0.001\n            })\n        \n        model = XGBRegressor(**params)\n        model.fit(X, y)\n        return model\n    \n    def identify_overfit_samples(self, X, y, n_folds=5):\n        \"\"\"Identify samples where model tends to overfit\"\"\"\n        kf = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n        \n        errors = np.zeros(len(X))\n        predictions = np.zeros(len(X))\n        \n        for fold, (train_idx, valid_idx) in enumerate(kf.split(X)):\n            X_fold_train = X.iloc[train_idx] if hasattr(X, 'iloc') else X[train_idx]\n            y_fold_train = y.iloc[train_idx] if hasattr(y, 'iloc') else y[train_idx]\n            X_fold_valid = X.iloc[valid_idx] if hasattr(X, 'iloc') else X[valid_idx]\n            y_fold_valid = y.iloc[valid_idx] if hasattr(y, 'iloc') else y[valid_idx]\n            \n            model = self.train_overfit_model(X_fold_train, y_fold_train, 'high')\n            pred = model.predict(X_fold_valid)\n            predictions[valid_idx] = pred\n            errors[valid_idx] = np.abs(pred - y_fold_valid)\n        \n        error_threshold = np.percentile(errors, 75)\n        overfit_mask = errors > error_threshold\n        \n        return overfit_mask, predictions, errors\n    \n    def train_adversarial_ensemble(self, X, y, X_test):\n        \"\"\"Train ensemble with models that overfit in opposite directions\"\"\"\n        print(\"Training high-overfitting model...\")\n        model_high = self.train_overfit_model(X, y, 'high')\n        pred_high_train = model_high.predict(X)\n        pred_high_test = model_high.predict(X_test)\n        \n        print(\"Training low-overfitting model...\")\n        model_low = self.train_overfit_model(X, y, 'low')\n        pred_low_train = model_low.predict(X)\n        pred_low_test = model_low.predict(X_test)\n        \n        print(\"Training residual model...\")\n        residuals = y - (pred_high_train + pred_low_train) / 2\n        params = self.base_params.copy()\n        params.update({\n            'max_depth': 6,\n            'learning_rate': 0.02,\n            'subsample': 0.7,\n            'colsample_bytree': 0.7\n        })\n        model_residual = XGBRegressor(**params)\n        model_residual.fit(X, residuals)\n        pred_residual_train = model_residual.predict(X)\n        pred_residual_test = model_residual.predict(X_test)\n        \n        final_train = (pred_high_train + pred_low_train) / 2 + pred_residual_train\n        final_test = (pred_high_test + pred_low_test) / 2 + pred_residual_test\n        \n        return final_train, final_test\n    \n    def train_with_sample_weights(self, X, y, overfit_mask):\n        \"\"\"Train model with reduced weights on overfit-prone samples\"\"\"\n        sample_weights = np.ones(len(X))\n        sample_weights[overfit_mask] = 0.5\n        \n        params = self.base_params.copy()\n        params.update({\n            'max_depth': 8,\n            'learning_rate': 0.02,\n            'subsample': 0.8,\n            'colsample_bytree': 0.8,\n            'reg_alpha': 1,\n            'reg_lambda': 1\n        })\n        \n        model = XGBRegressor(**params)\n        model.fit(X, y, sample_weight=sample_weights)\n        return model\n\n# ====================== Ensemble Optimization ======================\n\ndef optimize_ensemble_weights(predictions, y_true, method='advanced'):\n    \"\"\"Optimize ensemble weights using multiple methods\"\"\"\n    n_models = len(predictions)\n    \n    def objective(weights):\n        weights = weights / weights.sum()\n        blended = np.sum([w * p for w, p in zip(weights, predictions)], axis=0)\n        return -pearsonr(y_true, blended)[0]\n    \n    best_weights = None\n    best_score = float('inf')\n    \n    # Method 1: Equal weights\n    equal_weights = np.ones(n_models) / n_models\n    equal_score = objective(equal_weights)\n    if equal_score < best_score:\n        best_score = equal_score\n        best_weights = equal_weights\n    \n    # Method 2: SLSQP\n    constraints = {'type': 'eq', 'fun': lambda w: np.sum(w) - 1}\n    bounds = [(0, 1) for _ in range(n_models)]\n    result = minimize(objective, equal_weights, method='SLSQP', \n                     bounds=bounds, constraints=constraints)\n    if result.success and result.fun < best_score:\n        best_score = result.fun\n        best_weights = result.x\n    \n    # Method 3: Differential Evolution\n    if method == 'advanced':\n        try:\n            result_de = differential_evolution(objective, bounds, seed=42, maxiter=100)\n            de_weights = result_de.x / result_de.x.sum()\n            if result_de.fun < best_score:\n                best_score = result_de.fun\n                best_weights = de_weights\n        except:\n            pass\n    \n    return best_weights\n\n# ====================== Main Pipeline ======================\n\nprint(\"\\nLoading data...\")\ntrain = pd.read_parquet(CFG.train_path)\ntest = pd.read_parquet(CFG.test_path)\nsubmission = pd.read_csv(CFG.sample_sub_path)\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Test shape: {test.shape}\")\n\n# Create proprietary features\ntrain = create_proprietary_x_variables(train, CFG.n_proprietary_features)\ntest = create_proprietary_x_variables(test, CFG.n_proprietary_features)\n\n# Feature engineering\ntrain = add_features(train)\ntest = add_features(test)\n\n# Memory optimization\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\n# Select base features\nselected_x_features = [\n    \"X752\", \"X287\", \"X298\", \"X759\", \"X302\", \"X55\", \"X56\", \"X52\", \"X303\", \"X51\",\n    \"X344\", \"X598\", \"X385\", \"X603\", \"X674\", \"X415\", \"X345\", \"X137\", \"X174\", \"X178\"\n]\n\n# Add proprietary features\nproprietary_features = [f\"X_prop_{i}\" for i in range(1, CFG.n_proprietary_features + 1)]\nselected_x_features.extend(proprietary_features)\n\n# Get all X features and add more if needed\nall_x_features = [col for col in train.columns if col.startswith('X') and col[1:].isdigit()]\nadditional_x = [f for f in all_x_features if f not in selected_x_features][:CFG.max_x_features - len(selected_x_features)]\nselected_x_features.extend(additional_x)\n\navailable_x_features = [f for f in selected_x_features if f in train.columns]\n\n# Market and engineered features\nmarket_features = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\nengineered_features = [col for col in train.columns if col in [\n    'bid_ask_spread', 'bid_ask_ratio', 'buy_sell_ratio', 'order_flow_imbalance',\n    'buying_pressure', 'selling_pressure', 'net_pressure', 'total_liquidity', \n    'liquidity_imbalance', 'liquidity_ratio', 'log_volume', 'sqrt_volume',\n    'kyle_lambda', 'vpin', 'effective_spread', 'realized_spread', \n    'price_impact', 'trade_intensity'\n]]\n\n# Combine base features\nbase_selected_features = market_features + available_x_features + engineered_features\nbase_selected_features = list(dict.fromkeys(base_selected_features))\nbase_selected_features = [f for f in base_selected_features if f in train.columns]\n\nprint(f\"\\nBase features: {len(base_selected_features)}\")\n\n# Create interaction features\ninteraction_df_train = create_interaction_features(train, base_selected_features, CFG.n_interaction_features)\ninteraction_df_test = create_interaction_features(test, base_selected_features, CFG.n_interaction_features)\n\n# Add interaction features\nfor col in interaction_df_train.columns:\n    train[col] = interaction_df_train[col]\n    test[col] = interaction_df_test[col]\n\n# All features before selection\nall_features = base_selected_features + list(interaction_df_train.columns)\nprint(f\"Total features before selection: {len(all_features)}\")\n\n# Prepare data for feature selection\nX_train_all = train[all_features]\ny_train = train['label']\nX_test_all = test[all_features]\n\n# Feature selection\nif CFG.use_feature_selection:\n    selected_features, importance_df = select_features_by_importance(\n        X_train_all, y_train, all_features, CFG.feature_selection_threshold\n    )\n    print(f\"\\nTop 10 features by importance:\")\n    print(importance_df.head(10))\nelse:\n    selected_features = all_features\n\nX_train = X_train_all[selected_features]\nX_test = X_test_all[selected_features]\n\nprint(f\"\\nFinal features after selection: {len(selected_features)}\")\n\n# Initialize storage for predictions\nall_predictions_train = []\nall_predictions_test = []\nall_scores = []\nmodel_names = []\n\n# Initialize anti-overfitting trainer\nanti_overfit = AntiOverfitXGB()\n\n# ========== XGBoost Models ==========\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Training XGBoost Models\")\nprint(\"=\"*60)\n\n# Strategy 1: Standard scaling with adversarial ensemble\nprint(\"\\nStrategy 1: XGBoost Adversarial Ensemble (Standard Scaling)\")\nprint(\"-\"*60)\n\nscaler = StandardScaler()\nX_train_scaled = pd.DataFrame(\n    scaler.fit_transform(X_train),\n    columns=selected_features,\n    index=X_train.index\n)\nX_test_scaled = pd.DataFrame(\n    scaler.transform(X_test),\n    columns=selected_features,\n    index=X_test.index\n)\n\n# Identify overfit-prone samples\nprint(\"\\nIdentifying overfit-prone samples...\")\noverfit_mask, oof_predictions, errors = anti_overfit.identify_overfit_samples(\n    X_train_scaled, y_train, n_folds=5\n)\nprint(f\"Found {overfit_mask.sum()} overfit-prone samples ({100*overfit_mask.mean():.1f}%)\")\n\n# Train adversarial ensemble\nprint(\"\\nTraining adversarial ensemble...\")\nadv_train, adv_test = anti_overfit.train_adversarial_ensemble(\n    X_train_scaled, y_train, X_test_scaled\n)\n\nadv_score = pearsonr(y_train, adv_train)[0]\nprint(f\"XGB Adversarial ensemble score: {adv_score:.4f}\")\n\nall_predictions_train.append(adv_train)\nall_predictions_test.append(adv_test)\nall_scores.append(adv_score)\nmodel_names.append('xgb_adversarial_standard')\n\n# Train with sample weights\nprint(\"\\nTraining with adjusted sample weights...\")\nweighted_model = anti_overfit.train_with_sample_weights(\n    X_train_scaled, y_train, overfit_mask\n)\nweighted_train = weighted_model.predict(X_train_scaled)\nweighted_test = weighted_model.predict(X_test_scaled)\n\nweighted_score = pearsonr(y_train, weighted_train)[0]\nprint(f\"XGB Weighted model score: {weighted_score:.4f}\")\n\nall_predictions_train.append(weighted_train)\nall_predictions_test.append(weighted_test)\nall_scores.append(weighted_score)\nmodel_names.append('xgb_weighted_standard')\n\n# Strategy 2: Robust scaling with diverse models\nprint(\"\\nStrategy 2: XGBoost Diverse Models (Robust Scaling)\")\nprint(\"-\"*60)\n\nscaler_robust = RobustScaler()\nX_train_robust = pd.DataFrame(\n    scaler_robust.fit_transform(X_train),\n    columns=selected_features,\n    index=X_train.index\n)\nX_test_robust = pd.DataFrame(\n    scaler_robust.transform(X_test),\n    columns=selected_features,\n    index=X_test.index\n)\n\n# Train diverse models\ndiverse_configs = [\n    {\n        'name': 'conservative',\n        'params': {\n            'n_estimators': 800,\n            'max_depth': 5,\n            'learning_rate': 0.012,\n            'subsample': 0.65,\n            'colsample_bytree': 0.65,\n            'reg_alpha': 2.5,\n            'reg_lambda': 2.5,\n            'min_child_weight': 25,\n            'gamma': 0.3\n        }\n    },\n    {\n        'name': 'balanced',\n        'params': {\n            'n_estimators': 600,\n            'max_depth': 7,\n            'learning_rate': 0.018,\n            'subsample': 0.75,\n            'colsample_bytree': 0.75,\n            'reg_alpha': 1.0,\n            'reg_lambda': 1.0,\n            'min_child_weight': 10,\n            'gamma': 0.1\n        }\n    },\n    {\n        'name': 'aggressive',\n        'params': {\n            'n_estimators': 500,\n            'max_depth': 9,\n            'learning_rate': 0.025,\n            'subsample': 0.85,\n            'colsample_bytree': 0.85,\n            'reg_alpha': 0.3,\n            'reg_lambda': 0.3,\n            'min_child_weight': 5,\n            'gamma': 0.05\n        }\n    }\n]\n\nfor config in diverse_configs:\n    print(f\"\\nTraining {config['name']} model...\")\n    params = anti_overfit.base_params.copy()\n    params.update(config['params'])\n    \n    model = XGBRegressor(**params)\n    model.fit(X_train_robust, y_train)\n    \n    pred_train = model.predict(X_train_robust)\n    pred_test = model.predict(X_test_robust)\n    score = pearsonr(y_train, pred_train)[0]\n    \n    print(f\"  Score: {score:.4f}\")\n    \n    all_predictions_train.append(pred_train)\n    all_predictions_test.append(pred_test)\n    all_scores.append(score)\n    model_names.append(f'xgb_{config[\"name\"]}_robust')\n\n# ========== GANDALF Models ==========\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Training GANDALF Models\")\nprint(\"=\"*60)\n\n# Prepare data for GANDALF\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\ntrain_idx, val_idx = next(kf.split(X_train_scaled))\nX_train_gandalf = X_train_scaled.iloc[train_idx].values\ny_train_gandalf = y_train.iloc[train_idx].values\nX_val_gandalf = X_train_scaled.iloc[val_idx].values\ny_val_gandalf = y_train.iloc[val_idx].values\n\n# Train GANDALF with anti-overfitting\nprint(\"\\nTraining GANDALF with anti-overfitting discriminator...\")\ngandalf_trainer = GANDALFTrainer(X_train_gandalf.shape[1], CFG)\n\n# Train main GANDALF model\ngandalf_trainer.train_gandalf(X_train_gandalf, y_train_gandalf, X_val_gandalf, y_val_gandalf)\n\n# Train anti-overfitting discriminator\ngandalf_trainer.train_anti_overfit_discriminator(\n    X_train_gandalf, y_train_gandalf, X_val_gandalf, y_val_gandalf\n)\n\n# Get predictions\ngandalf_train_pred = gandalf_trainer.predict(X_train_scaled.values)\ngandalf_test_pred = gandalf_trainer.predict(X_test_scaled.values)\n\n# Get overfitting scores\noverfit_scores = gandalf_trainer.get_overfit_scores(X_train_scaled.values)\nprint(f\"Average overfitting score: {overfit_scores.mean():.4f}\")\n\ngandalf_score = pearsonr(y_train, gandalf_train_pred)[0]\nprint(f\"GANDALF model score: {gandalf_score:.4f}\")\n\nall_predictions_train.append(gandalf_train_pred)\nall_predictions_test.append(gandalf_test_pred)\nall_scores.append(gandalf_score)\nmodel_names.append('gandalf_with_discriminator')\n\n# Train GANDALF variant with different architecture\nprint(\"\\nTraining GANDALF variant (smaller architecture)...\")\nCFG_small = CFG()\nCFG_small.gandalf_hidden_dims = [128, 64, 32]\nCFG_small.gandalf_dropout = 0.4\n\ngandalf_small_trainer = GANDALFTrainer(X_train_gandalf.shape[1], CFG_small)\ngandalf_small_trainer.train_gandalf(X_train_gandalf, y_train_gandalf, X_val_gandalf, y_val_gandalf)\n\ngandalf_small_train_pred = gandalf_small_trainer.predict(X_train_scaled.values)\ngandalf_small_test_pred = gandalf_small_trainer.predict(X_test_scaled.values)\n\ngandalf_small_score = pearsonr(y_train, gandalf_small_train_pred)[0]\nprint(f\"GANDALF small model score: {gandalf_small_score:.4f}\")\n\nall_predictions_train.append(gandalf_small_train_pred)\nall_predictions_test.append(gandalf_small_test_pred)\nall_scores.append(gandalf_small_score)\nmodel_names.append('gandalf_small')\n\n# ========== Time Windows (for XGBoost) ==========\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Training Time-Window Models\")\nprint(\"=\"*60)\n\nwindows = [\n    {'name': 'recent_70', 'start': int(0.7 * len(train)), 'end': len(train)},\n    {'name': 'recent_50', 'start': int(0.5 * len(train)), 'end': len(train)},\n    {'name': 'middle', 'start': int(0.3 * len(train)), 'end': int(0.7 * len(train))}\n]\n\nfor window in windows:\n    print(f\"\\nTraining on {window['name']} window...\")\n    X_window = X_train_scaled.iloc[window['start']:window['end']]\n    y_window = y_train.iloc[window['start']:window['end']]\n    \n    params = anti_overfit.base_params.copy()\n    params.update({\n        'max_depth': 7,\n        'learning_rate': 0.02,\n        'subsample': 0.75,\n        'colsample_bytree': 0.75,\n        'n_estimators': 600\n    })\n    \n    model = XGBRegressor(**params)\n    model.fit(X_window, y_window)\n    \n    pred_test = model.predict(X_test_scaled)\n    \n    # For scoring, predict on the window\n    pred_window = model.predict(X_window)\n    window_score = pearsonr(y_window, pred_window)[0]\n    \n    print(f\"  Window score: {window_score:.4f}\")\n    \n    all_predictions_test.append(pred_test)\n    all_predictions_train.append(np.zeros_like(y_train))  # Placeholder\n    all_scores.append(window_score)\n    model_names.append(f'xgb_window_{window[\"name\"]}')\n\n# ========== Model Summary ==========\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Model Performance Summary\")\nprint(\"=\"*60)\nfor name, score in zip(model_names, all_scores):\n    model_type = \"XGBoost\" if \"xgb\" in name else \"GANDALF\"\n    print(f\"{name:30s} ({model_type:7s}): {score:.4f}\")\n\n# ========== Ensemble Optimization ==========\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"Creating Optimized XGBoost + GANDALF Ensemble\")\nprint(\"=\"*60)\n\n# Use models with valid scores for optimization\nvalid_indices = [i for i, score in enumerate(all_scores) if score > 0.5]\nvalid_predictions_train = [all_predictions_train[i] for i in valid_indices]\nvalid_model_names = [model_names[i] for i in valid_indices]\n\n# Optimize weights\nprint(\"\\nOptimizing ensemble weights...\")\noptimal_weights = optimize_ensemble_weights(valid_predictions_train, y_train, method='advanced')\n\nprint(\"\\nOptimal weights:\")\nfor name, weight in zip(valid_model_names, optimal_weights):\n    if weight > 0.01:\n        model_type = \"XGBoost\" if \"xgb\" in name else \"GANDALF\"\n        print(f\"  {name:30s} ({model_type:7s}): {weight:.3f}\")\n\n# Create final ensemble\nfinal_predictions = np.zeros_like(all_predictions_test[0])\n\n# Apply optimized weights to valid models\nfor i, idx in enumerate(valid_indices):\n    final_predictions += optimal_weights[i] * all_predictions_test[idx]\n\n# Add window predictions with fixed weight\nwindow_indices = [i for i, name in enumerate(model_names) if 'window' in name]\nif window_indices:\n    window_avg = np.mean([all_predictions_test[i] for i in window_indices], axis=0)\n    final_predictions = 0.7 * final_predictions + 0.3 * window_avg\n\n# Post-processing\np1, p99 = np.percentile(y_train, [1, 99])\nfinal_predictions = np.clip(final_predictions, p1, p99)\n\n# Create submission\nsubmission['prediction'] = final_predictions\nsubmission.to_csv('submission_xgb_gandalf.csv', index=False)\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"Submission saved to submission_xgb_gandalf.csv\")\nprint(submission.head())\n\n# Save detailed predictions and model analysis\npredictions_df = pd.DataFrame({\n    'xgb_adversarial': all_predictions_test[0],\n    'xgb_weighted': all_predictions_test[1],\n    'gandalf_main': all_predictions_test[model_names.index('gandalf_with_discriminator')],\n    'final': final_predictions\n})\npredictions_df.to_csv('prediction_components_xgb_gandalf.csv', index=False)\n\nprint(\"\\nPrediction statistics:\")\nprint(predictions_df.describe())\n\n# Model diversity analysis\nprint(\"\\n\" + \"=\"*60)\nprint(\"Model Diversity Analysis\")\nprint(\"=\"*60)\n\n# Separate XGBoost and GANDALF models\nxgb_indices = [i for i, name in enumerate(model_names) if 'xgb' in name and i in valid_indices]\ngandalf_indices = [i for i, name in enumerate(model_names) if 'gandalf' in name and i in valid_indices]\n\nif xgb_indices and gandalf_indices:\n    xgb_preds = [all_predictions_train[i] for i in xgb_indices]\n    gandalf_preds = [all_predictions_train[i] for i in gandalf_indices]\n    \n    # Inter-model correlation\n    inter_corr = np.corrcoef(\n        np.mean(xgb_preds, axis=0),\n        np.mean(gandalf_preds, axis=0)\n    )[0, 1]\n    print(f\"XGBoost-GANDALF correlation: {inter_corr:.4f}\")\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"XGBoost + GANDALF pipeline completed successfully!\")\nprint(f\"Total models trained: {len(model_names)}\")\nprint(f\"XGBoost models: {len([n for n in model_names if 'xgb' in n])}\")\nprint(f\"GANDALF models: {len([n for n in model_names if 'gandalf' in n])}\")\nprint(f\"Effective models in ensemble: {len(valid_indices)}\")\nprint(\"=\"*80)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}