{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport shutil\nfrom pathlib import Path\n\n# =========================\n# Create Working Directories\n# =========================\n# Create main module directory and subdirectories\nMODULE_DIR = Path(\"/kaggle/working/xgb_saint_backbone\")\nSUBMODELS_DIR = MODULE_DIR / \"submodels\"\nFINAL_SUBMISSIONS_DIR = MODULE_DIR / \"final_submissions\"\nMODEL_CHECKPOINTS_DIR = MODULE_DIR / \"model_checkpoints\"\nSAINT_RESULTS_DIR = MODULE_DIR / \"saint_results\"\n\n# Create all directories\nfor directory in [MODULE_DIR, SUBMODELS_DIR, FINAL_SUBMISSIONS_DIR, MODEL_CHECKPOINTS_DIR, SAINT_RESULTS_DIR]:\n    directory.mkdir(parents=True, exist_ok=True)\n    print(f\"Created directory: {directory}\")\n\n# Input data files are available in the read-only \"../input/\" directory\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nfrom sklearn.model_selection import KFold, train_test_split\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport numpy as np\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.feature_selection import mutual_info_regression, SelectKBest\nfrom tqdm import tqdm\nimport random\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\n# Deep Learning imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset, Dataset\nimport torch.nn.functional as F\nfrom torch.cuda.amp import autocast, GradScaler\nfrom torch.optim.lr_scheduler import CosineAnnealingLR, OneCycleLR\nimport matplotlib.pyplot as plt\nfrom collections import defaultdict\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# =========================\n# Configuration\n# =========================\nclass Config:\n    TRAIN_PATH = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    TEST_PATH = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    SUBMISSION_PATH = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n    FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\", \"bid_qty\",\n        \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\", \"X888\", \"X421\", \"X333\",\"X817\", \n        \"X586\",  \"X292\"\n    ]\n    \n    # Extended features for SAINT (selected high-importance features)\n    SAINT_FEATURES = [\n        \"X863\", \"X856\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n        \"X415\", \"X345\", \"X855\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        \"X425\", \"X132\", \"X691\", \"X593\", \"X377\", \"X285\", \"X126\", \"X419\", \"X604\",\n        \"X84\", \"X138\", \"X413\", \"X291\", \"X40\", \"X123\", \"X81\", \"X853\", \"X854\",\n        \"X777\", \"X219\", \"X776\", \"X180\", \"X781\", \"X445\", \"X444\", \"X384\", \"X466\",\n        \"X95\", \"X583\", \"X272\", \"X137\", \"X533\", \"X758\", \"X279\", \"X297\",\n        \"X21\", \"X20\", \"X28\", \"X29\", \"X19\", \"X27\", \"X22\", \"X198\", \"X89\", \"X90\",\n        \"X888\", \"X421\", \"X333\", \"X817\", \"X586\", \"X292\", \"X344\", \"X532\",\n        \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n    ]\n\n    LABEL_COLUMN = \"label\"\n    N_FOLDS = 3\n    RANDOM_STATE = 42\n    OUTLIER_FRACTION = 0.001  # 0.1% of records\n\nXGB_PARAMS = {\n    \"tree_method\": \"hist\",\n    \"device\": \"gpu\",\n    \"colsample_bylevel\": 0.4778,\n    \"colsample_bynode\": 0.3628,\n    \"colsample_bytree\": 0.7107,\n    \"gamma\": 1.7095,\n    \"learning_rate\": 0.02213,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"subsample\": 0.06567,\n    \"reg_alpha\": 39.3524,\n    \"reg_lambda\": 75.4484,\n    \"verbosity\": 0,\n    \"random_state\": Config.RANDOM_STATE,\n    \"n_jobs\": -1\n}\n\nLEARNERS = [\n    {\"name\": \"xgb\", \"Estimator\": XGBRegressor, \"params\": XGB_PARAMS}\n]\n\n# =========================\n# Deep Learning Components\n# =========================\ndef set_seed(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\n# =========================\n# SAINT Components\n# =========================\nclass GEGLU(nn.Module):\n    \"\"\"Gated Linear Unit with GELU activation\"\"\"\n    def forward(self, x):\n        x, gates = x.chunk(2, dim=-1)\n        return x * F.gelu(gates)\n\nclass FeedForwardGEGLU(nn.Module):\n    \"\"\"Feed Forward block with GEGLU activation\"\"\"\n    def __init__(self, d_model, d_ff, dropout=0.1):\n        super().__init__()\n        self.linear1 = nn.Linear(d_model, d_ff * 2)\n        self.geglu = GEGLU()\n        self.dropout = nn.Dropout(dropout)\n        self.linear2 = nn.Linear(d_ff, d_model)\n        \n    def forward(self, x):\n        x = self.linear1(x)\n        x = self.geglu(x)\n        x = self.dropout(x)\n        x = self.linear2(x)\n        return x\n\nclass MultiHeadAttention(nn.Module):\n    \"\"\"Multi-Head Attention mechanism\"\"\"\n    def __init__(self, d_model, n_heads, dropout=0.1):\n        super().__init__()\n        assert d_model % n_heads == 0\n        \n        self.d_model = d_model\n        self.n_heads = n_heads\n        self.d_k = d_model // n_heads\n        self.scale = self.d_k ** -0.5\n        \n        self.w_q = nn.Linear(d_model, d_model, bias=False)\n        self.w_k = nn.Linear(d_model, d_model, bias=False)\n        self.w_v = nn.Linear(d_model, d_model, bias=False)\n        self.w_o = nn.Linear(d_model, d_model)\n        \n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, query, key, value, mask=None):\n        batch_size = query.size(0)\n        \n        # Linear transformations\n        Q = self.w_q(query).view(batch_size, -1, self.n_heads, self.d_k).transpose(1, 2)\n        K = self.w_k(key).view(batch_size, -1, self.n_heads, self.d_k).transpose(1, 2)\n        V = self.w_v(value).view(batch_size, -1, self.n_heads, self.d_k).transpose(1, 2)\n        \n        # Attention scores\n        scores = torch.matmul(Q, K.transpose(-2, -1)) * self.scale\n        \n        if mask is not None:\n            scores = scores.masked_fill(mask == 0, -1e9)\n        \n        # Attention weights\n        attn = F.softmax(scores, dim=-1)\n        attn = self.dropout(attn)\n        \n        # Apply attention to values\n        context = torch.matmul(attn, V)\n        \n        # Concatenate heads\n        context = context.transpose(1, 2).contiguous().view(\n            batch_size, -1, self.d_model\n        )\n        \n        # Final linear layer\n        output = self.w_o(context)\n        \n        return output, attn\n\nclass SAINTLayer(nn.Module):\n    \"\"\"Single SAINT layer with self-attention and intersample attention\"\"\"\n    def __init__(self, d_model, n_heads, d_ff, dropout=0.1, use_intersample=True):\n        super().__init__()\n        self.use_intersample = use_intersample\n        \n        # Self-attention block\n        self.self_attention = MultiHeadAttention(d_model, n_heads, dropout)\n        self.norm1 = nn.LayerNorm(d_model)\n        self.dropout1 = nn.Dropout(dropout)\n        \n        # Intersample attention block (optional)\n        if use_intersample:\n            self.intersample_attention = MultiHeadAttention(d_model, n_heads, dropout)\n            self.norm2 = nn.LayerNorm(d_model)\n            self.dropout2 = nn.Dropout(dropout)\n        \n        # Feed forward block\n        self.ff = FeedForwardGEGLU(d_model, d_ff, dropout)\n        self.norm3 = nn.LayerNorm(d_model)\n        self.dropout3 = nn.Dropout(dropout)\n        \n    def forward(self, x, x_intersample=None):\n        # Self-attention\n        attn_output, _ = self.self_attention(x, x, x)\n        x = self.norm1(x + self.dropout1(attn_output))\n        \n        # Intersample attention\n        if self.use_intersample and x_intersample is not None:\n            intersample_output, _ = self.intersample_attention(x, x_intersample, x_intersample)\n            x = self.norm2(x + self.dropout2(intersample_output))\n        \n        # Feed forward\n        ff_output = self.ff(x)\n        x = self.norm3(x + self.dropout3(ff_output))\n        \n        return x\n\nclass CutMix(nn.Module):\n    \"\"\"CutMix augmentation for tabular data\"\"\"\n    def __init__(self, beta=1.0, prob=0.5):\n        super().__init__()\n        self.beta = beta\n        self.prob = prob\n        \n    def forward(self, x, y):\n        if not self.training or np.random.random() > self.prob:\n            return x, y\n            \n        batch_size = x.size(0)\n        lam = np.random.beta(self.beta, self.beta)\n        \n        # Random shuffle for mixing\n        index = torch.randperm(batch_size).to(x.device)\n        \n        # Create mask for features to mix\n        num_features = x.size(1)\n        mask = torch.rand(num_features) < lam\n        mask = mask.float().to(x.device)\n        \n        # Mix features\n        mixed_x = x * mask + x[index] * (1 - mask)\n        \n        # Mix targets\n        mixed_y = y * lam + y[index] * (1 - lam)\n        \n        return mixed_x, mixed_y\n\nclass SAINT(nn.Module):\n    \"\"\"SAINT: Self-Attention and Intersample Attention Transformer\"\"\"\n    def __init__(self, config):\n        super().__init__()\n        \n        self.num_features = config['num_features']\n        self.d_model = config['d_model']\n        self.n_heads = config['n_heads']\n        self.n_layers = config['n_layers']\n        self.d_ff = config['d_ff']\n        self.dropout = config['dropout']\n        self.use_intersample = config.get('use_intersample', True)\n        \n        # Feature embeddings\n        self.feature_embeddings = nn.ModuleList([\n            nn.Sequential(\n                nn.Linear(1, self.d_model),\n                nn.LayerNorm(self.d_model),\n                nn.ReLU(),\n                nn.Dropout(self.dropout)\n            ) for _ in range(self.num_features)\n        ])\n        \n        # Positional embeddings for features\n        self.feature_pos_encoding = nn.Parameter(\n            torch.randn(1, self.num_features, self.d_model) * 0.02\n        )\n        \n        # CLS token for aggregation\n        self.cls_token = nn.Parameter(torch.randn(1, 1, self.d_model) * 0.02)\n        \n        # SAINT layers\n        self.layers = nn.ModuleList([\n            SAINTLayer(\n                self.d_model, \n                self.n_heads, \n                self.d_ff, \n                self.dropout,\n                self.use_intersample and (i % 2 == 1)  # Alternate between self and intersample\n            ) for i in range(self.n_layers)\n        ])\n        \n        # Output head\n        self.head = nn.Sequential(\n            nn.LayerNorm(self.d_model),\n            nn.Linear(self.d_model, self.d_model // 2),\n            nn.ReLU(),\n            nn.Dropout(self.dropout),\n            nn.Linear(self.d_model // 2, 1)\n        )\n        \n        # CutMix augmentation\n        self.cutmix = CutMix(beta=config.get('cutmix_beta', 1.0), \n                            prob=config.get('cutmix_prob', 0.5))\n        \n        # Attention rollout for interpretability\n        self.attention_weights = []\n        \n    def embed_features(self, x):\n        \"\"\"Embed numerical features\"\"\"\n        batch_size = x.shape[0]\n        feature_embeds = []\n        \n        for i in range(self.num_features):\n            feat = x[:, i:i+1]\n            embed = self.feature_embeddings[i](feat)\n            feature_embeds.append(embed)\n        \n        # Stack embeddings\n        embeds = torch.stack(feature_embeds, dim=1)\n        \n        # Add positional encoding\n        embeds = embeds + self.feature_pos_encoding\n        \n        # Add CLS token\n        cls_tokens = self.cls_token.expand(batch_size, -1, -1)\n        embeds = torch.cat([cls_tokens, embeds], dim=1)\n        \n        return embeds\n    \n    def forward(self, x, targets=None):\n        # Apply CutMix augmentation during training\n        if targets is not None and self.training:\n            x, targets = self.cutmix(x, targets)\n        \n        # Embed features\n        x = self.embed_features(x)\n        \n        # Store attention weights\n        self.attention_weights = []\n        \n        # Apply SAINT layers\n        batch_size = x.shape[0]\n        \n        for i, layer in enumerate(self.layers):\n            if layer.use_intersample and batch_size > 1:\n                # Create intersample batch by shifting\n                roll_idx = torch.randint(1, batch_size, (1,)).item()\n                x_intersample = torch.roll(x, shifts=roll_idx, dims=0)\n                x = layer(x, x_intersample)\n            else:\n                x = layer(x)\n        \n        # Extract CLS token representation\n        cls_output = x[:, 0]\n        \n        # Final prediction\n        output = self.head(cls_output)\n        \n        if targets is not None:\n            return output, targets\n        return output\n\nclass TabularDataset(Dataset):\n    \"\"\"Custom dataset for tabular data with SAINT\"\"\"\n    def __init__(self, features, targets=None, transform=None):\n        self.features = torch.FloatTensor(features)\n        self.targets = torch.FloatTensor(targets).unsqueeze(1) if targets is not None else None\n        self.transform = transform\n        \n    def __len__(self):\n        return len(self.features)\n    \n    def __getitem__(self, idx):\n        x = self.features[idx]\n        \n        if self.transform:\n            x = self.transform(x)\n            \n        if self.targets is not None:\n            y = self.targets[idx]\n            return x, y\n        return x\n\n# =========================\n# Feature Engineering\n# =========================\ndef add_features(df):\n    # Original features\n    df['bid_ask_interaction'] = df['bid_qty'] * df['ask_qty']\n    df['bid_buy_interaction'] = df['bid_qty'] * df['buy_qty']\n    df['bid_sell_interaction'] = df['bid_qty'] * df['sell_qty']\n    df['ask_buy_interaction'] = df['ask_qty'] * df['buy_qty']\n    df['ask_sell_interaction'] = df['ask_qty'] * df['sell_qty']\n\n    df['volume_weighted_sell'] = df['sell_qty'] * df['volume']\n    df['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-10)\n    df['selling_pressure'] = df['sell_qty'] / (df['volume'] + 1e-10)\n    df['log_volume'] = np.log1p(df['volume'])\n\n    df['effective_spread_proxy'] = np.abs(df['buy_qty'] - df['sell_qty']) / (df['volume'] + 1e-10)\n    df['bid_ask_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['order_flow_imbalance'] = (df['buy_qty'] - df['sell_qty']) / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['liquidity_ratio'] = (df['bid_qty'] + df['ask_qty']) / (df['volume'] + 1e-10)\n    \n    # === NEW MICROSTRUCTURE FEATURES ===\n    \n    # Price Pressure Indicators\n    df['net_order_flow'] = df['buy_qty'] - df['sell_qty']\n    df['normalized_net_flow'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['buying_pressure'] = df['buy_qty'] / (df['volume'] + 1e-10)\n    df['volume_weighted_buy'] = df['buy_qty'] * df['volume']\n    \n    # Liquidity Depth Measures\n    df['total_depth'] = df['bid_qty'] + df['ask_qty']\n    df['depth_imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['relative_spread'] = np.abs(df['bid_qty'] - df['ask_qty']) / (df['total_depth'] + 1e-10)\n    df['log_depth'] = np.log1p(df['total_depth'])\n    \n    # Order Flow Toxicity Proxies\n    df['kyle_lambda'] = np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['flow_toxicity'] = np.abs(df['order_flow_imbalance']) * df['volume']\n    df['aggressive_flow_ratio'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    \n    # Market Activity Indicators\n    df['volume_depth_ratio'] = df['volume'] / (df['total_depth'] + 1e-10)\n    df['activity_intensity'] = (df['buy_qty'] + df['sell_qty']) / (df['volume'] + 1e-10)\n    df['log_buy_qty'] = np.log1p(df['buy_qty'])\n    df['log_sell_qty'] = np.log1p(df['sell_qty'])\n    df['log_bid_qty'] = np.log1p(df['bid_qty'])\n    df['log_ask_qty'] = np.log1p(df['ask_qty'])\n    \n    # Microstructure Volatility Proxies\n    df['realized_spread_proxy'] = 2 * np.abs(df['net_order_flow']) / (df['volume'] + 1e-10)\n    df['price_impact_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10)\n    df['quote_volatility_proxy'] = np.abs(df['depth_imbalance'])\n    \n    # Complex Interaction Terms\n    df['flow_depth_interaction'] = df['net_order_flow'] * df['total_depth']\n    df['imbalance_volume_interaction'] = df['order_flow_imbalance'] * df['volume']\n    df['depth_volume_interaction'] = df['total_depth'] * df['volume']\n    df['buy_sell_spread'] = np.abs(df['buy_qty'] - df['sell_qty'])\n    df['bid_ask_spread'] = np.abs(df['bid_qty'] - df['ask_qty'])\n    \n    # Information Asymmetry Measures\n    df['trade_informativeness'] = df['net_order_flow'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    df['execution_shortfall_proxy'] = df['buy_sell_spread'] / (df['volume'] + 1e-10)\n    df['adverse_selection_proxy'] = df['net_order_flow'] / (df['total_depth'] + 1e-10) * df['volume']\n    \n    # Market Efficiency Indicators\n    df['fill_probability'] = df['volume'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['execution_rate'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_efficiency'] = df['volume'] / (df['bid_ask_spread'] + 1e-10)\n    \n    # Non-linear Transformations\n    df['sqrt_volume'] = np.sqrt(df['volume'])\n    df['sqrt_depth'] = np.sqrt(df['total_depth'])\n    df['volume_squared'] = df['volume'] ** 2\n    df['imbalance_squared'] = df['order_flow_imbalance'] ** 2\n    \n    # Relative Measures\n    df['bid_ratio'] = df['bid_qty'] / (df['total_depth'] + 1e-10)\n    df['ask_ratio'] = df['ask_qty'] / (df['total_depth'] + 1e-10)\n    df['buy_ratio'] = df['buy_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    df['sell_ratio'] = df['sell_qty'] / (df['buy_qty'] + df['sell_qty'] + 1e-10)\n    \n    # Market Stress Indicators\n    df['liquidity_consumption'] = (df['buy_qty'] + df['sell_qty']) / (df['total_depth'] + 1e-10)\n    df['market_stress'] = df['volume'] / (df['total_depth'] + 1e-10) * np.abs(df['order_flow_imbalance'])\n    df['depth_depletion'] = df['volume'] / (df['bid_qty'] + df['ask_qty'] + 1e-10)\n    \n    # Directional Indicators\n    df['net_buying_ratio'] = df['net_order_flow'] / (df['volume'] + 1e-10)\n    df['directional_volume'] = df['net_order_flow'] * np.log1p(df['volume'])\n    df['signed_volume'] = np.sign(df['net_order_flow']) * df['volume']\n    \n    # Replace infinities and NaNs\n    df = df.replace([np.inf, -np.inf], 0).fillna(0)\n    \n    return df\n\ndef create_time_decay_weights(n: int, decay: float = 0.9) -> np.ndarray:\n    positions = np.arange(n)\n    normalized = positions / (n - 1)\n    weights = decay ** (1.0 - normalized)\n    return weights * n / weights.sum()\n\ndef detect_outliers_and_adjust_weights(X, y, sample_weights, outlier_fraction=0.001):\n    \"\"\"\n    Detect outliers based on prediction residuals and adjust their weights.\n    Only adjusts weights for the top outlier_fraction of records.\n    \"\"\"\n    # Train a simple model to get residuals\n    rf = RandomForestRegressor(n_estimators=50, max_depth=10, random_state=42, n_jobs=-1)\n    rf.fit(X, y, sample_weight=sample_weights)\n    \n    # Calculate residuals\n    predictions = rf.predict(X)\n    residuals = np.abs(y - predictions)\n    \n    # Find threshold for top outlier_fraction\n    n_outliers = max(1, int(len(residuals) * outlier_fraction))\n    threshold = np.sort(residuals)[-n_outliers]\n    \n    # Create outlier mask\n    outlier_mask = residuals >= threshold\n    \n    # Adjust weights for outliers\n    adjusted_weights = sample_weights.copy()\n    \n    if outlier_mask.any():\n        # Calculate weight reduction factor based on residual magnitude\n        outlier_residuals = residuals[outlier_mask]\n        \n        # Normalize residuals to [0, 1] range for outliers\n        min_outlier_res = outlier_residuals.min()\n        max_outlier_res = outlier_residuals.max()\n        \n        if max_outlier_res > min_outlier_res:\n            normalized_residuals = (outlier_residuals - min_outlier_res) / (max_outlier_res - min_outlier_res)\n        else:\n            normalized_residuals = np.ones_like(outlier_residuals)\n        \n        # Reduce weights proportionally (from 0.2 to 0.8 of original weight)\n        # The most extreme outliers get 0.2x weight, least extreme get 0.8x weight\n        weight_factors = 0.8 - 0.6 * normalized_residuals\n        \n        # Apply weight adjustments\n        adjusted_weights[outlier_mask] *= weight_factors\n        \n        print(f\"    Adjusted weights for {n_outliers} outliers ({outlier_fraction*100:.1f}% of data)\")\n    \n    return adjusted_weights\n\ndef load_data():\n    # Load data with all features available\n    all_features = list(set(Config.FEATURES + Config.SAINT_FEATURES))\n    train_df = pd.read_parquet(Config.TRAIN_PATH, columns=all_features + [Config.LABEL_COLUMN])\n    test_df = pd.read_parquet(Config.TEST_PATH, columns=all_features)\n    submission_df = pd.read_csv(Config.SUBMISSION_PATH)\n    print(f\"Loaded data - Train: {train_df.shape}, Test: {test_df.shape}, Submission: {submission_df.shape}\")\n\n    # Add features\n    train_df = add_features(train_df)\n    test_df = add_features(test_df)\n\n    # Update Config.FEATURES with new features\n    Config.FEATURES += [\n        \"log_volume\", 'bid_ask_interaction', 'bid_buy_interaction', 'bid_sell_interaction', \n        'ask_buy_interaction', 'ask_sell_interaction', 'net_order_flow', 'normalized_net_flow',\n        'buying_pressure', 'volume_weighted_buy', 'total_depth', 'depth_imbalance',\n        'relative_spread', 'log_depth', 'kyle_lambda', 'flow_toxicity', 'aggressive_flow_ratio',\n        'volume_depth_ratio', 'activity_intensity', 'log_buy_qty', 'log_sell_qty',\n        'log_bid_qty', 'log_ask_qty', 'realized_spread_proxy', 'price_impact_proxy',\n        'quote_volatility_proxy', 'flow_depth_interaction', 'imbalance_volume_interaction',\n        'depth_volume_interaction', 'buy_sell_spread', 'bid_ask_spread', 'trade_informativeness',\n        'execution_shortfall_proxy', 'adverse_selection_proxy', 'fill_probability',\n        'execution_rate', 'market_efficiency', 'sqrt_volume', 'sqrt_depth', 'volume_squared',\n        'imbalance_squared', 'bid_ratio', 'ask_ratio', 'buy_ratio', 'sell_ratio',\n        'liquidity_consumption', 'market_stress', 'depth_depletion', 'net_buying_ratio',\n        'directional_volume', 'signed_volume'\n    ]\n\n    return train_df.reset_index(drop=True), test_df.reset_index(drop=True), submission_df\n\ndef get_model_slices(n_samples: int):\n    # Original 5 slices\n    base_slices = [\n        {\"name\": \"full_data\", \"cutoff\": 0, \"is_oldest\": False, \"outlier_adjusted\": False},\n        {\"name\": \"last_90pct\", \"cutoff\": int(0.10 * n_samples), \"is_oldest\": False, \"outlier_adjusted\": False},\n        {\"name\": \"last_85pct\", \"cutoff\": int(0.15 * n_samples), \"is_oldest\": False, \"outlier_adjusted\": False},\n        {\"name\": \"last_80pct\", \"cutoff\": int(0.20 * n_samples), \"is_oldest\": False, \"outlier_adjusted\": False},\n        {\"name\": \"oldest_25pct\", \"cutoff\": int(0.25 * n_samples), \"is_oldest\": True, \"outlier_adjusted\": False},\n    ]\n    \n    # Duplicate slices with outlier adjustment\n    outlier_adjusted_slices = []\n    for slice_info in base_slices:\n        adjusted_slice = slice_info.copy()\n        adjusted_slice[\"name\"] = f\"{slice_info['name']}_outlier_adj\"\n        adjusted_slice[\"outlier_adjusted\"] = True\n        outlier_adjusted_slices.append(adjusted_slice)\n    \n    return base_slices + outlier_adjusted_slices\n\n# =========================\n# XGBoost Training and Evaluation\n# =========================\ndef train_and_evaluate_xgboost(train_df, test_df):\n    n_samples = len(train_df)\n    model_slices = get_model_slices(n_samples)\n\n    oof_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(n_samples) for s in model_slices}\n        for learner in LEARNERS\n    }\n    test_preds = {\n        learner[\"name\"]: {s[\"name\"]: np.zeros(len(test_df)) for s in model_slices}\n        for learner in LEARNERS\n    }\n\n    full_weights = create_time_decay_weights(n_samples)\n    kf = KFold(n_splits=Config.N_FOLDS, shuffle=False)\n\n    for fold, (train_idx, valid_idx) in enumerate(kf.split(train_df), start=1):\n        print(f\"\\n--- Fold {fold}/{Config.N_FOLDS} ---\")\n        X_valid = train_df.iloc[valid_idx][Config.FEATURES]\n        y_valid = train_df.iloc[valid_idx][Config.LABEL_COLUMN]\n\n        for s in model_slices:\n            cutoff = s[\"cutoff\"]\n            slice_name = s[\"name\"]\n            is_oldest = s[\"is_oldest\"]\n            outlier_adjusted = s[\"outlier_adjusted\"]\n            \n            if is_oldest:\n                # For oldest data, take only the first 25% of samples\n                subset = train_df.iloc[:cutoff].reset_index(drop=True)\n                rel_idx = train_idx[train_idx < cutoff]\n                # No time decay for oldest data - use uniform weights\n                sw = np.ones(len(rel_idx))\n            else:\n                # For recent data slices, take from cutoff to end\n                subset = train_df.iloc[cutoff:].reset_index(drop=True)\n                rel_idx = train_idx[train_idx >= cutoff] - cutoff\n                sw = create_time_decay_weights(len(subset))[rel_idx] if cutoff > 0 else full_weights[train_idx]\n\n            X_train = subset.iloc[rel_idx][Config.FEATURES]\n            y_train = subset.iloc[rel_idx][Config.LABEL_COLUMN]\n            \n            # Apply outlier detection and weight adjustment if needed\n            if outlier_adjusted and len(X_train) > 100:  # Only apply if we have enough samples\n                sw = detect_outliers_and_adjust_weights(\n                    X_train.values, \n                    y_train.values, \n                    sw, \n                    outlier_fraction=Config.OUTLIER_FRACTION\n                )\n\n            print(f\"  Training slice: {slice_name}, samples: {len(X_train)}\")\n\n            for learner in LEARNERS:\n                model = learner[\"Estimator\"](**learner[\"params\"])\n                model.fit(X_train, y_train, sample_weight=sw, eval_set=[(X_valid, y_valid)], verbose=False)\n\n                if is_oldest:\n                    # For oldest data slice, predict on all validation indices\n                    oof_preds[learner[\"name\"]][slice_name][valid_idx] = model.predict(\n                        train_df.iloc[valid_idx][Config.FEATURES]\n                    )\n                else:\n                    # For recent data slices, handle cutoff logic\n                    mask = valid_idx >= cutoff\n                    if mask.any():\n                        idxs = valid_idx[mask]\n                        oof_preds[learner[\"name\"]][slice_name][idxs] = model.predict(\n                            train_df.iloc[idxs][Config.FEATURES]\n                        )\n                    if cutoff > 0 and (~mask).any():\n                        base_slice_name = slice_name.replace(\"_outlier_adj\", \"\")\n                        if base_slice_name == slice_name:\n                            fallback_slice = \"full_data\"\n                        else:\n                            fallback_slice = \"full_data_outlier_adj\"\n                        oof_preds[learner[\"name\"]][slice_name][valid_idx[~mask]] = oof_preds[learner[\"name\"]][fallback_slice][\n                            valid_idx[~mask]\n                        ]\n\n                test_preds[learner[\"name\"]][slice_name] += model.predict(test_df[Config.FEATURES])\n\n    # Normalize test predictions\n    for learner_name in test_preds:\n        for slice_name in test_preds[learner_name]:\n            test_preds[learner_name][slice_name] /= (Config.N_FOLDS - 1)\n\n    return oof_preds, test_preds, model_slices\n\n# =========================\n# SAINT Training\n# =========================\ndef train_saint_model(model, train_loader, val_loader, config, device):\n    \"\"\"Train SAINT model\"\"\"\n    \n    # Loss and optimizer\n    criterion = nn.HuberLoss(delta=config.get('huber_delta', 1.0))\n    optimizer = optim.AdamW(\n        model.parameters(),\n        lr=config['learning_rate'],\n        weight_decay=config['weight_decay']\n    )\n    \n    # Scheduler\n    total_steps = len(train_loader) * config['num_epochs']\n    scheduler = OneCycleLR(\n        optimizer,\n        max_lr=config['learning_rate'],\n        total_steps=total_steps,\n        pct_start=0.3,\n        anneal_strategy='cos',\n        div_factor=25,\n        final_div_factor=10000\n    )\n    \n    # Training settings\n    best_val_pearson = -np.inf\n    patience_counter = 0\n    patience = config.get('patience', 15)\n    num_epochs = config.get('num_epochs', 50)\n    \n    # Mixed precision\n    use_amp = config.get('use_amp', True) and device.type == 'cuda'\n    scaler = GradScaler() if use_amp else None\n    \n    # Training history\n    train_losses = []\n    val_pearsons = []\n    \n    for epoch in range(num_epochs):\n        # Training\n        model.train()\n        train_loss = 0.0\n        train_batches = 0\n        \n        progress_bar = tqdm(train_loader, desc=f\"Epoch {epoch+1}/{num_epochs}\")\n        for inputs, targets in progress_bar:\n            inputs, targets = inputs.to(device), targets.to(device)\n            \n            optimizer.zero_grad()\n            \n            if use_amp:\n                with autocast():\n                    outputs, mixed_targets = model(inputs, targets)\n                    loss = criterion(outputs, mixed_targets)\n                \n                scaler.scale(loss).backward()\n                \n                # Gradient clipping\n                scaler.unscale_(optimizer)\n                torch.nn.utils.clip_grad_norm_(model.parameters(), config.get('grad_clip', 1.0))\n                \n                scaler.step(optimizer)\n                scaler.update()\n            else:\n                outputs, mixed_targets = model(inputs, targets)\n                loss = criterion(outputs, mixed_targets)\n                \n                loss.backward()\n                torch.nn.utils.clip_grad_norm_(model.parameters(), config.get('grad_clip', 1.0))\n                optimizer.step()\n            \n            scheduler.step()\n            \n            train_loss += loss.item()\n            train_batches += 1\n            \n            progress_bar.set_postfix({\n                'loss': f'{loss.item():.4f}',\n                'lr': f'{scheduler.get_last_lr()[0]:.2e}'\n            })\n        \n        # Validation\n        model.eval()\n        val_loss = 0.0\n        val_preds = []\n        val_targets = []\n        \n        with torch.no_grad():\n            for inputs, targets in val_loader:\n                inputs, targets = inputs.to(device), targets.to(device)\n                \n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                \n                val_loss += loss.item()\n                val_preds.extend(outputs.cpu().numpy().flatten())\n                val_targets.extend(targets.cpu().numpy().flatten())\n        \n        # Metrics\n        avg_train_loss = train_loss / train_batches\n        avg_val_loss = val_loss / len(val_loader)\n        val_pearson = pearsonr(val_targets, val_preds)[0]\n        \n        train_losses.append(avg_train_loss)\n        val_pearsons.append(val_pearson)\n        \n        print(f\"\\nTrain Loss: {avg_train_loss:.4f}, Val Loss: {avg_val_loss:.4f}\")\n        print(f\"Val Pearson: {val_pearson:.4f}\")\n        \n        # Save best model\n        if val_pearson > best_val_pearson:\n            best_val_pearson = val_pearson\n            patience_counter = 0\n            torch.save(model.state_dict(), MODEL_CHECKPOINTS_DIR / f\"best_saint_{config.get('model_id', 0)}.pt\")\n            print(f\"✅ New best model saved! Pearson: {best_val_pearson:.4f}\")\n        else:\n            patience_counter += 1\n            if patience_counter >= patience:\n                print(f\"Early stopping triggered after {epoch+1} epochs\")\n                break\n    \n    # Load best model\n    model.load_state_dict(torch.load(MODEL_CHECKPOINTS_DIR / f\"best_saint_{config.get('model_id', 0)}.pt\"))\n    \n    return model, best_val_pearson, {'train_losses': train_losses, 'val_pearsons': val_pearsons}\n\ndef train_saint(train_df, test_df):\n    print(\"\\n=== Training SAINT Model ===\")\n    \n    # Set seed\n    set_seed(42)\n    \n    # Get SAINT features\n    saint_features = Config.SAINT_FEATURES.copy()\n    \n    # Add selected engineered features\n    engineered_features = [\n        \"log_volume\", 'bid_ask_interaction', 'net_order_flow', 'normalized_net_flow',\n        'buying_pressure', 'total_depth', 'depth_imbalance', 'kyle_lambda', \n        'volume_depth_ratio', 'price_impact_proxy', 'market_stress', \n        'liquidity_consumption', 'order_flow_imbalance', 'bid_ask_imbalance',\n        'realized_spread_proxy', 'aggressive_flow_ratio'\n    ]\n    \n    all_saint_features = saint_features + engineered_features\n    all_saint_features = list(set(all_saint_features))  # Remove duplicates\n    \n    # Ensure all features exist\n    all_saint_features = [f for f in all_saint_features if f in train_df.columns]\n    \n    print(f\"Using {len(all_saint_features)} features for SAINT\")\n    \n    # Use recent data (last 85%)\n    train_size = int(0.85 * len(train_df))\n    train_data = train_df.iloc[-train_size:].reset_index(drop=True)\n    \n    # Split for validation\n    split_idx = int(0.8 * len(train_data))\n    train_split = train_data[:split_idx].copy()\n    val_split = train_data[split_idx:].copy()\n    \n    y_train = train_split[Config.LABEL_COLUMN].values\n    y_val = val_split[Config.LABEL_COLUMN].values\n    \n    X_train = train_split[all_saint_features].values\n    X_val = val_split[all_saint_features].values\n    \n    # Feature selection\n    print(\"\\nSelecting features...\")\n    selector = SelectKBest(score_func=mutual_info_regression, k=min(80, len(all_saint_features)))\n    X_train_selected = selector.fit_transform(X_train, y_train)\n    X_val_selected = selector.transform(X_val)\n    \n    selected_features = [all_saint_features[i] for i in selector.get_support(indices=True)]\n    print(f\"Selected {len(selected_features)} features\")\n    \n    # Transform data\n    print(\"\\nTransforming data...\")\n    transformer = RobustScaler()\n    X_train_transformed = transformer.fit_transform(X_train_selected)\n    X_val_transformed = transformer.transform(X_val_selected)\n    \n    # Create datasets\n    train_dataset = TabularDataset(X_train_transformed, y_train)\n    val_dataset = TabularDataset(X_val_transformed, y_val)\n    \n    # Model configurations for ensemble\n    configs = [\n        {\n            'model_id': 1,\n            'num_features': X_train_transformed.shape[1],\n            'd_model': 128,\n            'n_heads': 8,\n            'n_layers': 6,\n            'd_ff': 256,\n            'dropout': 0.2,\n            'use_intersample': True,\n            'cutmix_beta': 1.0,\n            'cutmix_prob': 0.5,\n            'learning_rate': 0.001,\n            'weight_decay': 0.01,\n            'huber_delta': 1.0,\n            'grad_clip': 1.0,\n            'num_epochs': 40,\n            'patience': 15,\n            'use_amp': True,\n            'batch_size': 256\n        },\n        {\n            'model_id': 2,\n            'd_model': 96,\n            'n_heads': 6,\n            'n_layers': 4,\n            'd_ff': 192,\n            'dropout': 0.3,\n            'use_intersample': True,\n            'cutmix_beta': 0.5,\n            'cutmix_prob': 0.3,\n            'learning_rate': 0.0005,\n            'weight_decay': 0.001,\n            'huber_delta': 0.5,\n            'grad_clip': 2.0,\n            'num_epochs': 40,\n            'patience': 15,\n            'use_amp': True,\n            'batch_size': 512,\n            'num_features': X_train_transformed.shape[1]\n        }\n    ]\n    \n    # Train ensemble of models\n    ensemble_models = []\n    ensemble_scores = []\n    all_histories = []\n    \n    for i, config in enumerate(configs):\n        print(f\"\\n=== Training SAINT Model {i+1}/{len(configs)} ===\")\n        \n        # Create data loaders with specific batch size\n        train_loader = DataLoader(\n            train_dataset, \n            batch_size=config['batch_size'], \n            shuffle=True,\n            num_workers=0,\n            pin_memory=True\n        )\n        val_loader = DataLoader(\n            val_dataset, \n            batch_size=config['batch_size']*2, \n            shuffle=False,\n            num_workers=0,\n            pin_memory=True\n        )\n        \n        # Create model\n        model = SAINT(config).to(device)\n        print(f\"Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n        \n        # Train\n        model, best_score, history = train_saint_model(model, train_loader, val_loader, config, device)\n        \n        ensemble_models.append(model)\n        ensemble_scores.append(best_score)\n        all_histories.append(history)\n        \n        print(f\"Model {i+1} best validation Pearson: {best_score:.4f}\")\n    \n    # Save training history plot\n    plt.figure(figsize=(12, 5))\n    for i, history in enumerate(all_histories):\n        plt.subplot(1, 2, 1)\n        plt.plot(history['train_losses'], label=f'Model {i+1}')\n        plt.xlabel('Epoch')\n        plt.ylabel('Training Loss')\n        plt.title('Training Loss History')\n        plt.legend()\n        \n        plt.subplot(1, 2, 2)\n        plt.plot(history['val_pearsons'], label=f'Model {i+1}')\n        plt.xlabel('Epoch')\n        plt.ylabel('Validation Pearson')\n        plt.title('Validation Pearson History')\n        plt.legend()\n    \n    plt.tight_layout()\n    plt.savefig(SAINT_RESULTS_DIR / 'training_history.png')\n    plt.close()\n    \n    # Make test predictions\n    print(\"\\n=== Making SAINT Test Predictions ===\")\n    \n    # Transform test data\n    X_test = test_df[all_saint_features].values\n    X_test_selected = selector.transform(X_test)\n    X_test_transformed = transformer.transform(X_test_selected)\n    \n    # Make predictions\n    all_predictions = []\n    \n    for model in ensemble_models:\n        model.eval()\n        test_dataset = TabularDataset(X_test_transformed)\n        test_loader = DataLoader(\n            test_dataset, \n            batch_size=2048, \n            shuffle=False,\n            num_workers=0,\n            pin_memory=True\n        )\n        \n        predictions = []\n        with torch.no_grad():\n            for (inputs,) in test_loader:\n                inputs = inputs.to(device)\n                outputs = model(inputs)\n                predictions.extend(outputs.cpu().numpy().flatten())\n        \n        all_predictions.append(np.array(predictions))\n    \n    # Ensemble predictions with weighted average\n    weights = np.array(ensemble_scores)\n    weights = weights / weights.sum()\n    \n    final_predictions = np.zeros_like(all_predictions[0])\n    for pred, weight in zip(all_predictions, weights):\n        final_predictions += weight * pred\n    \n    # Post-processing\n    pred_mean = train_df[Config.LABEL_COLUMN].mean()\n    pred_std = train_df[Config.LABEL_COLUMN].std()\n    final_predictions = np.clip(\n        final_predictions,\n        pred_mean - 4 * pred_std,\n        pred_mean + 4 * pred_std\n    )\n    \n    print(f\"\\nSAINT ensemble weights: {weights}\")\n    print(f\"SAINT prediction stats - Mean: {final_predictions.mean():.6f}, Std: {final_predictions.std():.6f}\")\n    \n    return final_predictions\n\n# =========================\n# Ensemble & Submission Functions\n# =========================\ndef create_xgboost_submission(train_df, oof_preds, test_preds, submission_df):\n    learner_name = 'xgb'\n    \n    # Weights for 10 slices\n    weights = np.array([\n        1.0,   # full_data\n        1.0,   # last_90pct\n        1.0,   # last_85pct\n        1.0,   # last_80pct\n        0.25,  # oldest_25pct\n        0.9,   # full_data_outlier_adj\n        0.9,   # last_90pct_outlier_adj\n        0.9,   # last_85pct_outlier_adj\n        0.9,   # last_80pct_outlier_adj\n        0.2    # oldest_25pct_outlier_adj\n    ])\n    \n    # Normalize weights\n    weights = weights / weights.sum()\n\n    oof_weighted = pd.DataFrame(oof_preds[learner_name]).values @ weights\n    test_weighted = pd.DataFrame(test_preds[learner_name]).values @ weights\n    score_weighted = pearsonr(train_df[Config.LABEL_COLUMN], oof_weighted)[0]\n    print(f\"\\n{learner_name.upper()} Weighted Ensemble Pearson: {score_weighted:.4f}\")\n\n    # Print individual slice scores and weights for analysis\n    print(\"\\nIndividual slice OOF scores and weights:\")\n    slice_names = list(oof_preds[learner_name].keys())\n    for i, slice_name in enumerate(slice_names):\n        score = pearsonr(train_df[Config.LABEL_COLUMN], oof_preds[learner_name][slice_name])[0]\n        print(f\"  {slice_name}: {score:.4f} (weight: {weights[i]:.3f})\")\n\n    # Save XGBoost submission to submodels folder\n    xgb_submission = submission_df.copy()\n    xgb_submission[\"prediction\"] = test_weighted\n    xgb_submission_path = SUBMODELS_DIR / \"submission_xgboost.csv\"\n    xgb_submission.to_csv(xgb_submission_path, index=False)\n    print(f\"\\nSaved: {xgb_submission_path}\")\n    \n    return test_weighted\n\ndef create_ensemble_submission(xgb_predictions, saint_predictions, submission_df, \n                             xgb_weight=0.82, saint_weight=0.18):\n    # Ensemble predictions\n    ensemble_predictions = (xgb_weight * xgb_predictions + \n                          saint_weight * saint_predictions)\n    \n    # Save ensemble submission to final_submissions folder\n    ensemble_submission = submission_df.copy()\n    ensemble_submission[\"prediction\"] = ensemble_predictions\n    ensemble_submission_path = FINAL_SUBMISSIONS_DIR / \"submission_ensemble.csv\"\n    ensemble_submission.to_csv(ensemble_submission_path, index=False)\n    print(f\"\\nSaved: {ensemble_submission_path} (XGBoost: {xgb_weight*100}%, SAINT: {saint_weight*100}%)\")\n    \n    return ensemble_predictions\n\n# =========================\n# Main Execution\n# =========================\nif __name__ == \"__main__\":\n    # Load data\n    train_df, test_df, submission_df = load_data()\n    \n    # Train XGBoost models\n    print(\"=== Training XGBoost Models ===\")\n    oof_preds, test_preds, model_slices = train_and_evaluate_xgboost(train_df, test_df)\n    \n    # Create XGBoost submission\n    xgb_predictions = create_xgboost_submission(train_df, oof_preds, test_preds, submission_df)\n    \n    # Train SAINT model\n    saint_predictions = train_saint(train_df, test_df)\n    \n    # Save SAINT submission to submodels folder\n    saint_submission = submission_df.copy()\n    saint_submission[\"prediction\"] = saint_predictions\n    saint_submission_path = SUBMODELS_DIR / \"submission_saint.csv\"\n    saint_submission.to_csv(saint_submission_path, index=False)\n    print(f\"\\nSaved: {saint_submission_path}\")\n    \n    # Create ensemble submission\n    ensemble_predictions = create_ensemble_submission(\n        xgb_predictions, saint_predictions, submission_df,\n        xgb_weight=0.82, saint_weight=0.18\n    )\n    \n    # Print summary\n    print(\"\\n=== Summary ===\")\n    print(f\"\\nModule Directory Structure:\")\n    print(f\"📁 {MODULE_DIR}\")\n    print(f\"  📁 submodels/\")\n    print(f\"    📄 submission_xgboost.csv\")\n    print(f\"    📄 submission_saint.csv\")\n    print(f\"  📁 final_submissions/\")\n    print(f\"    📄 submission_ensemble.csv\")\n    print(f\"  📁 model_checkpoints/\")\n    print(f\"    📄 best_saint_1.pt\")\n    print(f\"    📄 best_saint_2.pt\")\n    print(f\"  📁 saint_results/\")\n    print(f\"    📄 training_history.png\")\n    \n    # Show sample predictions\n    print(\"\\nSample predictions (first 10 rows):\")\n    comparison_df = pd.DataFrame({\n        'ID': submission_df['ID'][:10],\n        'XGBoost': xgb_predictions[:10],\n        'SAINT': saint_predictions[:10],\n        'Ensemble': ensemble_predictions[:10]\n    })\n    print(comparison_df)\n    \n    # List all created files\n    print(\"\\n=== Files Created ===\")\n    for directory in [SUBMODELS_DIR, FINAL_SUBMISSIONS_DIR, MODEL_CHECKPOINTS_DIR, SAINT_RESULTS_DIR]:\n        print(f\"\\n{directory}:\")\n        for file in sorted(directory.glob(\"*\")):\n            if file.is_file():\n                print(f\"  - {file.name}\")\n\n    # =========================\n    # Create averaged final submissions\n    # =========================\n    print(\"\\n\" + \"=\"*60)\n    print(\"Creating Final Ensemble Averages\")\n    print(\"=\"*60)\n\n    # Paths to all backbone ensembles\n    MODEL_ENSEMBLES = {\n        \"xgb_saint\": FINAL_SUBMISSIONS_DIR / \"submission_ensemble.csv\",\n        # Add paths to your other model ensembles here when you create them:\n        # \"xgb_mlp\": \"/kaggle/working/xgb_mlp_backbone/final_submissions/submission_ensemble.csv\",\n        # \"xgb_gandalf\": \"/kaggle/working/xgb_gandalf_backbone/final_submissions/submission_ensemble.csv\",\n        # \"xgb_autoint\": \"/kaggle/working/xgb_autoint_backbone/final_submissions/submission_ensemble.csv\",\n    }\n\n    OUTPUT_DIR = Path(\"/kaggle/working/averaged_final_submissions\")\n    OUTPUT_DIR.mkdir(exist_ok=True)\n\n    # Check and load available models\n    available_models = {}\n    print(\"\\n=== Checking Available Models ===\")\n    for name, path in MODEL_ENSEMBLES.items():\n        path = Path(path)\n        if path.exists():\n            df = pd.read_csv(path)\n            available_models[name] = df\n            print(f\"✓ {name}: Found - Mean: {df['prediction'].mean():.6f}, Std: {df['prediction'].std():.6f}\")\n        else:\n            print(f\"✗ {name}: Not found at {path}\")\n\n    # If you have multiple models, create averages\n    if len(available_models) >= 2:\n        model_names = list(available_models.keys())\n        print(f\"\\n=== Creating Ensembles from {len(model_names)} models ===\")\n        \n        # Get base submission format\n        base_submission = available_models[model_names[0]][['ID']].copy()\n        \n        # Create different weight combinations\n        weight_combinations = [\n            (0.5, 0.5, \"50_50\"),\n            (0.6, 0.4, \"60_40\"),\n            (0.7, 0.3, \"70_30\"),\n        ]\n        \n        for w1, w2, suffix in weight_combinations:\n            # Calculate weighted average\n            pred1 = available_models[model_names[0]][\"prediction\"].values\n            pred2 = available_models[model_names[1]][\"prediction\"].values\n            avg_pred = w1 * pred1 + w2 * pred2\n            \n            # Create submission\n            submission = base_submission.copy()\n            submission[\"prediction\"] = avg_pred\n            \n            # Clear filename\n            filename = f\"ensemble_{model_names[0]}{int(w1*100)}_{model_names[1]}{int(w2*100)}.csv\"\n            filepath = OUTPUT_DIR / filename\n            submission.to_csv(filepath, index=False)\n            \n            print(f\"\\n✓ {filename}\")\n            print(f\"  Weights: {model_names[0]}={w1:.0%}, {model_names[1]}={w2:.0%}\")\n            print(f\"  Mean: {avg_pred.mean():.6f}, Std: {avg_pred.std():.6f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}