{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Microsoft Malware Classification 2015 — Phase 2\n## True OOF Hybrid CNN + XGBoost (No Submission Generated)\n\n**Author:** Huynh Bao Khang, Nguyen Phuoc Khang, Pham Minh Dat\n\n**Objective:** Use the Phase 1 output to produce course-project results with a leakage-free, nested evaluation, comparing three models:\n\n- **CNN image-only**\n- **XGBoost tabular-only**\n- **Hybrid:** CNN embedding + tabular features\n\nKey design choices:\n- True Stratified K-Fold: each fold's CNN is trained only on that fold's training split; its embedding for the validation split is produced only after training.\n- No SMOTE and no feature scaling are applied for XGBoost.\n- Reports OOF accuracy, log-loss, macro-F1, a classification report, a confusion matrix, and a final summary file.\n\n**On Kaggle:** enable GPU, Add Data → select the Phase 1 Notebook Output, then Run All.\n","metadata":{}},{"cell_type":"markdown","source":"### Configuration\nSet the Phase 1 input path, output directory, run/CNN/XGBoost hyperparameters, and debug options used throughout the notebook.","metadata":{}},{"cell_type":"code","source":"# --- Phase 1 Input ---\n# Leave as None to let the notebook auto-detect `phase1_manifest.json` under /kaggle/input.\nPHASE1_ROOT = None\n\n# --- Paths ---\nLABEL_PATH = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\nOUTPUT_DIR = \"/kaggle/working/phase2_results\"\n\n# --- Run Settings ---\nSEED = 42\nN_SPLITS = 5\nEPOCHS = 6\nCNN_PATIENCE = 2\nBATCH_SIZE = 64\nEMBED_BATCH_SIZE = 128\nNUM_WORKERS = 2\nEMBED_DIM = 128\n\n# --- Class Balancing ---\n# sqrt inverse-frequency: moderate balancing, without over-amplifying rare classes like SMOTE would.\nCLASS_WEIGHT_POWER = 0.5\nMAX_CLASS_WEIGHT = 6.0\n\n# --- XGBoost Settings ---\n# XGBoost will early-stop before using all of these trees.\nXGB_N_ESTIMATORS = 1400\nXGB_EARLY_STOPPING = 60\n\nRUN_TABULAR_BASELINE = True\nSAVE_FOLD_MODELS = False\n\n# --- Debug ---\n# Set an integer for a quick debug run; None runs the full dataset.\nDEBUG_MAX_SAMPLES = None","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:41.171263Z","iopub.execute_input":"2026-07-20T11:04:41.171614Z","iopub.status.idle":"2026-07-20T11:04:41.177261Z","shell.execute_reply.started":"2026-07-20T11:04:41.171584Z","shell.execute_reply":"2026-07-20T11:04:41.176649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Imports, Seed, and Device\nImport the required libraries, fix all random seeds for reproducibility, and select the compute device.","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport json\nimport copy\nimport glob\nimport random\nimport shutil\nimport zipfile\nfrom pathlib import Path\nfrom collections import defaultdict\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport xgboost as xgb\nfrom packaging.version import Version\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.metrics import (\n    accuracy_score,\n    log_loss,\n    f1_score,\n    classification_report,\n    confusion_matrix,\n    ConfusionMatrixDisplay,\n)\n\n\n# --- Reproducibility ---\ndef seed_everything(seed: int):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cudnn.benchmark = True\n\n\nseed_everything(SEED)\n\n# --- Device and Output Setup ---\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nOUTPUT_PATH = Path(OUTPUT_DIR)\nOUTPUT_PATH.mkdir(parents=True, exist_ok=True)\n\nNUM_WORKERS = max(0, min(NUM_WORKERS, os.cpu_count() or 2))\nAMP_ENABLED = DEVICE.type == \"cuda\"\n\nprint(f\"PyTorch device : {DEVICE}\")\nprint(f\"XGBoost version: {xgb.__version__}\")\nprint(f\"Workers: {NUM_WORKERS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:41.178516Z","iopub.execute_input":"2026-07-20T11:04:41.178840Z","iopub.status.idle":"2026-07-20T11:04:41.209031Z","shell.execute_reply.started":"2026-07-20T11:04:41.178815Z","shell.execute_reply":"2026-07-20T11:04:41.208280Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Locate Phase 1 Output and Extract Images\nAuto-detect the Phase 1 manifest, resolve the image directory, and extract the image ZIP if the images aren't already unpacked.","metadata":{}},{"cell_type":"code","source":"# --- Locate the Phase 1 Manifest ---\ndef find_phase1_root(explicit_root=None) -> Path:\n    if explicit_root:\n        root = Path(explicit_root)\n        if not (root / \"phase1_manifest.json\").exists():\n            raise FileNotFoundError(f\"phase1_manifest.json not found in {root}\")\n        return root\n\n    candidates = []\n    for base in [Path(\"/kaggle/input\"), Path(\"/kaggle/working\")]:\n        if base.exists():\n            candidates.extend(base.rglob(\"phase1_manifest.json\"))\n\n    if not candidates:\n        raise FileNotFoundError(\n            \"phase1_manifest.json not found. Add Data -> Notebook Output of Phase 1, \"\n            \"or set PHASE1_ROOT manually.\"\n        )\n\n    # Prefer the manifest with the largest sample_count, then the most recently modified file.\n    scored = []\n    for manifest_path in candidates:\n        try:\n            info = json.loads(manifest_path.read_text(encoding=\"utf-8\"))\n            scored.append(\n                (\n                    int(info.get(\"sample_count\", 0)),\n                    manifest_path.stat().st_mtime,\n                    manifest_path.parent,\n                )\n            )\n        except Exception:\n            continue\n    if not scored:\n        raise RuntimeError(\"Found manifest files, but none of them could be read.\")\n    return sorted(scored, reverse=True)[0][2]\n\n\nPHASE1_PATH = find_phase1_root(PHASE1_ROOT)\nmanifest = json.loads((PHASE1_PATH / \"phase1_manifest.json\").read_text(encoding=\"utf-8\"))\n\n# --- Resolve the Image Directory ---\nimage_folder_name = manifest.get(\"image_folder_name\", \"train_malware_images\")\nsource_image_dir = PHASE1_PATH / image_folder_name\n\nif source_image_dir.exists():\n    IMAGE_DIR = source_image_dir\nelse:\n    zip_name = manifest.get(\"image_zip_name\") or f\"{image_folder_name}.zip\"\n    zip_path = PHASE1_PATH / zip_name\n    if not zip_path.exists():\n        raise FileNotFoundError(f\"Neither the image folder nor the image ZIP was found: {zip_path}\")\n\n    IMAGE_DIR = Path(\"/kaggle/working\") / image_folder_name\n    marker = IMAGE_DIR / \".extract_complete\"\n    if not marker.exists():\n        if IMAGE_DIR.exists():\n            shutil.rmtree(IMAGE_DIR)\n        IMAGE_DIR.mkdir(parents=True, exist_ok=True)\n        print(f\"Extracting {zip_path.name}...\")\n        with zipfile.ZipFile(zip_path, \"r\") as zf:\n            zf.extractall(IMAGE_DIR)\n        marker.touch()\n\nprint(f\"Phase 1 root: {PHASE1_PATH}\")\nprint(f\"Image dir: {IMAGE_DIR}\")\nprint(f\"Image size: {manifest.get('image_size', 'unknown')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:41.210148Z","iopub.execute_input":"2026-07-20T11:04:41.211149Z","iopub.status.idle":"2026-07-20T11:04:41.559261Z","shell.execute_reply.started":"2026-07-20T11:04:41.211107Z","shell.execute_reply":"2026-07-20T11:04:41.558606Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load and Validate Data\nLoad the Phase 1 tabular features, merge them with labels, drop rows with missing values or missing images, and build the final label/feature arrays.","metadata":{}},{"cell_type":"code","source":"# --- Load Tabular Features ---\nassert Path(LABEL_PATH).exists(), f\"Labels file not found: {LABEL_PATH}\"\n\nparquet_name = manifest.get(\"combined_parquet\")\ncsv_name = manifest.get(\"combined_csv\", \"train_tabular_features.csv\")\n\nif parquet_name and (PHASE1_PATH / parquet_name).exists():\n    features_df = pd.read_parquet(PHASE1_PATH / parquet_name)\nelif (PHASE1_PATH / csv_name).exists():\n    features_df = pd.read_csv(PHASE1_PATH / csv_name, dtype={\"Id\": str})\nelse:\n    batch_files = sorted(\n        (PHASE1_PATH / \"tabular_batches\").glob(\"train_tabular_features_batch_*.csv\")\n    )\n    if not batch_files:\n        raise FileNotFoundError(\"No tabular feature file from Phase 1 was found.\")\n    features_df = pd.concat(\n        [pd.read_csv(path, dtype={\"Id\": str}) for path in batch_files],\n        ignore_index=True,\n    )\n\n# --- Merge Features with Labels ---\nlabels_df = pd.read_csv(LABEL_PATH, dtype={\"Id\": str})\nfeatures_df[\"Id\"] = features_df[\"Id\"].astype(str)\nlabels_df[\"Id\"] = labels_df[\"Id\"].astype(str)\n\nfeatures_df = features_df.drop_duplicates(\"Id\", keep=\"last\")\ndata_df = features_df.merge(labels_df, on=\"Id\", how=\"inner\", validate=\"one_to_one\")\ndata_df = data_df.sort_values(\"Id\").reset_index(drop=True)\n\n# --- Clean Invalid Rows ---\nfeature_columns = [c for c in data_df.columns if c not in [\"Id\", \"Class\"]]\ndata_df[feature_columns] = data_df[feature_columns].replace([np.inf, -np.inf], np.nan)\n\nnan_rows = data_df[feature_columns].isna().any(axis=1)\nif nan_rows.any():\n    print(f\"Dropping {int(nan_rows.sum())} rows containing NaN/Inf\")\n    data_df = data_df.loc[~nan_rows].reset_index(drop=True)\n\nimage_exists = data_df[\"Id\"].map(lambda fid: (IMAGE_DIR / f\"{fid}.png\").exists())\nif not image_exists.all():\n    print(f\"Dropping {int((~image_exists).sum())} rows missing an image\")\n    data_df = data_df.loc[image_exists].reset_index(drop=True)\n\n# --- Optional Debug Subsampling ---\nif DEBUG_MAX_SAMPLES is not None and DEBUG_MAX_SAMPLES < len(data_df):\n    # Stratified sampling so every class still appears.\n    parts = []\n    per_class = max(2, int(DEBUG_MAX_SAMPLES) // data_df[\"Class\"].nunique())\n    for _, group in data_df.groupby(\"Class\"):\n        parts.append(group.sample(min(per_class, len(group)), random_state=SEED))\n    data_df = pd.concat(parts).sample(frac=1, random_state=SEED).reset_index(drop=True)\n\n# --- Build Final Arrays ---\nclasses = sorted(data_df[\"Class\"].unique().tolist())\nclass_to_index = {label: idx for idx, label in enumerate(classes)}\nindex_to_class = {idx: label for label, idx in class_to_index.items()}\n\ny = data_df[\"Class\"].map(class_to_index).to_numpy(dtype=np.int64)\nimage_ids = data_df[\"Id\"].astype(str).tolist()\nX_tabular = data_df[feature_columns].to_numpy(dtype=np.float32)\nnum_classes = len(classes)\n\nmin_class_count = int(np.bincount(y).min())\nif min_class_count < N_SPLITS:\n    raise ValueError(\n        f\"The smallest class only has {min_class_count} samples, not enough for {N_SPLITS}-fold CV.\"\n    )\n\nprint(f\"Samples: {len(y):,}\")\nprint(f\"Tabular features: {X_tabular.shape[1]:,}\")\nprint(f\"Classes: {classes}\")\nprint(\"Class counts:\")\nprint(data_df[\"Class\"].value_counts().sort_index().to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:41.559976Z","iopub.execute_input":"2026-07-20T11:04:41.560244Z","iopub.status.idle":"2026-07-20T11:04:42.108294Z","shell.execute_reply.started":"2026-07-20T11:04:41.560224Z","shell.execute_reply":"2026-07-20T11:04:42.107641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dataset and CNN\nDefine the image dataset and the adaptive CNN architecture used to produce a fixed-size embedding from each malware image.","metadata":{}},{"cell_type":"code","source":"# --- Image Transform ---\nIMAGE_SIZE = int(manifest.get(\"image_size\", 224))\n\nimage_transform = transforms.Compose(\n    [\n        transforms.Resize((IMAGE_SIZE, IMAGE_SIZE)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.5], std=[0.5]),\n    ]\n)\n\n\n# --- Dataset ---\nclass MalwareImageDataset(Dataset):\n    def __init__(self, image_dir, ids, labels=None):\n        self.image_dir = Path(image_dir)\n        self.ids = list(ids)\n        self.labels = None if labels is None else np.asarray(labels, dtype=np.int64)\n\n    def __len__(self):\n        return len(self.ids)\n\n    def __getitem__(self, index):\n        fid = self.ids[index]\n        path = self.image_dir / f\"{fid}.png\"\n        try:\n            with Image.open(path) as image:\n                image = image.convert(\"L\")\n                tensor = image_transform(image)\n        except Exception as exc:\n            raise RuntimeError(f\"Could not read image {path}: {exc}\") from exc\n\n        if self.labels is None:\n            return tensor\n        return tensor, torch.tensor(self.labels[index], dtype=torch.long)\n\n\n# --- CNN Architecture ---\nclass MalwareCNN(nn.Module):\n    def __init__(self, num_classes: int, embed_dim: int = 128):\n        super().__init__()\n        self.backbone = nn.Sequential(\n            nn.Conv2d(1, 32, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(32),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(32, 64, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(64),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(64, 128, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(128),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(128, 192, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(192),\n            nn.SiLU(inplace=True),\n            nn.AdaptiveAvgPool2d(1),\n        )\n        self.embedding = nn.Sequential(\n            nn.Flatten(),\n            nn.Linear(192, embed_dim),\n            nn.LayerNorm(embed_dim),\n            nn.SiLU(inplace=True),\n            nn.Dropout(0.25),\n        )\n        self.classifier = nn.Linear(embed_dim, num_classes)\n\n    def forward_features(self, x):\n        return self.embedding(self.backbone(x))\n\n    def forward(self, x):\n        return self.classifier(self.forward_features(x))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:42.109718Z","iopub.execute_input":"2026-07-20T11:04:42.110021Z","iopub.status.idle":"2026-07-20T11:04:42.121002Z","shell.execute_reply.started":"2026-07-20T11:04:42.109999Z","shell.execute_reply":"2026-07-20T11:04:42.120240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CNN Train/Evaluate Functions\nHelper functions for building data loaders, computing class-balancing weights, running CNN inference, and training a CNN for one outer fold with a nested inner-validation split used for early stopping.","metadata":{}},{"cell_type":"code","source":"# --- Data Loader Helper ---\ndef make_loader(ids, labels=None, shuffle=False, batch_size=BATCH_SIZE):\n    dataset = MalwareImageDataset(IMAGE_DIR, ids, labels)\n    return DataLoader(\n        dataset,\n        batch_size=batch_size,\n        shuffle=shuffle,\n        num_workers=NUM_WORKERS,\n        pin_memory=AMP_ENABLED,\n        persistent_workers=NUM_WORKERS > 0,\n        drop_last=False,\n    )\n\n\n# --- Class Balancing Weights ---\ndef compute_balancing_weights(labels):\n    counts = np.bincount(labels, minlength=num_classes).astype(np.float64)\n    class_weights = (len(labels) / (num_classes * np.maximum(counts, 1))) ** CLASS_WEIGHT_POWER\n    class_weights = np.clip(class_weights, 0.5, MAX_CLASS_WEIGHT)\n    sample_weights = class_weights[labels]\n    return class_weights.astype(np.float32), sample_weights.astype(np.float32)\n\n\n# --- CNN Inference ---\n@torch.no_grad()\ndef predict_cnn(model, loader, return_embeddings=False):\n    model.eval()\n    all_probs = []\n    all_embeddings = []\n    all_labels = []\n\n    for batch in loader:\n        if isinstance(batch, (tuple, list)):\n            inputs, labels = batch\n            all_labels.append(labels.numpy())\n        else:\n            inputs = batch\n\n        inputs = inputs.to(DEVICE, non_blocking=True)\n        with torch.amp.autocast(\"cuda\", enabled=AMP_ENABLED):\n            embeddings = model.forward_features(inputs)\n            logits = model.classifier(embeddings)\n            probs = torch.softmax(logits, dim=1)\n\n        all_probs.append(probs.cpu().numpy())\n        if return_embeddings:\n            all_embeddings.append(embeddings.cpu().numpy())\n\n    probs = np.concatenate(all_probs, axis=0)\n    embeddings = np.concatenate(all_embeddings, axis=0) if return_embeddings else None\n    labels = np.concatenate(all_labels, axis=0) if all_labels else None\n    return probs, embeddings, labels\n\n\n# --- CNN Training for One Fold ---\ndef train_cnn_fold(outer_train_ids, outer_train_y, outer_val_ids, fold_number):\n    \"\"\"\n    Early stopping only uses an inner-validation split carved out of the outer train set.\n    The outer validation fold never participates in epoch selection.\n    \"\"\"\n    positions = np.arange(len(outer_train_y))\n    inner_train_pos, inner_stop_pos = train_test_split(\n        positions,\n        test_size=0.12,\n        random_state=SEED + fold_number,\n        stratify=outer_train_y,\n    )\n\n    inner_train_ids = [outer_train_ids[i] for i in inner_train_pos]\n    inner_stop_ids = [outer_train_ids[i] for i in inner_stop_pos]\n    inner_train_y = outer_train_y[inner_train_pos]\n    inner_stop_y = outer_train_y[inner_stop_pos]\n\n    train_loader = make_loader(inner_train_ids, inner_train_y, shuffle=True, batch_size=BATCH_SIZE)\n\n    stop_loader = make_loader(\n        inner_stop_ids, inner_stop_y, shuffle=False, batch_size=EMBED_BATCH_SIZE\n    )\n\n    class_weights, _ = compute_balancing_weights(inner_train_y)\n    criterion = nn.CrossEntropyLoss(\n        weight=torch.tensor(class_weights, dtype=torch.float32, device=DEVICE)\n    )\n\n    model = MalwareCNN(num_classes=num_classes, embed_dim=EMBED_DIM).to(DEVICE)\n    optimizer = optim.AdamW(model.parameters(), lr=1e-3, weight_decay=1e-4)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer, mode=\"min\", factor=0.5, patience=1, min_lr=1e-5\n    )\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=AMP_ENABLED)\n\n    best_loss = float(\"inf\")\n    best_state = None\n    best_epoch = 0\n    epochs_without_improvement = 0\n    history = []\n\n    header = f\"    {'Epoch':^7} | {'Train Loss':>10} | {'Inner Loss':>10} | {'Inner Acc':>9}\"\n    print(f\"    Fold {fold_number} — CNN training\")\n    print(header)\n    print(f\"    {'-' * 7}-+-{'-' * 10}-+-{'-' * 10}-+-{'-' * 9}\")\n\n    for epoch in range(1, EPOCHS + 1):\n        model.train()\n        running_loss = 0.0\n        seen = 0\n\n        for inputs, labels in train_loader:\n            inputs = inputs.to(DEVICE, non_blocking=True)\n            labels = labels.to(DEVICE, non_blocking=True)\n            optimizer.zero_grad(set_to_none=True)\n\n            with torch.amp.autocast(\"cuda\", enabled=AMP_ENABLED):\n                logits = model(inputs)\n                loss = criterion(logits, labels)\n\n            scaler.scale(loss).backward()\n            scaler.step(optimizer)\n            scaler.update()\n\n            batch_n = labels.size(0)\n            running_loss += loss.item() * batch_n\n            seen += batch_n\n\n        train_loss = running_loss / max(seen, 1)\n        stop_probs, _, stop_labels = predict_cnn(model, stop_loader)\n        stop_loss = log_loss(stop_labels, stop_probs, labels=np.arange(num_classes))\n        stop_acc = accuracy_score(stop_labels, stop_probs.argmax(axis=1))\n        scheduler.step(stop_loss)\n\n        history.append(\n            {\n                \"fold\": fold_number,\n                \"epoch\": epoch,\n                \"train_loss\": train_loss,\n                \"inner_val_logloss\": stop_loss,\n                \"inner_val_accuracy\": stop_acc,\n                \"learning_rate\": optimizer.param_groups[0][\"lr\"],\n            }\n        )\n\n        is_best = stop_loss < best_loss - 1e-4\n        marker = \" *\" if is_best else \"  \"\n        epoch_label = f\"{epoch:>2d}/{EPOCHS:<2d}\"\n        print(\n            f\"    {epoch_label:^7} | {train_loss:>10.4f} | {stop_loss:>10.4f} | \"\n            f\"{stop_acc:>9.4f}{marker}\"\n        )\n\n        if is_best:\n            best_loss = stop_loss\n            best_epoch = epoch\n            best_state = copy.deepcopy(model.state_dict())\n            epochs_without_improvement = 0\n        else:\n            epochs_without_improvement += 1\n            if epochs_without_improvement >= CNN_PATIENCE:\n                print(\n                    f\"    Early stopping — best epoch was {best_epoch} (inner loss {best_loss:.4f})\"\n                )\n                break\n\n    if best_state is None:\n        raise RuntimeError(\"CNN failed to save a best state\")\n    model.load_state_dict(best_state)\n\n    if SAVE_FOLD_MODELS:\n        torch.save(model.state_dict(), OUTPUT_PATH / f\"cnn_fold_{fold_number}.pth\")\n\n    # Outer train and outer validation embeddings are produced by the same CNN.\n    # Outer validation was never used during training or early stopping.\n    outer_train_embed_loader = make_loader(\n        outer_train_ids, labels=None, shuffle=False, batch_size=EMBED_BATCH_SIZE\n    )\n\n    outer_val_loader = make_loader(\n        outer_val_ids, labels=None, shuffle=False, batch_size=EMBED_BATCH_SIZE\n    )\n\n    _, train_embeddings, _ = predict_cnn(model, outer_train_embed_loader, return_embeddings=True)\n\n    val_probs, val_embeddings, _ = predict_cnn(model, outer_val_loader, return_embeddings=True)\n\n    del train_loader, stop_loader, outer_train_embed_loader, outer_val_loader\n    torch.cuda.empty_cache()\n    gc.collect()\n    return model, train_embeddings, val_embeddings, val_probs, history, best_epoch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:42.122098Z","iopub.execute_input":"2026-07-20T11:04:42.122363Z","iopub.status.idle":"2026-07-20T11:04:42.144435Z","shell.execute_reply.started":"2026-07-20T11:04:42.122331Z","shell.execute_reply":"2026-07-20T11:04:42.143816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### XGBoost Helpers\nHelper functions for building an XGBoost model with automatic GPU/CPU device selection and CPU fallback, and for fitting a model whose tree count is chosen via a nested inner-validation split.","metadata":{}},{"cell_type":"code","source":"# --- Device Parameters ---\ndef xgb_device_params():\n    params = {\"tree_method\": \"hist\"}\n    if torch.cuda.is_available():\n        if Version(xgb.__version__) >= Version(\"2.0.0\"):\n            params[\"device\"] = \"cuda\"\n        else:\n            params[\"tree_method\"] = \"gpu_hist\"\n            params[\"predictor\"] = \"gpu_predictor\"\n    return params\n\n\n# --- Model Builder ---\ndef build_xgb_model(n_estimators, use_early_stopping=False):\n    params = {\n        \"n_estimators\": int(n_estimators),\n        \"max_depth\": 6,\n        \"learning_rate\": 0.03,\n        \"min_child_weight\": 1.0,\n        \"subsample\": 0.85,\n        \"colsample_bytree\": 0.80,\n        \"reg_alpha\": 0.05,\n        \"reg_lambda\": 2.0,\n        \"gamma\": 0.0,\n        \"max_bin\": 256,\n        \"objective\": \"multi:softprob\",\n        \"num_class\": num_classes,\n        \"eval_metric\": \"mlogloss\",\n        \"random_state\": SEED,\n        \"n_jobs\": -1,\n    }\n    if use_early_stopping:\n        params[\"early_stopping_rounds\"] = XGB_EARLY_STOPPING\n    params.update(xgb_device_params())\n    return xgb.XGBClassifier(**params)\n\n\n# --- GPU Fit with CPU Fallback ---\ndef fit_model_with_optional_cpu_fallback(model, *fit_args, **fit_kwargs):\n    try:\n        model.fit(*fit_args, **fit_kwargs)\n        return model\n    except xgb.core.XGBoostError as exc:\n        if not torch.cuda.is_available():\n            raise\n        print(f\"GPU XGBoost error, falling back to CPU: {str(exc)[:180]}\")\n        params = model.get_params()\n        params.pop(\"device\", None)\n        params.pop(\"predictor\", None)\n        params[\"tree_method\"] = \"hist\"\n        cpu_model = xgb.XGBClassifier(**params)\n        cpu_model.fit(*fit_args, **fit_kwargs)\n        return cpu_model\n\n\n# --- Nested Fit (Inner Validation for Tree Count) ---\ndef fit_xgb_nested(train_x, train_y, outer_val_x, sample_weight, model_name, fold_number):\n    \"\"\"\n    Selects the number of trees using an inner-validation split carved out of the outer train\n    set, then refits on the full outer train set. Outer validation never participates in\n    early stopping.\n    \"\"\"\n    positions = np.arange(len(train_y))\n    fit_pos, stop_pos = train_test_split(\n        positions,\n        test_size=0.12,\n        random_state=SEED + 100 + fold_number,\n        stratify=train_y,\n    )\n\n    tuning_model = build_xgb_model(XGB_N_ESTIMATORS, use_early_stopping=True)\n    tuning_model = fit_model_with_optional_cpu_fallback(\n        tuning_model,\n        train_x[fit_pos],\n        train_y[fit_pos],\n        sample_weight=sample_weight[fit_pos],\n        eval_set=[(train_x[stop_pos], train_y[stop_pos])],\n        verbose=False,\n    )\n\n    best_iteration = getattr(tuning_model, \"best_iteration\", None)\n    best_n_estimators = int(best_iteration + 1) if best_iteration is not None else XGB_N_ESTIMATORS\n    best_n_estimators = max(30, min(best_n_estimators, XGB_N_ESTIMATORS))\n    print(f\"{model_name}: selected {best_n_estimators} trees using inner validation\")\n\n    final_model = build_xgb_model(best_n_estimators, use_early_stopping=False)\n    final_model = fit_model_with_optional_cpu_fallback(\n        final_model,\n        train_x,\n        train_y,\n        sample_weight=sample_weight,\n        verbose=False,\n    )\n    probs = final_model.predict_proba(outer_val_x)\n    del tuning_model\n    gc.collect()\n    return final_model, probs, best_n_estimators","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:42.145275Z","iopub.execute_input":"2026-07-20T11:04:42.145601Z","iopub.status.idle":"2026-07-20T11:04:42.167087Z","shell.execute_reply.started":"2026-07-20T11:04:42.145557Z","shell.execute_reply":"2026-07-20T11:04:42.166441Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### True Stratified K-Fold\nRun the main cross-validation loop: for each fold, train a fold-specific CNN, optionally a tabular-only XGBoost baseline, and a hybrid XGBoost model on CNN embeddings + tabular features, collecting out-of-fold predictions for all three.","metadata":{}},{"cell_type":"code","source":"# --- Fold Result Table Helper ---\ndef print_fold_result(name, acc, ll):\n    print(f\"    {name:<10} | acc {acc:>7.4f} | logloss {ll:>7.4f}\")\n\n\nskf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)\n\noof_cnn = np.zeros((len(y), num_classes), dtype=np.float32)\noof_tabular = np.zeros((len(y), num_classes), dtype=np.float32)\noof_hybrid = np.zeros((len(y), num_classes), dtype=np.float32)\n\nfold_records = []\ncnn_history_records = []\nfeature_importance_sum = defaultdict(float)\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X_tabular, y), start=1):\n    print(\"\\n\" + \"=\" * 72)\n    print(f\"FOLD {fold}/{N_SPLITS}  (train={len(train_idx):,} | val={len(val_idx):,})\")\n    print(\"=\" * 72)\n\n    train_ids = [image_ids[i] for i in train_idx]\n    val_ids = [image_ids[i] for i in val_idx]\n    y_train = y[train_idx]\n    y_val = y[val_idx]\n\n    _, sample_weights = compute_balancing_weights(y_train)\n\n    # --- CNN: Sees Only the Train Fold ---\n    cnn_model, train_embeddings, val_embeddings, cnn_probs, history, cnn_best_epoch = (\n        train_cnn_fold(train_ids, y_train, val_ids, fold)\n    )\n    oof_cnn[val_idx] = cnn_probs\n    cnn_history_records.extend(history)\n\n    cnn_acc = accuracy_score(y_val, cnn_probs.argmax(axis=1))\n    cnn_ll = log_loss(y_val, cnn_probs, labels=np.arange(num_classes))\n\n    # --- Tabular-Only Baseline ---\n    if RUN_TABULAR_BASELINE:\n        tab_model, tab_probs, tab_best_trees = fit_xgb_nested(\n            X_tabular[train_idx],\n            y_train,\n            X_tabular[val_idx],\n            sample_weights,\n            model_name=\"Tabular XGBoost\",\n            fold_number=fold,\n        )\n        oof_tabular[val_idx] = tab_probs\n        tab_acc = accuracy_score(y_val, tab_probs.argmax(axis=1))\n        tab_ll = log_loss(y_val, tab_probs, labels=np.arange(num_classes))\n    else:\n        tab_model = None\n        tab_best_trees = np.nan\n        tab_acc = np.nan\n        tab_ll = np.nan\n\n    # --- Hybrid: Current Fold's CNN Embedding + Tabular ---\n    X_train_hybrid = np.hstack([train_embeddings, X_tabular[train_idx]])\n    X_val_hybrid = np.hstack([val_embeddings, X_tabular[val_idx]])\n\n    hybrid_model, hybrid_probs, hybrid_best_trees = fit_xgb_nested(\n        X_train_hybrid,\n        y_train,\n        X_val_hybrid,\n        sample_weights,\n        model_name=\"Hybrid XGBoost\",\n        fold_number=fold,\n    )\n    oof_hybrid[val_idx] = hybrid_probs\n\n    hybrid_acc = accuracy_score(y_val, hybrid_probs.argmax(axis=1))\n    hybrid_ll = log_loss(y_val, hybrid_probs, labels=np.arange(num_classes))\n\n    # --- Fold Result Summary ---\n    print(f\"    {'Model':<10} | {'Accuracy':>8} | {'Log-loss':>8}\")\n    print(f\"    {'-' * 10}-+-{'-' * 8}-+-{'-' * 8}\")\n    print_fold_result(\"CNN\", cnn_acc, cnn_ll)\n    if RUN_TABULAR_BASELINE:\n        print_fold_result(\"Tabular\", tab_acc, tab_ll)\n    print_fold_result(\"Hybrid\", hybrid_acc, hybrid_ll)\n\n    fold_records.append(\n        {\n            \"fold\": fold,\n            \"cnn_accuracy\": cnn_acc,\n            \"cnn_logloss\": cnn_ll,\n            \"tabular_accuracy\": tab_acc,\n            \"tabular_logloss\": tab_ll,\n            \"hybrid_accuracy\": hybrid_acc,\n            \"hybrid_logloss\": hybrid_ll,\n            \"cnn_best_epoch\": int(cnn_best_epoch),\n            \"tabular_best_trees\": float(tab_best_trees),\n            \"hybrid_best_trees\": int(hybrid_best_trees),\n        }\n    )\n\n    # --- Accumulate Feature Importance Across Folds ---\n    booster_importance = hybrid_model.get_booster().get_score(importance_type=\"gain\")\n    for key, value in booster_importance.items():\n        feature_importance_sum[int(key[1:])] += float(value)\n\n    if SAVE_FOLD_MODELS:\n        hybrid_model.save_model(OUTPUT_PATH / f\"hybrid_xgb_fold_{fold}.json\")\n        if tab_model is not None:\n            tab_model.save_model(OUTPUT_PATH / f\"tabular_xgb_fold_{fold}.json\")\n\n    del cnn_model, tab_model, hybrid_model\n    del train_embeddings, val_embeddings, X_train_hybrid, X_val_hybrid\n    torch.cuda.empty_cache()\n    gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T11:04:42.167889Z","iopub.execute_input":"2026-07-20T11:04:42.168191Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### OOF Evaluation and Save Results\nSummarize accuracy, log-loss, and F1 for all three models, generate a classification report and confusion matrix for the hybrid model, save all metrics/predictions/plots to disk, and write a final JSON summary.","metadata":{}},{"cell_type":"code","source":"# --- Summarize OOF Predictions ---\ndef summarize_predictions(name, probs):\n    pred = probs.argmax(axis=1)\n    metrics = {\n        \"model\": name,\n        \"accuracy\": float(accuracy_score(y, pred)),\n        \"log_loss\": float(log_loss(y, probs, labels=np.arange(num_classes))),\n        \"macro_f1\": float(f1_score(y, pred, average=\"macro\")),\n        \"weighted_f1\": float(f1_score(y, pred, average=\"weighted\")),\n    }\n    return metrics, pred\n\n\nsummaries = []\n\ncnn_summary, cnn_pred = summarize_predictions(\"CNN image-only\", oof_cnn)\nsummaries.append(cnn_summary)\n\nif RUN_TABULAR_BASELINE:\n    tab_summary, tab_pred = summarize_predictions(\"XGBoost tabular-only\", oof_tabular)\n    summaries.append(tab_summary)\n\nhybrid_summary, hybrid_pred = summarize_predictions(\"Hybrid CNN + tabular\", oof_hybrid)\nsummaries.append(hybrid_summary)\n\nsummary_df = pd.DataFrame(summaries).sort_values(\"log_loss\").reset_index(drop=True)\nfold_df = pd.DataFrame(fold_records)\nhistory_df = pd.DataFrame(cnn_history_records)\n\nprint(\"\\n\" + \"=\" * 72)\nprint(\"OOF MODEL COMPARISON\")\nprint(\"=\" * 72)\nprint(summary_df.to_string(index=False, float_format=lambda x: f\"{x:.5f}\"))\n\nprint(\"\\nClassification report — Hybrid:\")\ntarget_names = [f\"Class {label}\" for label in classes]\nprint(\n    classification_report(\n        y,\n        hybrid_pred,\n        labels=np.arange(num_classes),\n        target_names=target_names,\n        digits=4,\n        zero_division=0,\n    )\n)\n\n# --- Save Metrics and Predictions ---\nsummary_df.to_csv(OUTPUT_PATH / \"model_comparison.csv\", index=False)\nfold_df.to_csv(OUTPUT_PATH / \"fold_metrics.csv\", index=False)\nhistory_df.to_csv(OUTPUT_PATH / \"cnn_training_history.csv\", index=False)\n\noof_df = pd.DataFrame(\n    {\n        \"Id\": image_ids,\n        \"true_class\": [index_to_class[int(v)] for v in y],\n        \"cnn_pred_class\": [index_to_class[int(v)] for v in cnn_pred],\n        \"hybrid_pred_class\": [index_to_class[int(v)] for v in hybrid_pred],\n    }\n)\nif RUN_TABULAR_BASELINE:\n    oof_df[\"tabular_pred_class\"] = [index_to_class[int(v)] for v in tab_pred]\n\nfor idx, label in enumerate(classes):\n    oof_df[f\"cnn_prob_class_{label}\"] = oof_cnn[:, idx]\n    if RUN_TABULAR_BASELINE:\n        oof_df[f\"tabular_prob_class_{label}\"] = oof_tabular[:, idx]\n    oof_df[f\"hybrid_prob_class_{label}\"] = oof_hybrid[:, idx]\noof_df.to_csv(OUTPUT_PATH / \"oof_predictions.csv\", index=False)\n\n# --- Confusion Matrix (Hybrid Model) ---\ncm = confusion_matrix(y, hybrid_pred, labels=np.arange(num_classes))\nfig, ax = plt.subplots(figsize=(10, 9))\ndisplay = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=classes)\ndisplay.plot(ax=ax, values_format=\"d\", colorbar=False)\nax.set_title(\n    f\"Hybrid OOF Confusion Matrix\\n\"\n    f\"Accuracy = {hybrid_summary['accuracy']:.4f} | Log-loss = {hybrid_summary['log_loss']:.4f}\"\n)\nfig.tight_layout()\nfig.savefig(OUTPUT_PATH / \"hybrid_confusion_matrix.png\", dpi=160, bbox_inches=\"tight\")\nplt.show()\n\n# --- CNN Learning Curves ---\nif not history_df.empty:\n    fig, ax = plt.subplots(figsize=(10, 6))\n    for fold, group in history_df.groupby(\"fold\"):\n        ax.plot(group[\"epoch\"], group[\"inner_val_logloss\"], marker=\"o\", label=f\"Fold {fold}\")\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"Inner-validation log-loss\")\n    ax.set_title(\"CNN inner-validation log-loss by fold\")\n    ax.grid(alpha=0.25)\n    ax.legend()\n    fig.tight_layout()\n    fig.savefig(OUTPUT_PATH / \"cnn_validation_curves.png\", dpi=160, bbox_inches=\"tight\")\n    plt.show()\n\n# --- Average Feature Importance (Hybrid Model) ---\nhybrid_feature_names = [f\"cnn_embed_{i}\" for i in range(EMBED_DIM)] + feature_columns\nimportance_rows = []\nfor feature_idx, total_gain in feature_importance_sum.items():\n    if feature_idx < len(hybrid_feature_names):\n        importance_rows.append(\n            {\n                \"feature\": hybrid_feature_names[feature_idx],\n                \"mean_gain\": total_gain / N_SPLITS,\n            }\n        )\nimportance_df = pd.DataFrame(importance_rows).sort_values(\"mean_gain\", ascending=False)\nimportance_df.to_csv(OUTPUT_PATH / \"hybrid_feature_importance.csv\", index=False)\n\nprint(\"\\nTop 20 Hybrid features:\")\nprint(importance_df.head(20).to_string(index=False, float_format=lambda x: f\"{x:.3f}\"))\n\n# --- Write Final Summary ---\nfinal_summary = {\n    \"phase\": 2,\n    \"version\": \"2.0\",\n    \"sample_count\": int(len(y)),\n    \"n_splits\": int(N_SPLITS),\n    \"classes\": [int(v) if isinstance(v, (int, np.integer)) else v for v in classes],\n    \"metrics\": summaries,\n    \"primary_model\": \"Hybrid CNN + tabular\",\n    \"notes\": [\n        \"No submission was generated.\",\n        \"CNN embeddings are fold-specific; outer-validation samples are unseen during CNN training and early stopping.\",\n        \"XGBoost tree counts are selected on an inner-validation split; outer-validation is used only for final OOF prediction.\",\n        \"No SMOTE and no StandardScaler were used for XGBoost.\",\n    ],\n}\n(OUTPUT_PATH / \"results_summary.json\").write_text(\n    json.dumps(final_summary, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nprint(\"\\n\" + \"=\" * 72)\nprint(\"PHASE 2 COMPLETE\")\nprint(f\"Primary OOF Accuracy : {hybrid_summary['accuracy'] * 100:.2f}%\")\nprint(f\"Primary OOF Log-loss : {hybrid_summary['log_loss']:.5f}\")\nprint(f\"Primary OOF Macro-F1 : {hybrid_summary['macro_f1']:.5f}\")\nprint(f\"Artifacts: {OUTPUT_PATH}\")\nprint(\"No submission was generated, per the course project goal.\")\nprint(\"=\" * 72)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Model Comparison Charts and Analysis\nCompare OOF accuracy and log-loss across the three models, check how consistent each model is across folds, and summarize the findings in a short written analysis.","metadata":{}},{"cell_type":"code","source":"# --- Bar Chart: OOF Accuracy and Log-Loss by Model ---\ncompare_df = summary_df.set_index(\"model\").loc[[s[\"model\"] for s in summaries]]\n\nfig, axes = plt.subplots(1, 2, figsize=(13, 5))\n\naxes[0].bar(compare_df.index, compare_df[\"accuracy\"], color=\"#4C72B0\")\naxes[0].set_title(\"OOF Accuracy by Model\")\naxes[0].set_ylabel(\"Accuracy\")\naxes[0].set_ylim(0, 1.0)\naxes[0].tick_params(axis=\"x\", rotation=15)\nfor i, value in enumerate(compare_df[\"accuracy\"]):\n    axes[0].text(i, value + 0.01, f\"{value:.4f}\", ha=\"center\", va=\"bottom\", fontsize=9)\n\naxes[1].bar(compare_df.index, compare_df[\"log_loss\"], color=\"#DD8452\")\naxes[1].set_title(\"OOF Log-Loss by Model (lower is better)\")\naxes[1].set_ylabel(\"Log-loss\")\naxes[1].tick_params(axis=\"x\", rotation=15)\nfor i, value in enumerate(compare_df[\"log_loss\"]):\n    axes[1].text(i, value, f\"{value:.4f}\", ha=\"center\", va=\"bottom\", fontsize=9)\n\nfig.suptitle(\"Model Comparison — CNN vs Tabular vs Hybrid\", fontweight=\"bold\")\nfig.tight_layout()\nfig.savefig(OUTPUT_PATH / \"model_comparison_bars.png\", dpi=160, bbox_inches=\"tight\")\nplt.show()\n\n# --- Fold-Wise Comparison: Accuracy Across Folds ---\nfig, ax = plt.subplots(figsize=(10, 6))\nax.plot(fold_df[\"fold\"], fold_df[\"cnn_accuracy\"], marker=\"o\", label=\"CNN\")\nif RUN_TABULAR_BASELINE:\n    ax.plot(fold_df[\"fold\"], fold_df[\"tabular_accuracy\"], marker=\"s\", label=\"Tabular\")\nax.plot(fold_df[\"fold\"], fold_df[\"hybrid_accuracy\"], marker=\"^\", label=\"Hybrid\")\nax.set_xlabel(\"Fold\")\nax.set_ylabel(\"Accuracy\")\nax.set_xticks(fold_df[\"fold\"])\nax.set_title(\"Per-Fold Accuracy — CNN vs Tabular vs Hybrid\")\nax.grid(alpha=0.25)\nax.legend()\nfig.tight_layout()\nfig.savefig(OUTPUT_PATH / \"fold_accuracy_comparison.png\", dpi=160, bbox_inches=\"tight\")\nplt.show()\n\n# --- Written Analysis (Generated from the Computed Metrics) ---\nbest_model_row = compare_df[\"log_loss\"].idxmin()\nworst_model_row = compare_df[\"log_loss\"].idxmax()\n\nhybrid_vs_cnn_acc = (hybrid_summary[\"accuracy\"] - cnn_summary[\"accuracy\"]) * 100\nhybrid_vs_cnn_ll = cnn_summary[\"log_loss\"] - hybrid_summary[\"log_loss\"]\n\nfold_std_line = \"\"\nif RUN_TABULAR_BASELINE:\n    hybrid_vs_tab_acc = (hybrid_summary[\"accuracy\"] - tab_summary[\"accuracy\"]) * 100\n    hybrid_vs_tab_ll = tab_summary[\"log_loss\"] - hybrid_summary[\"log_loss\"]\n\ncnn_fold_std = fold_df[\"cnn_accuracy\"].std()\nhybrid_fold_std = fold_df[\"hybrid_accuracy\"].std()\nif RUN_TABULAR_BASELINE:\n    tab_fold_std = fold_df[\"tabular_accuracy\"].std()\n\nprint(\"\\n\" + \"=\" * 72)\nprint(\"ANALYSIS\")\nprint(\"=\" * 72)\nprint(\n    f\"The best-performing model by OOF log-loss is '{best_model_row}' \"\n    f\"({compare_df.loc[best_model_row, 'log_loss']:.4f}), while '{worst_model_row}' \"\n    f\"trails behind at {compare_df.loc[worst_model_row, 'log_loss']:.4f}.\"\n)\nprint(\n    f\"The hybrid model improves accuracy over the CNN-only model by \"\n    f\"{hybrid_vs_cnn_acc:+.2f} percentage points and reduces log-loss by {hybrid_vs_cnn_ll:+.4f}, \"\n    f\"suggesting the tabular features add information the CNN embedding does not fully capture.\"\n)\nif RUN_TABULAR_BASELINE:\n    print(\n        f\"Compared to the tabular-only baseline, the hybrid model changes accuracy by \"\n        f\"{hybrid_vs_tab_acc:+.2f} percentage points and log-loss by {hybrid_vs_tab_ll:+.4f}, \"\n        f\"indicating how much the CNN embedding contributes on top of the tabular features alone.\"\n    )\nprint(\n    f\"Fold-to-fold accuracy std: CNN={cnn_fold_std:.4f}\"\n    + (f\", Tabular={tab_fold_std:.4f}\" if RUN_TABULAR_BASELINE else \"\")\n    + f\", Hybrid={hybrid_fold_std:.4f}.\"\n)\nif hybrid_fold_std <= min(cnn_fold_std, tab_fold_std if RUN_TABULAR_BASELINE else cnn_fold_std):\n    print(\n        \"The hybrid model is also the most consistent across folds, which is a good sign for generalization.\"\n    )\nelse:\n    print(\n        \"The hybrid model is not the most consistent across folds; check individual fold results for outliers.\"\n    )\nprint(\"=\" * 72)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}