{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isGpuEnabled":false,"isInternetEnabled":false,"language":"python","sourceType":"notebook"},"papermill":{"default_parameters":{},"duration":955.052123,"end_time":"2026-07-20T11:41:53.991434+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-07-20T11:25:58.939311+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"aed5e548","cell_type":"markdown","source":"# Microsoft Malware Classification 2015 — Phase 2\n## Leakage-Free OOF Comparison: CNN, XGBoost, and Hybrid Learning\n\n**Authors:** Huynh Bao Khang, Nguyen Phuoc Khang, Pham Minh Dat\n\n**Objective:** Evaluate three models on exactly the same Stratified K-Fold splits:\n\n- **CNN image-only**\n- **XGBoost tabular-only**\n- **Hybrid:** fold-specific CNN embedding + tabular features\n\nThe outer validation fold is never used to train the CNN or select its epoch/tree count. The notebook selects the primary model from the computed OOF metric instead of assuming that the Hybrid model must win. Reports and confusion matrices are generated for all three models.\n\n**On Kaggle:** enable GPU, add the Phase 1 Notebook Output as input data, and select **Run All**.\n","metadata":{"papermill":{"duration":0.003967,"end_time":"2026-07-20T11:26:02.391654+00:00","exception":false,"start_time":"2026-07-20T11:26:02.387687+00:00","status":"completed"},"tags":[]}},{"id":"86d220cf","cell_type":"markdown","source":"### Configuration\nSet the Phase 1 input path, output directory, run/CNN/XGBoost hyperparameters, and debug options used throughout the notebook.","metadata":{"papermill":{"duration":0.003193,"end_time":"2026-07-20T11:26:02.398115+00:00","exception":false,"start_time":"2026-07-20T11:26:02.394922+00:00","status":"completed"},"tags":[]}},{"id":"bcdc62b2","cell_type":"code","source":"# --- Phase 1 Input ---\n# Leave as None to auto-detect `phase1_manifest.json` under /kaggle/input.\nPHASE1_ROOT = None\n\n# --- Paths ---\nLABEL_PATH = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\nOUTPUT_DIR = \"/kaggle/working/phase2_results\"\n\n# --- Run Settings ---\nSEED = 42\nN_SPLITS = 5\nEPOCHS = 6\nCNN_PATIENCE = 2\nBATCH_SIZE = 64\nEMBED_BATCH_SIZE = 128\nNUM_WORKERS = 2\nEMBED_DIM = 128\nTORCH_DETERMINISTIC = True\n\n# --- Class Balancing ---\nCLASS_WEIGHT_POWER = 0.5\nMAX_CLASS_WEIGHT = 6.0\n\n# --- XGBoost ---\nXGB_N_ESTIMATORS = 1400\nXGB_EARLY_STOPPING = 60\n# \"auto\" uses GPU when available, with a tested CPU fallback.\nXGB_DEVICE = \"auto\"  # choices: \"auto\", \"cuda\", \"cpu\"\n\n# --- Evaluation ---\nPRIMARY_METRIC = \"log_loss\"  # choices: log_loss, accuracy, macro_f1, weighted_f1\nRUN_TABULAR_BASELINE = True\nSAVE_FOLD_MODELS = False\n\n# --- Debug ---\n# Set an integer for a quick stratified test; None runs all samples.\nDEBUG_MAX_SAMPLES = None\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-07-20T11:26:02.405568Z","iopub.status.busy":"2026-07-20T11:26:02.405341Z","iopub.status.idle":"2026-07-20T11:26:02.412908Z","shell.execute_reply":"2026-07-20T11:26:02.412353Z"},"papermill":{"duration":0.013172,"end_time":"2026-07-20T11:26:02.414331+00:00","exception":false,"start_time":"2026-07-20T11:26:02.401159+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"53821d83","cell_type":"markdown","source":"### Imports, Reproducibility, and Device\nImport dependencies, configure deterministic PyTorch behavior, and record the runtime environment.\n","metadata":{"papermill":{"duration":0.003003,"end_time":"2026-07-20T11:26:02.420433+00:00","exception":false,"start_time":"2026-07-20T11:26:02.417430+00:00","status":"completed"},"tags":[]}},{"id":"5e303c9e","cell_type":"code","source":"import os\n\nos.environ.setdefault(\"CUBLAS_WORKSPACE_CONFIG\", \":4096:8\")\n\nimport copy\nimport gc\nimport json\nimport platform\nimport random\nimport shutil\nimport sys\nimport time\nimport warnings\nimport zipfile\nfrom collections import defaultdict\nfrom datetime import datetime, timezone\nfrom pathlib import Path\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom packaging.version import Version\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\n\nimport xgboost as xgb\nfrom sklearn.metrics import (\n    ConfusionMatrixDisplay,\n    accuracy_score,\n    classification_report,\n    confusion_matrix,\n    f1_score,\n    log_loss,\n)\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\n\n\ndef seed_everything(seed: int) -> None:\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\n    if TORCH_DETERMINISTIC:\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n        torch.use_deterministic_algorithms(True, warn_only=True)\n    else:\n        torch.backends.cudnn.deterministic = False\n        torch.backends.cudnn.benchmark = True\n\n\ndef seed_worker(worker_id: int) -> None:\n    worker_seed = torch.initial_seed() % (2**32)\n    np.random.seed(worker_seed)\n    random.seed(worker_seed)\n\n\nseed_everything(SEED)\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nOUTPUT_PATH = Path(OUTPUT_DIR)\nOUTPUT_PATH.mkdir(parents=True, exist_ok=True)\nNUM_WORKERS = max(0, min(int(NUM_WORKERS), os.cpu_count() or 2))\nAMP_ENABLED = DEVICE.type == \"cuda\"\n\nruntime_info = {\n    \"created_at_utc\": datetime.now(timezone.utc).isoformat(),\n    \"python\": platform.python_version(),\n    \"platform\": platform.platform(),\n    \"numpy\": np.__version__,\n    \"pandas\": pd.__version__,\n    \"pytorch\": torch.__version__,\n    \"torchvision\": getattr(sys.modules.get(\"torchvision\"), \"__version__\", None),\n    \"cuda_available\": bool(torch.cuda.is_available()),\n    \"cuda_device\": torch.cuda.get_device_name(0) if torch.cuda.is_available() else None,\n    \"xgboost\": xgb.__version__,\n    \"sklearn\": __import__(\"sklearn\").__version__,\n}\n\nprint(f\"PyTorch device : {DEVICE}\")\nprint(f\"XGBoost version: {xgb.__version__}\")\nprint(f\"Workers        : {NUM_WORKERS}\")\nprint(f\"Deterministic  : {TORCH_DETERMINISTIC}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:02.427927Z","iopub.status.busy":"2026-07-20T11:26:02.427416Z","iopub.status.idle":"2026-07-20T11:26:23.018945Z","shell.execute_reply":"2026-07-20T11:26:23.017903Z"},"papermill":{"duration":20.597264,"end_time":"2026-07-20T11:26:23.020866+00:00","exception":false,"start_time":"2026-07-20T11:26:02.423602+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9ce2d81b","cell_type":"markdown","source":"### Locate and Validate Phase 1 Output\nAuto-detect the Phase 1 manifest, reject incomplete output, and safely extract the image ZIP using a signature marker that prevents stale-folder reuse.\n","metadata":{"papermill":{"duration":0.004372,"end_time":"2026-07-20T11:26:23.028799+00:00","exception":false,"start_time":"2026-07-20T11:26:23.024427+00:00","status":"completed"},"tags":[]}},{"id":"cb73bb5f","cell_type":"code","source":"def find_phase1_root(explicit_root=None) -> Path:\n    if explicit_root:\n        root = Path(explicit_root)\n        if not (root / \"phase1_manifest.json\").exists():\n            raise FileNotFoundError(f\"phase1_manifest.json not found in {root}\")\n        return root\n\n    candidates = []\n    for base in [Path(\"/kaggle/input\"), Path(\"/kaggle/working\")]:\n        if base.exists():\n            candidates.extend(base.rglob(\"phase1_manifest.json\"))\n\n    if not candidates:\n        raise FileNotFoundError(\n            \"phase1_manifest.json not found. Add the Phase 1 Notebook Output \"\n            \"or set PHASE1_ROOT manually.\"\n        )\n\n    scored = []\n    for manifest_path in candidates:\n        try:\n            info = json.loads(manifest_path.read_text(encoding=\"utf-8\"))\n            status_score = 1 if info.get(\"status\") == \"complete\" else 0\n            scored.append(\n                (\n                    status_score,\n                    int(info.get(\"sample_count\", 0)),\n                    manifest_path.stat().st_mtime,\n                    manifest_path.parent,\n                )\n            )\n        except Exception:\n            continue\n\n    if not scored:\n        raise RuntimeError(\"Manifest files were found, but none could be read.\")\n    return sorted(scored, reverse=True)[0][3]\n\n\nPHASE1_PATH = find_phase1_root(PHASE1_ROOT)\nmanifest_path = PHASE1_PATH / \"phase1_manifest.json\"\nmanifest = json.loads(manifest_path.read_text(encoding=\"utf-8\"))\n\nif manifest.get(\"status\") not in (None, \"complete\"):\n    raise RuntimeError(f\"Phase 1 manifest status is not complete: {manifest.get('status')}\")\nif int(manifest.get(\"missing_feature_count\", 0)) != 0:\n    raise RuntimeError(\"Phase 1 reports missing tabular features.\")\nif int(manifest.get(\"missing_image_count\", 0)) != 0:\n    raise RuntimeError(\"Phase 1 reports missing images.\")\n\nimage_folder_name = manifest.get(\"image_folder_name\", \"train_malware_images\")\nsource_image_dir = PHASE1_PATH / image_folder_name\n\nif source_image_dir.exists() and any(source_image_dir.glob(\"*.png\")):\n    IMAGE_DIR = source_image_dir\nelse:\n    zip_name = manifest.get(\"image_zip_name\") or f\"{image_folder_name}.zip\"\n    zip_path = PHASE1_PATH / zip_name\n    if not zip_path.exists():\n        raise FileNotFoundError(\n            f\"Neither the image folder nor the image ZIP was found: {zip_path}\"\n        )\n\n    IMAGE_DIR = Path(\"/kaggle/working\") / image_folder_name\n    marker_path = IMAGE_DIR / \".extract_signature.json\"\n    signature = {\n        \"zip_name\": zip_path.name,\n        \"zip_size\": zip_path.stat().st_size,\n        \"zip_mtime_ns\": zip_path.stat().st_mtime_ns,\n        \"expected_images\": int(manifest.get(\"sample_count\", 0)),\n    }\n\n    marker_matches = False\n    if marker_path.exists():\n        try:\n            marker_matches = json.loads(marker_path.read_text(encoding=\"utf-8\")) == signature\n        except Exception:\n            marker_matches = False\n\n    if not marker_matches:\n        if IMAGE_DIR.exists():\n            shutil.rmtree(IMAGE_DIR)\n        IMAGE_DIR.mkdir(parents=True, exist_ok=True)\n        print(f\"Extracting {zip_path.name}...\")\n        with zipfile.ZipFile(zip_path, \"r\") as archive:\n            bad_member = archive.testzip()\n            if bad_member is not None:\n                raise RuntimeError(f\"Corrupted image archive member: {bad_member}\")\n            archive.extractall(IMAGE_DIR)\n        marker_path.write_text(json.dumps(signature, indent=2), encoding=\"utf-8\")\n\nprint(f\"Phase 1 root: {PHASE1_PATH}\")\nprint(f\"Image dir   : {IMAGE_DIR}\")\nprint(f\"Image size  : {manifest.get('image_size', 'unknown')}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:23.038265Z","iopub.status.busy":"2026-07-20T11:26:23.037563Z","iopub.status.idle":"2026-07-20T11:26:27.266610Z","shell.execute_reply":"2026-07-20T11:26:27.265725Z"},"papermill":{"duration":4.235477,"end_time":"2026-07-20T11:26:27.268178+00:00","exception":false,"start_time":"2026-07-20T11:26:23.032701+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"347cb96a","cell_type":"markdown","source":"### Load and Strictly Validate Data\nLoad the manifest-defined feature schema, merge it one-to-one with labels, remove zero-variance columns defensively, and stop instead of silently dropping invalid rows or missing images.\n","metadata":{"papermill":{"duration":0.003342,"end_time":"2026-07-20T11:26:27.274945+00:00","exception":false,"start_time":"2026-07-20T11:26:27.271603+00:00","status":"completed"},"tags":[]}},{"id":"c9615fc5","cell_type":"code","source":"if not Path(LABEL_PATH).exists():\n    raise FileNotFoundError(f\"Labels file not found: {LABEL_PATH}\")\n\nparquet_name = manifest.get(\"combined_parquet\")\ncsv_name = manifest.get(\"combined_csv\", \"train_tabular_features.csv\")\n\nif parquet_name and (PHASE1_PATH / parquet_name).exists():\n    features_df = pd.read_parquet(PHASE1_PATH / parquet_name)\nelif (PHASE1_PATH / csv_name).exists():\n    features_df = pd.read_csv(PHASE1_PATH / csv_name, dtype={\"Id\": str})\nelse:\n    batch_files = sorted(\n        (PHASE1_PATH / \"tabular_batches\").glob(\n            \"train_tabular_features_batch_*.csv\"\n        )\n    )\n    if not batch_files:\n        raise FileNotFoundError(\"No tabular feature file from Phase 1 was found.\")\n    features_df = pd.concat(\n        [pd.read_csv(path, dtype={\"Id\": str}) for path in batch_files],\n        ignore_index=True,\n        sort=False,\n    )\n\nlabels_df = pd.read_csv(LABEL_PATH, dtype={\"Id\": str})\nfor dataframe, name in [(features_df, \"features\"), (labels_df, \"labels\")]:\n    if \"Id\" not in dataframe.columns:\n        raise ValueError(f\"The {name} table has no `Id` column.\")\n    dataframe[\"Id\"] = dataframe[\"Id\"].astype(str)\n\nif labels_df[\"Id\"].duplicated().any():\n    raise ValueError(\"Duplicate IDs were found in trainLabels.csv.\")\n\nfeatures_df = features_df.drop_duplicates(\"Id\", keep=\"last\")\ndata_df = features_df.merge(labels_df, on=\"Id\", how=\"inner\", validate=\"one_to_one\")\ndata_df = data_df.sort_values(\"Id\").reset_index(drop=True)\n\nexpected_sample_count = int(manifest.get(\"sample_count\", len(data_df)))\nif DEBUG_MAX_SAMPLES is None and len(data_df) != expected_sample_count:\n    raise RuntimeError(\n        f\"Merged sample count is {len(data_df):,}; Phase 1 manifest expects \"\n        f\"{expected_sample_count:,}.\"\n    )\n\nmanifest_feature_columns = manifest.get(\"feature_columns\")\nif manifest_feature_columns:\n    missing_schema_columns = [\n        column for column in manifest_feature_columns if column not in data_df.columns\n    ]\n    if missing_schema_columns:\n        raise RuntimeError(\n            f\"Phase 1 feature schema is incomplete, examples: {missing_schema_columns[:10]}\"\n        )\n    candidate_feature_columns = list(manifest_feature_columns)\nelse:\n    candidate_feature_columns = [\n        column for column in data_df.columns if column not in [\"Id\", \"Class\"]\n    ]\n\n# Defensive removal also allows Phase 2 to consume an older Phase 1 output safely.\nlegacy_constant_columns = [\n    column\n    for column in [\"has_bytes\", \"has_asm\"]\n    if column in candidate_feature_columns\n]\ncandidate_feature_columns = [\n    column for column in candidate_feature_columns if column not in legacy_constant_columns\n]\n\nfor column in candidate_feature_columns:\n    data_df[column] = pd.to_numeric(data_df[column], errors=\"coerce\")\n\ndata_df[candidate_feature_columns] = data_df[candidate_feature_columns].replace(\n    [np.inf, -np.inf], np.nan\n)\ninvalid_rows = data_df[candidate_feature_columns].isna().any(axis=1)\nif invalid_rows.any():\n    examples = data_df.loc[invalid_rows, \"Id\"].head(10).tolist()\n    raise RuntimeError(\n        f\"Found {int(invalid_rows.sum())} samples with NaN/Inf features, examples: {examples}\"\n    )\n\nconstant_feature_columns = [\n    column\n    for column in candidate_feature_columns\n    if data_df[column].nunique(dropna=False) <= 1\n]\nfeature_columns = [\n    column for column in candidate_feature_columns if column not in constant_feature_columns\n]\nif not feature_columns:\n    raise RuntimeError(\"No usable tabular features remain after validation.\")\n\nmissing_image_ids = [\n    fid\n    for fid in data_df[\"Id\"].astype(str)\n    if not (IMAGE_DIR / f\"{fid}.png\").is_file()\n    or (IMAGE_DIR / f\"{fid}.png\").stat().st_size == 0\n]\nif missing_image_ids:\n    raise RuntimeError(\n        f\"Found {len(missing_image_ids)} missing/empty images, examples: \"\n        f\"{missing_image_ids[:10]}\"\n    )\n\n# --- Optional Debug Subsampling ---\nif DEBUG_MAX_SAMPLES is not None and DEBUG_MAX_SAMPLES < len(data_df):\n    parts = []\n    per_class = max(2, int(DEBUG_MAX_SAMPLES) // data_df[\"Class\"].nunique())\n    for _, group in data_df.groupby(\"Class\"):\n        parts.append(group.sample(min(per_class, len(group)), random_state=SEED))\n    data_df = (\n        pd.concat(parts)\n        .sample(frac=1, random_state=SEED)\n        .reset_index(drop=True)\n    )\n\nclasses = sorted(data_df[\"Class\"].unique().tolist())\nclass_to_index = {label: index for index, label in enumerate(classes)}\nindex_to_class = {index: label for label, index in class_to_index.items()}\n\ny = data_df[\"Class\"].map(class_to_index).to_numpy(dtype=np.int64)\nimage_ids = data_df[\"Id\"].astype(str).tolist()\nX_tabular = data_df[feature_columns].to_numpy(dtype=np.float32)\nnum_classes = len(classes)\n\nmin_class_count = int(np.bincount(y).min())\nif min_class_count < N_SPLITS:\n    raise ValueError(\n        f\"The smallest class has {min_class_count} samples, insufficient for \"\n        f\"{N_SPLITS}-fold CV.\"\n    )\n\nremoved_features = legacy_constant_columns + constant_feature_columns\nprint(f\"Samples                   : {len(y):,}\")\nprint(f\"Tabular features          : {X_tabular.shape[1]:,}\")\nprint(f\"Removed constant features : {len(removed_features):,}\")\nif removed_features:\n    print(f\"Removed columns           : {removed_features[:20]}\")\nprint(f\"Classes                   : {classes}\")\nprint(\"Class counts:\")\nprint(data_df[\"Class\"].value_counts().sort_index().to_string())\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:27.282946Z","iopub.status.busy":"2026-07-20T11:26:27.282565Z","iopub.status.idle":"2026-07-20T11:26:28.642743Z","shell.execute_reply":"2026-07-20T11:26:28.641829Z"},"papermill":{"duration":1.366119,"end_time":"2026-07-20T11:26:28.644498+00:00","exception":false,"start_time":"2026-07-20T11:26:27.278379+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7697350f","cell_type":"markdown","source":"### Dataset and CNN\nDefine the image dataset and the adaptive CNN architecture used to produce a fixed-size embedding from each malware image.","metadata":{"papermill":{"duration":0.003416,"end_time":"2026-07-20T11:26:28.651577+00:00","exception":false,"start_time":"2026-07-20T11:26:28.648161+00:00","status":"completed"},"tags":[]}},{"id":"ab4d4e88","cell_type":"code","source":"IMAGE_SIZE = int(manifest.get(\"image_size\", 224))\n\nimage_transform = transforms.Compose(\n    [\n        transforms.Resize((IMAGE_SIZE, IMAGE_SIZE)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.5], std=[0.5]),\n    ]\n)\n\n\nclass MalwareImageDataset(Dataset):\n    def __init__(self, image_dir, ids, labels=None):\n        self.image_dir = Path(image_dir)\n        self.ids = list(ids)\n        self.labels = None if labels is None else np.asarray(labels, dtype=np.int64)\n\n    def __len__(self):\n        return len(self.ids)\n\n    def __getitem__(self, index):\n        fid = self.ids[index]\n        path = self.image_dir / f\"{fid}.png\"\n        try:\n            with Image.open(path) as image:\n                tensor = image_transform(image.convert(\"L\"))\n        except Exception as exc:\n            raise RuntimeError(f\"Could not read image {path}: {exc}\") from exc\n\n        if self.labels is None:\n            return tensor\n        return tensor, torch.tensor(self.labels[index], dtype=torch.long)\n\n\nclass MalwareCNN(nn.Module):\n    def __init__(self, num_classes: int, embed_dim: int = 128):\n        super().__init__()\n        self.backbone = nn.Sequential(\n            nn.Conv2d(1, 32, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(32),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(32, 64, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(64),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(64, 128, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(128),\n            nn.SiLU(inplace=True),\n            nn.MaxPool2d(2),\n            nn.Conv2d(128, 192, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(192),\n            nn.SiLU(inplace=True),\n            nn.AdaptiveAvgPool2d(1),\n        )\n        self.embedding = nn.Sequential(\n            nn.Flatten(),\n            nn.Linear(192, embed_dim),\n            nn.LayerNorm(embed_dim),\n            nn.SiLU(inplace=True),\n            nn.Dropout(0.25),\n        )\n        self.classifier = nn.Linear(embed_dim, num_classes)\n\n    def forward_features(self, inputs):\n        return self.embedding(self.backbone(inputs))\n\n    def forward(self, inputs):\n        return self.classifier(self.forward_features(inputs))\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:28.659675Z","iopub.status.busy":"2026-07-20T11:26:28.659317Z","iopub.status.idle":"2026-07-20T11:26:28.670202Z","shell.execute_reply":"2026-07-20T11:26:28.669594Z"},"papermill":{"duration":0.016291,"end_time":"2026-07-20T11:26:28.671518+00:00","exception":false,"start_time":"2026-07-20T11:26:28.655227+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"14e2705c","cell_type":"markdown","source":"### CNN Training and Inference\nBuild deterministic data loaders, compute moderate class weights, and train one fold-specific CNN using only an inner split for early stopping.\n","metadata":{"papermill":{"duration":0.003468,"end_time":"2026-07-20T11:26:28.678125+00:00","exception":false,"start_time":"2026-07-20T11:26:28.674657+00:00","status":"completed"},"tags":[]}},{"id":"379f4e97","cell_type":"code","source":"def make_loader(\n    ids,\n    labels=None,\n    shuffle=False,\n    batch_size=BATCH_SIZE,\n    seed_offset=0,\n):\n    dataset = MalwareImageDataset(IMAGE_DIR, ids, labels)\n    generator = torch.Generator()\n    generator.manual_seed(SEED + int(seed_offset))\n    return DataLoader(\n        dataset,\n        batch_size=batch_size,\n        shuffle=shuffle,\n        num_workers=NUM_WORKERS,\n        pin_memory=AMP_ENABLED,\n        persistent_workers=NUM_WORKERS > 0,\n        drop_last=False,\n        worker_init_fn=seed_worker if NUM_WORKERS > 0 else None,\n        generator=generator,\n    )\n\n\ndef compute_balancing_weights(labels):\n    counts = np.bincount(labels, minlength=num_classes).astype(np.float64)\n    class_weights = (\n        len(labels) / (num_classes * np.maximum(counts, 1))\n    ) ** CLASS_WEIGHT_POWER\n    class_weights = np.clip(class_weights, 0.5, MAX_CLASS_WEIGHT)\n    sample_weights = class_weights[labels]\n    return class_weights.astype(np.float32), sample_weights.astype(np.float32)\n\n\n@torch.no_grad()\ndef predict_cnn(model, loader, return_embeddings=False):\n    model.eval()\n    all_probs, all_embeddings, all_labels = [], [], []\n\n    for batch in loader:\n        if isinstance(batch, (tuple, list)):\n            inputs, labels = batch\n            all_labels.append(labels.numpy())\n        else:\n            inputs = batch\n\n        inputs = inputs.to(DEVICE, non_blocking=True)\n        with torch.autocast(device_type=DEVICE.type, enabled=AMP_ENABLED):\n            embeddings = model.forward_features(inputs)\n            logits = model.classifier(embeddings)\n            probabilities = torch.softmax(logits, dim=1)\n\n        all_probs.append(probabilities.cpu().numpy())\n        if return_embeddings:\n            all_embeddings.append(embeddings.cpu().numpy())\n\n    probabilities = np.concatenate(all_probs, axis=0)\n    embeddings = (\n        np.concatenate(all_embeddings, axis=0) if return_embeddings else None\n    )\n    labels = np.concatenate(all_labels, axis=0) if all_labels else None\n    return probabilities, embeddings, labels\n\n\ndef train_cnn_fold(outer_train_ids, outer_train_y, outer_val_ids, fold_number):\n    \"\"\"Train one CNN without using the outer validation fold for model selection.\"\"\"\n    positions = np.arange(len(outer_train_y))\n    inner_train_pos, inner_stop_pos = train_test_split(\n        positions,\n        test_size=0.12,\n        random_state=SEED + fold_number,\n        stratify=outer_train_y,\n    )\n\n    inner_train_ids = [outer_train_ids[index] for index in inner_train_pos]\n    inner_stop_ids = [outer_train_ids[index] for index in inner_stop_pos]\n    inner_train_y = outer_train_y[inner_train_pos]\n    inner_stop_y = outer_train_y[inner_stop_pos]\n\n    train_loader = make_loader(\n        inner_train_ids,\n        inner_train_y,\n        shuffle=True,\n        batch_size=BATCH_SIZE,\n        seed_offset=fold_number,\n    )\n    stop_loader = make_loader(\n        inner_stop_ids,\n        inner_stop_y,\n        shuffle=False,\n        batch_size=EMBED_BATCH_SIZE,\n        seed_offset=100 + fold_number,\n    )\n\n    class_weights, _ = compute_balancing_weights(inner_train_y)\n    criterion = nn.CrossEntropyLoss(\n        weight=torch.tensor(class_weights, dtype=torch.float32, device=DEVICE)\n    )\n\n    model = MalwareCNN(num_classes=num_classes, embed_dim=EMBED_DIM).to(DEVICE)\n    optimizer = optim.AdamW(model.parameters(), lr=1e-3, weight_decay=1e-4)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer,\n        mode=\"min\",\n        factor=0.5,\n        patience=1,\n        min_lr=1e-5,\n    )\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=AMP_ENABLED)\n\n    best_loss = float(\"inf\")\n    best_state = None\n    best_epoch = 0\n    epochs_without_improvement = 0\n    history = []\n\n    print(f\"    Fold {fold_number} — CNN training\")\n    print(f\"    {'Epoch':^7} | {'Train Loss':>10} | {'Inner Loss':>10} | {'Inner Acc':>9}\")\n    print(f\"    {'-' * 7}-+-{'-' * 10}-+-{'-' * 10}-+-{'-' * 9}\")\n\n    for epoch in range(1, EPOCHS + 1):\n        model.train()\n        running_loss = 0.0\n        seen = 0\n\n        for inputs, labels in train_loader:\n            inputs = inputs.to(DEVICE, non_blocking=True)\n            labels = labels.to(DEVICE, non_blocking=True)\n            optimizer.zero_grad(set_to_none=True)\n\n            with torch.autocast(device_type=DEVICE.type, enabled=AMP_ENABLED):\n                logits = model(inputs)\n                loss = criterion(logits, labels)\n\n            scaler.scale(loss).backward()\n            scaler.step(optimizer)\n            scaler.update()\n\n            batch_size = labels.size(0)\n            running_loss += loss.item() * batch_size\n            seen += batch_size\n\n        train_loss = running_loss / max(seen, 1)\n        stop_probs, _, stop_labels = predict_cnn(model, stop_loader)\n        stop_loss = log_loss(\n            stop_labels,\n            stop_probs,\n            labels=np.arange(num_classes),\n        )\n        stop_accuracy = accuracy_score(stop_labels, stop_probs.argmax(axis=1))\n        scheduler.step(stop_loss)\n\n        history.append(\n            {\n                \"fold\": fold_number,\n                \"epoch\": epoch,\n                \"train_loss\": train_loss,\n                \"inner_val_logloss\": stop_loss,\n                \"inner_val_accuracy\": stop_accuracy,\n                \"learning_rate\": optimizer.param_groups[0][\"lr\"],\n            }\n        )\n\n        is_best = stop_loss < best_loss - 1e-4\n        marker = \" *\" if is_best else \"\"\n        print(\n            f\"    {f'{epoch}/{EPOCHS}':^7} | {train_loss:>10.4f} | \"\n            f\"{stop_loss:>10.4f} | {stop_accuracy:>9.4f}{marker}\"\n        )\n\n        if is_best:\n            best_loss = stop_loss\n            best_epoch = epoch\n            best_state = copy.deepcopy(model.state_dict())\n            epochs_without_improvement = 0\n        else:\n            epochs_without_improvement += 1\n            if epochs_without_improvement >= CNN_PATIENCE:\n                print(\n                    f\"    Early stopping — best epoch {best_epoch}, \"\n                    f\"inner log-loss {best_loss:.4f}\"\n                )\n                break\n\n    if best_state is None:\n        raise RuntimeError(\"CNN did not produce a valid best state.\")\n    model.load_state_dict(best_state)\n\n    if SAVE_FOLD_MODELS:\n        torch.save(model.state_dict(), OUTPUT_PATH / f\"cnn_fold_{fold_number}.pth\")\n\n    outer_train_loader = make_loader(\n        outer_train_ids,\n        labels=None,\n        shuffle=False,\n        batch_size=EMBED_BATCH_SIZE,\n        seed_offset=200 + fold_number,\n    )\n    outer_val_loader = make_loader(\n        outer_val_ids,\n        labels=None,\n        shuffle=False,\n        batch_size=EMBED_BATCH_SIZE,\n        seed_offset=300 + fold_number,\n    )\n\n    _, train_embeddings, _ = predict_cnn(\n        model,\n        outer_train_loader,\n        return_embeddings=True,\n    )\n    val_probs, val_embeddings, _ = predict_cnn(\n        model,\n        outer_val_loader,\n        return_embeddings=True,\n    )\n\n    del train_loader, stop_loader, outer_train_loader, outer_val_loader\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n    gc.collect()\n    return model, train_embeddings, val_embeddings, val_probs, history, best_epoch\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:28.685889Z","iopub.status.busy":"2026-07-20T11:26:28.685471Z","iopub.status.idle":"2026-07-20T11:26:28.703443Z","shell.execute_reply":"2026-07-20T11:26:28.702826Z"},"papermill":{"duration":0.02347,"end_time":"2026-07-20T11:26:28.704743+00:00","exception":false,"start_time":"2026-07-20T11:26:28.681273+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8554a9b2","cell_type":"markdown","source":"### XGBoost Helpers\nChoose the number of trees on an inner split, refit on the full outer-training fold, use device-aware prediction to avoid CPU/GPU mismatch warnings, and fall back to CPU only if necessary.\n","metadata":{"papermill":{"duration":0.003032,"end_time":"2026-07-20T11:26:28.711008+00:00","exception":false,"start_time":"2026-07-20T11:26:28.707976+00:00","status":"completed"},"tags":[]}},{"id":"faa545ff","cell_type":"code","source":"def xgb_device_params():\n    requested = str(XGB_DEVICE).lower()\n    if requested not in {\"auto\", \"cuda\", \"cpu\"}:\n        raise ValueError(\"XGB_DEVICE must be one of: auto, cuda, cpu\")\n\n    use_cuda = requested == \"cuda\" or (\n        requested == \"auto\" and torch.cuda.is_available()\n    )\n    params = {\"tree_method\": \"hist\"}\n    if use_cuda:\n        if Version(xgb.__version__) >= Version(\"2.0.0\"):\n            params[\"device\"] = \"cuda\"\n        else:\n            params[\"tree_method\"] = \"gpu_hist\"\n            params[\"predictor\"] = \"gpu_predictor\"\n    return params\n\n\ndef build_xgb_model(n_estimators, use_early_stopping=False):\n    params = {\n        \"n_estimators\": int(n_estimators),\n        \"max_depth\": 6,\n        \"learning_rate\": 0.03,\n        \"min_child_weight\": 1.0,\n        \"subsample\": 0.85,\n        \"colsample_bytree\": 0.80,\n        \"reg_alpha\": 0.05,\n        \"reg_lambda\": 2.0,\n        \"gamma\": 0.0,\n        \"max_bin\": 256,\n        \"objective\": \"multi:softprob\",\n        \"num_class\": num_classes,\n        \"eval_metric\": \"mlogloss\",\n        \"random_state\": SEED,\n        \"n_jobs\": -1,\n    }\n    if use_early_stopping:\n        params[\"early_stopping_rounds\"] = XGB_EARLY_STOPPING\n    params.update(xgb_device_params())\n    return xgb.XGBClassifier(**params)\n\n\ndef fit_model_with_optional_cpu_fallback(model, *fit_args, **fit_kwargs):\n    try:\n        model.fit(*fit_args, **fit_kwargs)\n        return model\n    except xgb.core.XGBoostError as exc:\n        model_params = model.get_params()\n        was_gpu = (\n            model_params.get(\"device\") == \"cuda\"\n            or model_params.get(\"tree_method\") == \"gpu_hist\"\n        )\n        if not was_gpu:\n            raise\n\n        print(f\"XGBoost GPU error; retrying on CPU: {str(exc)[:180]}\")\n        model_params.pop(\"device\", None)\n        model_params.pop(\"predictor\", None)\n        model_params[\"tree_method\"] = \"hist\"\n        cpu_model = xgb.XGBClassifier(**model_params)\n        cpu_model.fit(*fit_args, **fit_kwargs)\n        return cpu_model\n\n\ndef predict_proba_device_aware(model, features):\n    \"\"\"Use CuPy input for a CUDA model, preventing the device-mismatch warning.\"\"\"\n    params = model.get_params()\n    uses_gpu = (\n        params.get(\"device\") == \"cuda\"\n        or params.get(\"tree_method\") == \"gpu_hist\"\n    )\n\n    if uses_gpu:\n        try:\n            import cupy as cp\n\n            probabilities = model.predict_proba(cp.asarray(features))\n            if isinstance(probabilities, cp.ndarray):\n                probabilities = cp.asnumpy(probabilities)\n            return np.asarray(probabilities, dtype=np.float32)\n        except (ImportError, ModuleNotFoundError):\n            warnings.warn(\n                \"CuPy is unavailable; moving this XGBoost model to CPU for prediction.\",\n                RuntimeWarning,\n            )\n            model.get_booster().set_param({\"device\": \"cpu\"})\n        except xgb.core.XGBoostError as exc:\n            warnings.warn(\n                f\"CUDA prediction failed; using CPU prediction: {str(exc)[:160]}\",\n                RuntimeWarning,\n            )\n            model.get_booster().set_param({\"device\": \"cpu\"})\n\n    return np.asarray(model.predict_proba(features), dtype=np.float32)\n\n\ndef fit_xgb_nested(\n    train_x,\n    train_y,\n    outer_val_x,\n    sample_weight,\n    model_name,\n    fold_number,\n):\n    \"\"\"Select tree count on an inner split, then refit on the full outer train fold.\"\"\"\n    positions = np.arange(len(train_y))\n    fit_pos, stop_pos = train_test_split(\n        positions,\n        test_size=0.12,\n        random_state=SEED + 100 + fold_number,\n        stratify=train_y,\n    )\n\n    tuning_model = build_xgb_model(\n        XGB_N_ESTIMATORS,\n        use_early_stopping=True,\n    )\n    tuning_model = fit_model_with_optional_cpu_fallback(\n        tuning_model,\n        train_x[fit_pos],\n        train_y[fit_pos],\n        sample_weight=sample_weight[fit_pos],\n        eval_set=[(train_x[stop_pos], train_y[stop_pos])],\n        verbose=False,\n    )\n\n    best_iteration = getattr(tuning_model, \"best_iteration\", None)\n    best_n_estimators = (\n        int(best_iteration + 1)\n        if best_iteration is not None\n        else XGB_N_ESTIMATORS\n    )\n    best_n_estimators = max(30, min(best_n_estimators, XGB_N_ESTIMATORS))\n    print(f\"    {model_name}: selected {best_n_estimators} trees\")\n\n    final_model = build_xgb_model(\n        best_n_estimators,\n        use_early_stopping=False,\n    )\n    final_model = fit_model_with_optional_cpu_fallback(\n        final_model,\n        train_x,\n        train_y,\n        sample_weight=sample_weight,\n        verbose=False,\n    )\n    probabilities = predict_proba_device_aware(final_model, outer_val_x)\n\n    del tuning_model\n    gc.collect()\n    return final_model, probabilities, best_n_estimators\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:28.718615Z","iopub.status.busy":"2026-07-20T11:26:28.718244Z","iopub.status.idle":"2026-07-20T11:26:28.728118Z","shell.execute_reply":"2026-07-20T11:26:28.727497Z"},"papermill":{"duration":0.015379,"end_time":"2026-07-20T11:26:28.729558+00:00","exception":false,"start_time":"2026-07-20T11:26:28.714179+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7ee3bb59","cell_type":"markdown","source":"### Leakage-Free Stratified K-Fold Evaluation\nTrain the three models on the same five outer folds, collect complete OOF predictions, timing, fold metrics, and feature importance.\n","metadata":{"papermill":{"duration":0.003139,"end_time":"2026-07-20T11:26:28.735864+00:00","exception":false,"start_time":"2026-07-20T11:26:28.732725+00:00","status":"completed"},"tags":[]}},{"id":"adf2d89c","cell_type":"code","source":"def metric_values(y_true, probabilities):\n    predictions = probabilities.argmax(axis=1)\n    return {\n        \"accuracy\": float(accuracy_score(y_true, predictions)),\n        \"log_loss\": float(\n            log_loss(y_true, probabilities, labels=np.arange(num_classes))\n        ),\n        \"macro_f1\": float(f1_score(y_true, predictions, average=\"macro\")),\n        \"weighted_f1\": float(\n            f1_score(y_true, predictions, average=\"weighted\")\n        ),\n    }\n\n\ndef print_fold_result(name, metrics, seconds):\n    print(\n        f\"    {name:<10} | acc {metrics['accuracy']:.4f} | \"\n        f\"logloss {metrics['log_loss']:.4f} | \"\n        f\"macro-F1 {metrics['macro_f1']:.4f} | {seconds / 60:.1f} min\"\n    )\n\n\nskf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)\n\noof_cnn = np.zeros((len(y), num_classes), dtype=np.float32)\noof_tabular = np.zeros((len(y), num_classes), dtype=np.float32)\noof_hybrid = np.zeros((len(y), num_classes), dtype=np.float32)\n\nfold_records = []\ncnn_history_records = []\ntabular_importance_sum = defaultdict(float)\nhybrid_importance_sum = defaultdict(float)\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X_tabular, y), start=1):\n    print(\"\\n\" + \"=\" * 72)\n    print(f\"FOLD {fold}/{N_SPLITS}  (train={len(train_idx):,} | val={len(val_idx):,})\")\n    print(\"=\" * 72)\n\n    train_ids = [image_ids[index] for index in train_idx]\n    val_ids = [image_ids[index] for index in val_idx]\n    y_train = y[train_idx]\n    y_val = y[val_idx]\n    _, sample_weights = compute_balancing_weights(y_train)\n\n    # --- CNN Image-Only ---\n    cnn_start = time.perf_counter()\n    (\n        cnn_model,\n        train_embeddings,\n        val_embeddings,\n        cnn_probs,\n        history,\n        cnn_best_epoch,\n    ) = train_cnn_fold(train_ids, y_train, val_ids, fold)\n    cnn_seconds = time.perf_counter() - cnn_start\n    oof_cnn[val_idx] = cnn_probs\n    cnn_history_records.extend(history)\n    cnn_metrics = metric_values(y_val, cnn_probs)\n\n    # --- XGBoost Tabular-Only ---\n    if RUN_TABULAR_BASELINE:\n        tabular_start = time.perf_counter()\n        tabular_model, tabular_probs, tabular_best_trees = fit_xgb_nested(\n            X_tabular[train_idx],\n            y_train,\n            X_tabular[val_idx],\n            sample_weights,\n            model_name=\"Tabular XGBoost\",\n            fold_number=fold,\n        )\n        tabular_seconds = time.perf_counter() - tabular_start\n        oof_tabular[val_idx] = tabular_probs\n        tabular_metrics = metric_values(y_val, tabular_probs)\n\n        tabular_importance = tabular_model.get_booster().get_score(\n            importance_type=\"gain\"\n        )\n        for key, value in tabular_importance.items():\n            if key.startswith(\"f\") and key[1:].isdigit():\n                tabular_importance_sum[int(key[1:])] += float(value)\n    else:\n        tabular_model = None\n        tabular_best_trees = np.nan\n        tabular_seconds = np.nan\n        tabular_metrics = {\n            key: np.nan\n            for key in [\"accuracy\", \"log_loss\", \"macro_f1\", \"weighted_f1\"]\n        }\n\n    # --- Hybrid: Fold-Specific CNN Embedding + Tabular Features ---\n    hybrid_train = np.hstack([train_embeddings, X_tabular[train_idx]])\n    hybrid_val = np.hstack([val_embeddings, X_tabular[val_idx]])\n\n    hybrid_start = time.perf_counter()\n    hybrid_model, hybrid_probs, hybrid_best_trees = fit_xgb_nested(\n        hybrid_train,\n        y_train,\n        hybrid_val,\n        sample_weights,\n        model_name=\"Hybrid XGBoost\",\n        fold_number=fold,\n    )\n    hybrid_seconds = time.perf_counter() - hybrid_start\n    oof_hybrid[val_idx] = hybrid_probs\n    hybrid_metrics = metric_values(y_val, hybrid_probs)\n\n    hybrid_importance = hybrid_model.get_booster().get_score(\n        importance_type=\"gain\"\n    )\n    for key, value in hybrid_importance.items():\n        if key.startswith(\"f\") and key[1:].isdigit():\n            hybrid_importance_sum[int(key[1:])] += float(value)\n\n    print(f\"    {'Model':<10} | Fold results\")\n    print(f\"    {'-' * 10}-+-{'-' * 56}\")\n    print_fold_result(\"CNN\", cnn_metrics, cnn_seconds)\n    if RUN_TABULAR_BASELINE:\n        print_fold_result(\"Tabular\", tabular_metrics, tabular_seconds)\n    print_fold_result(\"Hybrid\", hybrid_metrics, hybrid_seconds)\n\n    record = {\n        \"fold\": fold,\n        \"cnn_best_epoch\": int(cnn_best_epoch),\n        \"tabular_best_trees\": float(tabular_best_trees),\n        \"hybrid_best_trees\": int(hybrid_best_trees),\n        \"cnn_seconds\": float(cnn_seconds),\n        \"tabular_seconds\": float(tabular_seconds),\n        \"hybrid_seconds\": float(hybrid_seconds),\n    }\n    for prefix, metrics in [\n        (\"cnn\", cnn_metrics),\n        (\"tabular\", tabular_metrics),\n        (\"hybrid\", hybrid_metrics),\n    ]:\n        for metric_name, value in metrics.items():\n            record[f\"{prefix}_{metric_name}\"] = value\n    fold_records.append(record)\n\n    if SAVE_FOLD_MODELS:\n        torch.save(cnn_model.state_dict(), OUTPUT_PATH / f\"cnn_fold_{fold}.pth\")\n        hybrid_model.save_model(OUTPUT_PATH / f\"hybrid_xgb_fold_{fold}.json\")\n        if tabular_model is not None:\n            tabular_model.save_model(OUTPUT_PATH / f\"tabular_xgb_fold_{fold}.json\")\n\n    del cnn_model, tabular_model, hybrid_model\n    del train_embeddings, val_embeddings, hybrid_train, hybrid_val\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n    gc.collect()\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:26:28.743673Z","iopub.status.busy":"2026-07-20T11:26:28.743223Z","iopub.status.idle":"2026-07-20T11:41:47.932836Z","shell.execute_reply":"2026-07-20T11:41:47.932076Z"},"papermill":{"duration":919.195488,"end_time":"2026-07-20T11:41:47.934636+00:00","exception":false,"start_time":"2026-07-20T11:26:28.739148+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"105cbeb7","cell_type":"markdown","source":"### OOF Evaluation and Artifacts\nSelect the primary model from the actual OOF result, save reports and confusion matrices for every model, and write all metrics, predictions, configuration, and feature importance files.\n","metadata":{"papermill":{"duration":0.005392,"end_time":"2026-07-20T11:41:47.945452+00:00","exception":false,"start_time":"2026-07-20T11:41:47.940060+00:00","status":"completed"},"tags":[]}},{"id":"dad2568f","cell_type":"code","source":"def summarize_predictions(name, probabilities):\n    predictions = probabilities.argmax(axis=1)\n    summary = {\"model\": name, **metric_values(y, probabilities)}\n    return summary, predictions\n\n\nmodel_results = {\n    \"CNN image-only\": {\"probs\": oof_cnn},\n    \"Hybrid CNN + tabular\": {\"probs\": oof_hybrid},\n}\nif RUN_TABULAR_BASELINE:\n    model_results[\"XGBoost tabular-only\"] = {\"probs\": oof_tabular}\n\nsummaries = []\nfor model_name, result in model_results.items():\n    summary, predictions = summarize_predictions(model_name, result[\"probs\"])\n    result[\"summary\"] = summary\n    result[\"predictions\"] = predictions\n    summaries.append(summary)\n\nsummary_df = pd.DataFrame(summaries)\nmetric_directions = {\n    \"log_loss\": \"min\",\n    \"accuracy\": \"max\",\n    \"macro_f1\": \"max\",\n    \"weighted_f1\": \"max\",\n}\nif PRIMARY_METRIC not in metric_directions:\n    raise ValueError(f\"Unsupported PRIMARY_METRIC: {PRIMARY_METRIC}\")\n\nascending = metric_directions[PRIMARY_METRIC] == \"min\"\nsummary_df = summary_df.sort_values(\n    PRIMARY_METRIC,\n    ascending=ascending,\n).reset_index(drop=True)\nprimary_model_name = str(summary_df.iloc[0][\"model\"])\nprimary_summary = model_results[primary_model_name][\"summary\"]\n\nfold_df = pd.DataFrame(fold_records)\nhistory_df = pd.DataFrame(cnn_history_records)\n\nprint(\"\\n\" + \"=\" * 72)\nprint(\"OOF MODEL COMPARISON\")\nprint(\"=\" * 72)\nprint(summary_df.to_string(index=False, float_format=lambda value: f\"{value:.5f}\"))\nprint(f\"\\nPrimary model by {PRIMARY_METRIC}: {primary_model_name}\")\n\n# --- Save Core Tables ---\nsummary_df.to_csv(OUTPUT_PATH / \"model_comparison.csv\", index=False)\nfold_df.to_csv(OUTPUT_PATH / \"fold_metrics.csv\", index=False)\nhistory_df.to_csv(OUTPUT_PATH / \"cnn_training_history.csv\", index=False)\n\n# Mean and standard deviation across folds.\nfold_statistics = []\nmodel_prefixes = {\n    \"CNN image-only\": \"cnn\",\n    \"XGBoost tabular-only\": \"tabular\",\n    \"Hybrid CNN + tabular\": \"hybrid\",\n}\nfor model_name in model_results:\n    prefix = model_prefixes[model_name]\n    row = {\"model\": model_name}\n    for metric_name in [\"accuracy\", \"log_loss\", \"macro_f1\", \"weighted_f1\"]:\n        values = fold_df[f\"{prefix}_{metric_name}\"]\n        row[f\"{metric_name}_mean\"] = float(values.mean())\n        row[f\"{metric_name}_std\"] = float(values.std(ddof=1))\n    row[\"training_seconds_mean\"] = float(fold_df[f\"{prefix}_seconds\"].mean())\n    row[\"training_seconds_total\"] = float(fold_df[f\"{prefix}_seconds\"].sum())\n    fold_statistics.append(row)\nfold_statistics_df = pd.DataFrame(fold_statistics)\nfold_statistics_df.to_csv(OUTPUT_PATH / \"fold_summary_mean_std.csv\", index=False)\n\n# --- OOF Predictions and Probabilities ---\noof_df = pd.DataFrame(\n    {\n        \"Id\": image_ids,\n        \"true_class\": [index_to_class[int(value)] for value in y],\n    }\n)\nshort_names = {\n    \"CNN image-only\": \"cnn\",\n    \"XGBoost tabular-only\": \"tabular\",\n    \"Hybrid CNN + tabular\": \"hybrid\",\n}\nfor model_name, result in model_results.items():\n    prefix = short_names[model_name]\n    oof_df[f\"{prefix}_pred_class\"] = [\n        index_to_class[int(value)] for value in result[\"predictions\"]\n    ]\n    for class_index, class_label in enumerate(classes):\n        oof_df[f\"{prefix}_prob_class_{class_label}\"] = result[\"probs\"][:, class_index]\noof_df.to_csv(OUTPUT_PATH / \"oof_predictions.csv\", index=False)\n\n# --- Per-Class Reports and Confusion Matrices for Every Model ---\ntarget_names = [f\"Class {label}\" for label in classes]\nfor model_name, result in model_results.items():\n    prefix = short_names[model_name]\n    report = classification_report(\n        y,\n        result[\"predictions\"],\n        labels=np.arange(num_classes),\n        target_names=target_names,\n        output_dict=True,\n        digits=6,\n        zero_division=0,\n    )\n    report_df = pd.DataFrame(report).transpose()\n    report_df.to_csv(OUTPUT_PATH / f\"{prefix}_classification_report.csv\")\n\n    print(f\"\\nClassification report — {model_name}:\")\n    print(report_df.to_string(float_format=lambda value: f\"{value:.4f}\"))\n\n    matrix = confusion_matrix(\n        y,\n        result[\"predictions\"],\n        labels=np.arange(num_classes),\n    )\n    fig, ax = plt.subplots(figsize=(10, 9))\n    display = ConfusionMatrixDisplay(\n        confusion_matrix=matrix,\n        display_labels=classes,\n    )\n    display.plot(ax=ax, values_format=\"d\", colorbar=False)\n    model_summary = result[\"summary\"]\n    ax.set_title(\n        f\"{model_name} — OOF Confusion Matrix\\n\"\n        f\"Accuracy={model_summary['accuracy']:.4f} | \"\n        f\"Log-loss={model_summary['log_loss']:.4f} | \"\n        f\"Macro-F1={model_summary['macro_f1']:.4f}\"\n    )\n    fig.tight_layout()\n    fig.savefig(\n        OUTPUT_PATH / f\"{prefix}_confusion_matrix.png\",\n        dpi=160,\n        bbox_inches=\"tight\",\n    )\n    plt.show()\n    plt.close(fig)\n\n# --- CNN Learning Curves ---\nif not history_df.empty:\n    fig, ax = plt.subplots(figsize=(10, 6))\n    for fold, group in history_df.groupby(\"fold\"):\n        ax.plot(\n            group[\"epoch\"],\n            group[\"inner_val_logloss\"],\n            marker=\"o\",\n            label=f\"Fold {fold}\",\n        )\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"Inner-validation log-loss\")\n    ax.set_title(\"CNN inner-validation log-loss by fold\")\n    ax.grid(alpha=0.25)\n    ax.legend()\n    fig.tight_layout()\n    fig.savefig(\n        OUTPUT_PATH / \"cnn_validation_curves.png\",\n        dpi=160,\n        bbox_inches=\"tight\",\n    )\n    plt.show()\n    plt.close(fig)\n\n# --- Feature Importance ---\ndef save_feature_importance(importance_sum, names, filename):\n    rows = []\n    for feature_index, total_gain in importance_sum.items():\n        if 0 <= feature_index < len(names):\n            rows.append(\n                {\n                    \"feature\": names[feature_index],\n                    \"mean_gain\": total_gain / N_SPLITS,\n                }\n            )\n    dataframe = pd.DataFrame(rows)\n    if not dataframe.empty:\n        dataframe = dataframe.sort_values(\"mean_gain\", ascending=False).reset_index(drop=True)\n    dataframe.to_csv(OUTPUT_PATH / filename, index=False)\n    return dataframe\n\n\ntabular_importance_df = save_feature_importance(\n    tabular_importance_sum,\n    feature_columns,\n    \"tabular_feature_importance.csv\",\n)\nhybrid_feature_names = [f\"cnn_embed_{index}\" for index in range(EMBED_DIM)] + feature_columns\nhybrid_importance_df = save_feature_importance(\n    hybrid_importance_sum,\n    hybrid_feature_names,\n    \"hybrid_feature_importance.csv\",\n)\n\nif not tabular_importance_df.empty:\n    print(\"\\nTop 20 tabular features:\")\n    print(tabular_importance_df.head(20).to_string(index=False, float_format=lambda x: f\"{x:.3f}\"))\nif not hybrid_importance_df.empty:\n    print(\"\\nTop 20 hybrid features:\")\n    print(hybrid_importance_df.head(20).to_string(index=False, float_format=lambda x: f\"{x:.3f}\"))\n\n# --- Configuration and Final Summary ---\nexperiment_config = {\n    \"seed\": SEED,\n    \"n_splits\": N_SPLITS,\n    \"epochs\": EPOCHS,\n    \"cnn_patience\": CNN_PATIENCE,\n    \"batch_size\": BATCH_SIZE,\n    \"embedding_batch_size\": EMBED_BATCH_SIZE,\n    \"embedding_dimension\": EMBED_DIM,\n    \"class_weight_power\": CLASS_WEIGHT_POWER,\n    \"max_class_weight\": MAX_CLASS_WEIGHT,\n    \"xgb_n_estimators\": XGB_N_ESTIMATORS,\n    \"xgb_early_stopping\": XGB_EARLY_STOPPING,\n    \"xgb_device\": XGB_DEVICE,\n    \"primary_metric\": PRIMARY_METRIC,\n    \"torch_deterministic\": TORCH_DETERMINISTIC,\n    \"feature_count\": len(feature_columns),\n    \"removed_constant_features\": removed_features,\n}\n(OUTPUT_PATH / \"experiment_config.json\").write_text(\n    json.dumps(experiment_config, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n(OUTPUT_PATH / \"runtime_environment.json\").write_text(\n    json.dumps(runtime_info, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nfinal_summary = {\n    \"phase\": 2,\n    \"version\": \"3.0\",\n    \"status\": \"complete\",\n    \"completed_at_utc\": datetime.now(timezone.utc).isoformat(),\n    \"sample_count\": int(len(y)),\n    \"feature_count\": int(len(feature_columns)),\n    \"n_splits\": int(N_SPLITS),\n    \"classes\": [\n        int(value) if isinstance(value, (int, np.integer)) else value\n        for value in classes\n    ],\n    \"primary_metric\": PRIMARY_METRIC,\n    \"primary_model\": primary_model_name,\n    \"primary_metrics\": primary_summary,\n    \"metrics\": summaries,\n    \"notes\": [\n        \"No competition submission was generated.\",\n        \"Outer-validation samples are unseen during CNN training and early stopping.\",\n        \"XGBoost tree counts are selected using an inner-validation split.\",\n        \"No SMOTE or StandardScaler was used for XGBoost.\",\n        \"Constant tabular features were removed before training.\",\n    ],\n}\n(OUTPUT_PATH / \"results_summary.json\").write_text(\n    json.dumps(final_summary, ensure_ascii=False, indent=2),\n    encoding=\"utf-8\",\n)\n\nprint(\"\\n\" + \"=\" * 72)\nprint(\"PHASE 2 CORE EVALUATION COMPLETE\")\nprint(f\"Primary model       : {primary_model_name}\")\nprint(f\"Primary OOF Accuracy: {primary_summary['accuracy'] * 100:.2f}%\")\nprint(f\"Primary OOF Log-loss: {primary_summary['log_loss']:.5f}\")\nprint(f\"Primary OOF Macro-F1: {primary_summary['macro_f1']:.5f}\")\nprint(f\"Artifacts           : {OUTPUT_PATH}\")\nprint(\"=\" * 72)\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:41:47.967124Z","iopub.status.busy":"2026-07-20T11:41:47.966817Z","iopub.status.idle":"2026-07-20T11:41:49.480964Z","shell.execute_reply":"2026-07-20T11:41:49.480167Z"},"papermill":{"duration":1.532712,"end_time":"2026-07-20T11:41:49.483165+00:00","exception":false,"start_time":"2026-07-20T11:41:47.950453+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3e70ee95","cell_type":"markdown","source":"### Model Comparison Charts and Data-Driven Analysis\nVisualize all three OOF results and print a neutral interpretation based only on the computed metrics. Finally package the complete Phase 2 artifact directory.\n","metadata":{"papermill":{"duration":0.006363,"end_time":"2026-07-20T11:41:49.496724+00:00","exception":false,"start_time":"2026-07-20T11:41:49.490361+00:00","status":"completed"},"tags":[]}},{"id":"59700cae","cell_type":"code","source":"ordered_models = list(model_results.keys())\ncompare_df = summary_df.set_index(\"model\").loc[ordered_models]\n\nfig, axes = plt.subplots(1, 2, figsize=(13, 5))\naxes[0].bar(compare_df.index, compare_df[\"accuracy\"])\naxes[0].set_title(\"OOF Accuracy by Model\")\naxes[0].set_ylabel(\"Accuracy\")\naxes[0].set_ylim(0, 1.0)\naxes[0].tick_params(axis=\"x\", rotation=15)\nfor index, value in enumerate(compare_df[\"accuracy\"]):\n    axes[0].text(index, value + 0.008, f\"{value:.4f}\", ha=\"center\", fontsize=9)\n\naxes[1].bar(compare_df.index, compare_df[\"log_loss\"])\naxes[1].set_title(\"OOF Log-Loss by Model (lower is better)\")\naxes[1].set_ylabel(\"Log-loss\")\naxes[1].tick_params(axis=\"x\", rotation=15)\nfor index, value in enumerate(compare_df[\"log_loss\"]):\n    axes[1].text(index, value, f\"{value:.4f}\", ha=\"center\", va=\"bottom\", fontsize=9)\n\nfig.suptitle(\"Model Comparison — CNN vs Tabular vs Hybrid\", fontweight=\"bold\")\nfig.tight_layout()\nfig.savefig(OUTPUT_PATH / \"model_comparison_bars.png\", dpi=160, bbox_inches=\"tight\")\nplt.show()\nplt.close(fig)\n\n# --- Fold-Wise Accuracy ---\nfig, ax = plt.subplots(figsize=(10, 6))\nax.plot(fold_df[\"fold\"], fold_df[\"cnn_accuracy\"], marker=\"o\", label=\"CNN\")\nif RUN_TABULAR_BASELINE:\n    ax.plot(\n        fold_df[\"fold\"],\n        fold_df[\"tabular_accuracy\"],\n        marker=\"s\",\n        label=\"Tabular\",\n    )\nax.plot(\n    fold_df[\"fold\"],\n    fold_df[\"hybrid_accuracy\"],\n    marker=\"^\",\n    label=\"Hybrid\",\n)\nax.set_xlabel(\"Fold\")\nax.set_ylabel(\"Accuracy\")\nax.set_xticks(fold_df[\"fold\"])\nax.set_title(\"Per-Fold Accuracy\")\nax.grid(alpha=0.25)\nax.legend()\nfig.tight_layout()\nfig.savefig(OUTPUT_PATH / \"fold_accuracy_comparison.png\", dpi=160, bbox_inches=\"tight\")\nplt.show()\nplt.close(fig)\n\n# --- Data-Driven Written Analysis ---\nbest_name = primary_model_name\nbest_metrics = model_results[best_name][\"summary\"]\nprint(\"\\n\" + \"=\" * 72)\nprint(\"DATA-DRIVEN ANALYSIS\")\nprint(\"=\" * 72)\nprint(\n    f\"The primary model selected by {PRIMARY_METRIC} is '{best_name}'. \"\n    f\"Its OOF accuracy is {best_metrics['accuracy']:.4f}, log-loss is \"\n    f\"{best_metrics['log_loss']:.4f}, and macro-F1 is {best_metrics['macro_f1']:.4f}.\"\n)\n\ncnn_metrics = model_results[\"CNN image-only\"][\"summary\"]\nhybrid_metrics = model_results[\"Hybrid CNN + tabular\"][\"summary\"]\nprint(\n    \"Hybrid minus CNN: \"\n    f\"accuracy {(hybrid_metrics['accuracy'] - cnn_metrics['accuracy']) * 100:+.2f} pp, \"\n    f\"log-loss {hybrid_metrics['log_loss'] - cnn_metrics['log_loss']:+.4f}, \"\n    f\"macro-F1 {hybrid_metrics['macro_f1'] - cnn_metrics['macro_f1']:+.4f}.\"\n)\n\nif RUN_TABULAR_BASELINE:\n    tabular_metrics = model_results[\"XGBoost tabular-only\"][\"summary\"]\n    print(\n        \"Hybrid minus tabular: \"\n        f\"accuracy {(hybrid_metrics['accuracy'] - tabular_metrics['accuracy']) * 100:+.2f} pp, \"\n        f\"log-loss {hybrid_metrics['log_loss'] - tabular_metrics['log_loss']:+.4f}, \"\n        f\"macro-F1 {hybrid_metrics['macro_f1'] - tabular_metrics['macro_f1']:+.4f}.\"\n    )\n\nfor model_name, prefix in model_prefixes.items():\n    if model_name not in model_results:\n        continue\n    print(\n        f\"{model_name} fold accuracy std: \"\n        f\"{fold_df[f'{prefix}_accuracy'].std(ddof=1):.4f}.\"\n    )\nprint(\"=\" * 72)\n\n# Package every generated Phase 2 artifact for convenient download.\narchive_base = Path(\"/kaggle/working\") / \"phase2_results\"\narchive_path = Path(shutil.make_archive(str(archive_base), \"zip\", root_dir=OUTPUT_PATH))\nprint(f\"Packaged artifacts: {archive_path}\")\n","metadata":{"execution":{"iopub.execute_input":"2026-07-20T11:41:49.511068Z","iopub.status.busy":"2026-07-20T11:41:49.510599Z","iopub.status.idle":"2026-07-20T11:41:50.417299Z","shell.execute_reply":"2026-07-20T11:41:50.416526Z"},"papermill":{"duration":0.91633,"end_time":"2026-07-20T11:41:50.419393+00:00","exception":false,"start_time":"2026-07-20T11:41:49.503063+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}