{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BIG 2015 Malware Classification - Stage 1\n\nFirst Kaggle run: validate the dataset, selectively extract only .bytes files, run EDA, test the parser, and benchmark NPY versus PNG cache formats. Full preprocessing and training are intentionally deferred until the cache decision is recorded.\n","metadata":{}},{"cell_type":"code","source":"import json, math, os, platform, random, shutil, signal, subprocess, tempfile, time\nfrom pathlib import Path\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision.models import resnet18\nfrom tqdm import tqdm\n\nSEED = 42\nIMAGE_SIZE = 128\nINPUT_DIR = Path(\"/kaggle/input/competitions/malware-classification\")\nWORK_DIR = Path(\"/kaggle/working/malware_big2015\")\nTRAIN_ARCHIVE = INPUT_DIR / \"train.7z\"\nLABELS_FILE = INPUT_DIR / \"trainLabels.csv\"\nTRAIN_BYTES_DIR = WORK_DIR / \"train_bytes\"\nARTIFACT_DIR = WORK_DIR / \"artifacts\"\nRUN_STAGE1 = True\nRUN_EXTRACT_TRAIN = True\nRUN_CACHE_BENCHMARK = True\nBENCHMARK_SAMPLES = 24\nRUN_BUILD_TRAIN_CACHE = True\nSTAGING_LIMIT_GIB = 4.0\nMIN_FREE_GIB = 5.0\nPROCESS_BATCH_FILES = 256\nRUN_PUBLISH_CACHE = False\nKAGGLE_DATASET_HANDLE = \"t4lf4n/big2015-grayscale-128-cache\"\nRUN_TRAIN_BASELINE = True\nRUN_CV_BASELINE = True\nCV_FOLDS = 5\nRUN_WEIGHTED_CV = True\nCLASS_WEIGHT_SCHEME = \"inverse_sqrt_frequency\"\nRUN_RESNET18_CV = True\nRESNET18_BATCH_SIZE = 64\nRESNET18_EPOCHS = 10\nRESNET18_LR = 3e-4\nRESNET18_WEIGHT_DECAY = 1e-4\nRESNET18_PATIENCE = 3\nTRAIN_BATCH_SIZE = 64\nTRAIN_EPOCHS = 15\nTRAIN_LR = 1e-3\nTRAIN_WEIGHT_DECAY = 1e-4\nVAL_SIZE = 0.20\nEARLY_STOPPING_PATIENCE = 4\nNUM_WORKERS = 0\nRUN_STAGE7 = True\nTEST_STAGING_LIMIT_GIB = 4.0\nTEST_MIN_FREE_GIB = 5.0\nTEST_PROCESS_BATCH_FILES = 256\nSTAGE7_DIR = ARTIFACT_DIR / \"stage7_inference\"\nTEST_BYTES_DIR = WORK_DIR / \"test_bytes\"\nTEST_CACHE_PATH = WORK_DIR / f\"test_images_{IMAGE_SIZE}.npy\"\nTEST_PROGRESS_PATH = WORK_DIR / f\"test_images_{IMAGE_SIZE}_progress.npy\"\nTEST_MANIFEST_PATH = STAGE7_DIR / f\"test_manifest_{IMAGE_SIZE}.csv\"\nTEST_CACHE_META_PATH = STAGE7_DIR / f\"test_cache_meta_{IMAGE_SIZE}.json\"\nTEST_EXTRACTION_LOG_PATH = STAGE7_DIR / \"test_extraction.log\"\nSUBMISSION_PATH = Path(\"/kaggle/working/submission.csv\")\n\nSTAGE7_DIR.mkdir(parents=True, exist_ok=True)\nTEST_BYTES_DIR.mkdir(parents=True, exist_ok=True)\n\nCLASS_NAMES = [\n    \"Ramnit\", \"Lollipop\", \"Kelihos_ver3\", \"Vundo\", \"Simda\",\n    \"Tracur\", \"Kelihos_ver1\", \"Obfuscator.ACY\", \"Gatak\",\n]\nfor directory in (WORK_DIR, TRAIN_BYTES_DIR, ARTIFACT_DIR):\n    directory.mkdir(parents=True, exist_ok=True)\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\ntorch.backends.cudnn.deterministic = True\ntorch.backends.cudnn.benchmark = False\n\nassert nn is torch.nn\n\nenvironment = {\n    \"python\": platform.python_version(),\n    \"pytorch\": torch.__version__,\n    \"cuda_available\": torch.cuda.is_available(),\n    \"cuda_version\": torch.version.cuda,\n    \"gpus\": [torch.cuda.get_device_name(i) for i in range(torch.cuda.device_count())],\n}\n(ARTIFACT_DIR / \"environment.json\").write_text(json.dumps(environment, indent=2), encoding=\"utf-8\")\nprint(json.dumps(environment, indent=2))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T16:13:06.982042Z","iopub.execute_input":"2026-07-21T16:13:06.982613Z","iopub.status.idle":"2026-07-21T16:13:17.735121Z","shell.execute_reply.started":"2026-07-21T16:13:06.982580Z","shell.execute_reply":"2026-07-21T16:13:17.734329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Selective batch extraction\n\nThe full train .bytes payload is larger than Kaggle working storage. Stage 1 lists the archive but extracts only the benchmark sample IDs. Full preprocessing will later use extract -> convert -> cache -> delete batches.","metadata":{}},{"cell_type":"code","source":"if RUN_STAGE1:\r\n    def require_file(path, label):\r\n        if not path.exists():\r\n            raise FileNotFoundError(f\"Missing {label}: {path}. Add the BIG 2015 competition data.\")\r\n    \r\n    def find_byte_files(root):\r\n        return [\r\n            path for path in root.rglob(\"*\")\r\n            if path.is_file() and path.suffix.lower() == \".bytes\"\r\n        ]\r\n    \r\n    def archive_inventory(archive):\r\n        require_file(archive, \"train.7z\")\r\n        result = subprocess.run(\r\n            [\"7z\", \"l\", \"-slt\", \"-r\", str(archive), \"*.bytes\"],\r\n            capture_output=True, text=True, check=True,\r\n        )\r\n        total = 0\r\n        entry_count = 0\r\n        current_path = \"\"\r\n        for raw in result.stdout.splitlines():\r\n            line = raw.strip()\r\n            if line.startswith(\"Path = \"):\r\n                current_path = line.split(\"=\", 1)[1].strip()\r\n            elif line.startswith(\"Size = \") and current_path.lower().endswith(\".bytes\"):\r\n                total += int(line.split(\"=\", 1)[1].strip())\r\n                entry_count += 1\r\n        return total, entry_count\r\n    \r\n    def extract_selected_bytes(archive, output_dir, sample_ids):\r\n        output_dir.mkdir(parents=True, exist_ok=True)\r\n        existing = {path.stem: path for path in find_byte_files(output_dir)}\r\n        missing_ids = [sample_id for sample_id in sample_ids if sample_id not in existing]\r\n        if not missing_ids:\r\n            print(f\"Reuse {len(sample_ids):,} selected .bytes files.\")\r\n            return [existing[sample_id] for sample_id in sample_ids]\r\n    \r\n        patterns = [f\"{sample_id}.bytes\" for sample_id in missing_ids]\r\n        result = subprocess.run(\r\n            [\"7z\", \"x\", \"-r\", str(archive), *patterns, f\"-o{output_dir}\", \"-y\"],\r\n            capture_output=True, text=True,\r\n        )\r\n        print(result.stdout[-4000:])\r\n        if result.returncode != 0:\r\n            print(result.stderr[-4000:])\r\n            raise subprocess.CalledProcessError(result.returncode, result.args)\r\n    \r\n        extracted = {path.stem: path for path in find_byte_files(output_dir)}\r\n        still_missing = [sample_id for sample_id in sample_ids if sample_id not in extracted]\r\n        if still_missing:\r\n            raise RuntimeError(\r\n                f\"7z did not extract {len(still_missing)} selected IDs. \"\r\n                f\"First missing IDs: {still_missing[:5]}\"\r\n            )\r\n        selected_paths = [extracted[sample_id] for sample_id in sample_ids]\r\n        selected_bytes = sum(path.stat().st_size for path in selected_paths)\r\n        print(f\"Extracted {len(selected_paths):,} benchmark files ({selected_bytes / 1024**2:.2f} MiB).\")\r\n        return selected_paths\r\n    \r\n    require_file(LABELS_FILE, \"trainLabels.csv\")\r\n    stage_labels = pd.read_csv(LABELS_FILE, dtype={\"Id\": str})\r\n    selected_ids = stage_labels[\"Id\"].sample(\r\n        min(BENCHMARK_SAMPLES, len(stage_labels)),\r\n        random_state=SEED,\r\n    ).tolist()\r\n    \r\n    full_bytes, archive_entries = archive_inventory(TRAIN_ARCHIVE)\r\n    free = shutil.disk_usage(TRAIN_BYTES_DIR).free\r\n    print(f\"Archive .bytes entries: {archive_entries:,}\")\r\n    print(f\"Full .bytes size:       {full_bytes / 1024**3:.2f} GiB\")\r\n    print(f\"Free working space:     {free / 1024**3:.2f} GiB\")\r\n    print(\"Full extraction skipped; extracting benchmark IDs only.\")\r\n    \r\n    if RUN_EXTRACT_TRAIN:\r\n        train_files = extract_selected_bytes(\r\n            TRAIN_ARCHIVE, TRAIN_BYTES_DIR, selected_ids\r\n        )\r\n    else:\r\n        existing = {path.stem: path for path in find_byte_files(TRAIN_BYTES_DIR)}\r\n        missing_ids = [sample_id for sample_id in selected_ids if sample_id not in existing]\r\n        if missing_ids:\r\n            raise FileNotFoundError(\r\n                \"Benchmark extraction is disabled but selected files are missing.\"\r\n            )\r\n        train_files = [existing[sample_id] for sample_id in selected_ids]\r\n    \r\n    train_index = {path.stem: path for path in train_files}\r\n    print(f\"Indexed benchmark samples: {len(train_index):,}\")\r\nelse:\r\n    print(\"Skip Stage 1 extraction (RUN_STAGE1=False).\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA and byte-to-image parser\n","metadata":{}},{"cell_type":"code","source":"if RUN_STAGE1:\r\n    CLASS_NAMES = [\r\n        \"Ramnit\", \"Lollipop\", \"Kelihos_ver3\", \"Vundo\", \"Simda\",\r\n        \"Tracur\", \"Kelihos_ver1\", \"Obfuscator.ACY\", \"Gatak\",\r\n    ]\r\n    labels = pd.read_csv(LABELS_FILE, dtype={\"Id\": str})\r\n    labels[\"Class\"] = labels[\"Class\"].astype(int)\r\n    counts = labels[\"Class\"].value_counts().sort_index().reindex(range(1, 10))\r\n    \r\n    class_table = pd.DataFrame({\r\n        \"Class\": range(1, 10),\r\n        \"Family\": CLASS_NAMES,\r\n        \"Count\": counts.values,\r\n    })\r\n    display(class_table)\r\n    class_table.to_csv(ARTIFACT_DIR / \"class_distribution.csv\", index=False)\r\n    \r\n    plt.figure(figsize=(11, 4))\r\n    sns.barplot(data=class_table, x=\"Family\", y=\"Count\", hue=\"Family\", legend=False)\r\n    plt.xticks(rotation=35, ha=\"right\")\r\n    plt.tight_layout()\r\n    plt.savefig(ARTIFACT_DIR / \"class_distribution.png\", dpi=160)\r\n    plt.show()\r\n    \r\n    missing = sorted(set(selected_ids) - set(train_index))\r\n    print(f\"Selected benchmark IDs without extracted .bytes: {len(missing):,}\")\r\n    if missing:\r\n        print(\"First missing IDs:\", missing[:5])\r\n    \r\n    def image_width(num_bytes):\r\n        size_kb = num_bytes / 1024\r\n        for limit, width in [\r\n            (10, 32), (30, 64), (60, 128), (100, 256),\r\n            (200, 384), (500, 512), (1000, 768),\r\n        ]:\r\n            if size_kb < limit:\r\n                return width\r\n        return 1024\r\n    \r\n    def bytes_to_vector(path):\r\n        values = bytearray()\r\n        with open(path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as handle:\r\n            for line_number, line in enumerate(handle, start=1):\r\n                for token in line.split()[1:]:\r\n                    try:\r\n                        values.append(0 if token == \"??\" else int(token, 16))\r\n                    except ValueError as error:\r\n                        raise ValueError(f\"Invalid token at {path}:{line_number}: {token}\") from error\r\n        if not values:\r\n            raise ValueError(f\"No byte values found in {path}\")\r\n        return np.frombuffer(values, dtype=np.uint8)\r\n    \r\n    def byte_to_image(path, output_size):\r\n        vector = bytes_to_vector(path)\r\n        width = image_width(vector.size)\r\n        height = math.ceil(vector.size / width)\r\n        padded = np.pad(vector, (0, height * width - vector.size))\r\n        raw_image = padded.reshape(height, width)\r\n        interpolation = cv2.INTER_AREA if max(raw_image.shape) > output_size else cv2.INTER_LINEAR\r\n        return cv2.resize(raw_image, (output_size, output_size), interpolation=interpolation)\r\n    \r\n    def parser_smoke_test():\r\n        with tempfile.TemporaryDirectory() as temp_dir:\r\n            sample = Path(temp_dir) / \"sample.bytes\"\r\n            sample.write_text(\"00401000 4D 5A ?? FF\\n00401004 00 10 20 30\\n\", encoding=\"utf-8\")\r\n            assert bytes_to_vector(sample).tolist() == [77, 90, 0, 255, 0, 16, 32, 48]\r\n            image = byte_to_image(sample, 32)\r\n            assert image.shape == (32, 32) and image.dtype == np.uint8\r\n        print(\"Parser smoke test passed.\")\r\n    \r\n    parser_smoke_test()\r\n    \r\n    available_ids = labels.loc[labels[\"Id\"].isin(train_index), \"Id\"]\r\n    if available_ids.empty:\r\n        raise RuntimeError(\"No labeled .bytes files available.\")\r\n    example_id = available_ids.iloc[0]\r\n    example = byte_to_image(train_index[example_id], IMAGE_SIZE)\r\n    plt.figure(figsize=(5, 5))\r\n    plt.imshow(example, cmap=\"gray\", vmin=0, vmax=255)\r\n    plt.title(f\"Sample {example_id}\")\r\n    plt.axis(\"off\")\r\n    plt.tight_layout()\r\n    plt.savefig(ARTIFACT_DIR / \"sample_image.png\", dpi=160)\r\n    plt.show()\r\n    \r\nelse:\r\n    print(\"Skip Stage 1 EDA/parser (RUN_STAGE1=False).\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cache benchmark\n\nCompare storage size and read/write time using the same converted images. Send the resulting table back before full preprocessing is implemented.\n","metadata":{}},{"cell_type":"code","source":"if RUN_STAGE1:\r\n    def benchmark_cache_formats(sample_ids):\r\n        started = time.perf_counter()\r\n        images = np.stack([\r\n            byte_to_image(train_index[sample_id], IMAGE_SIZE)\r\n            for sample_id in tqdm(sample_ids)\r\n        ])\r\n        conversion_seconds = time.perf_counter() - started\r\n        rows = []\r\n    \r\n        with tempfile.TemporaryDirectory(dir=WORK_DIR) as temp_dir:\r\n            temp_path = Path(temp_dir)\r\n    \r\n            npy_path = temp_path / \"sample.npy\"\r\n            started = time.perf_counter()\r\n            np.save(npy_path, images)\r\n            npy_write = time.perf_counter() - started\r\n            started = time.perf_counter()\r\n            loaded = np.load(npy_path, mmap_mode=\"r\")\r\n            _ = loaded[:, 0, 0].sum()\r\n            npy_read = time.perf_counter() - started\r\n            rows.append([\"npy\", npy_path.stat().st_size, npy_write, npy_read])\r\n    \r\n            png_dir = temp_path / \"png\"\r\n            png_dir.mkdir()\r\n            started = time.perf_counter()\r\n            for sample_id, image in zip(sample_ids, images):\r\n                if not cv2.imwrite(str(png_dir / f\"{sample_id}.png\"), image):\r\n                    raise RuntimeError(f\"Failed to write PNG for {sample_id}\")\r\n            png_write = time.perf_counter() - started\r\n            png_paths = list(png_dir.glob(\"*.png\"))\r\n            started = time.perf_counter()\r\n            for path in png_paths:\r\n                _ = cv2.imread(str(path), cv2.IMREAD_GRAYSCALE)\r\n            png_read = time.perf_counter() - started\r\n            rows.append([\"png\", sum(path.stat().st_size for path in png_paths), png_write, png_read])\r\n    \r\n        result = pd.DataFrame(rows, columns=[\"format\", \"bytes\", \"write_seconds\", \"read_seconds\"])\r\n        result[\"sample_count\"] = len(sample_ids)\r\n        result[\"conversion_seconds\"] = conversion_seconds\r\n        result[\"estimated_full_GiB\"] = result[\"bytes\"] / len(sample_ids) * len(labels) / 1024**3\r\n        return result\r\n    \r\n    if RUN_CACHE_BENCHMARK:\r\n        benchmark_ids = available_ids.sample(\r\n            min(BENCHMARK_SAMPLES, len(available_ids)),\r\n            random_state=SEED,\r\n        ).tolist()\r\n        benchmark = benchmark_cache_formats(benchmark_ids)\r\n        display(benchmark)\r\n        benchmark.to_csv(ARTIFACT_DIR / \"cache_benchmark.csv\", index=False)\r\n        print(\"Stage 1 complete. Save cache_benchmark.csv and record the cache decision.\")\r\n    else:\r\n        print(\"Cache benchmark disabled.\")\r\n    \r\nelse:\r\n    print(\"Skip Stage 1 cache benchmark (RUN_STAGE1=False).\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 2 - one-pass extraction to NPY memory-map\n\nThe full archive is never left on disk. A single 7-Zip process extracts .bytes files while the notebook periodically pauses it, converts completed files into the fixed NPY cache, deletes the staged files, and resumes extraction. Progress is checkpointed so an interrupted Kaggle session can resume.","metadata":{}},{"cell_type":"code","source":"CACHE_PATH = WORK_DIR / f\"train_images_{IMAGE_SIZE}.npy\"\nCACHE_PROGRESS_PATH = WORK_DIR / f\"train_images_{IMAGE_SIZE}_progress.npy\"\nCACHE_MANIFEST_PATH = ARTIFACT_DIR / f\"train_manifest_{IMAGE_SIZE}.csv\"\nCACHE_META_PATH = ARTIFACT_DIR / f\"train_cache_meta_{IMAGE_SIZE}.json\"\nEXTRACTION_LOG_PATH = ARTIFACT_DIR / \"full_extraction.log\"\n\ndef open_or_create_cache(frame):\n    shape = (len(frame), IMAGE_SIZE, IMAGE_SIZE)\n    if CACHE_PATH.exists():\n        cache = np.load(CACHE_PATH, mmap_mode=\"r+\")\n        if cache.shape != shape or cache.dtype != np.uint8:\n            raise ValueError(f\"Existing cache has incompatible shape/dtype: {cache.shape}, {cache.dtype}\")\n    else:\n        cache = np.lib.format.open_memmap(\n            CACHE_PATH, mode=\"w+\", dtype=np.uint8, shape=shape\n        )\n\n    if CACHE_PROGRESS_PATH.exists():\n        progress = np.load(CACHE_PROGRESS_PATH)\n        if progress.shape != (len(frame),):\n            raise ValueError(\"Existing progress file has incompatible shape.\")\n    else:\n        progress = np.zeros(len(frame), dtype=bool)\n        np.save(CACHE_PROGRESS_PATH, progress)\n\n    manifest = pd.DataFrame({\n        \"cache_index\": np.arange(len(frame)),\n        \"Id\": frame[\"Id\"].astype(str),\n        \"label\": frame[\"Class\"].astype(int) - 1,\n    })\n    manifest.to_csv(CACHE_MANIFEST_PATH, index=False)\n    return cache, progress, {sample_id: i for i, sample_id in enumerate(manifest[\"Id\"])}\n\ndef staged_bytes_size(files):\n    return sum(path.stat().st_size for path in files if path.exists())\n\ndef convert_and_remove(paths, cache, progress, row_by_id):\n    converted = 0\n    errors = []\n    for path in tqdm(paths, desc=\"Convert staged bytes\", leave=False):\n        sample_id = path.stem\n        row = row_by_id.get(sample_id)\n        if row is None:\n            errors.append((sample_id, \"ID is absent from trainLabels.csv\"))\n            continue\n        if progress[row]:\n            path.unlink(missing_ok=True)\n            continue\n        try:\n            cache[row] = byte_to_image(path, IMAGE_SIZE)\n            progress[row] = True\n            path.unlink()\n            converted += 1\n        except Exception as error:\n            errors.append((sample_id, repr(error)))\n    cache.flush()\n    np.save(CACHE_PROGRESS_PATH, progress)\n    if errors:\n        pd.DataFrame(errors, columns=[\"Id\", \"error\"]).to_csv(\n            ARTIFACT_DIR / \"preprocessing_errors.csv\", index=False\n        )\n        raise RuntimeError(f\"Failed to convert {len(errors)} staged files.\")\n    return converted\n\ndef build_train_cache_one_pass(frame):\n    if platform.system() != \"Linux\":\n        raise RuntimeError(\"Controlled one-pass extraction requires Linux/Kaggle process signals.\")\n\n    cache, progress, row_by_id = open_or_create_cache(frame)\n    if progress.all() and CACHE_META_PATH.exists():\n        meta = json.loads(CACHE_META_PATH.read_text(encoding=\"utf-8\"))\n        print(\"Reuse complete Stage 2 cache.\")\n        print(json.dumps(meta, indent=2))\n        return meta\n    started = time.perf_counter()\n    peak_staging_bytes = 0\n    total_converted = 0\n\n    existing = find_byte_files(TRAIN_BYTES_DIR)\n    if existing:\n        converted = convert_and_remove(existing, cache, progress, row_by_id)\n        total_converted += converted\n        print(f\"Consumed {converted:,} existing benchmark files.\")\n\n    command = [\n        \"7z\", \"x\", \"-r\", str(TRAIN_ARCHIVE), \"*.bytes\",\n        f\"-o{TRAIN_BYTES_DIR}\", \"-y\",\n    ]\n    print(\"Starting one-pass extraction. Progress is saved after every conversion batch.\")\n    log_handle = open(EXTRACTION_LOG_PATH, \"w\", encoding=\"utf-8\")\n    process = subprocess.Popen(\n        command, stdout=log_handle, stderr=subprocess.STDOUT, text=True\n    )\n    paused = False\n\n    try:\n        while process.poll() is None:\n            time.sleep(1.0)\n            files = find_byte_files(TRAIN_BYTES_DIR)\n            staging_bytes = staged_bytes_size(files)\n            peak_staging_bytes = max(peak_staging_bytes, staging_bytes)\n            free_bytes = shutil.disk_usage(TRAIN_BYTES_DIR).free\n            should_drain = (\n                len(files) >= PROCESS_BATCH_FILES\n                or staging_bytes >= STAGING_LIMIT_GIB * 1024**3\n                or free_bytes <= MIN_FREE_GIB * 1024**3\n            )\n            if not should_drain:\n                continue\n\n            os.kill(process.pid, signal.SIGSTOP)\n            paused = True\n            time.sleep(0.25)\n            files = sorted(find_byte_files(TRAIN_BYTES_DIR), key=lambda p: p.stat().st_mtime)\n            ready = files[:-1] if len(files) > 1 else []\n            if ready:\n                converted = convert_and_remove(ready, cache, progress, row_by_id)\n                total_converted += converted\n                current_staging = staged_bytes_size(find_byte_files(TRAIN_BYTES_DIR))\n                print(\n                    f\"Cached {int(progress.sum()):,}/{len(progress):,}; \"\n                    f\"staging {current_staging/1024**3:.2f} GiB\"\n                )\n            os.kill(process.pid, signal.SIGCONT)\n            paused = False\n\n        log_handle.flush()\n        if process.returncode != 0:\n            tail = EXTRACTION_LOG_PATH.read_text(encoding=\"utf-8\", errors=\"ignore\")[-4000:]\n            print(tail)\n            raise subprocess.CalledProcessError(process.returncode, command)\n\n        remaining = find_byte_files(TRAIN_BYTES_DIR)\n        if remaining:\n            converted = convert_and_remove(remaining, cache, progress, row_by_id)\n            total_converted += converted\n    except Exception:\n        if process.poll() is None:\n            if paused:\n                os.kill(process.pid, signal.SIGCONT)\n            process.terminate()\n            try:\n                process.wait(timeout=10)\n            except subprocess.TimeoutExpired:\n                process.kill()\n        raise\n    finally:\n        log_handle.close()\n        cache.flush()\n        np.save(CACHE_PROGRESS_PATH, progress)\n\n    missing_rows = np.flatnonzero(~progress)\n    if len(missing_rows):\n        missing_ids = frame.iloc[missing_rows][\"Id\"].head().tolist()\n        raise RuntimeError(\n            f\"Cache incomplete: {len(missing_rows)} samples missing. First IDs: {missing_ids}\"\n        )\n\n    meta = {\n        \"format\": \"npy_memmap\",\n        \"image_size\": IMAGE_SIZE,\n        \"sample_count\": len(frame),\n        \"cache_path\": str(CACHE_PATH),\n        \"cache_bytes\": CACHE_PATH.stat().st_size,\n        \"elapsed_seconds\": time.perf_counter() - started,\n        \"peak_staging_bytes\": peak_staging_bytes,\n        \"converted_this_run\": total_converted,\n        \"completed\": True,\n    }\n    CACHE_META_PATH.write_text(json.dumps(meta, indent=2), encoding=\"utf-8\")\n    print(json.dumps(meta, indent=2))\n    return meta\n\ntrain_cache_meta = None\nif RUN_BUILD_TRAIN_CACHE:\n    train_cache_meta = build_train_cache_one_pass(labels)\nelse:\n    print(\"Stage 2 cache build disabled.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Publish the completed Stage 2 cache as a private Kaggle Dataset\n\nRun this cell in the same live session that contains the completed NPY cache. Replace YOUR_USERNAME in KAGGLE_DATASET_HANDLE first. The cell copies only reusable files, creates SHA-256 checksums, and uploads the directory with kagglehub.","metadata":{}},{"cell_type":"code","source":"import hashlib\n\nPUBLISH_DIR = WORK_DIR / \"publish_stage2_cache\"\n\ndef sha256_file(path, chunk_size=1024 * 1024):\n    digest = hashlib.sha256()\n    with open(path, \"rb\") as handle:\n        while chunk := handle.read(chunk_size):\n            digest.update(chunk)\n    return digest.hexdigest()\n\ndef prepare_stage2_dataset():\n    required = {\n        \"train_images_128.npy\": CACHE_PATH,\n        \"train_manifest_128.csv\": CACHE_MANIFEST_PATH,\n        \"train_cache_meta_128.json\": CACHE_META_PATH,\n        \"environment.json\": ARTIFACT_DIR / \"environment.json\",\n        \"class_distribution.csv\": ARTIFACT_DIR / \"class_distribution.csv\",\n        \"cache_benchmark.csv\": ARTIFACT_DIR / \"cache_benchmark.csv\",\n    }\n    missing = [str(path) for path in required.values() if not path.exists()]\n    if missing:\n        raise FileNotFoundError(f\"Missing Stage 2 outputs: {missing}\")\n\n    PUBLISH_DIR.mkdir(parents=True, exist_ok=True)\n    checksums = {}\n    for output_name, source_path in required.items():\n        destination = PUBLISH_DIR / output_name\n        shutil.copy2(source_path, destination)\n        checksums[output_name] = {\n            \"bytes\": destination.stat().st_size,\n            \"sha256\": sha256_file(destination),\n        }\n\n    readme = \"\"\"# BIG 2015 grayscale cache\n\nPrivate derived cache for the Malware Classification from Binary Images project.\n\n- Source: Microsoft Malware Classification Challenge (BIG 2015)\n- Samples: 10,868 training files\n- Representation: uint8 grayscale images\n- Shape: (10868, 128, 128)\n- Labels: zero-based values in train_manifest_128.csv\n- Unknown byte token ??: replaced by 0\n- Cache format: NumPy NPY memory-map compatible\n\nKeep this dataset private unless the competition data terms explicitly permit redistribution.\n\"\"\"\n    (PUBLISH_DIR / \"README.md\").write_text(readme, encoding=\"utf-8\")\n    (PUBLISH_DIR / \"checksums.json\").write_text(\n        json.dumps(checksums, indent=2), encoding=\"utf-8\"\n    )\n    print(json.dumps(checksums, indent=2))\n    return PUBLISH_DIR\n\nif RUN_PUBLISH_CACHE:\n    if KAGGLE_DATASET_HANDLE.startswith(\"YOUR_USERNAME/\"):\n        raise ValueError(\"Replace YOUR_USERNAME in KAGGLE_DATASET_HANDLE before upload.\")\n    publish_dir = prepare_stage2_dataset()\n    import kagglehub\n    kagglehub.dataset_upload(\n        KAGGLE_DATASET_HANDLE,\n        str(publish_dir),\n        version_notes=\"Stage 2 complete: 10,868 BIG 2015 grayscale images at 128x128.\",\n    )\n    print(f\"Uploaded: https://www.kaggle.com/datasets/{KAGGLE_DATASET_HANDLE}\")\n    print(\"Verify that the dataset visibility is PRIVATE.\")\nelse:\n    print(\"Set RUN_PUBLISH_CACHE=True and replace YOUR_USERNAME, then run this cell.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 3 - stratified holdout and Custom CNN baseline\n\nThis baseline intentionally uses unweighted Cross-Entropy and no sampler. Its per-class metrics establish how class imbalance affects minority families before mitigation experiments.","metadata":{}},{"cell_type":"code","source":"import kagglehub\nfrom sklearn.metrics import (\n    accuracy_score,\n    classification_report,\n    confusion_matrix,\n    log_loss,\n    precision_recall_fscore_support,\n)\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import DataLoader, Dataset\n\nPERSISTED_CACHE_PATH = CACHE_PATH\nPERSISTED_MANIFEST_PATH = CACHE_MANIFEST_PATH\n\nif not PERSISTED_CACHE_PATH.exists() or not PERSISTED_MANIFEST_PATH.exists():\n    raise FileNotFoundError(\n        f\"Expected cache files in {CACHE_INPUT_DIR}; found {list(CACHE_INPUT_DIR.iterdir())}\"\n    )\n\ntrain_manifest = pd.read_csv(PERSISTED_MANIFEST_PATH, dtype={\"Id\": str})\ncached_images = np.load(PERSISTED_CACHE_PATH, mmap_mode=\"r\")\nif cached_images.shape != (len(train_manifest), IMAGE_SIZE, IMAGE_SIZE):\n    raise ValueError(\n        f\"Cache/manifest mismatch: cache={cached_images.shape}, rows={len(train_manifest)}\"\n    )\n\nall_indices = np.arange(len(train_manifest))\ntrain_indices, val_indices = train_test_split(\n    all_indices,\n    test_size=VAL_SIZE,\n    random_state=SEED,\n    stratify=train_manifest[\"label\"],\n)\nsplit_manifest = train_manifest.copy()\nsplit_manifest[\"split\"] = \"\"\nsplit_manifest.loc[train_indices, \"split\"] = \"train\"\nsplit_manifest.loc[val_indices, \"split\"] = \"validation\"\nsplit_manifest.to_csv(ARTIFACT_DIR / \"stage3_data_split.csv\", index=False)\n\nsplit_counts = (\n    split_manifest.groupby([\"split\", \"label\"]).size().unstack(fill_value=0)\n)\ndisplay(split_counts)\nprint(f\"Train: {len(train_indices):,}; validation: {len(val_indices):,}\")\n\nclass MalwareImageDataset(Dataset):\n    def __init__(self, cache_path, manifest, indices):\n        self.cache_path = str(cache_path)\n        self.manifest = manifest.reset_index(drop=True)\n        self.indices = np.asarray(indices)\n        self.images = None\n\n    def _cache(self):\n        if self.images is None:\n            self.images = np.load(self.cache_path, mmap_mode=\"r\")\n        return self.images\n\n    def __len__(self):\n        return len(self.indices)\n\n    def __getitem__(self, position):\n        row_index = int(self.indices[position])\n        row = self.manifest.iloc[row_index]\n        image = np.array(self._cache()[int(row[\"cache_index\"])], copy=True)\n        tensor = torch.from_numpy(image).float().unsqueeze(0) / 255.0\n        tensor = (tensor - 0.5) / 0.5\n        return tensor, int(row[\"label\"]), str(row[\"Id\"])\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nif RUN_TRAIN_BASELINE and device.type != \"cuda\":\n    raise RuntimeError(\"Enable a Kaggle GPU accelerator before running Stage 3 training.\")\n\nloader_options = {\n    \"batch_size\": TRAIN_BATCH_SIZE,\n    \"num_workers\": NUM_WORKERS,\n    \"pin_memory\": device.type == \"cuda\",\n    \"persistent_workers\": NUM_WORKERS > 0,\n}\ntrain_loader = DataLoader(\n    MalwareImageDataset(PERSISTED_CACHE_PATH, train_manifest, train_indices),\n    shuffle=True,\n    **loader_options,\n)\nval_loader = DataLoader(\n    MalwareImageDataset(PERSISTED_CACHE_PATH, train_manifest, val_indices),\n    shuffle=False,\n    **loader_options,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T13:23:08.280406Z","iopub.execute_input":"2026-07-21T13:23:08.280894Z","iopub.status.idle":"2026-07-21T13:23:09.091160Z","shell.execute_reply.started":"2026-07-21T13:23:08.280863Z","shell.execute_reply":"2026-07-21T13:23:09.090145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ConvBlock(nn.Sequential):\n    def __init__(self, in_channels, out_channels):\n        super().__init__(\n            nn.Conv2d(in_channels, out_channels, 3, padding=1, bias=False),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_channels, out_channels, 3, padding=1, bias=False),\n            nn.BatchNorm2d(out_channels),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(2),\n        )\n\nclass MalwareCNN(nn.Module):\n    def __init__(self, num_classes=9):\n        super().__init__()\n        self.features = nn.Sequential(\n            ConvBlock(1, 32),\n            ConvBlock(32, 64),\n            ConvBlock(64, 128),\n            ConvBlock(128, 256),\n        )\n        self.classifier = nn.Sequential(\n            nn.AdaptiveAvgPool2d(1),\n            nn.Flatten(),\n            nn.Dropout(0.30),\n            nn.Linear(256, num_classes),\n        )\n\n    def forward(self, inputs):\n        return self.classifier(self.features(inputs))\n\ndef calculate_metrics(targets, probabilities):\n    probabilities = np.asarray(probabilities, dtype=np.float64)\n    row_sums = probabilities.sum(axis=1, keepdims=True)\n    if not np.isfinite(probabilities).all() or (row_sums <= 0).any():\n        raise ValueError(\"Probabilities contain non-finite values or invalid row sums.\")\n    probabilities = probabilities / row_sums\n    predictions = probabilities.argmax(axis=1)\n    macro = precision_recall_fscore_support(\n        targets, predictions, average=\"macro\", zero_division=0\n    )\n    weighted = precision_recall_fscore_support(\n        targets, predictions, average=\"weighted\", zero_division=0\n    )\n    return {\n        \"accuracy\": accuracy_score(targets, predictions),\n        \"macro_precision\": macro[0],\n        \"macro_recall\": macro[1],\n        \"macro_f1\": macro[2],\n        \"weighted_f1\": weighted[2],\n        \"log_loss\": log_loss(targets, probabilities, labels=np.arange(9)),\n    }\n\ndef run_training_epoch(model, loader, criterion, optimizer=None, scaler=None):\n    training = optimizer is not None\n    model.train(training)\n    total_loss = 0.0\n    targets_all = []\n    probabilities_all = []\n    amp_enabled = device.type == \"cuda\"\n\n    for images, targets, _ in tqdm(loader, leave=False):\n        images = images.to(device, non_blocking=True)\n        targets = targets.to(device, non_blocking=True)\n        if training:\n            optimizer.zero_grad(set_to_none=True)\n\n        with torch.set_grad_enabled(training):\n            with torch.autocast(\n                device_type=device.type, dtype=torch.float16, enabled=amp_enabled\n            ):\n                logits = model(images)\n                loss = criterion(logits, targets)\n            if training:\n                scaler.scale(loss).backward()\n                scaler.step(optimizer)\n                scaler.update()\n\n        total_loss += loss.item() * targets.size(0)\n        targets_all.append(targets.detach().cpu().numpy())\n        probabilities_all.append(torch.softmax(logits.detach().float(), dim=1).cpu().numpy())\n\n    targets_all = np.concatenate(targets_all)\n    probabilities_all = np.concatenate(probabilities_all)\n    output = calculate_metrics(targets_all, probabilities_all)\n    output[\"loss\"] = total_loss / len(loader.dataset)\n    return output, targets_all, probabilities_all\n\ndef train_custom_cnn():\n    model = MalwareCNN().to(device)\n    criterion = nn.CrossEntropyLoss()\n    optimizer = torch.optim.AdamW(\n        model.parameters(), lr=TRAIN_LR, weight_decay=TRAIN_WEIGHT_DECAY\n    )\n    try:\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=True)\n    except TypeError:\n        scaler = torch.cuda.amp.GradScaler(enabled=True)\n\n    checkpoint_path = ARTIFACT_DIR / \"stage3_best_custom_cnn.pt\"\n    best_macro_f1 = -1.0\n    stale_epochs = 0\n    history = []\n    run_started = time.perf_counter()\n    torch.cuda.reset_peak_memory_stats()\n\n    for epoch_number in range(1, TRAIN_EPOCHS + 1):\n        epoch_started = time.perf_counter()\n        train_metrics, _, _ = run_training_epoch(\n            model, train_loader, criterion, optimizer, scaler\n        )\n        with torch.no_grad():\n            val_metrics, _, _ = run_training_epoch(model, val_loader, criterion)\n\n        row = {\n            \"epoch\": epoch_number,\n            \"epoch_seconds\": time.perf_counter() - epoch_started,\n            **{f\"train_{key}\": value for key, value in train_metrics.items()},\n            **{f\"val_{key}\": value for key, value in val_metrics.items()},\n        }\n        history.append(row)\n        pd.DataFrame(history).to_csv(\n            ARTIFACT_DIR / \"stage3_training_history.csv\", index=False\n        )\n        print(\n            f\"Epoch {epoch_number:02d} | \"\n            f\"train loss={train_metrics['loss']:.4f} | \"\n            f\"val loss={val_metrics['loss']:.4f} | \"\n            f\"val macro-F1={val_metrics['macro_f1']:.4f}\"\n        )\n\n        if val_metrics[\"macro_f1\"] > best_macro_f1:\n            best_macro_f1 = val_metrics[\"macro_f1\"]\n            stale_epochs = 0\n            torch.save(\n                {\n                    \"model_state\": model.state_dict(),\n                    \"epoch\": epoch_number,\n                    \"val_macro_f1\": best_macro_f1,\n                    \"class_names\": CLASS_NAMES,\n                    \"image_size\": IMAGE_SIZE,\n                },\n                checkpoint_path,\n            )\n        else:\n            stale_epochs += 1\n            if stale_epochs >= EARLY_STOPPING_PATIENCE:\n                print(\"Early stopping.\")\n                break\n\n    summary = {\n        \"experiment\": \"EXP-005\",\n        \"model\": \"Custom CNN baseline\",\n        \"best_val_macro_f1\": best_macro_f1,\n        \"total_seconds\": time.perf_counter() - run_started,\n        \"peak_vram_bytes\": torch.cuda.max_memory_allocated(),\n        \"checkpoint\": str(checkpoint_path),\n        \"gpu\": torch.cuda.get_device_name(0),\n        \"seed\": SEED,\n        \"image_size\": IMAGE_SIZE,\n        \"train_samples\": len(train_indices),\n        \"validation_samples\": len(val_indices),\n        \"batch_size\": TRAIN_BATCH_SIZE,\n        \"max_epochs\": TRAIN_EPOCHS,\n        \"learning_rate\": TRAIN_LR,\n        \"weight_decay\": TRAIN_WEIGHT_DECAY,\n        \"loss\": \"unweighted CrossEntropyLoss\",\n        \"sampler\": \"none\",\n        \"amp\": True,\n    }\n    (ARTIFACT_DIR / \"stage3_run_summary.json\").write_text(\n        json.dumps(summary, indent=2), encoding=\"utf-8\"\n    )\n    return pd.DataFrame(history), checkpoint_path, summary\n\nif RUN_TRAIN_BASELINE:\n    stage3_history, stage3_checkpoint, stage3_summary = train_custom_cnn()\n    display(pd.DataFrame([stage3_summary]))\nelse:\n    print(\"Stage 3 training disabled.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_stage3(checkpoint_path):\n    checkpoint = torch.load(checkpoint_path, map_location=device)\n    model = MalwareCNN().to(device)\n    model.load_state_dict(checkpoint[\"model_state\"])\n    criterion = nn.CrossEntropyLoss()\n    with torch.no_grad():\n        final_metrics, targets, probabilities = run_training_epoch(\n            model, val_loader, criterion\n        )\n    predictions = probabilities.argmax(axis=1)\n\n    report = pd.DataFrame(\n        classification_report(\n            targets,\n            predictions,\n            labels=np.arange(9),\n            target_names=CLASS_NAMES,\n            output_dict=True,\n            zero_division=0,\n        )\n    ).T\n    matrix = pd.DataFrame(\n        confusion_matrix(targets, predictions, labels=np.arange(9)),\n        index=CLASS_NAMES,\n        columns=CLASS_NAMES,\n    )\n    report.to_csv(ARTIFACT_DIR / \"stage3_classification_report.csv\")\n    matrix.to_csv(ARTIFACT_DIR / \"stage3_confusion_matrix.csv\")\n    (ARTIFACT_DIR / \"stage3_validation_metrics.json\").write_text(\n        json.dumps(final_metrics, indent=2), encoding=\"utf-8\"\n    )\n\n    figure, axes = plt.subplots(1, 2, figsize=(13, 4))\n    axes[0].plot(stage3_history[\"epoch\"], stage3_history[\"train_loss\"], label=\"train\")\n    axes[0].plot(stage3_history[\"epoch\"], stage3_history[\"val_loss\"], label=\"validation\")\n    axes[0].set_title(\"Loss\")\n    axes[0].legend()\n    axes[1].plot(stage3_history[\"epoch\"], stage3_history[\"train_macro_f1\"], label=\"train\")\n    axes[1].plot(stage3_history[\"epoch\"], stage3_history[\"val_macro_f1\"], label=\"validation\")\n    axes[1].set_title(\"Macro F1\")\n    axes[1].legend()\n    figure.tight_layout()\n    figure.savefig(ARTIFACT_DIR / \"stage3_training_curves.png\", dpi=160)\n    plt.show()\n\n    plt.figure(figsize=(11, 9))\n    sns.heatmap(matrix, annot=True, fmt=\"d\", cmap=\"Blues\")\n    plt.title(\"Stage 3 validation confusion matrix\")\n    plt.ylabel(\"True\")\n    plt.xlabel(\"Predicted\")\n    plt.tight_layout()\n    plt.savefig(ARTIFACT_DIR / \"stage3_confusion_matrix.png\", dpi=160)\n    plt.show()\n\n    display(report)\n    print(json.dumps(final_metrics, indent=2))\n    return model, final_metrics, report, matrix\n\nif RUN_TRAIN_BASELINE:\n    stage3_model, stage3_metrics, stage3_report, stage3_matrix = evaluate_stage3(\n        stage3_checkpoint\n    )","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 4 - Stratified 5-Fold Cross-Validation\n\nThis stage repeats the same unweighted Custom CNN baseline across five stratified folds. It saves fold metrics, mean and sample standard deviation, per-class stability, out-of-fold predictions, an aggregate confusion matrix, histories, and one best checkpoint per fold.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nif RUN_CV_BASELINE and device.type != \"cuda\":\n    raise RuntimeError(\"Enable a Kaggle GPU accelerator before running Stage 4.\")\n\nCV_DIR = ARTIFACT_DIR / \"stage4_cv\"\nCV_DIR.mkdir(parents=True, exist_ok=True)\n\ndef make_fold_loaders(fold_train_indices, fold_val_indices, fold_seed):\n    generator = torch.Generator()\n    generator.manual_seed(fold_seed)\n    common = {\n        \"batch_size\": TRAIN_BATCH_SIZE,\n        \"num_workers\": NUM_WORKERS,\n        \"pin_memory\": True,\n        \"persistent_workers\": False,\n    }\n    fold_train_loader = DataLoader(\n        MalwareImageDataset(PERSISTED_CACHE_PATH, train_manifest, fold_train_indices),\n        shuffle=True,\n        generator=generator,\n        **common,\n    )\n    fold_val_loader = DataLoader(\n        MalwareImageDataset(PERSISTED_CACHE_PATH, train_manifest, fold_val_indices),\n        shuffle=False,\n        **common,\n    )\n    return fold_train_loader, fold_val_loader\n\ndef train_one_fold(fold_number, fold_train_indices, fold_val_indices):\n    fold_seed = SEED + fold_number\n    random.seed(fold_seed)\n    np.random.seed(fold_seed)\n    torch.manual_seed(fold_seed)\n    torch.cuda.manual_seed_all(fold_seed)\n\n    fold_train_loader, fold_val_loader = make_fold_loaders(\n        fold_train_indices, fold_val_indices, fold_seed\n    )\n    model = MalwareCNN().to(device)\n    criterion = nn.CrossEntropyLoss()\n    optimizer = torch.optim.AdamW(\n        model.parameters(), lr=TRAIN_LR, weight_decay=TRAIN_WEIGHT_DECAY\n    )\n    try:\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=True)\n    except TypeError:\n        scaler = torch.cuda.amp.GradScaler(enabled=True)\n\n    checkpoint_path = CV_DIR / f\"fold_{fold_number}_best.pt\"\n    best_macro_f1 = -1.0\n    best_epoch = 0\n    stale_epochs = 0\n    fold_history = []\n    fold_started = time.perf_counter()\n\n    for epoch_number in range(1, TRAIN_EPOCHS + 1):\n        epoch_started = time.perf_counter()\n        train_metrics, _, _ = run_training_epoch(\n            model, fold_train_loader, criterion, optimizer, scaler\n        )\n        with torch.no_grad():\n            val_metrics, _, _ = run_training_epoch(\n                model, fold_val_loader, criterion\n            )\n\n        row = {\n            \"fold\": fold_number,\n            \"epoch\": epoch_number,\n            \"epoch_seconds\": time.perf_counter() - epoch_started,\n            **{f\"train_{key}\": value for key, value in train_metrics.items()},\n            **{f\"val_{key}\": value for key, value in val_metrics.items()},\n        }\n        fold_history.append(row)\n        print(\n            f\"Fold {fold_number}/{CV_FOLDS} epoch {epoch_number:02d} | \"\n            f\"val loss={val_metrics['loss']:.4f} | \"\n            f\"val macro-F1={val_metrics['macro_f1']:.4f}\"\n        )\n\n        if val_metrics[\"macro_f1\"] > best_macro_f1:\n            best_macro_f1 = val_metrics[\"macro_f1\"]\n            best_epoch = epoch_number\n            stale_epochs = 0\n            torch.save(\n                {\n                    \"model_state\": model.state_dict(),\n                    \"fold\": fold_number,\n                    \"epoch\": epoch_number,\n                    \"val_macro_f1\": best_macro_f1,\n                    \"class_names\": CLASS_NAMES,\n                    \"image_size\": IMAGE_SIZE,\n                    \"seed\": fold_seed,\n                },\n                checkpoint_path,\n            )\n        else:\n            stale_epochs += 1\n            if stale_epochs >= EARLY_STOPPING_PATIENCE:\n                print(f\"Fold {fold_number}: early stopping.\")\n                break\n\n    checkpoint = torch.load(checkpoint_path, map_location=device)\n    model.load_state_dict(checkpoint[\"model_state\"])\n    with torch.no_grad():\n        final_metrics, targets, probabilities = run_training_epoch(\n            model, fold_val_loader, criterion\n        )\n\n    fold_summary = {\n        \"fold\": fold_number,\n        \"seed\": fold_seed,\n        \"train_samples\": len(fold_train_indices),\n        \"validation_samples\": len(fold_val_indices),\n        \"best_epoch\": best_epoch,\n        \"elapsed_seconds\": time.perf_counter() - fold_started,\n        **final_metrics,\n    }\n\n    predictions = probabilities.argmax(axis=1)\n    report = classification_report(\n        targets,\n        predictions,\n        labels=np.arange(9),\n        target_names=CLASS_NAMES,\n        output_dict=True,\n        zero_division=0,\n    )\n    per_class_rows = []\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        values = report[class_name]\n        per_class_rows.append({\n            \"fold\": fold_number,\n            \"class_id\": class_index,\n            \"class_name\": class_name,\n            \"precision\": values[\"precision\"],\n            \"recall\": values[\"recall\"],\n            \"f1\": values[\"f1-score\"],\n            \"support\": values[\"support\"],\n        })\n\n    del model, optimizer, scaler, fold_train_loader, fold_val_loader\n    torch.cuda.empty_cache()\n    return (\n        fold_summary,\n        fold_history,\n        per_class_rows,\n        targets,\n        probabilities,\n    )\n\ndef run_stratified_cv():\n    labels_array = train_manifest[\"label\"].to_numpy(dtype=np.int64)\n    splitter = StratifiedKFold(\n        n_splits=CV_FOLDS, shuffle=True, random_state=SEED\n    )\n    oof_probabilities = np.zeros(\n        (len(train_manifest), len(CLASS_NAMES)), dtype=np.float32\n    )\n    oof_fold = np.full(len(train_manifest), -1, dtype=np.int8)\n    fold_summaries = []\n    all_history = []\n    all_per_class = []\n\n    torch.cuda.reset_peak_memory_stats()\n    cv_started = time.perf_counter()\n    for fold_number, (fold_train_indices, fold_val_indices) in enumerate(\n        splitter.split(np.zeros(len(labels_array)), labels_array), start=1\n    ):\n        (\n            fold_summary,\n            fold_history,\n            per_class_rows,\n            targets,\n            probabilities,\n        ) = train_one_fold(fold_number, fold_train_indices, fold_val_indices)\n\n        expected_targets = labels_array[fold_val_indices]\n        if not np.array_equal(targets, expected_targets):\n            raise RuntimeError(f\"Fold {fold_number}: validation order mismatch.\")\n        oof_probabilities[fold_val_indices] = probabilities\n        oof_fold[fold_val_indices] = fold_number\n        fold_summaries.append(fold_summary)\n        all_history.extend(fold_history)\n        all_per_class.extend(per_class_rows)\n\n        pd.DataFrame(fold_summaries).to_csv(\n            CV_DIR / \"fold_metrics.csv\", index=False\n        )\n        pd.DataFrame(all_history).to_csv(\n            CV_DIR / \"training_history.csv\", index=False\n        )\n        pd.DataFrame(all_per_class).to_csv(\n            CV_DIR / \"per_class_fold_metrics.csv\", index=False\n        )\n\n    if (oof_fold < 0).any():\n        raise RuntimeError(\"Some samples never received an out-of-fold prediction.\")\n\n    fold_frame = pd.DataFrame(fold_summaries)\n    metric_columns = [\n        \"accuracy\", \"macro_precision\", \"macro_recall\",\n        \"macro_f1\", \"weighted_f1\", \"log_loss\", \"loss\",\n    ]\n    aggregate_rows = []\n    for metric_name in metric_columns:\n        aggregate_rows.append({\n            \"metric\": metric_name,\n            \"mean\": fold_frame[metric_name].mean(),\n            \"std\": fold_frame[metric_name].std(ddof=1),\n            \"min\": fold_frame[metric_name].min(),\n            \"max\": fold_frame[metric_name].max(),\n        })\n    aggregate_frame = pd.DataFrame(aggregate_rows)\n    aggregate_frame.to_csv(CV_DIR / \"aggregate_metrics.csv\", index=False)\n\n    per_class_frame = pd.DataFrame(all_per_class)\n    per_class_summary = (\n        per_class_frame.groupby([\"class_id\", \"class_name\"])\n        .agg(\n            precision_mean=(\"precision\", \"mean\"),\n            precision_std=(\"precision\", \"std\"),\n            recall_mean=(\"recall\", \"mean\"),\n            recall_std=(\"recall\", \"std\"),\n            f1_mean=(\"f1\", \"mean\"),\n            f1_std=(\"f1\", \"std\"),\n            support_total=(\"support\", \"sum\"),\n        )\n        .reset_index()\n    )\n    per_class_summary.to_csv(\n        CV_DIR / \"per_class_summary.csv\", index=False\n    )\n\n    oof_targets = labels_array\n    oof_predictions = oof_probabilities.argmax(axis=1)\n    oof_metrics = calculate_metrics(oof_targets, oof_probabilities)\n    oof_frame = pd.DataFrame({\n        \"Id\": train_manifest[\"Id\"],\n        \"true_label\": oof_targets,\n        \"predicted_label\": oof_predictions,\n        \"fold\": oof_fold,\n    })\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        oof_frame[f\"prob_{class_index + 1}_{class_name}\"] = oof_probabilities[:, class_index]\n    oof_frame.to_csv(CV_DIR / \"oof_predictions.csv\", index=False)\n\n    aggregate_matrix = pd.DataFrame(\n        confusion_matrix(\n            oof_targets, oof_predictions, labels=np.arange(9)\n        ),\n        index=CLASS_NAMES,\n        columns=CLASS_NAMES,\n    )\n    aggregate_matrix.to_csv(CV_DIR / \"oof_confusion_matrix.csv\")\n    plt.figure(figsize=(11, 9))\n    sns.heatmap(aggregate_matrix, annot=True, fmt=\"d\", cmap=\"Blues\")\n    plt.title(\"Stage 4 out-of-fold confusion matrix\")\n    plt.ylabel(\"True\")\n    plt.xlabel(\"Predicted\")\n    plt.tight_layout()\n    plt.savefig(CV_DIR / \"oof_confusion_matrix.png\", dpi=160)\n    plt.show()\n\n    history_frame = pd.DataFrame(all_history)\n    plt.figure(figsize=(10, 5))\n    for fold_number in range(1, CV_FOLDS + 1):\n        fold_history = history_frame[history_frame[\"fold\"] == fold_number]\n        plt.plot(\n            fold_history[\"epoch\"],\n            fold_history[\"val_macro_f1\"],\n            marker=\"o\",\n            label=f\"Fold {fold_number}\",\n        )\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Validation macro-F1\")\n    plt.title(\"Stage 4 validation macro-F1 by fold\")\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(CV_DIR / \"fold_macro_f1_curves.png\", dpi=160)\n    plt.show()\n\n    cv_summary = {\n        \"experiment\": \"EXP-008\",\n        \"model\": \"Custom CNN baseline\",\n        \"folds\": CV_FOLDS,\n        \"total_seconds\": time.perf_counter() - cv_started,\n        \"peak_vram_bytes\": torch.cuda.max_memory_allocated(),\n        \"gpu\": torch.cuda.get_device_name(0),\n        \"seed\": SEED,\n        \"image_size\": IMAGE_SIZE,\n        \"batch_size\": TRAIN_BATCH_SIZE,\n        \"max_epochs\": TRAIN_EPOCHS,\n        \"learning_rate\": TRAIN_LR,\n        \"weight_decay\": TRAIN_WEIGHT_DECAY,\n        \"loss\": \"unweighted CrossEntropyLoss\",\n        \"sampler\": \"none\",\n        \"oof_metrics\": {key: float(value) for key, value in oof_metrics.items()},\n        \"fold_macro_f1_mean\": float(fold_frame[\"macro_f1\"].mean()),\n        \"fold_macro_f1_std\": float(fold_frame[\"macro_f1\"].std(ddof=1)),\n    }\n    (CV_DIR / \"run_summary.json\").write_text(\n        json.dumps(cv_summary, indent=2), encoding=\"utf-8\"\n    )\n\n    display(fold_frame)\n    display(aggregate_frame)\n    display(per_class_summary)\n    print(json.dumps(cv_summary, indent=2))\n    return fold_frame, aggregate_frame, per_class_summary, oof_metrics\n\nif RUN_CV_BASELINE:\n    (\n        stage4_fold_metrics,\n        stage4_aggregate_metrics,\n        stage4_per_class_summary,\n        stage4_oof_metrics,\n    ) = run_stratified_cv()\nelse:\n    print(\"Stage 4 cross-validation disabled.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 5 - Inverse-sqrt class-weighted 5-Fold CV\n\nEXP-009 changes only the loss weighting relative to EXP-008. Class weights are computed from each training fold as inverse square-root frequency and normalized to mean 1. This avoids validation leakage and limits over-correction for the 42-sample Simda class.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nif RUN_WEIGHTED_CV and device.type != \"cuda\":\n    raise RuntimeError(\"Enable a Kaggle GPU accelerator before running Stage 5.\")\n\nWEIGHTED_CV_DIR = ARTIFACT_DIR / \"stage5_weighted_cv\"\nWEIGHTED_CV_DIR.mkdir(parents=True, exist_ok=True)\n\n\ndef get_inverse_sqrt_weights(fold_train_indices):\n    fold_labels = train_manifest.iloc[fold_train_indices][\"label\"].to_numpy(dtype=np.int64)\n    counts = np.bincount(fold_labels, minlength=len(CLASS_NAMES)).astype(np.int64)\n    if (counts == 0).any():\n        raise RuntimeError(f\"Training fold is missing a class: {counts.tolist()}\")\n    raw_weights = 1.0 / np.sqrt(counts.astype(np.float64))\n    normalized_weights = raw_weights / raw_weights.mean()\n    return counts, normalized_weights.astype(np.float32)\n\n\ndef train_one_weighted_fold(fold_number, fold_train_indices, fold_val_indices):\n    fold_seed = SEED + fold_number\n    random.seed(fold_seed)\n    np.random.seed(fold_seed)\n    torch.manual_seed(fold_seed)\n    torch.cuda.manual_seed_all(fold_seed)\n\n    fold_train_loader, fold_val_loader = make_fold_loaders(\n        fold_train_indices, fold_val_indices, fold_seed\n    )\n    class_counts, class_weights = get_inverse_sqrt_weights(fold_train_indices)\n    criterion = nn.CrossEntropyLoss(\n        weight=torch.tensor(class_weights, dtype=torch.float32, device=device)\n    )\n    model = MalwareCNN().to(device)\n    optimizer = torch.optim.AdamW(\n        model.parameters(), lr=TRAIN_LR, weight_decay=TRAIN_WEIGHT_DECAY\n    )\n    try:\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=True)\n    except TypeError:\n        scaler = torch.cuda.amp.GradScaler(enabled=True)\n\n    checkpoint_path = WEIGHTED_CV_DIR / f\"fold_{fold_number}_best.pt\"\n    best_macro_f1 = -1.0\n    best_epoch = 0\n    stale_epochs = 0\n    fold_history = []\n    fold_started = time.perf_counter()\n\n    for epoch_number in range(1, TRAIN_EPOCHS + 1):\n        epoch_started = time.perf_counter()\n        train_metrics, _, _ = run_training_epoch(\n            model, fold_train_loader, criterion, optimizer, scaler\n        )\n        with torch.no_grad():\n            val_metrics, _, _ = run_training_epoch(\n                model, fold_val_loader, criterion\n            )\n\n        row = {\n            \"fold\": fold_number,\n            \"epoch\": epoch_number,\n            \"epoch_seconds\": time.perf_counter() - epoch_started,\n            **{f\"train_{key}\": value for key, value in train_metrics.items()},\n            **{f\"val_{key}\": value for key, value in val_metrics.items()},\n        }\n        fold_history.append(row)\n        print(\n            f\"Weighted fold {fold_number}/{CV_FOLDS} epoch {epoch_number:02d} | \"\n            f\"val loss={val_metrics['loss']:.4f} | \"\n            f\"val macro-F1={val_metrics['macro_f1']:.4f}\"\n        )\n\n        if val_metrics[\"macro_f1\"] > best_macro_f1:\n            best_macro_f1 = val_metrics[\"macro_f1\"]\n            best_epoch = epoch_number\n            stale_epochs = 0\n            torch.save(\n                {\n                    \"model_state\": model.state_dict(),\n                    \"fold\": fold_number,\n                    \"epoch\": epoch_number,\n                    \"val_macro_f1\": best_macro_f1,\n                    \"class_names\": CLASS_NAMES,\n                    \"class_counts\": class_counts.tolist(),\n                    \"class_weights\": class_weights.tolist(),\n                    \"class_weight_scheme\": CLASS_WEIGHT_SCHEME,\n                    \"image_size\": IMAGE_SIZE,\n                    \"seed\": fold_seed,\n                },\n                checkpoint_path,\n            )\n        else:\n            stale_epochs += 1\n            if stale_epochs >= EARLY_STOPPING_PATIENCE:\n                print(f\"Weighted fold {fold_number}: early stopping.\")\n                break\n\n    checkpoint = torch.load(checkpoint_path, map_location=device)\n    model.load_state_dict(checkpoint[\"model_state\"])\n    with torch.no_grad():\n        final_metrics, targets, probabilities = run_training_epoch(\n            model, fold_val_loader, criterion\n        )\n\n    fold_summary = {\n        \"fold\": fold_number,\n        \"seed\": fold_seed,\n        \"train_samples\": len(fold_train_indices),\n        \"validation_samples\": len(fold_val_indices),\n        \"best_epoch\": best_epoch,\n        \"elapsed_seconds\": time.perf_counter() - fold_started,\n        **final_metrics,\n    }\n    predictions = probabilities.argmax(axis=1)\n    report = classification_report(\n        targets,\n        predictions,\n        labels=np.arange(len(CLASS_NAMES)),\n        target_names=CLASS_NAMES,\n        output_dict=True,\n        zero_division=0,\n    )\n    per_class_rows = []\n    weight_rows = []\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        values = report[class_name]\n        per_class_rows.append({\n            \"fold\": fold_number,\n            \"class_id\": class_index,\n            \"class_name\": class_name,\n            \"precision\": values[\"precision\"],\n            \"recall\": values[\"recall\"],\n            \"f1\": values[\"f1-score\"],\n            \"support\": values[\"support\"],\n        })\n        weight_rows.append({\n            \"fold\": fold_number,\n            \"class_id\": class_index,\n            \"class_name\": class_name,\n            \"train_count\": int(class_counts[class_index]),\n            \"weight\": float(class_weights[class_index]),\n        })\n\n    del model, optimizer, scaler, criterion, fold_train_loader, fold_val_loader\n    torch.cuda.empty_cache()\n    return (\n        fold_summary,\n        fold_history,\n        per_class_rows,\n        weight_rows,\n        targets,\n        probabilities,\n    )\n\n\ndef run_weighted_cv():\n    labels_array = train_manifest[\"label\"].to_numpy(dtype=np.int64)\n    splitter = StratifiedKFold(\n        n_splits=CV_FOLDS, shuffle=True, random_state=SEED\n    )\n    oof_probabilities = np.zeros(\n        (len(train_manifest), len(CLASS_NAMES)), dtype=np.float32\n    )\n    oof_fold = np.full(len(train_manifest), -1, dtype=np.int8)\n    fold_summaries = []\n    all_history = []\n    all_per_class = []\n    all_weights = []\n\n    torch.cuda.reset_peak_memory_stats()\n    cv_started = time.perf_counter()\n    for fold_number, (fold_train_indices, fold_val_indices) in enumerate(\n        splitter.split(np.zeros(len(labels_array)), labels_array), start=1\n    ):\n        (\n            fold_summary,\n            fold_history,\n            per_class_rows,\n            weight_rows,\n            targets,\n            probabilities,\n        ) = train_one_weighted_fold(\n            fold_number, fold_train_indices, fold_val_indices\n        )\n\n        expected_targets = labels_array[fold_val_indices]\n        if not np.array_equal(targets, expected_targets):\n            raise RuntimeError(f\"Weighted fold {fold_number}: validation order mismatch.\")\n        oof_probabilities[fold_val_indices] = probabilities\n        oof_fold[fold_val_indices] = fold_number\n        fold_summaries.append(fold_summary)\n        all_history.extend(fold_history)\n        all_per_class.extend(per_class_rows)\n        all_weights.extend(weight_rows)\n\n        pd.DataFrame(fold_summaries).to_csv(\n            WEIGHTED_CV_DIR / \"fold_metrics.csv\", index=False\n        )\n        pd.DataFrame(all_history).to_csv(\n            WEIGHTED_CV_DIR / \"training_history.csv\", index=False\n        )\n        pd.DataFrame(all_per_class).to_csv(\n            WEIGHTED_CV_DIR / \"per_class_fold_metrics.csv\", index=False\n        )\n        pd.DataFrame(all_weights).to_csv(\n            WEIGHTED_CV_DIR / \"class_weights.csv\", index=False\n        )\n\n    if (oof_fold < 0).any():\n        raise RuntimeError(\"Some samples never received a weighted OOF prediction.\")\n\n    fold_frame = pd.DataFrame(fold_summaries)\n    metric_columns = [\n        \"accuracy\", \"macro_precision\", \"macro_recall\",\n        \"macro_f1\", \"weighted_f1\", \"log_loss\", \"loss\",\n    ]\n    aggregate_frame = pd.DataFrame([\n        {\n            \"metric\": metric_name,\n            \"mean\": fold_frame[metric_name].mean(),\n            \"std\": fold_frame[metric_name].std(ddof=1),\n            \"min\": fold_frame[metric_name].min(),\n            \"max\": fold_frame[metric_name].max(),\n        }\n        for metric_name in metric_columns\n    ])\n    aggregate_frame.to_csv(\n        WEIGHTED_CV_DIR / \"aggregate_metrics.csv\", index=False\n    )\n\n    per_class_frame = pd.DataFrame(all_per_class)\n    per_class_summary = (\n        per_class_frame.groupby([\"class_id\", \"class_name\"])\n        .agg(\n            precision_mean=(\"precision\", \"mean\"),\n            precision_std=(\"precision\", \"std\"),\n            recall_mean=(\"recall\", \"mean\"),\n            recall_std=(\"recall\", \"std\"),\n            f1_mean=(\"f1\", \"mean\"),\n            f1_std=(\"f1\", \"std\"),\n            support_total=(\"support\", \"sum\"),\n        )\n        .reset_index()\n    )\n    per_class_summary.to_csv(\n        WEIGHTED_CV_DIR / \"per_class_summary.csv\", index=False\n    )\n\n    oof_targets = labels_array\n    oof_predictions = oof_probabilities.argmax(axis=1)\n    oof_metrics = calculate_metrics(oof_targets, oof_probabilities)\n    oof_frame = pd.DataFrame({\n        \"Id\": train_manifest[\"Id\"],\n        \"true_label\": oof_targets,\n        \"predicted_label\": oof_predictions,\n        \"fold\": oof_fold,\n    })\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        oof_frame[f\"prob_{class_index + 1}_{class_name}\"] = oof_probabilities[:, class_index]\n    oof_frame.to_csv(\n        WEIGHTED_CV_DIR / \"oof_predictions.csv\", index=False\n    )\n\n    aggregate_matrix = pd.DataFrame(\n        confusion_matrix(\n            oof_targets, oof_predictions,\n            labels=np.arange(len(CLASS_NAMES)),\n        ),\n        index=CLASS_NAMES,\n        columns=CLASS_NAMES,\n    )\n    aggregate_matrix.to_csv(\n        WEIGHTED_CV_DIR / \"oof_confusion_matrix.csv\"\n    )\n    plt.figure(figsize=(11, 9))\n    sns.heatmap(aggregate_matrix, annot=True, fmt=\"d\", cmap=\"Purples\")\n    plt.title(\"Stage 5 weighted out-of-fold confusion matrix\")\n    plt.ylabel(\"True\")\n    plt.xlabel(\"Predicted\")\n    plt.tight_layout()\n    plt.savefig(\n        WEIGHTED_CV_DIR / \"oof_confusion_matrix.png\", dpi=160\n    )\n    plt.show()\n\n    history_frame = pd.DataFrame(all_history)\n    plt.figure(figsize=(10, 5))\n    for fold_number in range(1, CV_FOLDS + 1):\n        fold_history = history_frame[history_frame[\"fold\"] == fold_number]\n        plt.plot(\n            fold_history[\"epoch\"],\n            fold_history[\"val_macro_f1\"],\n            marker=\"o\",\n            label=f\"Fold {fold_number}\",\n        )\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Validation macro-F1\")\n    plt.title(\"Stage 5 weighted validation macro-F1 by fold\")\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(\n        WEIGHTED_CV_DIR / \"fold_macro_f1_curves.png\", dpi=160\n    )\n    plt.show()\n\n    cv_summary = {\n        \"experiment\": \"EXP-009\",\n        \"model\": \"Custom CNN with class-weighted loss\",\n        \"folds\": CV_FOLDS,\n        \"total_seconds\": time.perf_counter() - cv_started,\n        \"peak_vram_bytes\": torch.cuda.max_memory_allocated(),\n        \"gpu\": torch.cuda.get_device_name(0),\n        \"seed\": SEED,\n        \"fold_seeds\": [SEED + fold for fold in range(1, CV_FOLDS + 1)],\n        \"image_size\": IMAGE_SIZE,\n        \"batch_size\": TRAIN_BATCH_SIZE,\n        \"max_epochs\": TRAIN_EPOCHS,\n        \"learning_rate\": TRAIN_LR,\n        \"weight_decay\": TRAIN_WEIGHT_DECAY,\n        \"loss\": \"class-weighted CrossEntropyLoss\",\n        \"class_weight_scheme\": CLASS_WEIGHT_SCHEME,\n        \"class_weight_scope\": \"computed from each training fold only\",\n        \"sampler\": \"none\",\n        \"num_workers\": NUM_WORKERS,\n        \"oof_metrics\": {key: float(value) for key, value in oof_metrics.items()},\n        \"fold_macro_f1_mean\": float(fold_frame[\"macro_f1\"].mean()),\n        \"fold_macro_f1_std\": float(fold_frame[\"macro_f1\"].std(ddof=1)),\n    }\n    (WEIGHTED_CV_DIR / \"run_summary.json\").write_text(\n        json.dumps(cv_summary, indent=2), encoding=\"utf-8\"\n    )\n\n    display(pd.DataFrame(all_weights))\n    display(fold_frame)\n    display(aggregate_frame)\n    display(per_class_summary)\n    print(json.dumps(cv_summary, indent=2))\n    return fold_frame, aggregate_frame, per_class_summary, oof_metrics\n\n\nif RUN_WEIGHTED_CV:\n    (\n        stage5_fold_metrics,\n        stage5_aggregate_metrics,\n        stage5_per_class_summary,\n        stage5_oof_metrics,\n    ) = run_weighted_cv()\nelse:\n    print(\"Stage 5 weighted cross-validation disabled.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 6 - Pretrained ResNet18 5-Fold CV\n\nEXP-010 compares architecture and transfer learning against the Custom CNN baseline. Grayscale images are repeated to three channels and normalized with ImageNet statistics. All ResNet18 layers are fine-tuned with unweighted Cross-Entropy on the same stratified folds and fold seeds as EXP-008. Kaggle Internet must be enabled for the official torchvision weights unless they are already cached.\n","metadata":{}},{"cell_type":"code","source":"from torchvision.models import ResNet18_Weights, resnet18\n\nif RUN_RESNET18_CV and device.type != \"cuda\":\n    raise RuntimeError(\"Enable a Kaggle GPU accelerator before running Stage 6.\")\n\nRESNET18_CV_DIR = ARTIFACT_DIR / \"stage6_resnet18_cv\"\nRESNET18_CV_DIR.mkdir(parents=True, exist_ok=True)\nRESNET18_WEIGHTS = ResNet18_Weights.DEFAULT\nIMAGENET_MEAN = torch.tensor([0.485, 0.456, 0.406]).view(3, 1, 1)\nIMAGENET_STD = torch.tensor([0.229, 0.224, 0.225]).view(3, 1, 1)\n\n\nclass ResNetMalwareImageDataset(Dataset):\n    def __init__(self, cache_path, manifest, indices):\n        self.cache_path = str(cache_path)\n        self.manifest = manifest.reset_index(drop=True)\n        self.indices = np.asarray(indices)\n        self.images = None\n\n    def _cache(self):\n        if self.images is None:\n            self.images = np.load(self.cache_path, mmap_mode=\"r\")\n        return self.images\n\n    def __len__(self):\n        return len(self.indices)\n\n    def __getitem__(self, position):\n        row_index = int(self.indices[position])\n        row = self.manifest.iloc[row_index]\n        image = np.array(self._cache()[int(row[\"cache_index\"])], copy=True)\n        tensor = torch.from_numpy(image).float().unsqueeze(0) / 255.0\n        tensor = tensor.repeat(3, 1, 1)\n        tensor = (tensor - IMAGENET_MEAN) / IMAGENET_STD\n        return tensor, int(row[\"label\"]), str(row[\"Id\"])\n\n\ndef make_resnet18_fold_loaders(fold_train_indices, fold_val_indices, fold_seed):\n    generator = torch.Generator()\n    generator.manual_seed(fold_seed)\n    common = {\n        \"batch_size\": RESNET18_BATCH_SIZE,\n        \"num_workers\": NUM_WORKERS,\n        \"pin_memory\": True,\n        \"persistent_workers\": False,\n    }\n    fold_train_loader = DataLoader(\n        ResNetMalwareImageDataset(\n            PERSISTED_CACHE_PATH, train_manifest, fold_train_indices\n        ),\n        shuffle=True,\n        generator=generator,\n        **common,\n    )\n    fold_val_loader = DataLoader(\n        ResNetMalwareImageDataset(\n            PERSISTED_CACHE_PATH, train_manifest, fold_val_indices\n        ),\n        shuffle=False,\n        **common,\n    )\n    return fold_train_loader, fold_val_loader\n\n\ndef build_pretrained_resnet18():\n    try:\n        model = resnet18(weights=RESNET18_WEIGHTS)\n    except Exception as error:\n        raise RuntimeError(\n            \"Unable to load pretrained ResNet18 weights. Enable Internet in \"\n            \"Kaggle or attach the official torchvision weights cache before \"\n            \"running EXP-010. The experiment must not silently fall back to \"\n            \"random initialization.\"\n        ) from error\n    model.fc = nn.Linear(model.fc.in_features, len(CLASS_NAMES))\n    return model\n\n\ndef train_one_resnet18_fold(fold_number, fold_train_indices, fold_val_indices):\n    fold_seed = SEED + fold_number\n    random.seed(fold_seed)\n    np.random.seed(fold_seed)\n    torch.manual_seed(fold_seed)\n    torch.cuda.manual_seed_all(fold_seed)\n\n    fold_train_loader, fold_val_loader = make_resnet18_fold_loaders(\n        fold_train_indices, fold_val_indices, fold_seed\n    )\n    model = build_pretrained_resnet18().to(device)\n    criterion = nn.CrossEntropyLoss()\n    optimizer = torch.optim.AdamW(\n        model.parameters(),\n        lr=RESNET18_LR,\n        weight_decay=RESNET18_WEIGHT_DECAY,\n    )\n    try:\n        scaler = torch.amp.GradScaler(\"cuda\", enabled=True)\n    except TypeError:\n        scaler = torch.cuda.amp.GradScaler(enabled=True)\n\n    checkpoint_path = RESNET18_CV_DIR / f\"fold_{fold_number}_best.pt\"\n    best_macro_f1 = -1.0\n    best_epoch = 0\n    stale_epochs = 0\n    fold_history = []\n    fold_started = time.perf_counter()\n\n    for epoch_number in range(1, RESNET18_EPOCHS + 1):\n        epoch_started = time.perf_counter()\n        train_metrics, _, _ = run_training_epoch(\n            model, fold_train_loader, criterion, optimizer, scaler\n        )\n        with torch.no_grad():\n            val_metrics, _, _ = run_training_epoch(\n                model, fold_val_loader, criterion\n            )\n\n        row = {\n            \"fold\": fold_number,\n            \"epoch\": epoch_number,\n            \"epoch_seconds\": time.perf_counter() - epoch_started,\n            **{f\"train_{key}\": value for key, value in train_metrics.items()},\n            **{f\"val_{key}\": value for key, value in val_metrics.items()},\n        }\n        fold_history.append(row)\n        print(\n            f\"ResNet18 fold {fold_number}/{CV_FOLDS} epoch {epoch_number:02d} | \"\n            f\"val loss={val_metrics['loss']:.4f} | \"\n            f\"val macro-F1={val_metrics['macro_f1']:.4f}\"\n        )\n\n        if val_metrics[\"macro_f1\"] > best_macro_f1:\n            best_macro_f1 = val_metrics[\"macro_f1\"]\n            best_epoch = epoch_number\n            stale_epochs = 0\n            torch.save(\n                {\n                    \"model_state\": model.state_dict(),\n                    \"fold\": fold_number,\n                    \"epoch\": epoch_number,\n                    \"val_macro_f1\": best_macro_f1,\n                    \"class_names\": CLASS_NAMES,\n                    \"architecture\": \"resnet18\",\n                    \"pretrained_weights\": str(RESNET18_WEIGHTS),\n                    \"input_channels\": 3,\n                    \"input_normalization\": \"ImageNet mean/std\",\n                    \"image_size\": IMAGE_SIZE,\n                    \"seed\": fold_seed,\n                },\n                checkpoint_path,\n            )\n        else:\n            stale_epochs += 1\n            if (\n                stale_epochs >= RESNET18_PATIENCE\n                and epoch_number < RESNET18_EPOCHS\n            ):\n                print(f\"ResNet18 fold {fold_number}: early stopping.\")\n                break\n\n    checkpoint = torch.load(checkpoint_path, map_location=device)\n    model.load_state_dict(checkpoint[\"model_state\"])\n    with torch.no_grad():\n        final_metrics, targets, probabilities = run_training_epoch(\n            model, fold_val_loader, criterion\n        )\n\n    parameter_count = sum(parameter.numel() for parameter in model.parameters())\n    trainable_parameter_count = sum(\n        parameter.numel() for parameter in model.parameters()\n        if parameter.requires_grad\n    )\n    fold_summary = {\n        \"fold\": fold_number,\n        \"parameter_count\": parameter_count,\n        \"trainable_parameter_count\": trainable_parameter_count,\n        \"seed\": fold_seed,\n        \"train_samples\": len(fold_train_indices),\n        \"validation_samples\": len(fold_val_indices),\n        \"best_epoch\": best_epoch,\n        \"elapsed_seconds\": time.perf_counter() - fold_started,\n        **final_metrics,\n    }\n    predictions = probabilities.argmax(axis=1)\n    report = classification_report(\n        targets,\n        predictions,\n        labels=np.arange(len(CLASS_NAMES)),\n        target_names=CLASS_NAMES,\n        output_dict=True,\n        zero_division=0,\n    )\n    per_class_rows = []\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        values = report[class_name]\n        per_class_rows.append({\n            \"fold\": fold_number,\n            \"class_id\": class_index,\n            \"class_name\": class_name,\n            \"precision\": values[\"precision\"],\n            \"recall\": values[\"recall\"],\n            \"f1\": values[\"f1-score\"],\n            \"support\": values[\"support\"],\n        })\n\n    del model, optimizer, scaler, criterion, fold_train_loader, fold_val_loader\n    torch.cuda.empty_cache()\n    return (\n        fold_summary,\n        fold_history,\n        per_class_rows,\n        targets,\n        probabilities,\n    )\n\n\ndef run_resnet18_cv():\n    labels_array = train_manifest[\"label\"].to_numpy(dtype=np.int64)\n    splitter = StratifiedKFold(\n        n_splits=CV_FOLDS, shuffle=True, random_state=SEED\n    )\n    oof_probabilities = np.zeros(\n        (len(train_manifest), len(CLASS_NAMES)), dtype=np.float32\n    )\n    oof_fold = np.full(len(train_manifest), -1, dtype=np.int8)\n    fold_summaries = []\n    all_history = []\n    all_per_class = []\n\n    torch.cuda.reset_peak_memory_stats()\n    cv_started = time.perf_counter()\n    for fold_number, (fold_train_indices, fold_val_indices) in enumerate(\n        splitter.split(np.zeros(len(labels_array)), labels_array), start=1\n    ):\n        (\n            fold_summary,\n            fold_history,\n            per_class_rows,\n            targets,\n            probabilities,\n        ) = train_one_resnet18_fold(\n            fold_number, fold_train_indices, fold_val_indices\n        )\n\n        expected_targets = labels_array[fold_val_indices]\n        if not np.array_equal(targets, expected_targets):\n            raise RuntimeError(f\"ResNet18 fold {fold_number}: validation order mismatch.\")\n        oof_probabilities[fold_val_indices] = probabilities\n        oof_fold[fold_val_indices] = fold_number\n        fold_summaries.append(fold_summary)\n        all_history.extend(fold_history)\n        all_per_class.extend(per_class_rows)\n\n        pd.DataFrame(fold_summaries).to_csv(\n            RESNET18_CV_DIR / \"fold_metrics.csv\", index=False\n        )\n        pd.DataFrame(all_history).to_csv(\n            RESNET18_CV_DIR / \"training_history.csv\", index=False\n        )\n        pd.DataFrame(all_per_class).to_csv(\n            RESNET18_CV_DIR / \"per_class_fold_metrics.csv\", index=False\n        )\n\n    if (oof_fold < 0).any():\n        raise RuntimeError(\"Some samples never received a ResNet18 OOF prediction.\")\n\n    fold_frame = pd.DataFrame(fold_summaries)\n    metric_columns = [\n        \"accuracy\", \"macro_precision\", \"macro_recall\",\n        \"macro_f1\", \"weighted_f1\", \"log_loss\", \"loss\",\n    ]\n    aggregate_frame = pd.DataFrame([\n        {\n            \"metric\": metric_name,\n            \"mean\": fold_frame[metric_name].mean(),\n            \"std\": fold_frame[metric_name].std(ddof=1),\n            \"min\": fold_frame[metric_name].min(),\n            \"max\": fold_frame[metric_name].max(),\n        }\n        for metric_name in metric_columns\n    ])\n    aggregate_frame.to_csv(\n        RESNET18_CV_DIR / \"aggregate_metrics.csv\", index=False\n    )\n\n    per_class_frame = pd.DataFrame(all_per_class)\n    per_class_summary = (\n        per_class_frame.groupby([\"class_id\", \"class_name\"])\n        .agg(\n            precision_mean=(\"precision\", \"mean\"),\n            precision_std=(\"precision\", \"std\"),\n            recall_mean=(\"recall\", \"mean\"),\n            recall_std=(\"recall\", \"std\"),\n            f1_mean=(\"f1\", \"mean\"),\n            f1_std=(\"f1\", \"std\"),\n            support_total=(\"support\", \"sum\"),\n        )\n        .reset_index()\n    )\n    per_class_summary.to_csv(\n        RESNET18_CV_DIR / \"per_class_summary.csv\", index=False\n    )\n\n    oof_targets = labels_array\n    oof_predictions = oof_probabilities.argmax(axis=1)\n    oof_metrics = calculate_metrics(oof_targets, oof_probabilities)\n    oof_frame = pd.DataFrame({\n        \"Id\": train_manifest[\"Id\"],\n        \"true_label\": oof_targets,\n        \"predicted_label\": oof_predictions,\n        \"fold\": oof_fold,\n    })\n    for class_index, class_name in enumerate(CLASS_NAMES):\n        oof_frame[f\"prob_{class_index + 1}_{class_name}\"] = oof_probabilities[:, class_index]\n    oof_frame.to_csv(\n        RESNET18_CV_DIR / \"oof_predictions.csv\", index=False\n    )\n\n    aggregate_matrix = pd.DataFrame(\n        confusion_matrix(\n            oof_targets, oof_predictions,\n            labels=np.arange(len(CLASS_NAMES)),\n        ),\n        index=CLASS_NAMES,\n        columns=CLASS_NAMES,\n    )\n    aggregate_matrix.to_csv(\n        RESNET18_CV_DIR / \"oof_confusion_matrix.csv\"\n    )\n    plt.figure(figsize=(11, 9))\n    sns.heatmap(aggregate_matrix, annot=True, fmt=\"d\", cmap=\"Greens\")\n    plt.title(\"Stage 6 ResNet18 out-of-fold confusion matrix\")\n    plt.ylabel(\"True\")\n    plt.xlabel(\"Predicted\")\n    plt.tight_layout()\n    plt.savefig(\n        RESNET18_CV_DIR / \"oof_confusion_matrix.png\", dpi=160\n    )\n    plt.show()\n\n    history_frame = pd.DataFrame(all_history)\n    plt.figure(figsize=(10, 5))\n    for fold_number in range(1, CV_FOLDS + 1):\n        fold_history = history_frame[history_frame[\"fold\"] == fold_number]\n        plt.plot(\n            fold_history[\"epoch\"],\n            fold_history[\"val_macro_f1\"],\n            marker=\"o\",\n            label=f\"Fold {fold_number}\",\n        )\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Validation macro-F1\")\n    plt.title(\"Stage 6 ResNet18 validation macro-F1 by fold\")\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(\n        RESNET18_CV_DIR / \"fold_macro_f1_curves.png\", dpi=160\n    )\n    plt.show()\n\n    cv_summary = {\n        \"experiment\": \"EXP-010\",\n        \"model\": \"pretrained ResNet18\",\n        \"parameter_count\": int(fold_frame[\"parameter_count\"].iloc[0]),\n        \"trainable_parameter_count\": int(\n            fold_frame[\"trainable_parameter_count\"].iloc[0]\n        ),\n        \"pretrained_weights\": str(RESNET18_WEIGHTS),\n        \"fine_tuning\": \"all layers\",\n        \"input\": \"grayscale repeated to 3 channels\",\n        \"normalization\": \"ImageNet mean/std\",\n        \"folds\": CV_FOLDS,\n        \"total_seconds\": time.perf_counter() - cv_started,\n        \"peak_vram_bytes\": torch.cuda.max_memory_allocated(),\n        \"gpu\": torch.cuda.get_device_name(0),\n        \"seed\": SEED,\n        \"fold_seeds\": [SEED + fold for fold in range(1, CV_FOLDS + 1)],\n        \"image_size\": IMAGE_SIZE,\n        \"batch_size\": RESNET18_BATCH_SIZE,\n        \"max_epochs\": RESNET18_EPOCHS,\n        \"learning_rate\": RESNET18_LR,\n        \"weight_decay\": RESNET18_WEIGHT_DECAY,\n        \"loss\": \"unweighted CrossEntropyLoss\",\n        \"sampler\": \"none\",\n        \"num_workers\": NUM_WORKERS,\n        \"oof_metrics\": {key: float(value) for key, value in oof_metrics.items()},\n        \"fold_macro_f1_mean\": float(fold_frame[\"macro_f1\"].mean()),\n        \"fold_macro_f1_std\": float(fold_frame[\"macro_f1\"].std(ddof=1)),\n    }\n    (RESNET18_CV_DIR / \"run_summary.json\").write_text(\n        json.dumps(cv_summary, indent=2), encoding=\"utf-8\"\n    )\n\n    display(fold_frame)\n    display(aggregate_frame)\n    display(per_class_summary)\n    print(json.dumps(cv_summary, indent=2))\n    return fold_frame, aggregate_frame, per_class_summary, oof_metrics\n\n\nif RUN_RESNET18_CV:\n    (\n        stage6_fold_metrics,\n        stage6_aggregate_metrics,\n        stage6_per_class_summary,\n        stage6_oof_metrics,\n    ) = run_resnet18_cv()\nelse:\n    print(\"Stage 6 ResNet18 cross-validation disabled.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Stage 7 - Test cache and 5-fold ResNet18 ensemble inference\n\nRun this cell only in the still-active EXP-010 Kaggle session so the five checkpoints remain available. Test .bytes files are extracted, converted and deleted in controlled batches; the resulting submission averages normalized probabilities from all five folds.\n","metadata":{}},{"cell_type":"code","source":"# Stage 7 - same-session test cache and ResNet18 5-fold ensemble inference\n\ndef resolve_kaggle_input_file(filename):\n    preferred = INPUT_DIR / filename\n    if preferred.exists():\n        return preferred\n    matches = sorted(Path(\"/kaggle/input\").rglob(filename))\n    if not matches:\n        raise FileNotFoundError(\n            f\"Could not find {filename} under /kaggle/input. \"\n            \"Attach the BIG 2015 competition dataset.\"\n        )\n    if len(matches) > 1:\n        print(f\"Multiple {filename} files found; using {matches[0]}\")\n    return matches[0]\n\n\ndef stage7_image_width(num_bytes):\n    size_kb = num_bytes / 1024\n    for limit, width in [\n        (10, 32), (30, 64), (60, 128), (100, 256),\n        (200, 384), (500, 512), (1000, 768),\n    ]:\n        if size_kb < limit:\n            return width\n    return 1024\n\n\ndef stage7_bytes_to_vector(path):\n    values = bytearray()\n    with open(path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as handle:\n        for line_number, line in enumerate(handle, start=1):\n            for token in line.split()[1:]:\n                try:\n                    values.append(0 if token == \"??\" else int(token, 16))\n                except ValueError as error:\n                    raise ValueError(\n                        f\"Invalid token at {path}:{line_number}: {token}\"\n                    ) from error\n    if not values:\n        raise ValueError(f\"No byte values found in {path}\")\n    return np.frombuffer(values, dtype=np.uint8)\n\n\ndef stage7_byte_to_image(path, output_size):\n    vector = stage7_bytes_to_vector(path)\n    width = stage7_image_width(vector.size)\n    height = math.ceil(vector.size / width)\n    padded = np.pad(vector, (0, height * width - vector.size))\n    raw_image = padded.reshape(height, width)\n    interpolation = (\n        cv2.INTER_AREA if max(raw_image.shape) > output_size\n        else cv2.INTER_LINEAR\n    )\n    return cv2.resize(\n        raw_image, (output_size, output_size), interpolation=interpolation\n    )\n\n\ndef stage7_parser_smoke_test():\n    fixture = STAGE7_DIR / \"parser_fixture.bytes\"\n    try:\n        fixture.write_text(\n            \"00401000 4D 5A ?? FF\\n00401004 00 10 20 30\\n\",\n            encoding=\"utf-8\",\n        )\n        assert stage7_bytes_to_vector(fixture).tolist() == [\n            77, 90, 0, 255, 0, 16, 32, 48,\n        ]\n        image = stage7_byte_to_image(fixture, 32)\n        assert image.shape == (32, 32) and image.dtype == np.uint8\n    finally:\n        fixture.unlink(missing_ok=True)\n    print(\"Stage 7 parser smoke test passed.\")\n\n\ndef find_test_byte_files():\n    return list(TEST_BYTES_DIR.rglob(\"*.bytes\"))\n\n\ndef open_or_create_test_cache(test_manifest):\n    shape = (len(test_manifest), IMAGE_SIZE, IMAGE_SIZE)\n    if TEST_CACHE_PATH.exists():\n        cache = np.load(TEST_CACHE_PATH, mmap_mode=\"r+\")\n        if cache.shape != shape or cache.dtype != np.uint8:\n            raise ValueError(\n                f\"Existing test cache is incompatible: {cache.shape}, {cache.dtype}\"\n            )\n    else:\n        cache = np.lib.format.open_memmap(\n            TEST_CACHE_PATH, mode=\"w+\", dtype=np.uint8, shape=shape\n        )\n\n    if TEST_PROGRESS_PATH.exists():\n        progress = np.load(TEST_PROGRESS_PATH)\n        if progress.shape != (len(test_manifest),):\n            raise ValueError(\"Existing test progress file has incompatible shape.\")\n    else:\n        progress = np.zeros(len(test_manifest), dtype=bool)\n        np.save(TEST_PROGRESS_PATH, progress)\n\n    row_by_id = {\n        sample_id: int(index)\n        for index, sample_id in enumerate(test_manifest[\"Id\"])\n    }\n    return cache, progress, row_by_id\n\n\ndef convert_test_batch(paths, cache, progress, row_by_id):\n    converted = 0\n    errors = []\n    for path in tqdm(paths, desc=\"Convert test bytes\", leave=False):\n        sample_id = path.stem\n        row = row_by_id.get(sample_id)\n        if row is None:\n            errors.append((sample_id, \"ID absent from sampleSubmission.csv\"))\n            continue\n        if progress[row]:\n            path.unlink(missing_ok=True)\n            continue\n        try:\n            cache[row] = stage7_byte_to_image(path, IMAGE_SIZE)\n            progress[row] = True\n            path.unlink()\n            converted += 1\n        except Exception as error:\n            errors.append((sample_id, repr(error)))\n    cache.flush()\n    np.save(TEST_PROGRESS_PATH, progress)\n    if errors:\n        pd.DataFrame(errors, columns=[\"Id\", \"error\"]).to_csv(\n            STAGE7_DIR / \"test_preprocessing_errors.csv\", index=False\n        )\n        raise RuntimeError(f\"Failed to convert {len(errors)} test files.\")\n    return converted\n\n\ndef build_test_cache_one_pass(test_archive, test_manifest):\n    if platform.system() != \"Linux\":\n        raise RuntimeError(\"Stage 7 controlled extraction requires Linux/Kaggle.\")\n\n    cache, progress, row_by_id = open_or_create_test_cache(test_manifest)\n    if progress.all():\n        print(\"Reuse complete test image cache.\")\n        if TEST_CACHE_META_PATH.exists():\n            return json.loads(TEST_CACHE_META_PATH.read_text(encoding=\"utf-8\"))\n        return {\n            \"completed\": True,\n            \"sample_count\": len(test_manifest),\n            \"cache_path\": str(TEST_CACHE_PATH),\n        }\n\n    started = time.perf_counter()\n    peak_staging_bytes = 0\n    converted_this_run = 0\n\n    existing = find_test_byte_files()\n    if existing:\n        converted_this_run += convert_test_batch(\n            existing, cache, progress, row_by_id\n        )\n        print(f\"Consumed existing test files; cached {int(progress.sum()):,}.\")\n\n    command = [\n        \"7z\", \"x\", \"-r\", str(test_archive), \"*.bytes\",\n        f\"-o{TEST_BYTES_DIR}\", \"-y\",\n    ]\n    log_handle = open(TEST_EXTRACTION_LOG_PATH, \"w\", encoding=\"utf-8\")\n    process = subprocess.Popen(\n        command,\n        stdout=log_handle,\n        stderr=subprocess.STDOUT,\n        text=True,\n    )\n    paused = False\n    print(\"Stage 7 extraction started; cache progress is resumable.\")\n\n    try:\n        while process.poll() is None:\n            time.sleep(1.0)\n            files = find_test_byte_files()\n            staging_bytes = sum(\n                path.stat().st_size for path in files if path.exists()\n            )\n            peak_staging_bytes = max(peak_staging_bytes, staging_bytes)\n            free_bytes = shutil.disk_usage(TEST_BYTES_DIR).free\n            should_drain = (\n                len(files) >= TEST_PROCESS_BATCH_FILES\n                or staging_bytes >= TEST_STAGING_LIMIT_GIB * 1024**3\n                or free_bytes <= TEST_MIN_FREE_GIB * 1024**3\n            )\n            if not should_drain:\n                continue\n\n            os.kill(process.pid, signal.SIGSTOP)\n            paused = True\n            time.sleep(0.25)\n            files = sorted(\n                find_test_byte_files(), key=lambda path: path.stat().st_mtime\n            )\n            ready = files[:-1] if len(files) > 1 else []\n            if ready:\n                converted_this_run += convert_test_batch(\n                    ready, cache, progress, row_by_id\n                )\n                print(\n                    f\"Test cache {int(progress.sum()):,}/{len(progress):,}; \"\n                    f\"free disk {shutil.disk_usage(TEST_BYTES_DIR).free/1024**3:.2f} GiB\"\n                )\n            os.kill(process.pid, signal.SIGCONT)\n            paused = False\n\n        log_handle.flush()\n        if process.returncode != 0:\n            tail = TEST_EXTRACTION_LOG_PATH.read_text(\n                encoding=\"utf-8\", errors=\"ignore\"\n            )[-4000:]\n            print(tail)\n            raise subprocess.CalledProcessError(process.returncode, command)\n\n        remaining = find_test_byte_files()\n        if remaining:\n            converted_this_run += convert_test_batch(\n                remaining, cache, progress, row_by_id\n            )\n    except Exception:\n        if process.poll() is None:\n            if paused:\n                os.kill(process.pid, signal.SIGCONT)\n            process.terminate()\n            try:\n                process.wait(timeout=10)\n            except subprocess.TimeoutExpired:\n                process.kill()\n        raise\n    finally:\n        log_handle.close()\n        cache.flush()\n        np.save(TEST_PROGRESS_PATH, progress)\n\n    missing = np.flatnonzero(~progress)\n    if len(missing):\n        missing_ids = test_manifest.iloc[missing][\"Id\"].head().tolist()\n        raise RuntimeError(\n            f\"Test cache incomplete: {len(missing)} missing; first IDs {missing_ids}\"\n        )\n\n    meta = {\n        \"format\": \"npy_memmap\",\n        \"image_size\": IMAGE_SIZE,\n        \"sample_count\": len(test_manifest),\n        \"cache_path\": str(TEST_CACHE_PATH),\n        \"cache_bytes\": TEST_CACHE_PATH.stat().st_size,\n        \"elapsed_seconds\": time.perf_counter() - started,\n        \"peak_staging_bytes\": peak_staging_bytes,\n        \"converted_this_run\": converted_this_run,\n        \"completed\": True,\n    }\n    TEST_CACHE_META_PATH.write_text(\n        json.dumps(meta, indent=2), encoding=\"utf-8\"\n    )\n    print(json.dumps(meta, indent=2))\n    return meta\n\n\nclass Stage7TestDataset(Dataset):\n    def __init__(self, cache_path, manifest):\n        self.cache_path = str(cache_path)\n        self.manifest = manifest.reset_index(drop=True)\n        self.images = None\n        self.mean = torch.tensor([0.485, 0.456, 0.406]).view(3, 1, 1)\n        self.std = torch.tensor([0.229, 0.224, 0.225]).view(3, 1, 1)\n\n    def _cache(self):\n        if self.images is None:\n            self.images = np.load(self.cache_path, mmap_mode=\"r\")\n        return self.images\n\n    def __len__(self):\n        return len(self.manifest)\n\n    def __getitem__(self, index):\n        row = self.manifest.iloc[index]\n        image = np.array(\n            self._cache()[int(row[\"cache_index\"])], copy=True\n        )\n        tensor = torch.from_numpy(image).float().unsqueeze(0) / 255.0\n        tensor = tensor.repeat(3, 1, 1)\n        tensor = (tensor - self.mean) / self.std\n        return tensor, str(row[\"Id\"])\n\n\ndef build_resnet18_from_checkpoint(checkpoint_path):\n    checkpoint = torch.load(checkpoint_path, map_location=\"cpu\")\n    model = resnet18(weights=None)\n    model.fc = nn.Linear(model.fc.in_features, len(CLASS_NAMES))\n    model.load_state_dict(checkpoint[\"model_state\"], strict=True)\n    return model, checkpoint\n\n\ndef run_stage7_ensemble(test_manifest, sample_submission):\n    checkpoint_dir = ARTIFACT_DIR / \"stage6_resnet18_cv\"\n    checkpoint_paths = sorted(checkpoint_dir.glob(\"fold_*_best.pt\"))\n    if len(checkpoint_paths) != 5:\n        raise FileNotFoundError(\n            f\"Expected 5 ResNet18 checkpoints in {checkpoint_dir}, \"\n            f\"found {len(checkpoint_paths)}. Keep the current training session alive.\"\n        )\n\n    probability_columns = [\n        column for column in sample_submission.columns if column != \"Id\"\n    ]\n    if len(probability_columns) != len(CLASS_NAMES):\n        raise ValueError(\n            f\"Expected {len(CLASS_NAMES)} probability columns, \"\n            f\"found {probability_columns}\"\n        )\n    if sample_submission[\"Id\"].astype(str).tolist() != test_manifest[\"Id\"].tolist():\n        raise RuntimeError(\"Submission and test manifest ID order mismatch.\")\n\n    dataset = Stage7TestDataset(TEST_CACHE_PATH, test_manifest)\n    loader = DataLoader(\n        dataset,\n        batch_size=RESNET18_BATCH_SIZE,\n        shuffle=False,\n        num_workers=0,\n        pin_memory=True,\n    )\n    ensemble = np.zeros(\n        (len(test_manifest), len(CLASS_NAMES)), dtype=np.float64\n    )\n    fold_predictions = np.zeros(\n        (len(test_manifest), len(checkpoint_paths)), dtype=np.int8\n    )\n    started = time.perf_counter()\n    torch.cuda.reset_peak_memory_stats()\n\n    for fold_index, checkpoint_path in enumerate(checkpoint_paths):\n        model, checkpoint = build_resnet18_from_checkpoint(checkpoint_path)\n        model = model.to(device).eval()\n        fold_batches = []\n        with torch.inference_mode():\n            for images, _ in tqdm(\n                loader,\n                desc=f\"Inference fold {fold_index + 1}/5\",\n                leave=False,\n            ):\n                images = images.to(device, non_blocking=True)\n                with torch.autocast(\n                    device_type=device.type,\n                    dtype=torch.float16,\n                    enabled=device.type == \"cuda\",\n                ):\n                    logits = model(images)\n                probabilities = torch.softmax(\n                    logits.float(), dim=1\n                ).cpu().numpy().astype(np.float64)\n                probabilities /= probabilities.sum(axis=1, keepdims=True)\n                fold_batches.append(probabilities)\n\n        fold_probabilities = np.vstack(fold_batches)\n        if fold_probabilities.shape != ensemble.shape:\n            raise RuntimeError(\n                f\"Fold {fold_index + 1} prediction shape mismatch: \"\n                f\"{fold_probabilities.shape}\"\n            )\n        ensemble += fold_probabilities / len(checkpoint_paths)\n        fold_predictions[:, fold_index] = fold_probabilities.argmax(axis=1)\n        print(\n            f\"Loaded fold {checkpoint.get('fold')} checkpoint \"\n            f\"from epoch {checkpoint.get('epoch')}.\"\n        )\n        del model, checkpoint, fold_probabilities, fold_batches\n        torch.cuda.empty_cache()\n\n    ensemble /= ensemble.sum(axis=1, keepdims=True)\n    if not np.isfinite(ensemble).all() or (ensemble < 0).any():\n        raise RuntimeError(\"Ensemble contains invalid probabilities.\")\n    max_sum_error = float(np.abs(ensemble.sum(axis=1) - 1.0).max())\n\n    submission = sample_submission.copy()\n    submission[probability_columns] = ensemble\n    submission.to_csv(SUBMISSION_PATH, index=False)\n\n    predicted_labels = ensemble.argmax(axis=1)\n    agreement = (fold_predictions == predicted_labels[:, None]).sum(axis=1)\n    prediction_frame = pd.DataFrame({\n        \"Id\": test_manifest[\"Id\"],\n        \"predicted_class\": predicted_labels + 1,\n        \"predicted_family\": [CLASS_NAMES[index] for index in predicted_labels],\n        \"confidence\": ensemble.max(axis=1),\n        \"agreeing_folds\": agreement,\n    })\n    prediction_frame.to_csv(\n        STAGE7_DIR / \"test_predictions.csv\", index=False\n    )\n    distribution = (\n        prediction_frame.groupby([\"predicted_class\", \"predicted_family\"])\n        .size()\n        .rename(\"count\")\n        .reset_index()\n    )\n    distribution.to_csv(\n        STAGE7_DIR / \"test_prediction_distribution.csv\", index=False\n    )\n    plt.figure(figsize=(11, 4))\n    plt.bar(distribution[\"predicted_family\"], distribution[\"count\"])\n    plt.xticks(rotation=35, ha=\"right\")\n    plt.ylabel(\"Predicted samples\")\n    plt.title(\"Stage 7 test prediction distribution\")\n    plt.tight_layout()\n    plt.savefig(\n        STAGE7_DIR / \"test_prediction_distribution.png\", dpi=160\n    )\n    plt.show()\n\n    summary = {\n        \"experiment\": \"EXP-011\",\n        \"model\": \"5-fold pretrained ResNet18 probability ensemble\",\n        \"sample_count\": len(test_manifest),\n        \"checkpoint_count\": len(checkpoint_paths),\n        \"checkpoints\": [path.name for path in checkpoint_paths],\n        \"inference_seconds\": time.perf_counter() - started,\n        \"peak_vram_bytes\": torch.cuda.max_memory_allocated(),\n        \"gpu\": torch.cuda.get_device_name(0),\n        \"image_size\": IMAGE_SIZE,\n        \"batch_size\": RESNET18_BATCH_SIZE,\n        \"probability_sum_max_abs_error\": max_sum_error,\n        \"mean_confidence\": float(ensemble.max(axis=1).mean()),\n        \"unanimous_samples\": int((agreement == len(checkpoint_paths)).sum()),\n        \"submission_path\": str(SUBMISSION_PATH),\n    }\n    (STAGE7_DIR / \"inference_summary.json\").write_text(\n        json.dumps(summary, indent=2), encoding=\"utf-8\"\n    )\n    display(distribution)\n    print(json.dumps(summary, indent=2))\n    print(f\"Submission ready: {SUBMISSION_PATH}\")\n    return submission, prediction_frame, summary\n\n\nif RUN_STAGE7:\n    if device.type != \"cuda\":\n        raise RuntimeError(\"Keep the Kaggle GPU enabled for Stage 7 inference.\")\n    test_archive = resolve_kaggle_input_file(\"test.7z\")\n    sample_submission_path = resolve_kaggle_input_file(\"sampleSubmission.csv\")\n    sample_submission = pd.read_csv(\n        sample_submission_path, dtype={\"Id\": str}\n    )\n    if sample_submission[\"Id\"].duplicated().any():\n        raise ValueError(\"sampleSubmission.csv contains duplicate IDs.\")\n    test_manifest = pd.DataFrame({\n        \"cache_index\": np.arange(len(sample_submission), dtype=np.int64),\n        \"Id\": sample_submission[\"Id\"].astype(str),\n    })\n    test_manifest.to_csv(TEST_MANIFEST_PATH, index=False)\n    stage7_parser_smoke_test()\n    stage7_cache_meta = build_test_cache_one_pass(\n        test_archive, test_manifest\n    )\n    stage7_submission, stage7_predictions, stage7_summary = (\n        run_stage7_ensemble(test_manifest, sample_submission)\n    )\nelse:\n    print(\"Stage 7 inference disabled.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-21T13:21:16.211547Z","iopub.execute_input":"2026-07-21T13:21:16.212371Z","iopub.status.idle":"2026-07-21T13:21:16.448810Z","shell.execute_reply.started":"2026-07-21T13:21:16.212329Z","shell.execute_reply":"2026-07-21T13:21:16.447669Z"}},"outputs":[],"execution_count":null}]}