{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"d00b1d65-f6b8-45c3-9f2b-8a81af2e5676","cell_type":"markdown","source":"# PIXEL Camera-ready Runtime Audit\n\nThis notebook/script implements **Task 1** in\n`yeu_cau_dieu_chinh_thuc_nghiem_camera_ready.md`.\n\nIt does not train or fine-tune any model. It produces a reproducible runtime\naudit for Microsoft BIG-2015 with:\n\n- a fixed list of 500 test samples;\n- three independent preprocessing and inference timing repeats;\n- raw per-sample and per-batch timing records;\n- grayscale-versus-PPS ResNet-50 inference;\n- PPS EfficientNet-B3 inference;\n- environment, summary, and estimated end-to-end reports.\n\nTasks 2 and 3 (channel occlusion and XAI checkpoint verification) are outside\nthis runtime notebook and must be handled by their corresponding analysis code.\n\n","metadata":{}},{"id":"cf77ba56-ce73-4f21-afcf-870925cb93bc","cell_type":"code","source":"# ============================================================\n# CELL 1: Imports, paths, and benchmark configuration\n# ============================================================\nimport csv\nimport gc\nimport glob\nimport hashlib\nimport importlib.metadata as importlib_metadata\nimport json\nimport logging\nimport os\nimport platform\nimport random\nimport subprocess\nimport sys\nimport time\nfrom contextlib import nullcontext\nfrom pathlib import Path\n\n# Keep the CPU thread settings used by the original experiments.\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\nos.environ[\"OMP_NUM_THREADS\"] = \"2\"\nos.environ[\"MKL_NUM_THREADS\"] = \"2\"\n\nimport numpy as np\nimport pandas as pd\nimport psutil\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nfrom skimage.feature import local_binary_pattern\nfrom skimage.filters.rank import entropy\nfrom skimage.morphology import disk\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import models, transforms\n\n\n# ---------- Input data ----------\nRAW_BYTES_DIR = \"/kaggle/input/datasets/songwonmin/malware-only-byte/only_byte\"\nMICROSOFT_RGB_DIR = \"/kaggle/input/datasets/vnhtbo/microsoft/train_rgb/train_rgb\"\nMICROSOFT_LABELS_CSV = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\n\n# Use the exact saved split when available. If it is unavailable, the notebook\n# deterministically recreates the original 70/15/15 split with seed 42 and saves\n# the resulting split snapshot in the audit output.\nSPLIT_JSON_OVERRIDE = None\nSPLIT_JSON_CANDIDATES = [\n    \"/kaggle/working/checkpoints/split_microsoft_rgb.json\",\n    \"/kaggle/input/datasets/vnhtbo/eb3-train/split_microsoft_rgb.json\",\n]\n\n# ---------- Model checkpoints ----------\n# Set overrides to the exact checkpoints used in the paper. Random weights are\n# deliberately not allowed because the audit must identify the measured model.\nRESNET50_GRAYSCALE_CKPT_OVERRIDE = None\nRESNET50_PPS_CKPT_OVERRIDE = None\nEB3_PPS_CKPT_OVERRIDE = None\n\nRESNET50_GRAYSCALE_CKPT_CANDIDATES = [\n    \"/kaggle/input/datasets/vnhtbo/microsoftgray/best_microsoft_gray_run1.pt\",\n    \"/kaggle/working/checkpoints/best_microsoft_grayscale_run1.pt\",\n]\nRESNET50_PPS_CKPT_CANDIDATES = [\n    \"/kaggle/input/datasets/vnhtbo/microsoft-rgb/best_microsoft_rgb_RGB_run1.pt\",\n]\nEB3_PPS_CKPT_CANDIDATES = [\n    \"/kaggle/input/datasets/vnhtbo/eb3-train/best_eb3_teacher_run1.pt\",\n    \"/kaggle/working/checkpoints/best_eb3_teacher_run1.pt\",\n]\n\nRESNET50_GRAYSCALE_CKPT_PATTERNS = [\n    \"best_microsoft_gray*_run*.pt\",\n    \"best_microsoft_grayscale*_run*.pt\",\n]\nRESNET50_PPS_CKPT_PATTERNS = [\n    \"best_microsoft_rgb_RGB_run*.pt\",\n    \"best_microsoft_rgb_rgb_run*.pt\",\n]\nEB3_PPS_CKPT_PATTERNS = [\n    \"best_eb3_teacher_run1.pt\",\n    \"*eb3*teacher*run1*.pt\",\n]\n\n# ---------- Audit output ----------\nAUDIT_ROOT = Path(\"/kaggle/working/camera_ready_experiment_audit\")\nENV_DIR = AUDIT_ROOT / \"environment\"\nRUNTIME_DIR = AUDIT_ROOT / \"runtime\"\nLOG_DIR = AUDIT_ROOT / \"logs\"\nfor directory in (AUDIT_ROOT, ENV_DIR, RUNTIME_DIR, LOG_DIR):\n    directory.mkdir(parents=True, exist_ok=True)\n\nSAMPLE_IDS_CSV = RUNTIME_DIR / \"runtime_sample_ids.csv\"\nSAMPLE_MANIFEST_CSV = RUNTIME_DIR / \"runtime_sample_manifest.csv\"\nSPLIT_SNAPSHOT_JSON = RUNTIME_DIR / \"runtime_split_snapshot.json\"\nPREPROCESS_RAW_CSV = RUNTIME_DIR / \"preprocessing_runtime_raw.csv\"\nINFERENCE_RAW_CSV = RUNTIME_DIR / \"inference_runtime_raw.csv\"\nSUMMARY_CSV = RUNTIME_DIR / \"runtime_summary.csv\"\nEND_TO_END_CSV = RUNTIME_DIR / \"end_to_end_runtime.csv\"\nENVIRONMENT_TXT = ENV_DIR / \"runtime_environment.txt\"\nCONFIG_JSON = ENV_DIR / \"runtime_config.json\"\nREADME_PATH = AUDIT_ROOT / \"README.md\"\nRUNTIME_LOG = LOG_DIR / \"runtime.log\"\n\n# ---------- Measurement protocol ----------\nDATASET_NAME = \"Microsoft BIG-2015\"\nSPLIT_NAME = \"microsoft_rgb\"\nSPLIT_SEED = 42\nVAL_RATIO = 0.15\nTEST_RATIO = 0.15\nN_RUNTIME_SAMPLES = 500\nN_REPEATS = 3\n\nRESNET50_IMAGE_SIZE = 224\nEB3_IMAGE_SIZE = 300\nINFERENCE_BATCH_SIZES = [1, 32]\nWARMUP_BATCHES = 20\nMEASURE_BATCHES = 100\nNUM_WORKERS = 2\nPIN_MEMORY = True\nUSE_AMP = True\nINCLUDE_H2D_IN_TIMING = False\n\n# Full PPS means R+G+B. Set this to False only for an explicitly documented\n# architecture-latency check using an RG/RB/GB checkpoint.\nREQUIRE_FULL_PPS_RGB = True\nRUN_PREPROCESSING = True\nRUN_INFERENCE = True\nFORCE_RERUN_PREPROCESSING = False\nFORCE_RERUN_INFERENCE = False\n\n# The notebook reopens every .bytes file for every end-to-end observation, but\n# does not clear the operating-system page cache. This is recorded explicitly.\nDATA_LOCATION_NOTE = \"Kaggle mounted dataset; physical storage backend managed by Kaggle\"\nCACHE_POLICY_NOTE = (\n    \"Files are reopened for every end-to-end observation; OS page cache is not \"\n    \"explicitly cleared or controlled\"\n)\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nAMP_ENABLED = bool(USE_AMP and DEVICE.type == \"cuda\")\n\n\ndef configure_logger():\n    logger = logging.getLogger(\"pixel_runtime_audit\")\n    logger.setLevel(logging.INFO)\n    for handler in list(logger.handlers):\n        handler.close()\n        logger.removeHandler(handler)\n    formatter = logging.Formatter(\"%(asctime)s | %(levelname)s | %(message)s\")\n    file_handler = logging.FileHandler(RUNTIME_LOG, mode=\"a\", encoding=\"utf-8\")\n    file_handler.setFormatter(formatter)\n    stream_handler = logging.StreamHandler(sys.stdout)\n    stream_handler.setFormatter(formatter)\n    logger.addHandler(file_handler)\n    logger.addHandler(stream_handler)\n    return logger\n\n\nLOGGER = configure_logger()\nLOGGER.info(\"Runtime audit configured on device=%s\", DEVICE)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-11T02:51:25.388720Z","iopub.execute_input":"2026-07-11T02:51:25.389521Z","iopub.status.idle":"2026-07-11T02:51:25.685432Z","shell.execute_reply.started":"2026-07-11T02:51:25.389490Z","shell.execute_reply":"2026-07-11T02:51:25.684594Z"}},"outputs":[],"execution_count":null},{"id":"89a36da7-1617-40b8-b455-9f790c2a7ca9","cell_type":"markdown","source":"## Environment and preflight\n\nThis section records CPU, GPU, memory, thread, library, storage, and cache\ninformation before any timing begins. It also resolves the fixed split and all\nrequired checkpoints before starting the expensive preprocessing benchmark.\n\n","metadata":{}},{"id":"46f87de4-c6d6-4404-91bd-26ddfa6b77b9","cell_type":"code","source":"# ============================================================\n# CELL 2: Environment capture and file-resolution helpers\n# ============================================================\ndef command_output(command):\n    try:\n        completed = subprocess.run(\n            command,\n            check=False,\n            capture_output=True,\n            text=True,\n            timeout=30,\n        )\n        output = (completed.stdout or completed.stderr).strip()\n        return output or \"Unavailable\"\n    except Exception as exc:\n        return f\"Unavailable ({type(exc).__name__}: {exc})\"\n\n\ndef package_version(package_name):\n    try:\n        return importlib_metadata.version(package_name)\n    except importlib_metadata.PackageNotFoundError:\n        return \"not installed\"\n\n\ndef cpu_model_name():\n    model = platform.processor().strip()\n    if model:\n        return model\n    try:\n        for line in Path(\"/proc/cpuinfo\").read_text(errors=\"ignore\").splitlines():\n            if line.lower().startswith(\"model name\"):\n                return line.split(\":\", 1)[1].strip()\n    except OSError:\n        pass\n    return \"Unavailable\"\n\n\ndef collect_environment():\n    gpu_name = \"Unavailable\"\n    gpu_driver = \"Unavailable\"\n    gpu_memory = \"Unavailable\"\n    if torch.cuda.is_available():\n        gpu_name = torch.cuda.get_device_name(0)\n        query = command_output([\n            \"nvidia-smi\",\n            \"--query-gpu=name,driver_version,memory.total\",\n            \"--format=csv,noheader\",\n        ])\n        if query != \"Unavailable\":\n            parts = [part.strip() for part in query.splitlines()[0].split(\",\")]\n            if len(parts) >= 3:\n                gpu_name, gpu_driver, gpu_memory = parts[:3]\n\n    environment = {\n        \"timestamp_local\": time.strftime(\"%Y-%m-%d %H:%M:%S %z\"),\n        \"operating_system\": platform.platform(),\n        \"python_version\": sys.version.replace(\"\\n\", \" \"),\n        \"cpu_model\": cpu_model_name(),\n        \"physical_cpu_cores\": psutil.cpu_count(logical=False),\n        \"logical_cpu_threads\": psutil.cpu_count(logical=True),\n        \"ram_total_gib\": psutil.virtual_memory().total / (1024 ** 3),\n        \"gpu_name\": gpu_name,\n        \"gpu_driver\": gpu_driver,\n        \"gpu_memory\": gpu_memory,\n        \"cuda_version_reported_by_pytorch\": torch.version.cuda or \"Unavailable\",\n        \"cudnn_version\": torch.backends.cudnn.version(),\n        \"pytorch_version\": torch.__version__,\n        \"torchvision_version\": package_version(\"torchvision\"),\n        \"numpy_version\": np.__version__,\n        \"pandas_version\": pd.__version__,\n        \"pillow_version\": package_version(\"Pillow\"),\n        \"scikit_image_version\": package_version(\"scikit-image\"),\n        \"psutil_version\": psutil.__version__,\n        \"data_loader_workers\": NUM_WORKERS,\n        \"pytorch_cpu_threads\": torch.get_num_threads(),\n        \"pytorch_interop_threads\": torch.get_num_interop_threads(),\n        \"OMP_NUM_THREADS\": os.environ.get(\"OMP_NUM_THREADS\", \"unset\"),\n        \"MKL_NUM_THREADS\": os.environ.get(\"MKL_NUM_THREADS\", \"unset\"),\n        \"raw_bytes_directory\": RAW_BYTES_DIR,\n        \"rgb_image_directory\": MICROSOFT_RGB_DIR,\n        \"data_location\": DATA_LOCATION_NOTE,\n        \"cache_policy\": CACHE_POLICY_NOTE,\n        \"raw_data_mount\": command_output([\"df\", \"-T\", RAW_BYTES_DIR]),\n        \"inference_amp_enabled\": AMP_ENABLED,\n        \"host_to_device_transfer_timed\": INCLUDE_H2D_IN_TIMING,\n        \"png_writing_timed\": False,\n        \"zip_writing_timed\": False,\n    }\n\n    lines = [f\"{key}: {value}\" for key, value in environment.items()]\n    ENVIRONMENT_TXT.write_text(\"\\n\".join(lines) + \"\\n\", encoding=\"utf-8\")\n    LOGGER.info(\"Saved environment report: %s\", ENVIRONMENT_TXT)\n    return environment\n\n\ndef resolve_file(override, candidates, patterns, label, reject_tokens=()):\n    paths = []\n    if override:\n        paths.append(override)\n    paths.extend(candidates)\n    for path in paths:\n        if path and os.path.isfile(path):\n            name = os.path.basename(path).lower()\n            if not any(token.lower() in name for token in reject_tokens):\n                LOGGER.info(\"%s resolved to %s\", label, path)\n                return path\n\n    for root in (\"/kaggle/working\", \"/kaggle/input\"):\n        if not os.path.isdir(root):\n            continue\n        for pattern in patterns:\n            for path in sorted(glob.glob(os.path.join(root, \"**\", pattern), recursive=True)):\n                name = os.path.basename(path).lower()\n                if not any(token.lower() in name for token in reject_tokens):\n                    LOGGER.info(\"%s resolved by pattern to %s\", label, path)\n                    return path\n    LOGGER.warning(\"%s could not be resolved\", label)\n    return None\n\n\ndef sha256_file(path, chunk_size=1024 * 1024):\n    digest = hashlib.sha256()\n    with open(path, \"rb\") as handle:\n        while True:\n            chunk = handle.read(chunk_size)\n            if not chunk:\n                break\n            digest.update(chunk)\n    return digest.hexdigest()\n\n\nENVIRONMENT = collect_environment()\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"30799d40-5db0-41ec-84a9-9377259b510e","cell_type":"code","source":"# ============================================================\n# CELL 3: Original byte-to-PPS preprocessing implementation\n# ============================================================\n# These functions preserve the implementation in malwarecuathay.ipynb:\n# - address-like tokens are skipped because they are not two characters long;\n# - ?? is replaced with zero;\n# - dynamic width follows the original thresholds;\n# - entropy uses disk radius 8;\n# - LBP uses P=8, R=1, method=\"uniform\" (riu2);\n# - no PNG or ZIP writing is included.\n\ndef get_dynamic_width(num_bytes):\n    kb = num_bytes / 1024.0\n    if kb < 10:\n        return 32\n    if kb < 30:\n        return 64\n    if kb < 60:\n        return 128\n    if kb < 100:\n        return 256\n    if kb < 200:\n        return 384\n    if kb < 500:\n        return 512\n    if kb < 1000:\n        return 768\n    return 1024\n\n\ndef parse_bytes_file(filepath):\n    byte_list = []\n    with open(filepath, \"r\", encoding=\"utf-8\", errors=\"ignore\") as handle:\n        for line in handle:\n            for token in line.strip().split():\n                if len(token) != 2:\n                    continue\n                if token == \"??\":\n                    byte_list.append(0)\n                else:\n                    try:\n                        byte_list.append(int(token, 16))\n                    except ValueError:\n                        pass\n    if len(byte_list) < 256:\n        raise ValueError(f\"Too few bytes parsed from {filepath}: {len(byte_list)}\")\n    return np.asarray(byte_list, dtype=np.uint8)\n\n\ndef byte_array_to_gray(byte_array):\n    width = get_dynamic_width(len(byte_array))\n    height = len(byte_array) // width\n    if height == 0:\n        raise ValueError(\"Byte array is too short for the selected dynamic width\")\n    trimmed = byte_array[: height * width]\n    return trimmed.reshape(height, width)\n\n\ndef compute_entropy_map(gray_image):\n    entropy_image = entropy(gray_image, disk(8))\n    maximum = entropy_image.max()\n    if maximum > 0:\n        return (entropy_image / maximum * 255).astype(np.uint8)\n    return entropy_image.astype(np.uint8)\n\n\ndef compute_lbp_map(gray_image):\n    lbp_image = local_binary_pattern(gray_image, P=8, R=1, method=\"uniform\")\n    maximum = lbp_image.max()\n    if maximum > 0:\n        return (lbp_image / maximum * 255).astype(np.uint8)\n    return lbp_image.astype(np.uint8)\n\n\ndef gray_to_pps(gray_image):\n    raw_channel = gray_image.copy()\n    entropy_channel = compute_entropy_map(gray_image)\n    lbp_channel = compute_lbp_map(gray_image)\n    return np.stack((raw_channel, entropy_channel, lbp_channel), axis=-1)\n\n\nGRAYSCALE_MODEL_TRANSFORM = transforms.Compose([\n    transforms.Resize((RESNET50_IMAGE_SIZE, RESNET50_IMAGE_SIZE)),\n    transforms.Grayscale(num_output_channels=3),\n    transforms.ToTensor(),\n    transforms.Normalize([0.5, 0.5, 0.5], [0.5, 0.5, 0.5]),\n])\n\nPPS_RESNET_TRANSFORM = transforms.Compose([\n    transforms.Resize((RESNET50_IMAGE_SIZE, RESNET50_IMAGE_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n])\n\nPPS_EB3_TRANSFORM = transforms.Compose([\n    transforms.Resize((EB3_IMAGE_SIZE, EB3_IMAGE_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225]),\n])\n\n\ndef gray_matrix_from_file(filepath):\n    return byte_array_to_gray(parse_bytes_file(filepath))\n\n\ndef grayscale_preprocessing_e2e(filepath):\n    gray_image = gray_matrix_from_file(filepath)\n    pil_image = Image.fromarray(gray_image, mode=\"L\")\n    return GRAYSCALE_MODEL_TRANSFORM(pil_image)\n\n\ndef pps_preprocessing_e2e(filepath):\n    gray_image = gray_matrix_from_file(filepath)\n    pps_image = gray_to_pps(gray_image)\n    pil_image = Image.fromarray(pps_image, mode=\"RGB\")\n    return PPS_RESNET_TRANSFORM(pil_image)\n\n\nLOGGER.info(\"Preprocessing functions loaded without algorithm changes\")\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"6edcdb09-dd6a-41fb-a5c3-ec6ba63b6564","cell_type":"code","source":"# ============================================================\n# CELL 4: Fixed BIG-2015 test sample list and checkpoint preflight\n# ============================================================\ndef build_rgb_dataset_index():\n    if not os.path.isdir(MICROSOFT_RGB_DIR):\n        raise FileNotFoundError(f\"RGB image directory not found: {MICROSOFT_RGB_DIR}\")\n    if not os.path.isfile(MICROSOFT_LABELS_CSV):\n        raise FileNotFoundError(f\"Label CSV not found: {MICROSOFT_LABELS_CSV}\")\n\n    labels = pd.read_csv(MICROSOFT_LABELS_CSV)\n    existing_images = set(os.listdir(MICROSOFT_RGB_DIR))\n    records = []\n    for row in labels.itertuples(index=False):\n        sample_id = str(row.Id)\n        filename = f\"{sample_id}.png\"\n        if filename in existing_images:\n            records.append({\n                \"sample_id\": sample_id,\n                \"class_id\": int(row.Class),\n                \"rgb_path\": os.path.join(MICROSOFT_RGB_DIR, filename),\n            })\n    if not records:\n        raise RuntimeError(\"No labeled BIG-2015 RGB images were found\")\n    return records\n\n\ndef resolve_or_recreate_split(num_samples):\n    split_path = resolve_file(\n        SPLIT_JSON_OVERRIDE,\n        SPLIT_JSON_CANDIDATES,\n        [\"split_microsoft_rgb.json\"],\n        \"BIG-2015 split\",\n    )\n    if split_path:\n        with open(split_path, \"r\", encoding=\"utf-8\") as handle:\n            split = json.load(handle)\n        source = f\"loaded:{split_path}\"\n    else:\n        n_test = int(num_samples * TEST_RATIO)\n        n_val = int(num_samples * VAL_RATIO)\n        n_train = num_samples - n_val - n_test\n        generator = torch.Generator().manual_seed(SPLIT_SEED)\n        permutation = torch.randperm(num_samples, generator=generator).tolist()\n        split = {\n            \"train\": permutation[:n_train],\n            \"val\": permutation[n_train:n_train + n_val],\n            \"test\": permutation[n_train + n_val:],\n            \"seed\": SPLIT_SEED,\n            \"total\": num_samples,\n        }\n        source = \"deterministically_recreated_from_original_seed_and_ratios\"\n        LOGGER.warning(\"Saved split was unavailable; recreated deterministic seed-42 split\")\n\n    if int(split.get(\"total\", num_samples)) != num_samples:\n        raise ValueError(\n            f\"Split total={split.get('total')} does not match dataset size={num_samples}\"\n        )\n    test_indices = [int(index) for index in split[\"test\"]]\n    if not test_indices or min(test_indices) < 0 or max(test_indices) >= num_samples:\n        raise ValueError(\"Split contains invalid test indices\")\n\n    snapshot = dict(split)\n    snapshot[\"audit_split_source\"] = source\n    with open(SPLIT_SNAPSHOT_JSON, \"w\", encoding=\"utf-8\") as handle:\n        json.dump(snapshot, handle, indent=2)\n    return split, source\n\n\ndef build_raw_bytes_lookup():\n    paths = sorted(glob.glob(os.path.join(RAW_BYTES_DIR, \"**\", \"*.bytes\"), recursive=True))\n    if not paths:\n        raise FileNotFoundError(f\"No .bytes files found under {RAW_BYTES_DIR}\")\n    lookup = {Path(path).stem: path for path in paths}\n    if len(lookup) != len(paths):\n        raise ValueError(\"Duplicate .bytes sample IDs found under RAW_BYTES_DIR\")\n    return lookup\n\n\ndef create_or_load_fixed_samples(dataset_records, test_indices, raw_lookup):\n    record_by_id = {record[\"sample_id\"]: record for record in dataset_records}\n    test_ids = [dataset_records[index][\"sample_id\"] for index in test_indices]\n    test_id_set = set(test_ids)\n\n    if SAMPLE_IDS_CSV.exists():\n        saved = pd.read_csv(SAMPLE_IDS_CSV, dtype={\"sample_id\": str})\n        sample_ids = saved[\"sample_id\"].astype(str).tolist()\n        if len(sample_ids) != N_RUNTIME_SAMPLES or len(set(sample_ids)) != N_RUNTIME_SAMPLES:\n            raise ValueError(\n                f\"Existing {SAMPLE_IDS_CSV} must contain {N_RUNTIME_SAMPLES} unique IDs\"\n            )\n        invalid = [sample_id for sample_id in sample_ids if sample_id not in test_id_set]\n        if invalid:\n            raise ValueError(f\"Saved runtime IDs are not in the fixed test split: {invalid[:5]}\")\n        selection_source = \"reused_existing_runtime_sample_ids.csv\"\n    else:\n        available_test_ids = [\n            sample_id\n            for sample_id in test_ids\n            if sample_id in raw_lookup and sample_id in record_by_id\n        ]\n        if len(available_test_ids) < N_RUNTIME_SAMPLES:\n            raise RuntimeError(\n                f\"Only {len(available_test_ids)} test samples have both .bytes and RGB inputs\"\n            )\n        sample_ids = available_test_ids[:N_RUNTIME_SAMPLES]\n        pd.DataFrame({\"sample_id\": sample_ids}).to_csv(SAMPLE_IDS_CSV, index=False)\n        selection_source = \"first_500_in_fixed_test_split_order\"\n\n    manifest = []\n    for sample_id in sample_ids:\n        if sample_id not in raw_lookup:\n            raise FileNotFoundError(f\"Missing raw .bytes file for {sample_id}\")\n        record = record_by_id[sample_id]\n        manifest.append({\n            \"sample_id\": sample_id,\n            \"class_id\": record[\"class_id\"],\n            \"bytes_path\": raw_lookup[sample_id],\n            \"rgb_path\": record[\"rgb_path\"],\n        })\n    manifest_df = pd.DataFrame(manifest)\n    manifest_df.to_csv(SAMPLE_MANIFEST_CSV, index=False)\n    LOGGER.info(\"Fixed runtime sample list: %s (%s)\", SAMPLE_IDS_CSV, selection_source)\n    return manifest_df, selection_source\n\n\ndef infer_channel_config_from_checkpoint(path, default=\"RGB\"):\n    if not path:\n        return default\n    name = os.path.basename(path).upper()\n    for config in (\"RGB\", \"RG\", \"RB\", \"GB\", \"R\", \"G\", \"B\"):\n        if f\"_{config}_RUN\" in name or f\"_{config}.PT\" in name:\n            return config\n    return default\n\n\nDATASET_RECORDS = build_rgb_dataset_index()\nSPLIT, SPLIT_SOURCE = resolve_or_recreate_split(len(DATASET_RECORDS))\nRAW_BYTES_LOOKUP = build_raw_bytes_lookup()\nSAMPLE_MANIFEST, SAMPLE_SELECTION_SOURCE = create_or_load_fixed_samples(\n    DATASET_RECORDS,\n    SPLIT[\"test\"],\n    RAW_BYTES_LOOKUP,\n)\n\nRESOLVED_CHECKPOINTS = {\n    \"resnet50_grayscale\": resolve_file(\n        RESNET50_GRAYSCALE_CKPT_OVERRIDE,\n        RESNET50_GRAYSCALE_CKPT_CANDIDATES,\n        RESNET50_GRAYSCALE_CKPT_PATTERNS,\n        \"ResNet-50 grayscale checkpoint\",\n        reject_tokens=(\"rgb\",),\n    ),\n    \"resnet50_pps\": resolve_file(\n        RESNET50_PPS_CKPT_OVERRIDE,\n        RESNET50_PPS_CKPT_CANDIDATES,\n        RESNET50_PPS_CKPT_PATTERNS,\n        \"ResNet-50 PPS checkpoint\",\n    ),\n    \"efficientnet_b3_pps\": resolve_file(\n        EB3_PPS_CKPT_OVERRIDE,\n        EB3_PPS_CKPT_CANDIDATES,\n        EB3_PPS_CKPT_PATTERNS,\n        \"EfficientNet-B3 PPS checkpoint\",\n    ),\n}\n\nif RUN_INFERENCE:\n    if DEVICE.type != \"cuda\":\n        raise RuntimeError(\"Inference audit requires CUDA for synchronized GPU timing\")\n    missing = [name for name, path in RESOLVED_CHECKPOINTS.items() if not path]\n    if missing:\n        raise FileNotFoundError(\n            \"Missing required checkpoint(s): \" + \", \".join(missing) +\n            \". Attach the checkpoint dataset or set the corresponding *_CKPT_OVERRIDE.\"\n        )\n    for name in (\"resnet50_pps\", \"efficientnet_b3_pps\"):\n        channel_config = infer_channel_config_from_checkpoint(\n            RESOLVED_CHECKPOINTS[name], default=\"RGB\"\n        )\n        if REQUIRE_FULL_PPS_RGB and channel_config != \"RGB\":\n            raise ValueError(\n                f\"{name} resolved to channel config {channel_config}, but full RGB PPS is required. \"\n                \"Use the RGB checkpoint, or set REQUIRE_FULL_PPS_RGB=False and report the deviation.\"\n            )\n\nCONFIG_RECORD = {\n    \"dataset\": DATASET_NAME,\n    \"split_source\": SPLIT_SOURCE,\n    \"split_seed\": SPLIT.get(\"seed\", SPLIT_SEED),\n    \"sample_selection\": SAMPLE_SELECTION_SOURCE,\n    \"num_runtime_samples\": len(SAMPLE_MANIFEST),\n    \"num_repeats\": N_REPEATS,\n    \"entropy_radius\": 8,\n    \"lbp_P\": 8,\n    \"lbp_R\": 1,\n    \"lbp_method\": \"uniform (riu2)\",\n    \"resnet50_image_size\": RESNET50_IMAGE_SIZE,\n    \"efficientnet_b3_image_size\": EB3_IMAGE_SIZE,\n    \"warmup_batches\": WARMUP_BATCHES,\n    \"measured_batches\": MEASURE_BATCHES,\n    \"inference_batch_sizes\": INFERENCE_BATCH_SIZES,\n    \"include_h2d_in_timing\": INCLUDE_H2D_IN_TIMING,\n    \"amp_enabled\": AMP_ENABLED,\n    \"cache_policy\": CACHE_POLICY_NOTE,\n    \"grayscale_input_format\": (\n        \"PPS R/raw-byte channel converted to L, resized to 224x224, repeated to \"\n        \"three channels, then normalized with mean=std=0.5\"\n    ),\n    \"checkpoints\": {\n        name: {\n            \"path\": path,\n            \"sha256\": sha256_file(path) if path else None,\n        }\n        for name, path in RESOLVED_CHECKPOINTS.items()\n    },\n}\nwith open(CONFIG_JSON, \"w\", encoding=\"utf-8\") as handle:\n    json.dump(CONFIG_RECORD, handle, indent=2)\n\nprint(SAMPLE_MANIFEST.head())\nprint(f\"Fixed samples: {len(SAMPLE_MANIFEST)}\")\nprint(json.dumps(RESOLVED_CHECKPOINTS, indent=2))\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"3a7be158-5e8c-4b4a-b006-a3d2dbfdad9c","cell_type":"markdown","source":"## Preprocessing timing\n\nEnd-to-end grayscale and PPS observations include `.bytes` file reading,\nparsing, dynamic-width reshaping, resize, tensor conversion, and normalization.\nPNG writing, ZIP writing, and GPU transfer are excluded. Entropy and LBP rows\nare compute-only and start from an already parsed grayscale matrix.\n\n","metadata":{}},{"id":"be1bdf12-e9dc-46a3-bae1-1f012a990221","cell_type":"code","source":"# ============================================================\n# CELL 5: Raw per-sample preprocessing timing, three repeats\n# ============================================================\nPREPROCESS_COLUMNS = [\n    \"sample_id\",\n    \"repeat\",\n    \"stage\",\n    \"time_ms\",\n    \"include_io\",\n    \"input_type\",\n    \"tensor_size\",\n    \"notes\",\n]\n\n\ndef read_existing_preprocessing_rows():\n    if FORCE_RERUN_PREPROCESSING and PREPROCESS_RAW_CSV.exists():\n        PREPROCESS_RAW_CSV.unlink()\n    if not PREPROCESS_RAW_CSV.exists():\n        return pd.DataFrame(columns=PREPROCESS_COLUMNS)\n    frame = pd.read_csv(PREPROCESS_RAW_CSV, dtype={\"sample_id\": str})\n    missing_columns = set(PREPROCESS_COLUMNS) - set(frame.columns)\n    if missing_columns:\n        raise ValueError(\n            f\"Existing preprocessing CSV has incompatible columns: {sorted(missing_columns)}\"\n        )\n    return frame[PREPROCESS_COLUMNS]\n\n\ndef timed_call(function, argument):\n    start_ns = time.perf_counter_ns()\n    result = function(argument)\n    end_ns = time.perf_counter_ns()\n    return (end_ns - start_ns) / 1_000_000.0, result\n\n\ndef run_preprocessing_benchmark():\n    existing = read_existing_preprocessing_rows()\n    done = {\n        (str(row.sample_id), int(row.repeat), str(row.stage))\n        for row in existing.itertuples(index=False)\n    }\n    mode = \"a\" if PREPROCESS_RAW_CSV.exists() else \"w\"\n    needs_header = mode == \"w\"\n\n    with open(PREPROCESS_RAW_CSV, mode, newline=\"\", encoding=\"utf-8\") as handle:\n        writer = csv.DictWriter(handle, fieldnames=PREPROCESS_COLUMNS)\n        if needs_header:\n            writer.writeheader()\n\n        for repeat in range(1, N_REPEATS + 1):\n            LOGGER.info(\"Preprocessing repeat %d/%d\", repeat, N_REPEATS)\n\n            # End-to-end stages. Each observation reopens and reparses .bytes.\n            e2e_stages = [\n                (\n                    \"grayscale_preprocessing\",\n                    \"grayscale\",\n                    grayscale_preprocessing_e2e,\n                    \"raw .bytes to normalized 3-channel grayscale tensor; PNG/ZIP/H2D excluded\",\n                ),\n                (\n                    \"pps_preprocessing\",\n                    \"pps\",\n                    pps_preprocessing_e2e,\n                    \"raw .bytes to normalized R+G+B PPS tensor; PNG/ZIP/H2D excluded\",\n                ),\n            ]\n            for stage, input_type, function, notes in e2e_stages:\n                for row in tqdm(\n                    SAMPLE_MANIFEST.itertuples(index=False),\n                    total=len(SAMPLE_MANIFEST),\n                    desc=f\"{stage} repeat={repeat}\",\n                ):\n                    key = (str(row.sample_id), repeat, stage)\n                    if key in done:\n                        continue\n                    elapsed_ms, result = timed_call(function, row.bytes_path)\n                    del result\n                    output = {\n                        \"sample_id\": str(row.sample_id),\n                        \"repeat\": repeat,\n                        \"stage\": stage,\n                        \"time_ms\": elapsed_ms,\n                        \"include_io\": True,\n                        \"input_type\": input_type,\n                        \"tensor_size\": RESNET50_IMAGE_SIZE,\n                        \"notes\": notes,\n                    }\n                    writer.writerow(output)\n                    handle.flush()\n                    done.add(key)\n\n            # Compute-only stages. Parsing happens before the timers start.\n            for row in tqdm(\n                SAMPLE_MANIFEST.itertuples(index=False),\n                total=len(SAMPLE_MANIFEST),\n                desc=f\"compute-only repeat={repeat}\",\n            ):\n                entropy_key = (str(row.sample_id), repeat, \"entropy_map\")\n                lbp_key = (str(row.sample_id), repeat, \"lbp_map\")\n                if entropy_key in done and lbp_key in done:\n                    continue\n                gray_image = gray_matrix_from_file(row.bytes_path)\n                compute_stages = [\n                    (\n                        \"entropy_map\",\n                        \"entropy\",\n                        compute_entropy_map,\n                        \"compute-only from parsed grayscale matrix; disk radius 8\",\n                        entropy_key,\n                    ),\n                    (\n                        \"lbp_map\",\n                        \"lbp\",\n                        compute_lbp_map,\n                        \"compute-only from parsed grayscale matrix; P=8, R=1, uniform (riu2)\",\n                        lbp_key,\n                    ),\n                ]\n                for stage, input_type, function, notes, key in compute_stages:\n                    if key in done:\n                        continue\n                    elapsed_ms, result = timed_call(function, gray_image)\n                    del result\n                    output = {\n                        \"sample_id\": str(row.sample_id),\n                        \"repeat\": repeat,\n                        \"stage\": stage,\n                        \"time_ms\": elapsed_ms,\n                        \"include_io\": False,\n                        \"input_type\": input_type,\n                        \"tensor_size\": \"\",\n                        \"notes\": notes,\n                    }\n                    writer.writerow(output)\n                    handle.flush()\n                    done.add(key)\n                del gray_image\n            gc.collect()\n\n    frame = pd.read_csv(PREPROCESS_RAW_CSV, dtype={\"sample_id\": str})\n    expected = len(SAMPLE_MANIFEST) * N_REPEATS * 4\n    if len(frame) != expected:\n        raise RuntimeError(f\"Expected {expected} preprocessing rows, found {len(frame)}\")\n    LOGGER.info(\"Saved %d raw preprocessing observations\", len(frame))\n    return frame\n\n\nif RUN_PREPROCESSING:\n    PREPROCESS_RAW = run_preprocessing_benchmark()\nelse:\n    PREPROCESS_RAW = pd.read_csv(PREPROCESS_RAW_CSV, dtype={\"sample_id\": str})\n\nprint(PREPROCESS_RAW.groupby(\"stage\").size())\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"8a3adb2b-0f97-4736-aef6-f25fcc033886","cell_type":"markdown","source":"## Inference timing\n\nEach model is evaluated using already prepared input tensors. Data loading,\nimage transforms, and host-to-device transfer occur before the timed region.\nEvery repeat performs 20 warm-up batches followed by 100 measured batches.\n`torch.cuda.synchronize()` is called immediately before and after every timed\nforward pass.\n\n","metadata":{}},{"id":"3a5197aa-d45d-4035-9153-8ab29f0827e1","cell_type":"code","source":"# ============================================================\n# CELL 6: Inference datasets, models, and checkpoint loading\n# ============================================================\nCHANNEL_TO_INDEX = {\"R\": 0, \"G\": 1, \"B\": 2}\n\n\ndef normalize_channel_config(config):\n    normalized = \"\".join(dict.fromkeys(str(config).upper()))\n    if not normalized or any(channel not in CHANNEL_TO_INDEX for channel in normalized):\n        raise ValueError(f\"Invalid channel configuration: {config}\")\n    return normalized\n\n\ndef mask_pps_channels(image, channel_config):\n    channel_config = normalize_channel_config(channel_config)\n    if channel_config == \"RGB\":\n        return image.convert(\"RGB\")\n    array = np.asarray(image.convert(\"RGB\"), dtype=np.uint8).copy()\n    active = {CHANNEL_TO_INDEX[channel] for channel in channel_config}\n    for index in range(3):\n        if index not in active:\n            array[..., index] = 0\n    return Image.fromarray(array, mode=\"RGB\")\n\n\nclass FixedInferenceDataset(Dataset):\n    def __init__(self, manifest, input_type, image_size, channel_config=\"RGB\"):\n        self.records = manifest.to_dict(\"records\")\n        self.input_type = input_type\n        self.channel_config = normalize_channel_config(channel_config)\n        if input_type == \"grayscale\":\n            self.transform = GRAYSCALE_MODEL_TRANSFORM\n        elif input_type == \"pps\" and image_size == RESNET50_IMAGE_SIZE:\n            self.transform = PPS_RESNET_TRANSFORM\n        elif input_type == \"pps\" and image_size == EB3_IMAGE_SIZE:\n            self.transform = PPS_EB3_TRANSFORM\n        else:\n            raise ValueError(f\"Unsupported input_type/image_size: {input_type}/{image_size}\")\n\n    def __len__(self):\n        return len(self.records)\n\n    def __getitem__(self, index):\n        record = self.records[index]\n        with Image.open(record[\"rgb_path\"]) as source:\n            if self.input_type == \"grayscale\":\n                # PPS R is the original byte-amplitude grayscale image.\n                image = source.convert(\"RGB\").getchannel(\"R\")\n            else:\n                image = mask_pps_channels(source, self.channel_config)\n            tensor = self.transform(image)\n        return tensor, int(record[\"class_id\"]) - 1, str(record[\"sample_id\"])\n\n\ndef build_resnet50(num_classes=9):\n    model = models.resnet50(weights=None)\n    model.fc = nn.Sequential(\n        nn.Dropout(0.3),\n        nn.Linear(2048, num_classes),\n    )\n    return model.to(DEVICE)\n\n\ndef build_efficientnet_b3(num_classes=9):\n    model = models.efficientnet_b3(weights=None)\n    in_features = model.classifier[1].in_features\n    model.classifier = nn.Sequential(\n        nn.Dropout(p=0.3, inplace=True),\n        nn.Linear(in_features, num_classes),\n    )\n    return model.to(DEVICE)\n\n\ndef clean_state_dict(checkpoint):\n    state = checkpoint\n    if isinstance(state, dict) and \"model\" in state:\n        state = state[\"model\"]\n    if isinstance(state, dict) and \"state_dict\" in state:\n        state = state[\"state_dict\"]\n    if not isinstance(state, dict):\n        raise TypeError(\"Checkpoint does not contain a state dictionary\")\n    cleaned = {}\n    for key, value in state.items():\n        if key.startswith(\"module.\"):\n            key = key[len(\"module.\"):]\n        cleaned[key] = value\n    return cleaned\n\n\ndef load_checkpoint(model, checkpoint_path):\n    checkpoint = torch.load(checkpoint_path, map_location=DEVICE)\n    state = clean_state_dict(checkpoint)\n    model.load_state_dict(state, strict=True)\n    model.eval()\n    return model\n\n\ndef amp_context():\n    if AMP_ENABLED:\n        return torch.autocast(device_type=\"cuda\", dtype=torch.float16)\n    return nullcontext()\n\n\ndef cuda_sync():\n    if DEVICE.type == \"cuda\":\n        torch.cuda.synchronize()\n\n\ndef make_inference_loader(input_type, image_size, channel_config, batch_size):\n    dataset = FixedInferenceDataset(\n        SAMPLE_MANIFEST,\n        input_type=input_type,\n        image_size=image_size,\n        channel_config=channel_config,\n    )\n    return DataLoader(\n        dataset,\n        batch_size=batch_size,\n        shuffle=False,\n        num_workers=NUM_WORKERS,\n        pin_memory=bool(PIN_MEMORY and DEVICE.type == \"cuda\"),\n        drop_last=True,\n        persistent_workers=bool(NUM_WORKERS > 0),\n    )\n\n\ndef next_batch(iterator, loader):\n    try:\n        batch = next(iterator)\n    except StopIteration:\n        iterator = iter(loader)\n        batch = next(iterator)\n    return batch, iterator\n\n\nMODEL_SPECS = [\n    {\n        \"model\": \"resnet50\",\n        \"input_type\": \"grayscale\",\n        \"channel_config\": \"R\",\n        \"image_size\": RESNET50_IMAGE_SIZE,\n        \"checkpoint\": RESOLVED_CHECKPOINTS[\"resnet50_grayscale\"],\n        \"builder\": build_resnet50,\n    },\n    {\n        \"model\": \"resnet50\",\n        \"input_type\": \"pps\",\n        \"channel_config\": infer_channel_config_from_checkpoint(\n            RESOLVED_CHECKPOINTS[\"resnet50_pps\"], default=\"RGB\"\n        ),\n        \"image_size\": RESNET50_IMAGE_SIZE,\n        \"checkpoint\": RESOLVED_CHECKPOINTS[\"resnet50_pps\"],\n        \"builder\": build_resnet50,\n    },\n    {\n        \"model\": \"efficientnet_b3\",\n        \"input_type\": \"pps\",\n        \"channel_config\": infer_channel_config_from_checkpoint(\n            RESOLVED_CHECKPOINTS[\"efficientnet_b3_pps\"], default=\"RGB\"\n        ),\n        \"image_size\": EB3_IMAGE_SIZE,\n        \"checkpoint\": RESOLVED_CHECKPOINTS[\"efficientnet_b3_pps\"],\n        \"builder\": build_efficientnet_b3,\n    },\n]\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"df1c772e-a613-4e69-bab0-dd741a01fa21","cell_type":"code","source":"# ============================================================\n# CELL 7: Raw per-batch inference timing, three repeats\n# ============================================================\nINFERENCE_COLUMNS = [\n    \"model\",\n    \"input_type\",\n    \"batch_size\",\n    \"repeat\",\n    \"batch_index\",\n    \"num_samples\",\n    \"time_ms\",\n    \"channel_config\",\n    \"image_size\",\n    \"checkpoint_path\",\n    \"checkpoint_sha256\",\n    \"amp_enabled\",\n    \"h2d_included\",\n    \"notes\",\n]\n\n\ndef read_existing_inference_rows():\n    if FORCE_RERUN_INFERENCE and INFERENCE_RAW_CSV.exists():\n        INFERENCE_RAW_CSV.unlink()\n    if not INFERENCE_RAW_CSV.exists():\n        return pd.DataFrame(columns=INFERENCE_COLUMNS)\n    frame = pd.read_csv(INFERENCE_RAW_CSV)\n    missing_columns = set(INFERENCE_COLUMNS) - set(frame.columns)\n    if missing_columns:\n        raise ValueError(\n            f\"Existing inference CSV has incompatible columns: {sorted(missing_columns)}\"\n        )\n    return frame[INFERENCE_COLUMNS]\n\n\ndef run_one_inference_repeat(model, spec, batch_size, repeat):\n    loader = make_inference_loader(\n        spec[\"input_type\"],\n        spec[\"image_size\"],\n        spec[\"channel_config\"],\n        batch_size,\n    )\n    iterator = iter(loader)\n\n    with torch.no_grad():\n        for _ in range(WARMUP_BATCHES):\n            (images, _, _), iterator = next_batch(iterator, loader)\n            images = images.to(DEVICE, non_blocking=True)\n            with amp_context():\n                _ = model(images)\n            cuda_sync()\n\n        rows = []\n        for batch_index in tqdm(\n            range(1, MEASURE_BATCHES + 1),\n            desc=(\n                f\"{spec['model']} {spec['input_type']} \"\n                f\"bs={batch_size} repeat={repeat}\"\n            ),\n        ):\n            (images, _, _), iterator = next_batch(iterator, loader)\n\n            if INCLUDE_H2D_IN_TIMING:\n                cuda_sync()\n                start_ns = time.perf_counter_ns()\n                images = images.to(DEVICE, non_blocking=True)\n                with amp_context():\n                    _ = model(images)\n                cuda_sync()\n                end_ns = time.perf_counter_ns()\n            else:\n                images = images.to(DEVICE, non_blocking=True)\n                cuda_sync()\n                start_ns = time.perf_counter_ns()\n                with amp_context():\n                    _ = model(images)\n                cuda_sync()\n                end_ns = time.perf_counter_ns()\n\n            rows.append({\n                \"model\": spec[\"model\"],\n                \"input_type\": spec[\"input_type\"],\n                \"batch_size\": int(batch_size),\n                \"repeat\": int(repeat),\n                \"batch_index\": int(batch_index),\n                \"num_samples\": int(images.size(0)),\n                \"time_ms\": (end_ns - start_ns) / 1_000_000.0,\n                \"channel_config\": spec[\"channel_config\"],\n                \"image_size\": int(spec[\"image_size\"]),\n                \"checkpoint_path\": spec[\"checkpoint\"],\n                \"checkpoint_sha256\": sha256_file(spec[\"checkpoint\"]),\n                \"amp_enabled\": AMP_ENABLED,\n                \"h2d_included\": INCLUDE_H2D_IN_TIMING,\n                \"notes\": (\n                    \"forward pass only; CUDA synchronized immediately before/after timing; \"\n                    + (\n                        \"grayscale is PPS R/raw-byte channel repeated to three channels\"\n                        if spec[\"input_type\"] == \"grayscale\"\n                        else \"PPS channels follow channel_config\"\n                    )\n                ),\n            })\n            del images\n    del iterator, loader\n    return rows\n\n\ndef run_inference_benchmark():\n    frame = read_existing_inference_rows()\n\n    for spec in MODEL_SPECS:\n        if not spec[\"checkpoint\"]:\n            raise FileNotFoundError(\n                f\"Checkpoint missing for {spec['model']} / {spec['input_type']}\"\n            )\n        LOGGER.info(\n            \"Loading %s input=%s channel=%s checkpoint=%s\",\n            spec[\"model\"],\n            spec[\"input_type\"],\n            spec[\"channel_config\"],\n            spec[\"checkpoint\"],\n        )\n        model = load_checkpoint(spec[\"builder\"](9), spec[\"checkpoint\"])\n\n        for batch_size in INFERENCE_BATCH_SIZES:\n            for repeat in range(1, N_REPEATS + 1):\n                mask = (\n                    (frame[\"model\"] == spec[\"model\"])\n                    & (frame[\"input_type\"] == spec[\"input_type\"])\n                    & (pd.to_numeric(frame[\"batch_size\"], errors=\"coerce\") == batch_size)\n                    & (pd.to_numeric(frame[\"repeat\"], errors=\"coerce\") == repeat)\n                )\n                existing_count = int(mask.sum())\n                if existing_count == MEASURE_BATCHES:\n                    LOGGER.info(\n                        \"Skipping complete inference group: %s/%s bs=%d repeat=%d\",\n                        spec[\"model\"], spec[\"input_type\"], batch_size, repeat,\n                    )\n                    continue\n                if existing_count:\n                    LOGGER.warning(\n                        \"Discarding partial inference group (%d/%d rows) before rerun\",\n                        existing_count, MEASURE_BATCHES,\n                    )\n                    frame = frame.loc[~mask].copy()\n\n                rows = run_one_inference_repeat(model, spec, batch_size, repeat)\n                frame = pd.concat([frame, pd.DataFrame(rows)], ignore_index=True)\n                frame.to_csv(INFERENCE_RAW_CSV, index=False)\n\n        del model\n        gc.collect()\n        torch.cuda.empty_cache()\n        cuda_sync()\n\n    expected = len(MODEL_SPECS) * len(INFERENCE_BATCH_SIZES) * N_REPEATS * MEASURE_BATCHES\n    if len(frame) != expected:\n        raise RuntimeError(f\"Expected {expected} inference rows, found {len(frame)}\")\n    LOGGER.info(\"Saved %d raw inference observations\", len(frame))\n    return frame\n\n\nif RUN_INFERENCE:\n    INFERENCE_RAW = run_inference_benchmark()\nelse:\n    INFERENCE_RAW = pd.read_csv(INFERENCE_RAW_CSV)\n\nprint(INFERENCE_RAW.groupby([\"model\", \"input_type\", \"batch_size\", \"repeat\"]).size())\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"b104b729-2d9f-4b80-9d80-902d4e38cacd","cell_type":"markdown","source":"## Summary and end-to-end estimate\n\nFor inference, `mean_ms`, `std_ms`, quartiles, and median are calculated from\nper-sample batch latency (`time_ms / num_samples`). Throughput is calculated as\ntotal measured samples divided by total measured forward-pass time.\n\nEnd-to-end medians are explicitly marked as estimates because they sum medians\nfrom independently observed preprocessing and batch-size-1 inference timings.\n\n","metadata":{}},{"id":"de33a825-1b5e-4f3d-a6e8-cb2b3322b185","cell_type":"code","source":"# ============================================================\n# CELL 8: Required summaries and validation\n# ============================================================\nSUMMARY_COLUMNS = [\n    \"stage_or_model\",\n    \"input_type\",\n    \"batch_size\",\n    \"n\",\n    \"mean_ms\",\n    \"std_ms\",\n    \"median_ms\",\n    \"q25_ms\",\n    \"q75_ms\",\n    \"samples_per_second\",\n]\n\n\ndef summary_statistics(values):\n    values = np.asarray(values, dtype=np.float64)\n    return {\n        \"n\": int(len(values)),\n        \"mean_ms\": float(values.mean()),\n        \"std_ms\": float(values.std(ddof=1)) if len(values) > 1 else 0.0,\n        \"median_ms\": float(np.median(values)),\n        \"q25_ms\": float(np.percentile(values, 25)),\n        \"q75_ms\": float(np.percentile(values, 75)),\n    }\n\n\ndef generate_runtime_summary(preprocess_frame, inference_frame):\n    rows = []\n    for stage, group in preprocess_frame.groupby(\"stage\", sort=False):\n        stats = summary_statistics(group[\"time_ms\"].astype(float).to_numpy())\n        total_seconds = group[\"time_ms\"].astype(float).sum() / 1000.0\n        rows.append({\n            \"stage_or_model\": stage,\n            \"input_type\": str(group[\"input_type\"].iloc[0]),\n            \"batch_size\": \"\",\n            **stats,\n            \"samples_per_second\": float(len(group) / total_seconds),\n        })\n\n    inference_frame = inference_frame.copy()\n    inference_frame[\"per_sample_ms\"] = (\n        inference_frame[\"time_ms\"].astype(float)\n        / inference_frame[\"num_samples\"].astype(int)\n    )\n    group_columns = [\"model\", \"input_type\", \"batch_size\"]\n    for (model_name, input_type, batch_size), group in inference_frame.groupby(\n        group_columns, sort=False\n    ):\n        stats = summary_statistics(group[\"per_sample_ms\"].to_numpy())\n        total_seconds = group[\"time_ms\"].astype(float).sum() / 1000.0\n        rows.append({\n            \"stage_or_model\": model_name,\n            \"input_type\": input_type,\n            \"batch_size\": int(batch_size),\n            **stats,\n            \"samples_per_second\": float(group[\"num_samples\"].astype(int).sum() / total_seconds),\n        })\n\n    summary = pd.DataFrame(rows, columns=SUMMARY_COLUMNS)\n    summary.to_csv(SUMMARY_CSV, index=False)\n    return summary\n\n\ndef generate_end_to_end_summary(summary):\n    def median_for(stage_or_model, input_type, batch_size=None):\n        mask = (\n            (summary[\"stage_or_model\"] == stage_or_model)\n            & (summary[\"input_type\"] == input_type)\n        )\n        if batch_size is not None:\n            mask &= pd.to_numeric(summary[\"batch_size\"], errors=\"coerce\") == batch_size\n        matched = summary.loc[mask, \"median_ms\"]\n        if len(matched) != 1:\n            raise ValueError(\n                f\"Expected one summary row for {stage_or_model}/{input_type}/{batch_size}, \"\n                f\"found {len(matched)}\"\n            )\n        return float(matched.iloc[0])\n\n    pipelines = []\n    for pipeline, preprocessing_stage, input_type in [\n        (\"grayscale_resnet50\", \"grayscale_preprocessing\", \"grayscale\"),\n        (\"pps_resnet50\", \"pps_preprocessing\", \"pps\"),\n    ]:\n        preprocessing_median = median_for(preprocessing_stage, input_type)\n        inference_median = median_for(\"resnet50\", input_type, batch_size=1)\n        pipelines.append({\n            \"pipeline\": pipeline,\n            \"preprocessing_median_ms\": preprocessing_median,\n            \"inference_median_ms\": inference_median,\n            \"total_median_ms\": preprocessing_median + inference_median,\n            \"is_estimate\": True,\n            \"notes\": (\n                \"Estimated by summing medians from independent preprocessing and \"\n                \"batch-size-1 inference observations\"\n            ),\n        })\n    frame = pd.DataFrame(pipelines)\n    frame.to_csv(END_TO_END_CSV, index=False)\n    return frame\n\n\ndef validate_audit(preprocess_frame, inference_frame):\n    errors = []\n    expected_per_stage = len(SAMPLE_MANIFEST) * N_REPEATS\n    for stage in (\n        \"grayscale_preprocessing\",\n        \"entropy_map\",\n        \"lbp_map\",\n        \"pps_preprocessing\",\n    ):\n        count = int((preprocess_frame[\"stage\"] == stage).sum())\n        if count != expected_per_stage:\n            errors.append(f\"{stage}: expected {expected_per_stage}, found {count}\")\n\n    expected_per_inference_group = N_REPEATS * MEASURE_BATCHES\n    required_groups = [\n        (\"resnet50\", \"grayscale\", 1),\n        (\"resnet50\", \"grayscale\", 32),\n        (\"resnet50\", \"pps\", 1),\n        (\"resnet50\", \"pps\", 32),\n        (\"efficientnet_b3\", \"pps\", 1),\n        (\"efficientnet_b3\", \"pps\", 32),\n    ]\n    for model_name, input_type, batch_size in required_groups:\n        mask = (\n            (inference_frame[\"model\"] == model_name)\n            & (inference_frame[\"input_type\"] == input_type)\n            & (pd.to_numeric(inference_frame[\"batch_size\"], errors=\"coerce\") == batch_size)\n        )\n        count = int(mask.sum())\n        if count != expected_per_inference_group:\n            errors.append(\n                f\"{model_name}/{input_type}/bs={batch_size}: \"\n                f\"expected {expected_per_inference_group}, found {count}\"\n            )\n    if errors:\n        raise RuntimeError(\"Audit completeness check failed:\\n- \" + \"\\n- \".join(errors))\n\n\nvalidate_audit(PREPROCESS_RAW, INFERENCE_RAW)\nRUNTIME_SUMMARY = generate_runtime_summary(PREPROCESS_RAW, INFERENCE_RAW)\nEND_TO_END_SUMMARY = generate_end_to_end_summary(RUNTIME_SUMMARY)\n\nprint(RUNTIME_SUMMARY.to_string(index=False))\nprint()\nprint(END_TO_END_SUMMARY.to_string(index=False))\n\n","metadata":{},"outputs":[],"execution_count":null},{"id":"64a75519-d1b8-4386-a2fd-0869627b3926","cell_type":"code","source":"# ============================================================\n# CELL 9: README and final deliverable check\n# ============================================================\ndef write_readme():\n    checkpoint_lines = \"\\n\".join(\n        f\"- `{name}`: `{path}`\" for name, path in RESOLVED_CHECKPOINTS.items()\n    )\n    content = f\"\"\"# PIXEL camera-ready runtime audit\n\nThis directory contains Task 1 of the camera-ready experiment audit. No model\nwas trained or fine-tuned. The byte parsing, dynamic width, entropy radius,\nLBP variant, model input sizes, evaluation mode, and checkpoint architectures\nmatch the original experiment code.\n\n## How to run\n\n1. Use a Kaggle GPU notebook with the BIG-2015 competition labels, raw `.bytes`\n   dataset, pre-rendered PPS dataset, fixed split (when available), and all three\n   checkpoints attached.\n2. Set checkpoint overrides in `runtime_benchmark.ipynb` when auto-discovery does\n   not find the exact files.\n3. Run the notebook from top to bottom. The preflight cell fails before timing if\n   a required input is missing.\n4. Re-running resumes completed observations. Set the two `FORCE_RERUN_*` flags\n   only when a complete fresh audit is intended.\n\n## Required libraries\n\nPython, PyTorch, torchvision, NumPy, pandas, Pillow, psutil, scikit-image, tqdm.\nExact versions are recorded in `environment/runtime_environment.txt`.\n\n## Inputs\n\n- Raw bytes: `{RAW_BYTES_DIR}`\n- PPS RGB images: `{MICROSOFT_RGB_DIR}`\n- Labels: `{MICROSOFT_LABELS_CSV}`\n- Split source: `{SPLIT_SOURCE}`\n\nCheckpoints:\n\n{checkpoint_lines}\n\n## Protocol\n\n- Dataset: Microsoft BIG-2015.\n- Samples: 500 fixed test samples saved in `runtime/runtime_sample_ids.csv`.\n- Preprocessing: 3 repeats per sample and stage.\n- Inference: batch sizes 1 and 32; 3 repeats; 20 warm-up plus 100 measured\n  batches per repeat.\n- CUDA is synchronized immediately before and after every measured forward pass.\n- Host-to-device transfer included in inference timing: `{INCLUDE_H2D_IN_TIMING}`.\n- PNG writing and ZIP writing are excluded.\n- Grayscale inference uses the PPS R/raw-byte channel, resized to 224 x 224,\n  repeated to three channels, and normalized with mean = std = 0.5, matching the\n  original grayscale baseline input format.\n- Cache policy: {CACHE_POLICY_NOTE}.\n\n## Outputs\n\n- `environment/runtime_environment.txt`: hardware, software, threads, and storage.\n- `environment/runtime_config.json`: exact audit configuration and checkpoint hashes.\n- `runtime/runtime_sample_ids.csv`: fixed 500-sample list.\n- `runtime/runtime_sample_manifest.csv`: IDs, labels, and resolved input paths.\n- `runtime/runtime_split_snapshot.json`: exact split used by the audit.\n- `runtime/preprocessing_runtime_raw.csv`: one row per sample, repeat, and stage.\n- `runtime/inference_runtime_raw.csv`: one row per measured batch.\n- `runtime/runtime_summary.csv`: n, mean, standard deviation, median, quartiles,\n  and throughput.\n- `runtime/end_to_end_runtime.csv`: estimated grayscale and PPS ResNet-50 totals.\n- `logs/runtime.log`: execution and resume log.\n\n## Expected duration\n\nPreprocessing is CPU-bound and may take several hours for 500 samples and three\nrepeats. Inference is usually much shorter on an NVIDIA T4. Actual duration\ndepends on CPU allocation and the Kaggle storage/cache state.\n\n## Differences from the previous runtime notebook\n\nThe previous notebook saved only aggregate values. This revision adds fixed test\nIDs, three repeats, raw observations, environment capture, grayscale-versus-PPS\nResNet-50 inference, checkpoint hashes, resume support, quartiles, and explicit\nend-to-end estimates. No entropy, LBP, byte parsing, resize, or model algorithm\nwas changed to improve timing.\n\n## Scope\n\nThis runtime notebook covers Task 1 only. Channel-occlusion aggregation and XAI\ncheckpoint evaluation from Tasks 2 and 3 require their own source data and code.\n\"\"\"\n    README_PATH.write_text(content, encoding=\"utf-8\")\n\n\nwrite_readme()\n\nrequired_files = [\n    README_PATH,\n    ENVIRONMENT_TXT,\n    CONFIG_JSON,\n    SAMPLE_IDS_CSV,\n    SAMPLE_MANIFEST_CSV,\n    SPLIT_SNAPSHOT_JSON,\n    PREPROCESS_RAW_CSV,\n    INFERENCE_RAW_CSV,\n    SUMMARY_CSV,\n    END_TO_END_CSV,\n    RUNTIME_LOG,\n]\nmissing_files = [str(path) for path in required_files if not path.exists()]\nif missing_files:\n    raise RuntimeError(\"Missing audit outputs:\\n- \" + \"\\n- \".join(missing_files))\n\nLOGGER.info(\"Runtime audit complete: %s\", AUDIT_ROOT)\nprint(\"Runtime audit complete. Files:\")\nfor path in required_files:\n    print(f\"- {path}\")\n","metadata":{},"outputs":[],"execution_count":null}]}