{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# =============================================================================\n# STEP 13A — DETERMINISTIC XAI CASE SELECTION\n#\n# Selects representative final-test cases for later Grad-CAM++ analysis.\n#\n# Selection plan:\n#   Correct cases:\n#       1 case from each of the five DR grades\n#\n#   Error cases:\n#       Moderate -> Mild\n#       Severe -> Moderate\n#       Proliferative DR -> Moderate\n#       Proliferative DR -> Mild\n#       Moderate -> Severe\n#\n# Selection rule:\n#   - Median predicted-confidence case within each predefined stratum\n#   - Lexicographic midpoint fallback if probabilities are unavailable\n#\n# This stage:\n#   - performs no training\n#   - performs no model inference\n#   - does not decode image pixels\n#   - does not regenerate validation/test predictions\n#   - does not change the frozen model\n#\n# Run this cell only. Do not use Run All.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\nfrom io import BytesIO\n\nimport hashlib\nimport json\nimport os\nimport re\nimport zipfile\n\nimport numpy as np\nimport pandas as pd\n\n\n# =============================================================================\n# 1. Exact paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nSTEP10D_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_10d_final_test_state.json\"\n)\n\nSTEP12B_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_12b_locked_prediction_error_analysis_state.json\"\n)\n\nSTEP12B_SNAPSHOT_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_12b_locked_prediction_error_analysis\"\n    / \"step_12b_locked_prediction_snapshot.csv\"\n)\n\nFINAL_MODEL_CHECKPOINT_PATH = (\n    PROJECT\n    / \"06_checkpoints\"\n    / \"step_10b_registered_baseline_final_training\"\n    / \"step_10b_final_training_checkpoint.pt\"\n)\n\nTEST_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_10d_final_test_backup.zip\"\n)\n\nTEST_PREDICTION_MEMBER = (\n    \"07_predictions/step_10d_final_test/\"\n    \"step_10d_final_test_predictions.csv\"\n)\n\nIMAGE_ROOT = Path(\n    \"/kaggle/input/competitions/\"\n    \"aptos2019-blindness-detection/train_images\"\n)\n\n\nOUTPUT_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13a_xai_case_selection\"\n)\n\nOUTPUT_EVIDENCE_DIR = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13a_xai_case_selection\"\n)\n\nfor directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n]:\n    directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\nSELECTION_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13a_selected_xai_cases.csv\"\n)\n\nCANDIDATE_SUMMARY_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13a_candidate_pool_summary.csv\"\n)\n\nPROBABILITY_QA_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13a_probability_column_qa.csv\"\n)\n\nSOURCE_VERIFICATION_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_prediction_source_verification.csv\"\n)\n\nPUBLICATION_SELECTION_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_publication_xai_case_selection.csv\"\n)\n\nSELECTION_PLAN_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_xai_selection_plan.json\"\n)\n\nMANUSCRIPT_NOTE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_xai_case_selection_note.txt\"\n)\n\nSUMMARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_xai_case_selection_summary.json\"\n)\n\nMANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13a_xai_case_selection_manifest.csv\"\n)\n\nSTATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13a_xai_case_selection_state.json\"\n)\n\nBACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_13a_xai_case_selection_backup.zip\"\n)\n\n\n# =============================================================================\n# 2. Locked expectations\n# =============================================================================\n\nCLASS_NAMES = [\n    \"No_DR\",\n    \"Mild\",\n    \"Moderate\",\n    \"Severe\",\n    \"Proliferative_DR\",\n]\n\nEXPECTED_TEST_ROWS = 522\nEXPECTED_EXACT_CORRECT = 402\nEXPECTED_EXACT_ERRORS = 120\n\nEXPECTED_CLASS_COUNTS = {\n    0: 269,\n    1: 51,\n    2: 138,\n    3: 25,\n    4: 39,\n}\n\nEXPECTED_TRANSITION_COUNTS = {\n    (2, 1): 51,\n    (3, 2): 11,\n    (4, 2): 12,\n    (4, 1): 5,\n    (2, 3): 10,\n}\n\n\nSELECTION_STRATA = [\n    {\n        \"selection_order\": 1,\n        \"case_role\": \"correct\",\n        \"case_category\": \"Correct No DR\",\n        \"true_grade\": 0,\n        \"predicted_grade\": 0,\n        \"clinical_focus\": (\n            \"Representative correctly classified No-DR case\"\n        ),\n    },\n    {\n        \"selection_order\": 2,\n        \"case_role\": \"correct\",\n        \"case_category\": \"Correct Mild DR\",\n        \"true_grade\": 1,\n        \"predicted_grade\": 1,\n        \"clinical_focus\": (\n            \"Representative correctly classified Mild DR case\"\n        ),\n    },\n    {\n        \"selection_order\": 3,\n        \"case_role\": \"correct\",\n        \"case_category\": \"Correct Moderate DR\",\n        \"true_grade\": 2,\n        \"predicted_grade\": 2,\n        \"clinical_focus\": (\n            \"Representative correctly classified Moderate DR case\"\n        ),\n    },\n    {\n        \"selection_order\": 4,\n        \"case_role\": \"correct\",\n        \"case_category\": \"Correct Severe DR\",\n        \"true_grade\": 3,\n        \"predicted_grade\": 3,\n        \"clinical_focus\": (\n            \"Representative correctly classified Severe DR case\"\n        ),\n    },\n    {\n        \"selection_order\": 5,\n        \"case_role\": \"correct\",\n        \"case_category\": \"Correct Proliferative DR\",\n        \"true_grade\": 4,\n        \"predicted_grade\": 4,\n        \"clinical_focus\": (\n            \"Representative correctly classified Proliferative DR case\"\n        ),\n    },\n    {\n        \"selection_order\": 6,\n        \"case_role\": \"error\",\n        \"case_category\": \"Moderate to Mild undergrading\",\n        \"true_grade\": 2,\n        \"predicted_grade\": 1,\n        \"clinical_focus\": (\n            \"Most frequent final-test confusion transition\"\n        ),\n    },\n    {\n        \"selection_order\": 7,\n        \"case_role\": \"error\",\n        \"case_category\": \"Severe to Moderate undergrading\",\n        \"true_grade\": 3,\n        \"predicted_grade\": 2,\n        \"clinical_focus\": (\n            \"Adjacent undergrading of advanced disease\"\n        ),\n    },\n    {\n        \"selection_order\": 8,\n        \"case_role\": \"error\",\n        \"case_category\": \"PDR to Moderate undergrading\",\n        \"true_grade\": 4,\n        \"predicted_grade\": 2,\n        \"clinical_focus\": (\n            \"Two-grade undergrading of proliferative disease\"\n        ),\n    },\n    {\n        \"selection_order\": 9,\n        \"case_role\": \"error\",\n        \"case_category\": \"PDR to Mild large undergrading\",\n        \"true_grade\": 4,\n        \"predicted_grade\": 1,\n        \"clinical_focus\": (\n            \"Three-grade clinically important undergrading\"\n        ),\n    },\n    {\n        \"selection_order\": 10,\n        \"case_role\": \"error\",\n        \"case_category\": \"Moderate to Severe overgrading\",\n        \"true_grade\": 2,\n        \"predicted_grade\": 3,\n        \"clinical_focus\": (\n            \"Representative adjacent overgrading case\"\n        ),\n    },\n]\n\n\nTRUE_COLUMN_CANDIDATES = [\n    \"y_true\",\n    \"true_label\",\n    \"true_grade\",\n    \"actual_label\",\n    \"actual_grade\",\n    \"ground_truth\",\n    \"target\",\n    \"diagnosis\",\n    \"label\",\n]\n\nPRED_COLUMN_CANDIDATES = [\n    \"y_pred\",\n    \"pred_label\",\n    \"predicted_label\",\n    \"predicted_grade\",\n    \"predicted_class\",\n    \"prediction\",\n    \"final_prediction\",\n    \"argmax_prediction\",\n    \"pred\",\n]\n\nID_COLUMN_CANDIDATES = [\n    \"id_code\",\n    \"image_id\",\n    \"sample_id\",\n    \"filename\",\n    \"file_name\",\n    \"id\",\n]\n\nGROUP_COLUMN_CANDIDATES = [\n    \"leakage_group_id\",\n    \"group_id\",\n    \"duplicate_group_id\",\n]\n\nCONFIDENCE_COLUMN_CANDIDATES = [\n    \"predicted_confidence\",\n    \"prediction_confidence\",\n    \"max_probability\",\n    \"max_prob\",\n    \"confidence\",\n    \"predicted_probability\",\n]\n\n\n# =============================================================================\n# 3. Utility functions\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef read_json(path):\n\n    with open(\n        path,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        return json.load(\n            file\n        )\n\n\ndef atomic_json_save(\n    record,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        json.dump(\n            record,\n            file,\n            indent=2,\n            ensure_ascii=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_csv_save(\n    dataframe,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    dataframe.to_csv(\n        temporary_path,\n        index=False,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_text_save(\n    text,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        file.write(\n            text\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef sha256_bytes(data):\n\n    return hashlib.sha256(\n        data\n    ).hexdigest()\n\n\ndef sha256_file(path):\n\n    digest = hashlib.sha256()\n\n    with open(\n        path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef dataframe_fingerprint(\n    dataframe,\n):\n\n    canonical = dataframe.copy()\n\n    canonical = canonical.sort_values(\n        list(\n            canonical.columns\n        )\n    ).reset_index(\n        drop=True\n    )\n\n    csv_bytes = canonical.to_csv(\n        index=False,\n        float_format=\"%.10f\",\n        lineterminator=\"\\n\",\n    ).encode(\n        \"utf-8\"\n    )\n\n    return sha256_bytes(\n        csv_bytes\n    )\n\n\ndef normalized_column_map(\n    dataframe,\n):\n\n    return {\n        str(\n            column\n        ).strip().lower(): column\n\n        for column\n        in dataframe.columns\n    }\n\n\ndef find_column(\n    dataframe,\n    candidates,\n):\n\n    column_map = normalized_column_map(\n        dataframe\n    )\n\n    for candidate in candidates:\n\n        if candidate in column_map:\n\n            return column_map[\n                candidate\n            ]\n\n    return None\n\n\ndef discover_probability_columns(\n    dataframe,\n):\n\n    discovered = {}\n\n    normalized_columns = {\n        str(\n            column\n        ).strip().lower(): column\n\n        for column\n        in dataframe.columns\n    }\n\n\n    explicit_candidates = {\n        0: [\n            \"prob_0\",\n            \"prob0\",\n            \"probability_0\",\n            \"probability0\",\n            \"p_0\",\n            \"p0\",\n            \"prob_no_dr\",\n            \"prob_nodr\",\n            \"no_dr_probability\",\n        ],\n        1: [\n            \"prob_1\",\n            \"prob1\",\n            \"probability_1\",\n            \"probability1\",\n            \"p_1\",\n            \"p1\",\n            \"prob_mild\",\n            \"mild_probability\",\n        ],\n        2: [\n            \"prob_2\",\n            \"prob2\",\n            \"probability_2\",\n            \"probability2\",\n            \"p_2\",\n            \"p2\",\n            \"prob_moderate\",\n            \"moderate_probability\",\n        ],\n        3: [\n            \"prob_3\",\n            \"prob3\",\n            \"probability_3\",\n            \"probability3\",\n            \"p_3\",\n            \"p3\",\n            \"prob_severe\",\n            \"severe_probability\",\n        ],\n        4: [\n            \"prob_4\",\n            \"prob4\",\n            \"probability_4\",\n            \"probability4\",\n            \"p_4\",\n            \"p4\",\n            \"prob_proliferative_dr\",\n            \"prob_pdr\",\n            \"pdr_probability\",\n        ],\n    }\n\n\n    for class_index, candidates in explicit_candidates.items():\n\n        for candidate in candidates:\n\n            if candidate in normalized_columns:\n\n                discovered[\n                    class_index\n                ] = normalized_columns[\n                    candidate\n                ]\n\n                break\n\n\n    if len(\n        discovered\n    ) == 5:\n\n        return discovered\n\n\n    regex_discovered = {}\n\n\n    for column in dataframe.columns:\n\n        column_text = str(\n            column\n        ).strip().lower()\n\n\n        if not any(\n            token in column_text\n            for token in [\n                \"prob\",\n                \"softmax\",\n                \"class_p\",\n            ]\n        ):\n\n            continue\n\n\n        digit_matches = re.findall(\n            r\"(?<!\\d)([0-4])(?!\\d)\",\n            column_text,\n        )\n\n\n        if len(\n            digit_matches\n        ) == 1:\n\n            class_index = int(\n                digit_matches[\n                    0\n                ]\n            )\n\n            regex_discovered[\n                class_index\n            ] = column\n\n\n    if len(\n        regex_discovered\n    ) == 5:\n\n        return regex_discovered\n\n\n    return {}\n\n\ndef resolve_image_path(\n    sample_id,\n):\n\n    sample_text = str(\n        sample_id\n    ).strip()\n\n    original_name = Path(\n        sample_text\n    ).name\n\n    original_suffix = Path(\n        original_name\n    ).suffix.lower()\n\n\n    candidates = []\n\n\n    if original_suffix in {\n        \".png\",\n        \".jpg\",\n        \".jpeg\",\n    }:\n\n        candidates.append(\n            IMAGE_ROOT\n            / original_name\n        )\n\n\n    stem = Path(\n        original_name\n    ).stem\n\n\n    for extension in [\n        \".png\",\n        \".jpg\",\n        \".jpeg\",\n    ]:\n\n        candidates.append(\n            IMAGE_ROOT\n            / f\"{stem}{extension}\"\n        )\n\n\n    for candidate in candidates:\n\n        if candidate.exists():\n\n            return candidate\n\n\n    raise FileNotFoundError(\n        \"Selected XAI image was not found for sample: \"\n        f\"{sample_id}\"\n    )\n\n\ndef select_representative_case(\n    candidate_pool,\n    confidence_available,\n):\n\n    pool = candidate_pool.copy()\n\n\n    if pool.empty:\n\n        raise RuntimeError(\n            \"A required XAI selection stratum has no candidates.\"\n        )\n\n\n    pool[\n        \"sample_id\"\n    ] = pool[\n        \"sample_id\"\n    ].astype(\n        str\n    )\n\n\n    if (\n        confidence_available\n        and\n        pool[\n            \"predicted_confidence\"\n        ].notna().all()\n    ):\n\n        group_median = float(\n            pool[\n                \"predicted_confidence\"\n            ].median()\n        )\n\n\n        pool[\n            \"distance_from_group_median_confidence\"\n        ] = np.abs(\n            pool[\n                \"predicted_confidence\"\n            ]\n            -\n            group_median\n        )\n\n\n        pool = pool.sort_values(\n            [\n                \"distance_from_group_median_confidence\",\n                \"sample_id\",\n            ],\n            ascending=[\n                True,\n                True,\n            ],\n        ).reset_index(\n            drop=True\n        )\n\n\n        selected = pool.iloc[\n            0\n        ].copy()\n\n\n        selection_method = (\n            \"nearest_to_median_predicted_confidence\"\n        )\n\n\n    else:\n\n        pool = pool.sort_values(\n            \"sample_id\"\n        ).reset_index(\n            drop=True\n        )\n\n\n        selected_index = int(\n            len(\n                pool\n            )\n            //\n            2\n        )\n\n\n        selected = pool.iloc[\n            selected_index\n        ].copy()\n\n\n        group_median = None\n\n\n        selected[\n            \"distance_from_group_median_confidence\"\n        ] = np.nan\n\n\n        selection_method = (\n            \"lexicographic_midpoint_fallback\"\n        )\n\n\n    return (\n        selected,\n        group_median,\n        selection_method,\n    )\n\n\ndef create_verified_zip(\n    zip_path,\n    source_files,\n):\n\n    temporary_path = zip_path.with_suffix(\n        zip_path.suffix + \".tmp\"\n    )\n\n\n    if temporary_path.exists():\n\n        temporary_path.unlink()\n\n\n    unique_files = []\n\n\n    for source_file in source_files:\n\n        if (\n            source_file.exists()\n            and\n            source_file not in unique_files\n        ):\n\n            unique_files.append(\n                source_file\n            )\n\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"w\",\n        compression=zipfile.ZIP_DEFLATED,\n        compresslevel=6,\n    ) as archive:\n\n        for source_file in unique_files:\n\n            archive.write(\n                source_file,\n                arcname=str(\n                    source_file.relative_to(\n                        PROJECT\n                    )\n                ),\n            )\n\n\n    with zipfile.ZipFile(\n        temporary_path,\n        \"r\",\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            f\"Step 13A backup damaged at: {damaged_member}\"\n        )\n\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            \"Duplicate files detected in Step 13A backup.\"\n        )\n\n\n    os.replace(\n        temporary_path,\n        zip_path,\n    )\n\n\n    return members\n\n\n# =============================================================================\n# 4. Completed-stage protection\n# =============================================================================\n\nif STATE_PATH.exists():\n\n    existing_state = read_json(\n        STATE_PATH\n    )\n\n\n    if existing_state.get(\n        \"status\"\n    ) == \"complete\":\n\n        raise RuntimeError(\n            \"Step 13A is already complete. Do not rerun it.\"\n        )\n\n\n# =============================================================================\n# 5. Prerequisite checks\n# =============================================================================\n\nfor required_path in [\n    STEP10D_STATE_PATH,\n    STEP12B_STATE_PATH,\n    STEP12B_SNAPSHOT_PATH,\n    FINAL_MODEL_CHECKPOINT_PATH,\n    TEST_BACKUP_PATH,\n]:\n\n    if not required_path.exists():\n\n        raise FileNotFoundError(\n            f\"Required Step 13A evidence missing: {required_path}\"\n        )\n\n\nstep10d_state = read_json(\n    STEP10D_STATE_PATH\n)\n\nstep12b_state = read_json(\n    STEP12B_STATE_PATH\n)\n\n\nif step10d_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 10D final test is incomplete.\"\n    )\n\n\nif int(\n    step10d_state.get(\n        \"test_evaluation_count\",\n        -1,\n    )\n) != 1:\n\n    raise RuntimeError(\n        \"Final-test evaluation count is not exactly one.\"\n    )\n\n\nif step10d_state.get(\n    \"another_test_evaluation_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Final-test evaluation is not formally closed.\"\n    )\n\n\nif step10d_state.get(\n    \"model_change_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Frozen-model protection is not preserved.\"\n    )\n\n\nif (\n    step12b_state.get(\n        \"status\"\n    )\n    !=\n    \"complete\"\n    or\n    step12b_state.get(\n        \"locked_metrics_reproduced_from_saved_predictions\"\n    )\n    is not True\n):\n\n    raise RuntimeError(\n        \"Step 12B locked prediction audit is incomplete.\"\n    )\n\n\n# =============================================================================\n# 6. Load exact preserved final-test prediction table\n# =============================================================================\n\nwith zipfile.ZipFile(\n    TEST_BACKUP_PATH,\n    \"r\",\n) as archive:\n\n    if TEST_PREDICTION_MEMBER not in archive.namelist():\n\n        raise FileNotFoundError(\n            \"Expected final-test prediction member is missing \"\n            \"from the Step 10D backup.\"\n        )\n\n\n    prediction_bytes = archive.read(\n        TEST_PREDICTION_MEMBER\n    )\n\n\ntest_predictions_raw = pd.read_csv(\n    BytesIO(\n        prediction_bytes\n    )\n)\n\n\nprediction_source_sha256 = sha256_bytes(\n    prediction_bytes\n)\n\n\n# =============================================================================\n# 7. Resolve prediction columns\n# =============================================================================\n\ntrue_column = find_column(\n    test_predictions_raw,\n    TRUE_COLUMN_CANDIDATES,\n)\n\npred_column = find_column(\n    test_predictions_raw,\n    PRED_COLUMN_CANDIDATES,\n)\n\nid_column = find_column(\n    test_predictions_raw,\n    ID_COLUMN_CANDIDATES,\n)\n\ngroup_column = find_column(\n    test_predictions_raw,\n    GROUP_COLUMN_CANDIDATES,\n)\n\nconfidence_column = find_column(\n    test_predictions_raw,\n    CONFIDENCE_COLUMN_CANDIDATES,\n)\n\nprobability_columns = discover_probability_columns(\n    test_predictions_raw\n)\n\n\nif true_column is None:\n\n    raise RuntimeError(\n        \"True-label column was not found.\"\n    )\n\n\nif pred_column is None:\n\n    raise RuntimeError(\n        \"Prediction column was not found.\"\n    )\n\n\nif id_column is None:\n\n    raise RuntimeError(\n        \"Image/sample identifier column was not found.\"\n    )\n\n\nif len(\n    test_predictions_raw\n) != EXPECTED_TEST_ROWS:\n\n    raise RuntimeError(\n        \"Final-test prediction row count mismatch.\"\n    )\n\n\n# =============================================================================\n# 8. Normalize final-test prediction evidence\n# =============================================================================\n\ntest_df = pd.DataFrame({\n    \"sample_id\": (\n        test_predictions_raw[\n            id_column\n        ].astype(\n            str\n        )\n    ),\n\n    \"y_true\": pd.to_numeric(\n        test_predictions_raw[\n            true_column\n        ],\n        errors=\"raise\",\n    ).astype(\n        int\n    ),\n\n    \"y_pred\": pd.to_numeric(\n        test_predictions_raw[\n            pred_column\n        ],\n        errors=\"raise\",\n    ).astype(\n        int\n    ),\n})\n\n\nif group_column is not None:\n\n    test_df[\n        \"leakage_group_id\"\n    ] = test_predictions_raw[\n        group_column\n    ].astype(\n        str\n    )\n\nelse:\n\n    test_df[\n        \"leakage_group_id\"\n    ] = test_df[\n        \"sample_id\"\n    ]\n\n\nif test_df[\n    \"sample_id\"\n].duplicated().any():\n\n    raise RuntimeError(\n        \"Duplicate final-test sample IDs detected.\"\n    )\n\n\nif not set(\n    test_df[\n        \"y_true\"\n    ].unique()\n).issubset(\n    set(\n        range(\n            5\n        )\n    )\n):\n\n    raise RuntimeError(\n        \"Unexpected true-grade values detected.\"\n    )\n\n\nif not set(\n    test_df[\n        \"y_pred\"\n    ].unique()\n).issubset(\n    set(\n        range(\n            5\n        )\n    )\n):\n\n    raise RuntimeError(\n        \"Unexpected predicted-grade values detected.\"\n    )\n\n\n# =============================================================================\n# 9. Recover probabilities or predicted confidence\n# =============================================================================\n\nfull_probability_vector_available = bool(\n    len(\n        probability_columns\n    )\n    ==\n    5\n)\n\n\nprobability_qa_records = []\n\n\nif full_probability_vector_available:\n\n    probability_matrix = np.column_stack([\n        pd.to_numeric(\n            test_predictions_raw[\n                probability_columns[\n                    class_index\n                ]\n            ],\n            errors=\"raise\",\n        ).to_numpy(\n            dtype=np.float64\n        )\n\n        for class_index\n        in range(\n            5\n        )\n    ])\n\n\n    if not np.isfinite(\n        probability_matrix\n    ).all():\n\n        raise RuntimeError(\n            \"Non-finite probability values detected.\"\n        )\n\n\n    probability_row_sums = probability_matrix.sum(\n        axis=1\n    )\n\n\n    probability_argmax = np.argmax(\n        probability_matrix,\n        axis=1,\n    )\n\n\n    probability_argmax_matches = (\n        probability_argmax\n        ==\n        test_df[\n            \"y_pred\"\n        ].to_numpy(\n            dtype=np.int64\n        )\n    )\n\n\n    if not probability_argmax_matches.all():\n\n        raise RuntimeError(\n            \"Saved probability argmax does not reproduce \"\n            \"the locked final prediction labels.\"\n        )\n\n\n    test_df[\n        \"predicted_confidence\"\n    ] = probability_matrix[\n        np.arange(\n            len(\n                test_df\n            )\n        ),\n        test_df[\n            \"y_pred\"\n        ].to_numpy(\n            dtype=np.int64\n        ),\n    ]\n\n\n    for class_index in range(\n        5\n    ):\n\n        test_df[\n            f\"probability_grade_{class_index}\"\n        ] = probability_matrix[\n            :,\n            class_index,\n        ]\n\n\n    probability_qa_records.append({\n        \"probability_source\": (\n            \"full_saved_probability_vector\"\n        ),\n\n        \"full_probability_vector_available\": (\n            True\n        ),\n\n        \"probability_columns\": (\n            \"|\".join(\n                str(\n                    probability_columns[\n                        class_index\n                    ]\n                )\n                for class_index\n                in range(\n                    5\n                )\n            )\n        ),\n\n        \"maximum_row_sum_deviation_from_one\": float(\n            np.max(\n                np.abs(\n                    probability_row_sums\n                    -\n                    1.0\n                )\n            )\n        ),\n\n        \"argmax_prediction_disagreements\": int(\n            (\n                ~probability_argmax_matches\n            ).sum()\n        ),\n\n        \"confidence_available_for_selection\": (\n            True\n        ),\n    })\n\n\nelif confidence_column is not None:\n\n    test_df[\n        \"predicted_confidence\"\n    ] = pd.to_numeric(\n        test_predictions_raw[\n            confidence_column\n        ],\n        errors=\"raise\",\n    ).astype(\n        float\n    )\n\n\n    if not np.isfinite(\n        test_df[\n            \"predicted_confidence\"\n        ].to_numpy()\n    ).all():\n\n        raise RuntimeError(\n            \"Non-finite confidence values detected.\"\n        )\n\n\n    probability_qa_records.append({\n        \"probability_source\": (\n            \"saved_predicted_confidence_column\"\n        ),\n\n        \"full_probability_vector_available\": (\n            False\n        ),\n\n        \"probability_columns\": (\n            str(\n                confidence_column\n            )\n        ),\n\n        \"maximum_row_sum_deviation_from_one\": (\n            np.nan\n        ),\n\n        \"argmax_prediction_disagreements\": (\n            np.nan\n        ),\n\n        \"confidence_available_for_selection\": (\n            True\n        ),\n    })\n\n\nelse:\n\n    test_df[\n        \"predicted_confidence\"\n    ] = np.nan\n\n\n    probability_qa_records.append({\n        \"probability_source\": (\n            \"no_probability_or_confidence_column\"\n        ),\n\n        \"full_probability_vector_available\": (\n            False\n        ),\n\n        \"probability_columns\": (\n            \"\"\n        ),\n\n        \"maximum_row_sum_deviation_from_one\": (\n            np.nan\n        ),\n\n        \"argmax_prediction_disagreements\": (\n            np.nan\n        ),\n\n        \"confidence_available_for_selection\": (\n            False\n        ),\n    })\n\n\nprobability_qa_df = pd.DataFrame(\n    probability_qa_records\n)\n\n\natomic_csv_save(\n    probability_qa_df,\n    PROBABILITY_QA_PATH,\n)\n\n\nconfidence_available = bool(\n    probability_qa_df.iloc[\n        0\n    ][\n        \"confidence_available_for_selection\"\n    ]\n)\n\n\n# =============================================================================\n# 10. Verify locked final-test composition\n# =============================================================================\n\nobserved_class_counts = (\n    test_df[\n        \"y_true\"\n    ]\n    .value_counts()\n    .sort_index()\n    .to_dict()\n)\n\n\nif observed_class_counts != EXPECTED_CLASS_COUNTS:\n\n    raise RuntimeError(\n        \"Final-test class-count mismatch.\"\n    )\n\n\nobserved_correct = int(\n    (\n        test_df[\n            \"y_true\"\n        ]\n        ==\n        test_df[\n            \"y_pred\"\n        ]\n    ).sum()\n)\n\n\nobserved_errors = int(\n    len(\n        test_df\n    )\n    -\n    observed_correct\n)\n\n\nif observed_correct != EXPECTED_EXACT_CORRECT:\n\n    raise RuntimeError(\n        \"Final-test exact-correct count mismatch.\"\n    )\n\n\nif observed_errors != EXPECTED_EXACT_ERRORS:\n\n    raise RuntimeError(\n        \"Final-test error count mismatch.\"\n    )\n\n\nfor transition, expected_count in EXPECTED_TRANSITION_COUNTS.items():\n\n    true_grade, predicted_grade = transition\n\n    observed_count = int(\n        (\n            (\n                test_df[\n                    \"y_true\"\n                ]\n                ==\n                true_grade\n            )\n            &\n            (\n                test_df[\n                    \"y_pred\"\n                ]\n                ==\n                predicted_grade\n            )\n        ).sum()\n    )\n\n\n    if observed_count != expected_count:\n\n        raise RuntimeError(\n            \"Expected transition-count mismatch for \"\n            f\"{CLASS_NAMES[true_grade]} -> \"\n            f\"{CLASS_NAMES[predicted_grade]}.\"\n        )\n\n\n# =============================================================================\n# 11. Deterministic case selection\n# =============================================================================\n\nselected_records = []\ncandidate_summary_records = []\n\n\nfor stratum in SELECTION_STRATA:\n\n    true_grade = int(\n        stratum[\n            \"true_grade\"\n        ]\n    )\n\n    predicted_grade = int(\n        stratum[\n            \"predicted_grade\"\n        ]\n    )\n\n\n    candidate_pool = test_df[\n        (\n            test_df[\n                \"y_true\"\n            ]\n            ==\n            true_grade\n        )\n        &\n        (\n            test_df[\n                \"y_pred\"\n            ]\n            ==\n            predicted_grade\n        )\n    ].copy()\n\n\n    (\n        selected_row,\n        group_median_confidence,\n        selection_method,\n    ) = select_representative_case(\n        candidate_pool,\n        confidence_available,\n    )\n\n\n    image_path = resolve_image_path(\n        selected_row[\n            \"sample_id\"\n        ]\n    )\n\n\n    selected_record = {\n        \"selection_order\": int(\n            stratum[\n                \"selection_order\"\n            ]\n        ),\n\n        \"case_role\": (\n            stratum[\n                \"case_role\"\n            ]\n        ),\n\n        \"case_category\": (\n            stratum[\n                \"case_category\"\n            ]\n        ),\n\n        \"clinical_focus\": (\n            stratum[\n                \"clinical_focus\"\n            ]\n        ),\n\n        \"sample_id\": str(\n            selected_row[\n                \"sample_id\"\n            ]\n        ),\n\n        \"leakage_group_id\": str(\n            selected_row[\n                \"leakage_group_id\"\n            ]\n        ),\n\n        \"true_grade\": (\n            true_grade\n        ),\n\n        \"true_class\": (\n            CLASS_NAMES[\n                true_grade\n            ]\n        ),\n\n        \"predicted_grade\": (\n            predicted_grade\n        ),\n\n        \"predicted_class\": (\n            CLASS_NAMES[\n                predicted_grade\n            ]\n        ),\n\n        \"signed_grade_error\": (\n            predicted_grade\n            -\n            true_grade\n        ),\n\n        \"absolute_grade_error\": abs(\n            predicted_grade\n            -\n            true_grade\n        ),\n\n        \"candidate_pool_size\": int(\n            len(\n                candidate_pool\n            )\n        ),\n\n        \"selection_method\": (\n            selection_method\n        ),\n\n        \"group_median_predicted_confidence\": (\n            group_median_confidence\n        ),\n\n        \"selected_predicted_confidence\": (\n            float(\n                selected_row[\n                    \"predicted_confidence\"\n                ]\n            )\n            if pd.notna(\n                selected_row[\n                    \"predicted_confidence\"\n                ]\n            )\n            else\n            np.nan\n        ),\n\n        \"distance_from_group_median_confidence\": (\n            float(\n                selected_row.get(\n                    \"distance_from_group_median_confidence\",\n                    np.nan,\n                )\n            )\n            if pd.notna(\n                selected_row.get(\n                    \"distance_from_group_median_confidence\",\n                    np.nan,\n                )\n            )\n            else\n            np.nan\n        ),\n\n        \"image_path\": str(\n            image_path\n        ),\n\n        \"image_exists\": (\n            True\n        ),\n\n        \"image_size_bytes\": int(\n            image_path.stat().st_size\n        ),\n\n        \"selected_after_model_lock\": (\n            True\n        ),\n\n        \"used_for_model_selection\": (\n            False\n        ),\n\n        \"used_for_threshold_tuning\": (\n            False\n        ),\n    }\n\n\n    for class_index in range(\n        5\n    ):\n\n        probability_column = (\n            f\"probability_grade_{class_index}\"\n        )\n\n\n        selected_record[\n            probability_column\n        ] = (\n            float(\n                selected_row[\n                    probability_column\n                ]\n            )\n            if (\n                probability_column\n                in\n                selected_row.index\n                and\n                pd.notna(\n                    selected_row[\n                        probability_column\n                    ]\n                )\n            )\n            else\n            np.nan\n        )\n\n\n    selected_records.append(\n        selected_record\n    )\n\n\n    candidate_summary_records.append({\n        \"selection_order\": int(\n            stratum[\n                \"selection_order\"\n            ]\n        ),\n\n        \"case_category\": (\n            stratum[\n                \"case_category\"\n            ]\n        ),\n\n        \"true_grade\": (\n            true_grade\n        ),\n\n        \"predicted_grade\": (\n            predicted_grade\n        ),\n\n        \"candidate_pool_size\": int(\n            len(\n                candidate_pool\n            )\n        ),\n\n        \"confidence_available\": (\n            confidence_available\n        ),\n\n        \"group_median_predicted_confidence\": (\n            group_median_confidence\n        ),\n\n        \"selection_method\": (\n            selection_method\n        ),\n    })\n\n\nselected_cases_df = pd.DataFrame(\n    selected_records\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\ncandidate_summary_df = pd.DataFrame(\n    candidate_summary_records\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\nif len(\n    selected_cases_df\n) != 10:\n\n    raise RuntimeError(\n        \"Expected exactly 10 selected XAI cases.\"\n    )\n\n\nif selected_cases_df[\n    \"sample_id\"\n].duplicated().any():\n\n    raise RuntimeError(\n        \"Duplicate samples detected in XAI selection.\"\n    )\n\n\nif not selected_cases_df[\n    \"image_exists\"\n].all():\n\n    raise RuntimeError(\n        \"One or more selected XAI images are missing.\"\n    )\n\n\natomic_csv_save(\n    selected_cases_df,\n    SELECTION_PATH,\n)\n\n\natomic_csv_save(\n    candidate_summary_df,\n    CANDIDATE_SUMMARY_PATH,\n)\n\n\n# =============================================================================\n# 12. Selection fingerprint and source verification\n# =============================================================================\n\nselection_fingerprint = dataframe_fingerprint(\n    selected_cases_df[\n        [\n            \"selection_order\",\n            \"case_category\",\n            \"sample_id\",\n            \"true_grade\",\n            \"predicted_grade\",\n            \"selection_method\",\n            \"image_path\",\n        ]\n    ]\n)\n\n\nsource_verification_df = pd.DataFrame([\n    {\n        \"source_stage\": (\n            \"Step 10D final-test locked predictions\"\n        ),\n\n        \"source_type\": (\n            \"backup_member\"\n        ),\n\n        \"source_path\": str(\n            TEST_BACKUP_PATH\n        ),\n\n        \"source_member\": (\n            TEST_PREDICTION_MEMBER\n        ),\n\n        \"source_size_bytes\": int(\n            len(\n                prediction_bytes\n            )\n        ),\n\n        \"source_sha256\": (\n            prediction_source_sha256\n        ),\n\n        \"rows\": int(\n            len(\n                test_df\n            )\n        ),\n\n        \"unique_sample_ids\": int(\n            test_df[\n                \"sample_id\"\n            ].nunique()\n        ),\n\n        \"exact_correct\": (\n            observed_correct\n        ),\n\n        \"exact_errors\": (\n            observed_errors\n        ),\n\n        \"confidence_available\": (\n            confidence_available\n        ),\n\n        \"model_inference_performed\": (\n            False\n        ),\n    },\n\n    {\n        \"source_stage\": (\n            \"Frozen final-model checkpoint readiness\"\n        ),\n\n        \"source_type\": (\n            \"checkpoint_file\"\n        ),\n\n        \"source_path\": str(\n            FINAL_MODEL_CHECKPOINT_PATH\n        ),\n\n        \"source_member\": (\n            \"\"\n        ),\n\n        \"source_size_bytes\": int(\n            FINAL_MODEL_CHECKPOINT_PATH.stat().st_size\n        ),\n\n        \"source_sha256\": sha256_file(\n            FINAL_MODEL_CHECKPOINT_PATH\n        ),\n\n        \"rows\": (\n            np.nan\n        ),\n\n        \"unique_sample_ids\": (\n            np.nan\n        ),\n\n        \"exact_correct\": (\n            np.nan\n        ),\n\n        \"exact_errors\": (\n            np.nan\n        ),\n\n        \"confidence_available\": (\n            np.nan\n        ),\n\n        \"model_inference_performed\": (\n            False\n        ),\n    },\n])\n\n\natomic_csv_save(\n    source_verification_df,\n    SOURCE_VERIFICATION_PATH,\n)\n\n\n# =============================================================================\n# 13. Publication-ready case table\n# =============================================================================\n\npublication_selection_df = selected_cases_df[\n    [\n        \"selection_order\",\n        \"case_role\",\n        \"case_category\",\n        \"sample_id\",\n        \"true_class\",\n        \"predicted_class\",\n        \"signed_grade_error\",\n        \"candidate_pool_size\",\n        \"selected_predicted_confidence\",\n        \"selection_method\",\n        \"clinical_focus\",\n    ]\n].copy()\n\n\npublication_selection_df.columns = [\n    \"Order\",\n    \"Case role\",\n    \"Case category\",\n    \"Image ID\",\n    \"Reference grade\",\n    \"Predicted grade\",\n    \"Signed grade error\",\n    \"Candidate-pool size\",\n    \"Predicted confidence\",\n    \"Deterministic selection rule\",\n    \"Interpretive focus\",\n]\n\n\natomic_csv_save(\n    publication_selection_df,\n    PUBLICATION_SELECTION_PATH,\n)\n\n\n# =============================================================================\n# 14. Lock selection plan\n# =============================================================================\n\nselection_plan = {\n    \"step\": (\n        \"STEP_13A_DETERMINISTIC_XAI_CASE_SELECTION\"\n    ),\n\n    \"status\": (\n        \"locked\"\n    ),\n\n    \"created_utc\": (\n        utc_now()\n    ),\n\n    \"evaluation_source\": (\n        \"One-time locked final-test predictions\"\n    ),\n\n    \"selected_case_count\": (\n        10\n    ),\n\n    \"correct_case_count\": (\n        5\n    ),\n\n    \"error_case_count\": (\n        5\n    ),\n\n    \"selection_rule\": (\n        \"For each predefined true-grade/predicted-grade \"\n        \"stratum, select the case nearest to the median \"\n        \"saved predicted confidence. If saved confidence \"\n        \"is unavailable, select the lexicographic midpoint \"\n        \"sample deterministically.\"\n    ),\n\n    \"selection_strata\": (\n        SELECTION_STRATA\n    ),\n\n    \"probability_vector_available\": (\n        full_probability_vector_available\n    ),\n\n    \"confidence_available\": (\n        confidence_available\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        selection_fingerprint\n    ),\n\n    \"selection_changed_after_xai_review_allowed\": (\n        False\n    ),\n\n    \"selection_used_for_training\": (\n        False\n    ),\n\n    \"selection_used_for_model_selection\": (\n        False\n    ),\n\n    \"selection_used_for_threshold_tuning\": (\n        False\n    ),\n\n    \"next_stage\": (\n        \"STEP_13B_GRADCAM_PLUS_PLUS_GENERATION\"\n    ),\n}\n\n\natomic_json_save(\n    selection_plan,\n    SELECTION_PLAN_PATH,\n)\n\n\n# =============================================================================\n# 15. Formal summary and manuscript note\n# =============================================================================\n\ncorrect_case_count = int(\n    (\n        selected_cases_df[\n            \"case_role\"\n        ]\n        ==\n        \"correct\"\n    ).sum()\n)\n\n\nerror_case_count = int(\n    (\n        selected_cases_df[\n            \"case_role\"\n        ]\n        ==\n        \"error\"\n    ).sum()\n)\n\n\nlarge_undergrading_case_count = int(\n    (\n        selected_cases_df[\n            \"signed_grade_error\"\n        ]\n        <=\n        -2\n    ).sum()\n)\n\n\nsummary_record = {\n    \"step\": (\n        \"STEP_13A_DETERMINISTIC_XAI_CASE_SELECTION\"\n    ),\n\n    \"status\": (\n        \"completed\"\n    ),\n\n    \"completed_utc\": (\n        utc_now()\n    ),\n\n    \"selected_case_count\": (\n        len(\n            selected_cases_df\n        )\n    ),\n\n    \"correct_case_count\": (\n        correct_case_count\n    ),\n\n    \"error_case_count\": (\n        error_case_count\n    ),\n\n    \"large_undergrading_case_count\": (\n        large_undergrading_case_count\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        selection_fingerprint\n    ),\n\n    \"selection_method\": (\n        \"Predefined strata with median-confidence \"\n        \"representative selection\"\n        if confidence_available\n        else\n        \"Predefined strata with deterministic \"\n        \"lexicographic midpoint selection\"\n    ),\n\n    \"selected_categories\": (\n        selected_cases_df[\n            \"case_category\"\n        ].tolist()\n    ),\n\n    \"all_image_paths_verified\": (\n        True\n    ),\n\n    \"selection_locked_before_xai_generation\": (\n        True\n    ),\n\n    \"selection_change_after_heatmap_review_allowed\": (\n        False\n    ),\n\n    \"new_training_performed\": (\n        False\n    ),\n\n    \"model_inference_performed\": (\n        False\n    ),\n\n    \"raw_image_pixels_decoded\": (\n        0\n    ),\n\n    \"validation_predictions_regenerated\": (\n        False\n    ),\n\n    \"final_test_predictions_regenerated\": (\n        False\n    ),\n\n    \"model_change_performed\": (\n        False\n    ),\n\n    \"next_stage\": (\n        \"STEP_13B_GRADCAM_PLUS_PLUS_GENERATION\"\n    ),\n}\n\n\natomic_json_save(\n    summary_record,\n    SUMMARY_PATH,\n)\n\n\nmanuscript_note = f\"\"\"\nSTEP 13A — DETERMINISTIC XAI CASE SELECTION\n\nTen final-test cases were selected before visualization: one correctly\nclassified example from each of the five diabetic-retinopathy grades\nand five predefined error cases representing the principal confusion\npatterns identified in the locked prediction analysis.\n\nThe error strata comprised Moderate-to-Mild, Severe-to-Moderate,\nProliferative-DR-to-Moderate, Proliferative-DR-to-Mild, and\nModerate-to-Severe predictions. Within each stratum, the case nearest\nto the median saved predicted confidence was selected to reduce\nsubjective cherry-picking. When confidence was unavailable, a\ndeterministic lexicographic midpoint rule was specified as the fallback.\n\nThe selection contained {correct_case_count} correct cases and\n{error_case_count} error cases. Its locked SHA-256 fingerprint was:\n\n{selection_fingerprint}\n\nThe cases were selected only after model locking and were not used for\ntraining, model selection, calibration or threshold optimization.\nCase replacement after viewing the attribution maps is prohibited.\n\"\"\".strip()\n\n\natomic_text_save(\n    manuscript_note,\n    MANUSCRIPT_NOTE_PATH,\n)\n\n\n# =============================================================================\n# 16. Formal state\n# =============================================================================\n\nstate_record = {\n    \"step\": (\n        \"STEP_13A_DETERMINISTIC_XAI_CASE_SELECTION\"\n    ),\n\n    \"status\": (\n        \"complete\"\n    ),\n\n    \"updated_utc\": (\n        utc_now()\n    ),\n\n    \"xai_case_selection_completed\": (\n        True\n    ),\n\n    \"xai_case_selection_locked\": (\n        True\n    ),\n\n    \"selected_case_count\": (\n        10\n    ),\n\n    \"correct_case_count\": (\n        correct_case_count\n    ),\n\n    \"error_case_count\": (\n        error_case_count\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        selection_fingerprint\n    ),\n\n    \"case_replacement_after_heatmap_review_allowed\": (\n        False\n    ),\n\n    \"all_selected_images_exist\": (\n        True\n    ),\n\n    \"final_checkpoint_exists\": (\n        True\n    ),\n\n    \"new_training_performed\": (\n        False\n    ),\n\n    \"model_inference_performed\": (\n        False\n    ),\n\n    \"raw_image_pixels_decoded\": (\n        0\n    ),\n\n    \"validation_predictions_regenerated\": (\n        False\n    ),\n\n    \"final_test_predictions_regenerated\": (\n        False\n    ),\n\n    \"model_change_allowed\": (\n        False\n    ),\n\n    \"another_validation_evaluation_allowed\": (\n        False\n    ),\n\n    \"another_test_evaluation_allowed\": (\n        False\n    ),\n\n    \"authoritative_selection_table\": str(\n        SELECTION_PATH\n    ),\n\n    \"next_stage\": (\n        \"STEP_13B_GRADCAM_PLUS_PLUS_GENERATION\"\n    ),\n}\n\n\natomic_json_save(\n    state_record,\n    STATE_PATH,\n)\n\n\n# =============================================================================\n# 17. Manifest and verified backup\n# =============================================================================\n\nmanifest_sources = [\n    STEP10D_STATE_PATH,\n    STEP12B_STATE_PATH,\n    STEP12B_SNAPSHOT_PATH,\n    SOURCE_VERIFICATION_PATH,\n    PROBABILITY_QA_PATH,\n    CANDIDATE_SUMMARY_PATH,\n    SELECTION_PATH,\n    PUBLICATION_SELECTION_PATH,\n    SELECTION_PLAN_PATH,\n    SUMMARY_PATH,\n    MANUSCRIPT_NOTE_PATH,\n    STATE_PATH,\n]\n\n\nmanifest_records = []\n\n\nfor source_path in manifest_sources:\n\n    if not source_path.exists():\n\n        raise FileNotFoundError(\n            f\"Step 13A manifest source missing: {source_path}\"\n        )\n\n\n    manifest_records.append({\n        \"relative_path\": str(\n            source_path.relative_to(\n                PROJECT\n            )\n        ),\n\n        \"size_bytes\": int(\n            source_path.stat().st_size\n        ),\n\n        \"sha256\": sha256_file(\n            source_path\n        ),\n    })\n\n\natomic_csv_save(\n    pd.DataFrame(\n        manifest_records\n    ),\n    MANIFEST_PATH,\n)\n\n\nbackup_members = create_verified_zip(\n    BACKUP_PATH,\n    [\n        SOURCE_VERIFICATION_PATH,\n        PROBABILITY_QA_PATH,\n        CANDIDATE_SUMMARY_PATH,\n        SELECTION_PATH,\n        PUBLICATION_SELECTION_PATH,\n        SELECTION_PLAN_PATH,\n        SUMMARY_PATH,\n        MANUSCRIPT_NOTE_PATH,\n        STATE_PATH,\n        MANIFEST_PATH,\n    ],\n)\n\n\n# =============================================================================\n# 18. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"STEP 13A — DETERMINISTIC XAI \"\n    \"CASE SELECTION COMPLETED\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nEVIDENCE SAFETY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"New training performed                :\",\n    False\n)\n\nprint(\n    \"Model inference performed             :\",\n    False\n)\n\nprint(\n    \"Raw image pixels decoded              :\",\n    0\n)\n\nprint(\n    \"Validation predictions regenerated    :\",\n    False\n)\n\nprint(\n    \"Final-test predictions regenerated    :\",\n    False\n)\n\nprint(\n    \"Frozen final model changed            :\",\n    False\n)\n\nprint(\n    \"Selection locked before heatmaps      :\",\n    True\n)\n\nprint(\n    \"Case replacement after review allowed :\",\n    False\n)\n\n\nprint(\n    \"\\nSOURCE AND PROBABILITY QA\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Prediction source                     :\",\n    (\n        str(\n            TEST_BACKUP_PATH\n        )\n        +\n        \"::\"\n        +\n        TEST_PREDICTION_MEMBER\n    )\n)\n\nprint(\n    \"Prediction source SHA-256             :\",\n    prediction_source_sha256\n)\n\nprint(\n    \"Final-test rows                       :\",\n    len(\n        test_df\n    )\n)\n\nprint(\n    \"Unique sample IDs                     :\",\n    test_df[\n        \"sample_id\"\n    ].nunique()\n)\n\nprint(\n    \"Full probability vector available     :\",\n    full_probability_vector_available\n)\n\nprint(\n    \"Confidence available for selection    :\",\n    confidence_available\n)\n\n\nprint(\n    \"\\nSELECTED XAI CASES\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_columns = [\n    \"selection_order\",\n    \"case_role\",\n    \"case_category\",\n    \"sample_id\",\n    \"true_class\",\n    \"predicted_class\",\n    \"signed_grade_error\",\n    \"candidate_pool_size\",\n    \"selected_predicted_confidence\",\n    \"selection_method\",\n]\n\n\ndisplay_df = selected_cases_df[\n    display_columns\n].copy()\n\n\ndisplay_df[\n    \"selected_predicted_confidence\"\n] = display_df[\n    \"selected_predicted_confidence\"\n].map(\n    lambda value: (\n        f\"{float(value):.6f}\"\n        if pd.notna(\n            value\n        )\n        else\n        \"NA\"\n    )\n)\n\n\nprint(\n    display_df.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nSELECTION INTEGRITY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Selected case count                   :\",\n    len(\n        selected_cases_df\n    )\n)\n\nprint(\n    \"Correct cases                         :\",\n    correct_case_count\n)\n\nprint(\n    \"Error cases                           :\",\n    error_case_count\n)\n\nprint(\n    \"Large-undergrading cases              :\",\n    large_undergrading_case_count\n)\n\nprint(\n    \"Duplicate selected sample IDs         :\",\n    int(\n        selected_cases_df[\n            \"sample_id\"\n        ].duplicated().sum()\n    )\n)\n\nprint(\n    \"All selected image paths exist        :\",\n    bool(\n        selected_cases_df[\n            \"image_exists\"\n        ].all()\n    )\n)\n\nprint(\n    \"Selection fingerprint SHA-256         :\",\n    selection_fingerprint\n)\n\n\nprint(\n    \"\\nBACKUP\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup path                           :\",\n    BACKUP_PATH\n)\n\nprint(\n    \"Backup members                        :\",\n    len(\n        backup_members\n    )\n)\n\nprint(\n    \"ZIP integrity passed                  :\",\n    True\n)\n\n\nprint(\n    \"\\nNEXT STAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"READY FOR STEP 13B — GRAD-CAM++ GENERATION\"\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T15:24:10.942940Z","iopub.execute_input":"2026-07-18T15:24:10.943810Z","iopub.status.idle":"2026-07-18T15:24:11.194995Z","shell.execute_reply.started":"2026-07-18T15:24:10.943777Z","shell.execute_reply":"2026-07-18T15:24:11.194083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 13B-D0 — READ-ONLY PREDICTION-GATE FAILURE DIAGNOSTIC\n#\n# No training\n# No inference\n# No image decoding\n# No validation/test evaluation\n# No file modification\n# =============================================================================\n\nfrom pathlib import Path\nimport json\nimport pandas as pd\n\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nPREDICTION_QA_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13b_gradcam_plus_plus\"\n    / \"step_13b_selected_case_prediction_qa.csv\"\n)\n\nMODEL_EVIDENCE_PATH = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13b_gradcam_plus_plus\"\n    / \"step_13b_model_load_evidence.json\"\n)\n\nPREPROCESSING_EVIDENCE_PATH = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13b_gradcam_plus_plus\"\n    / \"step_13b_preprocessing_protocol.json\"\n)\n\nFIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_13b_gradcam_plus_plus\"\n)\n\nSTATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13b_gradcam_plus_plus_state.json\"\n)\n\n\nprint(\"=\" * 120)\nprint(\"STEP 13B-D0 — PREDICTION-GATE FAILURE DIAGNOSTIC\")\nprint(\"=\" * 120)\n\nprint(\"\\nSAFETY\")\nprint(\"-\" * 120)\nprint(\"Training performed                  :\", False)\nprint(\"Model inference performed           :\", False)\nprint(\"Images decoded                      :\", 0)\nprint(\"Validation/test evaluated           :\", False)\nprint(\"Files modified                      :\", False)\n\n\n# -------------------------------------------------------------------------\n# 1. Check incomplete-state protection\n# -------------------------------------------------------------------------\n\nprint(\"\\nCURRENT STEP 13B STATUS\")\nprint(\"-\" * 120)\n\nprint(\"Completed Step 13B state exists     :\", STATE_PATH.exists())\n\nif STATE_PATH.exists():\n\n    with open(\n        STATE_PATH,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        state = json.load(file)\n\n    print(\"State status                        :\", state.get(\"status\"))\n\nelse:\n\n    print(\"State status                        : incomplete / not created\")\n\n\n# -------------------------------------------------------------------------\n# 2. Prediction QA\n# -------------------------------------------------------------------------\n\nif not PREDICTION_QA_PATH.exists():\n\n    raise FileNotFoundError(\n        \"Prediction QA CSV was not created:\\n\"\n        f\"{PREDICTION_QA_PATH}\"\n    )\n\n\nqa_df = pd.read_csv(\n    PREDICTION_QA_PATH\n)\n\n\nrequired_columns = [\n    \"selection_order\",\n    \"sample_id\",\n    \"true_grade\",\n    \"saved_predicted_grade\",\n    \"amp_reproduced_grade\",\n    \"fp32_reproduced_grade\",\n    \"amp_prediction_matches\",\n    \"fp32_prediction_matches\",\n    \"maximum_amp_probability_difference\",\n    \"maximum_fp32_probability_difference\",\n    \"amp_probability_match_passed\",\n    \"crop_status\",\n]\n\n\nmissing_columns = [\n    column\n    for column in required_columns\n    if column not in qa_df.columns\n]\n\n\nif missing_columns:\n\n    raise RuntimeError(\n        \"Prediction QA CSV is missing columns:\\n\"\n        f\"{missing_columns}\"\n    )\n\n\n# Robust bool normalization in case CSV values were read as text.\ndef normalize_bool_series(series):\n\n    if series.dtype == bool:\n        return series\n\n    return (\n        series.astype(str)\n        .str.strip()\n        .str.lower()\n        .map({\n            \"true\": True,\n            \"false\": False,\n            \"1\": True,\n            \"0\": False,\n        })\n    )\n\n\nfor column in [\n    \"amp_prediction_matches\",\n    \"fp32_prediction_matches\",\n    \"amp_probability_match_passed\",\n]:\n\n    qa_df[column] = normalize_bool_series(\n        qa_df[column]\n    )\n\n\namp_argmax_failures = qa_df[\n    qa_df[\"amp_prediction_matches\"] != True\n].copy()\n\nfp32_argmax_failures = qa_df[\n    qa_df[\"fp32_prediction_matches\"] != True\n].copy()\n\nprobability_failures = qa_df[\n    qa_df[\"amp_probability_match_passed\"] != True\n].copy()\n\n\nprint(\"\\nPREDICTION QA — ALL 10 CASES\")\nprint(\"-\" * 120)\n\ndisplay_columns = [\n    \"selection_order\",\n    \"sample_id\",\n    \"true_grade\",\n    \"saved_predicted_grade\",\n    \"amp_reproduced_grade\",\n    \"fp32_reproduced_grade\",\n    \"amp_prediction_matches\",\n    \"fp32_prediction_matches\",\n    \"maximum_amp_probability_difference\",\n    \"maximum_fp32_probability_difference\",\n    \"amp_probability_match_passed\",\n    \"crop_status\",\n]\n\n\ndisplay_df = qa_df[\n    display_columns\n].copy()\n\n\nfor column in [\n    \"maximum_amp_probability_difference\",\n    \"maximum_fp32_probability_difference\",\n]:\n\n    display_df[column] = display_df[column].map(\n        lambda value: f\"{float(value):.8f}\"\n    )\n\n\nprint(\n    display_df.to_string(\n        index=False\n    )\n)\n\n\nprint(\"\\nFAILURE SUMMARY\")\nprint(\"-\" * 120)\n\nprint(\"Selected cases                      :\", len(qa_df))\n\nprint(\n    \"AMP argmax matches                  :\",\n    int((qa_df[\"amp_prediction_matches\"] == True).sum()),\n    \"/\",\n    len(qa_df),\n)\n\nprint(\n    \"FP32 argmax matches                 :\",\n    int((qa_df[\"fp32_prediction_matches\"] == True).sum()),\n    \"/\",\n    len(qa_df),\n)\n\nprint(\n    \"AMP probability matches             :\",\n    int((qa_df[\"amp_probability_match_passed\"] == True).sum()),\n    \"/\",\n    len(qa_df),\n)\n\nprint(\n    \"Maximum AMP probability difference  :\",\n    f\"{qa_df['maximum_amp_probability_difference'].max():.8f}\",\n)\n\nprint(\n    \"Maximum FP32 probability difference :\",\n    f\"{qa_df['maximum_fp32_probability_difference'].max():.8f}\",\n)\n\nprint(\n    \"AMP argmax failure count            :\",\n    len(amp_argmax_failures),\n)\n\nprint(\n    \"FP32 argmax failure count           :\",\n    len(fp32_argmax_failures),\n)\n\nprint(\n    \"Probability-tolerance failure count :\",\n    len(probability_failures),\n)\n\n\n# -------------------------------------------------------------------------\n# 3. Failed cases only\n# -------------------------------------------------------------------------\n\nfailed_mask = (\n    (qa_df[\"amp_prediction_matches\"] != True)\n    |\n    (qa_df[\"fp32_prediction_matches\"] != True)\n    |\n    (qa_df[\"amp_probability_match_passed\"] != True)\n)\n\nfailed_df = qa_df[\n    failed_mask\n].copy()\n\n\nprint(\"\\nFAILED CASES ONLY\")\nprint(\"-\" * 120)\n\nif failed_df.empty:\n\n    print(\"No failed rows found; review boolean parsing or gate logic.\")\n\nelse:\n\n    print(\n        failed_df[\n            display_columns\n        ].to_string(\n            index=False\n        )\n    )\n\n\n# -------------------------------------------------------------------------\n# 4. Model restoration evidence\n# -------------------------------------------------------------------------\n\nprint(\"\\nMODEL RESTORATION EVIDENCE\")\nprint(\"-\" * 120)\n\nif MODEL_EVIDENCE_PATH.exists():\n\n    with open(\n        MODEL_EVIDENCE_PATH,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        model_evidence = json.load(file)\n\n    print(\n        \"Checkpoint SHA-256                 :\",\n        model_evidence.get(\"checkpoint_sha256\"),\n    )\n\n    print(\n        \"EMA source key                     :\",\n        model_evidence.get(\"ema_source_key\"),\n    )\n\n    print(\n        \"State-dict normalization           :\",\n        model_evidence.get(\"state_dict_normalization_variant\"),\n    )\n\n    print(\n        \"Strict loading passed              :\",\n        model_evidence.get(\"strict_state_dict_loading_passed\"),\n    )\n\n    print(\n        \"Parameter count                    :\",\n        model_evidence.get(\"parameter_count\"),\n    )\n\n    print(\n        \"Target layer                       :\",\n        model_evidence.get(\"target_layer_name\"),\n    )\n\nelse:\n\n    print(\"Model-load evidence file is missing.\")\n\n\n# -------------------------------------------------------------------------\n# 5. Preprocessing evidence\n# -------------------------------------------------------------------------\n\nprint(\"\\nPREPROCESSING EVIDENCE USED BY FAILED CELL\")\nprint(\"-\" * 120)\n\nif PREPROCESSING_EVIDENCE_PATH.exists():\n\n    with open(\n        PREPROCESSING_EVIDENCE_PATH,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        preprocessing_evidence = json.load(file)\n\n    for key in [\n        \"variant\",\n        \"input_size\",\n        \"retinal_foreground_threshold\",\n        \"morphological_kernel\",\n        \"minimum_bounding_area_ratio\",\n        \"crop_margin_fraction\",\n        \"clahe_clip_limit\",\n        \"clahe_tile_grid_size\",\n        \"normalization_mean\",\n        \"normalization_std\",\n    ]:\n\n        print(\n            f\"{key:36s}:\",\n            preprocessing_evidence.get(key),\n        )\n\nelse:\n\n    print(\"Preprocessing evidence file is missing.\")\n\n\n# -------------------------------------------------------------------------\n# 6. Confirm that no Grad-CAM figures were generated\n# -------------------------------------------------------------------------\n\nfigure_files = []\n\nif FIGURE_DIR.exists():\n\n    figure_files = [\n        path\n        for path in FIGURE_DIR.rglob(\"*\")\n        if path.is_file()\n    ]\n\n\nprint(\"\\nPARTIAL-OUTPUT SAFETY\")\nprint(\"-\" * 120)\n\nprint(\"Grad-CAM figure files found          :\", len(figure_files))\n\nif figure_files:\n\n    for path in figure_files[:20]:\n        print(path)\n\nprint(\n    \"Safe interpretation                  :\",\n    (\n        \"No heatmaps were generated.\"\n        if len(figure_files) == 0\n        else\n        \"Partial files exist; they must not be used.\"\n    ),\n)\n\n\nprint(\"\\nNEXT ACTION\")\nprint(\"-\" * 120)\nprint(\n    \"SEND THIS COMPLETE OUTPUT. \"\n    \"DO NOT RERUN THE OLD STEP 13B CELL.\"\n)\nprint(\"=\" * 120)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T15:43:36.691649Z","iopub.execute_input":"2026-07-18T15:43:36.692003Z","iopub.status.idle":"2026-07-18T15:43:36.733959Z","shell.execute_reply.started":"2026-07-18T15:43:36.691976Z","shell.execute_reply":"2026-07-18T15:43:36.733171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 13B-R — CORRECTED LOCKED GRAD-CAM++ GENERATION\n#\n# Correction basis:\n#   Previous diagnostic confirmed:\n#       AMP argmax  = 10/10 matches\n#       FP32 argmax = 10/10 matches\n#       Maximum AMP probability difference = 0.00111536\n#\n# Revised numerical-equivalence gate:\n#   - Saved probability vectors must be valid\n#   - Saved probability argmax must reproduce locked prediction\n#   - AMP argmax must match for all 10 cases\n#   - FP32 argmax must match for all 10 cases\n#   - Maximum AMP probability difference must be <= 0.002\n#\n# No training.\n# No optimizer.\n# No model modification.\n# No complete validation/test evaluation.\n# No threshold or calibration tuning.\n# Only the 10 cases locked before XAI review are decoded.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\n\nimport gc\nimport hashlib\nimport json\nimport math\nimport os\nimport re\nimport zipfile\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torchvision.models import efficientnet_b0\n\n\n# =============================================================================\n# 1. Fixed paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nSTEP10D_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_10d_final_test_state.json\"\n)\n\nSTEP13A_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13a_xai_case_selection_state.json\"\n)\n\nSELECTION_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13a_xai_case_selection\"\n    / \"step_13a_selected_xai_cases.csv\"\n)\n\nFINAL_CHECKPOINT_PATH = (\n    PROJECT\n    / \"06_checkpoints\"\n    / \"step_10b_registered_baseline_final_training\"\n    / \"step_10b_final_training_checkpoint.pt\"\n)\n\nFAILED_STAGE_QA_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13b_gradcam_plus_plus\"\n    / \"step_13b_selected_case_prediction_qa.csv\"\n)\n\nFAILED_STAGE_FIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_13b_gradcam_plus_plus\"\n)\n\n\nOUTPUT_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13b_r_gradcam_plus_plus\"\n)\n\nOUTPUT_EVIDENCE_DIR = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13b_r_gradcam_plus_plus\"\n)\n\nFIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_13b_r_gradcam_plus_plus\"\n)\n\n\nPREDICTION_QA_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13b_r_prediction_reproduction_qa.csv\"\n)\n\nATTRIBUTION_METRICS_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13b_r_gradcampp_attribution_metrics.csv\"\n)\n\nCASE_OUTPUT_INDEX_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13b_r_case_output_index.csv\"\n)\n\nDIAGNOSTIC_AMENDMENT_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_numerical_gate_amendment.json\"\n)\n\nMODEL_LOAD_EVIDENCE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_model_load_evidence.json\"\n)\n\nPREPROCESSING_EVIDENCE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_preprocessing_protocol.json\"\n)\n\nPUBLICATION_CASE_TABLE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_publication_xai_case_table.csv\"\n)\n\nSUMMARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_gradcam_plus_plus_summary.json\"\n)\n\nMANUSCRIPT_NOTE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_gradcam_plus_plus_method_note.txt\"\n)\n\nMANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13b_r_gradcam_plus_plus_manifest.csv\"\n)\n\nSTATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13b_r_gradcam_plus_plus_state.json\"\n)\n\nBACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_13b_r_gradcam_plus_plus_backup.zip\"\n)\n\n\n# =============================================================================\n# 2. Locked constants\n# =============================================================================\n\nEXPECTED_CHECKPOINT_SHA256 = (\n    \"7b46f5562e2a5884027a16c20e2b15de\"\n    \"a2f185195cbdde4e9169528fc7c14f71\"\n)\n\nEXPECTED_SELECTION_FINGERPRINT = (\n    \"da2b47d1c8cce21ab4231edf96e27f227\"\n    \"bfb24fca7fdc8bb508b47ceb5a34f89\"\n)\n\nEXPECTED_PARAMETER_COUNT = 4_013_953\n\nEXPECTED_SELECTED_CASES = 10\nEXPECTED_CORRECT_CASES = 5\nEXPECTED_ERROR_CASES = 5\n\nEXPECTED_PREDICTED_MAPS = 10\nEXPECTED_REFERENCE_MAPS = 5\nEXPECTED_TOTAL_MAPS = 15\n\nTARGET_SIZE = 384\nNUM_CLASSES = 5\n\nCLASS_NAMES = [\n    \"No_DR\",\n    \"Mild\",\n    \"Moderate\",\n    \"Severe\",\n    \"Proliferative_DR\",\n]\n\nIMAGENET_MEAN = np.array(\n    [0.485, 0.456, 0.406],\n    dtype=np.float32,\n)\n\nIMAGENET_STD = np.array(\n    [0.229, 0.224, 0.225],\n    dtype=np.float32,\n)\n\n# Corrected numerical-equivalence limit.\nAMP_PROBABILITY_TOLERANCE = 0.002\n\n# Descriptive only; not used as a failure gate.\nFP32_REPORTING_REFERENCE = 0.005\n\nSAVED_ROW_SUM_TOLERANCE = 0.002\n\nHEATMAP_EPSILON = 1.0e-12\nOVERLAY_ALPHA = 0.42\n\n\n# =============================================================================\n# 3. General utilities\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef read_json(path):\n\n    with open(\n        path,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        return json.load(\n            file\n        )\n\n\ndef atomic_json_save(\n    record,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        json.dump(\n            record,\n            file,\n            indent=2,\n            ensure_ascii=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_csv_save(\n    dataframe,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    dataframe.to_csv(\n        temporary_path,\n        index=False,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_text_save(\n    text,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        file.write(\n            text\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_numpy_save(\n    array,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"wb\",\n    ) as file:\n\n        np.save(\n            file,\n            np.asarray(\n                array,\n                dtype=np.float32,\n            ),\n            allow_pickle=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef sha256_file(path):\n\n    digest = hashlib.sha256()\n\n    with open(\n        path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef sha256_bytes(data):\n\n    return hashlib.sha256(\n        data\n    ).hexdigest()\n\n\ndef dataframe_fingerprint(\n    dataframe,\n):\n\n    canonical = dataframe.copy()\n\n    canonical = canonical.sort_values(\n        list(\n            canonical.columns\n        )\n    ).reset_index(\n        drop=True\n    )\n\n    csv_bytes = canonical.to_csv(\n        index=False,\n        float_format=\"%.10f\",\n        lineterminator=\"\\n\",\n    ).encode(\n        \"utf-8\"\n    )\n\n    return sha256_bytes(\n        csv_bytes\n    )\n\n\ndef normalize_bool_series(series):\n\n    if series.dtype == bool:\n\n        return series\n\n    normalized = (\n        series.astype(str)\n        .str.strip()\n        .str.lower()\n        .map({\n            \"true\": True,\n            \"false\": False,\n            \"1\": True,\n            \"0\": False,\n        })\n    )\n\n    if normalized.isna().any():\n\n        raise RuntimeError(\n            \"A boolean QA column could not be normalized.\"\n        )\n\n    return normalized.astype(\n        bool\n    )\n\n\ndef save_rgb_png(\n    rgb_image,\n    path,\n):\n\n    image_array = np.asarray(\n        rgb_image,\n        dtype=np.uint8,\n    )\n\n    if (\n        image_array.ndim != 3\n        or\n        image_array.shape[\n            2\n        ] != 3\n    ):\n\n        raise RuntimeError(\n            \"Expected an RGB image for PNG saving.\"\n        )\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    image = Image.fromarray(\n        image_array\n    )\n\n    image.save(\n        temporary_path,\n        format=\"PNG\",\n        dpi=(\n            600,\n            600,\n        ),\n        compress_level=6,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef create_verified_zip(\n    zip_path,\n    source_files,\n):\n\n    temporary_path = zip_path.with_suffix(\n        zip_path.suffix + \".tmp\"\n    )\n\n    if temporary_path.exists():\n\n        temporary_path.unlink()\n\n    unique_files = []\n\n    for source_file in source_files:\n\n        source_file = Path(\n            source_file\n        )\n\n        if (\n            source_file.exists()\n            and\n            source_file.is_file()\n            and\n            source_file not in unique_files\n        ):\n\n            unique_files.append(\n                source_file\n            )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"w\",\n        compression=zipfile.ZIP_DEFLATED,\n        compresslevel=6,\n    ) as archive:\n\n        for source_file in unique_files:\n\n            archive.write(\n                source_file,\n                arcname=str(\n                    source_file.relative_to(\n                        PROJECT\n                    )\n                ),\n            )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"r\",\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            \"Step 13B-R backup ZIP is damaged at: \"\n            f\"{damaged_member}\"\n        )\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            \"Duplicate members detected in Step 13B-R backup.\"\n        )\n\n    os.replace(\n        temporary_path,\n        zip_path,\n    )\n\n    return members\n\n\n# =============================================================================\n# 4. Exact locked preprocessing\n# =============================================================================\n\ndef read_rgb_image(\n    image_path,\n):\n\n    bgr_image = cv2.imread(\n        str(\n            image_path\n        ),\n        cv2.IMREAD_COLOR,\n    )\n\n    if bgr_image is None:\n\n        raise ValueError(\n            f\"Could not decode image: {image_path}\"\n        )\n\n    return cv2.cvtColor(\n        bgr_image,\n        cv2.COLOR_BGR2RGB,\n    )\n\n\ndef crop_retinal_field(\n    rgb_image,\n):\n\n    original_height, original_width = (\n        rgb_image.shape[\n            :2\n        ]\n    )\n\n    gray_image = cv2.cvtColor(\n        rgb_image,\n        cv2.COLOR_RGB2GRAY,\n    )\n\n    foreground_mask = (\n        gray_image\n        >\n        7\n    ).astype(\n        np.uint8\n    ) * 255\n\n    kernel = cv2.getStructuringElement(\n        cv2.MORPH_ELLIPSE,\n        (\n            9,\n            9,\n        ),\n    )\n\n    foreground_mask = cv2.morphologyEx(\n        foreground_mask,\n        cv2.MORPH_CLOSE,\n        kernel,\n    )\n\n    contours, _ = cv2.findContours(\n        foreground_mask,\n        cv2.RETR_EXTERNAL,\n        cv2.CHAIN_APPROX_SIMPLE,\n    )\n\n    if not contours:\n\n        return (\n            rgb_image,\n            \"fallback_no_contour\",\n        )\n\n    largest_contour = max(\n        contours,\n        key=cv2.contourArea,\n    )\n\n    x, y, width, height = cv2.boundingRect(\n        largest_contour\n    )\n\n    bounding_area_ratio = float(\n        width\n        *\n        height\n        /\n        (\n            original_width\n            *\n            original_height\n        )\n    )\n\n    if bounding_area_ratio < 0.10:\n\n        return (\n            rgb_image,\n            \"fallback_small_contour\",\n        )\n\n    margin = int(\n        round(\n            0.02\n            *\n            max(\n                width,\n                height,\n            )\n        )\n    )\n\n    x_start = max(\n        0,\n        x - margin,\n    )\n\n    y_start = max(\n        0,\n        y - margin,\n    )\n\n    x_end = min(\n        original_width,\n        x + width + margin,\n    )\n\n    y_end = min(\n        original_height,\n        y + height + margin,\n    )\n\n    cropped_image = rgb_image[\n        y_start:y_end,\n        x_start:x_end,\n    ]\n\n    if cropped_image.size == 0:\n\n        return (\n            rgb_image,\n            \"fallback_empty_crop\",\n        )\n\n    return (\n        cropped_image,\n        \"largest_retinal_contour\",\n    )\n\n\ndef square_pad_and_resize(\n    rgb_image,\n):\n\n    height, width = rgb_image.shape[\n        :2\n    ]\n\n    square_size = max(\n        height,\n        width,\n    )\n\n    square_image = np.zeros(\n        (\n            square_size,\n            square_size,\n            3,\n        ),\n        dtype=np.uint8,\n    )\n\n    y_offset = (\n        square_size\n        -\n        height\n    ) // 2\n\n    x_offset = (\n        square_size\n        -\n        width\n    ) // 2\n\n    square_image[\n        y_offset:y_offset + height,\n        x_offset:x_offset + width,\n    ] = rgb_image\n\n    interpolation = (\n        cv2.INTER_AREA\n        if square_size > TARGET_SIZE\n        else\n        cv2.INTER_CUBIC\n    )\n\n    resized_image = cv2.resize(\n        square_image,\n        (\n            TARGET_SIZE,\n            TARGET_SIZE,\n        ),\n        interpolation=interpolation,\n    )\n\n    return resized_image\n\n\ndef apply_mild_lab_clahe(\n    rgb_image,\n):\n\n    lab_image = cv2.cvtColor(\n        rgb_image,\n        cv2.COLOR_RGB2LAB,\n    )\n\n    lightness, channel_a, channel_b = cv2.split(\n        lab_image\n    )\n\n    clahe = cv2.createCLAHE(\n        clipLimit=1.5,\n        tileGridSize=(\n            8,\n            8,\n        ),\n    )\n\n    enhanced_lightness = clahe.apply(\n        lightness\n    )\n\n    enhanced_lab = cv2.merge(\n        (\n            enhanced_lightness,\n            channel_a,\n            channel_b,\n        )\n    )\n\n    return cv2.cvtColor(\n        enhanced_lab,\n        cv2.COLOR_LAB2RGB,\n    )\n\n\ndef image_to_tensor(\n    rgb_image,\n):\n\n    image_float = (\n        rgb_image.astype(\n            np.float32\n        )\n        /\n        255.0\n    )\n\n    image_float = (\n        image_float\n        -\n        IMAGENET_MEAN\n    ) / IMAGENET_STD\n\n    tensor = torch.from_numpy(\n        image_float.transpose(\n            2,\n            0,\n            1,\n        )\n    ).float()\n\n    if not torch.isfinite(\n        tensor\n    ).all():\n\n        raise RuntimeError(\n            \"Non-finite normalized input tensor detected.\"\n        )\n\n    return tensor\n\n\ndef prepare_selected_image(\n    image_path,\n):\n\n    raw_rgb = read_rgb_image(\n        image_path\n    )\n\n    original_preview = square_pad_and_resize(\n        raw_rgb\n    )\n\n    cropped_rgb, crop_status = crop_retinal_field(\n        raw_rgb\n    )\n\n    crop_resize_rgb = square_pad_and_resize(\n        cropped_rgb\n    )\n\n    model_input_rgb = apply_mild_lab_clahe(\n        crop_resize_rgb\n    )\n\n    image_tensor = image_to_tensor(\n        model_input_rgb\n    )\n\n    return {\n        \"original_preview\": (\n            original_preview\n        ),\n\n        \"crop_resize_rgb\": (\n            crop_resize_rgb\n        ),\n\n        \"model_input_rgb\": (\n            model_input_rgb\n        ),\n\n        \"image_tensor\": (\n            image_tensor\n        ),\n\n        \"crop_status\": (\n            crop_status\n        ),\n    }\n\n\n# =============================================================================\n# 5. Frozen EMA model restoration\n# =============================================================================\n\ndef is_tensor_state_dict(\n    candidate,\n):\n\n    if (\n        not isinstance(\n            candidate,\n            dict,\n        )\n        or\n        len(\n            candidate\n        )\n        ==\n        0\n    ):\n\n        return False\n\n    tensor_count = sum(\n        torch.is_tensor(\n            value\n        )\n        for value\n        in candidate.values()\n    )\n\n    return bool(\n        tensor_count\n        >=\n        max(\n            1,\n            int(\n                0.90\n                *\n                len(\n                    candidate\n                )\n            ),\n        )\n    )\n\n\ndef find_ema_state_dict(\n    checkpoint,\n):\n\n    if not isinstance(\n        checkpoint,\n        dict,\n    ):\n\n        raise RuntimeError(\n            \"Final checkpoint is not a dictionary.\"\n        )\n\n    preferred_keys = [\n        \"ema_state_dict\",\n        \"model_ema_state_dict\",\n        \"ema_model_state_dict\",\n        \"ema\",\n    ]\n\n    for key in preferred_keys:\n\n        candidate = checkpoint.get(\n            key\n        )\n\n        if is_tensor_state_dict(\n            candidate\n        ):\n\n            return (\n                key,\n                candidate,\n            )\n\n        if isinstance(\n            candidate,\n            dict,\n        ):\n\n            for nested_key in [\n                \"state_dict\",\n                \"module\",\n                \"model_state_dict\",\n            ]:\n\n                nested_candidate = candidate.get(\n                    nested_key\n                )\n\n                if is_tensor_state_dict(\n                    nested_candidate\n                ):\n\n                    return (\n                        f\"{key}.{nested_key}\",\n                        nested_candidate,\n                    )\n\n    raise RuntimeError(\n        \"EMA state dictionary was not found in the \"\n        \"registered final checkpoint.\"\n    )\n\n\ndef remove_non_model_entries(\n    state_dict,\n):\n\n    excluded_keys = {\n        \"n_averaged\",\n        \"num_updates\",\n    }\n\n    cleaned = {}\n\n    for key, value in state_dict.items():\n\n        key_text = str(\n            key\n        )\n\n        if (\n            torch.is_tensor(\n                value\n            )\n            and\n            key_text not in excluded_keys\n        ):\n\n            cleaned[\n                key_text\n            ] = value\n\n    return cleaned\n\n\ndef strip_prefix(\n    state_dict,\n    prefix,\n):\n\n    return {\n        (\n            key[\n                len(\n                    prefix\n                ):\n            ]\n            if key.startswith(\n                prefix\n            )\n            else key\n        ): value\n\n        for key, value\n        in state_dict.items()\n    }\n\n\ndef create_registered_model():\n\n    model = efficientnet_b0(\n        weights=None\n    )\n\n    classifier_input_features = int(\n        model.classifier[\n            1\n        ].in_features\n    )\n\n    model.classifier[\n        1\n    ] = nn.Linear(\n        classifier_input_features,\n        NUM_CLASSES,\n    )\n\n    return model\n\n\ndef load_registered_ema_model(\n    raw_state_dict,\n):\n\n    base_state_dict = remove_non_model_entries(\n        raw_state_dict\n    )\n\n    variants = {\n        \"raw\": (\n            base_state_dict\n        ),\n\n        \"strip_module\": strip_prefix(\n            base_state_dict,\n            \"module.\",\n        ),\n\n        \"strip_model\": strip_prefix(\n            base_state_dict,\n            \"model.\",\n        ),\n\n        \"strip_ema_module\": strip_prefix(\n            base_state_dict,\n            \"ema.module.\",\n        ),\n\n        \"strip_module_module\": strip_prefix(\n            base_state_dict,\n            \"module.module.\",\n        ),\n    }\n\n    seen_signatures = set()\n    rejected_variants = {}\n\n    for variant_name, variant in variants.items():\n\n        key_signature = tuple(\n            sorted(\n                variant.keys()\n            )\n        )\n\n        if key_signature in seen_signatures:\n\n            continue\n\n        seen_signatures.add(\n            key_signature\n        )\n\n        candidate_model = create_registered_model()\n\n        try:\n\n            candidate_model.load_state_dict(\n                variant,\n                strict=True,\n            )\n\n            return (\n                candidate_model,\n                variant_name,\n                len(\n                    variant\n                ),\n                rejected_variants,\n            )\n\n        except Exception as error:\n\n            rejected_variants[\n                variant_name\n            ] = str(\n                error\n            )\n\n            del candidate_model\n\n    raise RuntimeError(\n        \"No EMA state-dictionary variant loaded strictly.\\n\"\n        +\n        json.dumps(\n            rejected_variants,\n            indent=2,\n        )\n    )\n\n\n# =============================================================================\n# 6. Grad-CAM++ implementation\n# =============================================================================\n\nclass ActivationGradientCapture:\n\n    def __init__(\n        self,\n        target_layer,\n    ):\n\n        self.activations = None\n        self.gradients = None\n\n        self.forward_handle = (\n            target_layer.register_forward_hook(\n                self._forward_hook\n            )\n        )\n\n\n    def _forward_hook(\n        self,\n        module,\n        inputs,\n        output,\n    ):\n\n        self.activations = output\n\n        if output.requires_grad:\n\n            output.register_hook(\n                self._gradient_hook\n            )\n\n\n    def _gradient_hook(\n        self,\n        gradient,\n    ):\n\n        self.gradients = gradient\n\n\n    def clear(\n        self,\n    ):\n\n        self.activations = None\n        self.gradients = None\n\n\n    def remove(\n        self,\n    ):\n\n        self.forward_handle.remove()\n\n\ndef generate_gradcam_plus_plus(\n    model,\n    capture,\n    image_tensor,\n    target_grade,\n    device,\n):\n\n    capture.clear()\n\n    model.zero_grad(\n        set_to_none=True\n    )\n\n    input_batch = (\n        image_tensor\n        .unsqueeze(\n            0\n        )\n        .to(\n            device\n        )\n        .requires_grad_(\n            True\n        )\n    )\n\n    # Grad-CAM++ is intentionally computed in FP32.\n    logits = model(\n        input_batch\n    )\n\n    probabilities = torch.softmax(\n        logits.float(),\n        dim=1,\n    )\n\n    target_score = logits[\n        0,\n        int(\n            target_grade\n        ),\n    ]\n\n    target_score.backward()\n\n    if (\n        capture.activations is None\n        or\n        capture.gradients is None\n    ):\n\n        raise RuntimeError(\n            \"Target-layer activations or gradients were not captured.\"\n        )\n\n    activations = (\n        capture.activations\n        .detach()\n        .float()\n    )\n\n    gradients = (\n        capture.gradients\n        .detach()\n        .float()\n    )\n\n    if (\n        activations.ndim != 4\n        or\n        gradients.ndim != 4\n    ):\n\n        raise RuntimeError(\n            \"Unexpected activation or gradient dimensions.\"\n        )\n\n    gradient_squared = gradients.pow(\n        2\n    )\n\n    gradient_cubed = gradient_squared * gradients\n\n    activation_sum = activations.sum(\n        dim=(\n            2,\n            3,\n        ),\n        keepdim=True,\n    )\n\n    alpha_denominator = (\n        2.0\n        *\n        gradient_squared\n        +\n        activation_sum\n        *\n        gradient_cubed\n    )\n\n    safe_denominator = torch.where(\n        torch.abs(\n            alpha_denominator\n        )\n        >\n        HEATMAP_EPSILON,\n        alpha_denominator,\n        torch.ones_like(\n            alpha_denominator\n        ),\n    )\n\n    alphas = (\n        gradient_squared\n        /\n        safe_denominator\n    )\n\n    positive_gradients = F.relu(\n        gradients\n    )\n\n    channel_weights = (\n        alphas\n        *\n        positive_gradients\n    ).sum(\n        dim=(\n            2,\n            3,\n        ),\n        keepdim=True,\n    )\n\n    cam = (\n        channel_weights\n        *\n        activations\n    ).sum(\n        dim=1,\n        keepdim=True,\n    )\n\n    cam = F.relu(\n        cam\n    )\n\n    cam = F.interpolate(\n        cam,\n        size=(\n            TARGET_SIZE,\n            TARGET_SIZE,\n        ),\n        mode=\"bilinear\",\n        align_corners=False,\n    )[\n        0,\n        0,\n    ]\n\n    if not torch.isfinite(\n        cam\n    ).all():\n\n        raise RuntimeError(\n            \"A non-finite Grad-CAM++ map was produced.\"\n        )\n\n    cam_minimum = cam.min()\n    cam_maximum = cam.max()\n    cam_range = cam_maximum - cam_minimum\n\n    if float(\n        cam_range.detach().cpu()\n    ) <= HEATMAP_EPSILON:\n\n        raise RuntimeError(\n            \"A degenerate Grad-CAM++ map was produced.\"\n        )\n\n    normalized_cam = (\n        cam\n        -\n        cam_minimum\n    ) / cam_range\n\n    cam_numpy = (\n        normalized_cam\n        .detach()\n        .cpu()\n        .numpy()\n        .astype(\n            np.float32\n        )\n    )\n\n    logits_numpy = (\n        logits[\n            0\n        ]\n        .detach()\n        .float()\n        .cpu()\n        .numpy()\n    )\n\n    probabilities_numpy = (\n        probabilities[\n            0\n        ]\n        .detach()\n        .cpu()\n        .numpy()\n        .astype(\n            np.float64\n        )\n    )\n\n    predicted_grade = int(\n        np.argmax(\n            probabilities_numpy\n        )\n    )\n\n    activation_shape = [\n        int(\n            value\n        )\n        for value\n        in activations.shape\n    ]\n\n    gradient_shape = [\n        int(\n            value\n        )\n        for value\n        in gradients.shape\n    ]\n\n    del input_batch\n    del logits\n    del probabilities\n    del target_score\n    del activations\n    del gradients\n    del gradient_squared\n    del gradient_cubed\n    del activation_sum\n    del alpha_denominator\n    del safe_denominator\n    del alphas\n    del positive_gradients\n    del channel_weights\n    del cam\n    del normalized_cam\n\n    return {\n        \"cam\": (\n            cam_numpy\n        ),\n\n        \"logits\": (\n            logits_numpy\n        ),\n\n        \"probabilities\": (\n            probabilities_numpy\n        ),\n\n        \"predicted_grade\": (\n            predicted_grade\n        ),\n\n        \"activation_shape\": (\n            activation_shape\n        ),\n\n        \"gradient_shape\": (\n            gradient_shape\n        ),\n    }\n\n\n# =============================================================================\n# 7. Visualization and spatial attribution QA\n# =============================================================================\n\ndef colourize_heatmap(\n    normalized_cam,\n):\n\n    heatmap_uint8 = np.clip(\n        normalized_cam\n        *\n        255.0,\n        0,\n        255,\n    ).astype(\n        np.uint8\n    )\n\n    heatmap_bgr = cv2.applyColorMap(\n        heatmap_uint8,\n        cv2.COLORMAP_TURBO,\n    )\n\n    return cv2.cvtColor(\n        heatmap_bgr,\n        cv2.COLOR_BGR2RGB,\n    )\n\n\ndef create_overlay(\n    base_rgb,\n    heatmap_rgb,\n):\n\n    overlay = (\n        (\n            1.0\n            -\n            OVERLAY_ALPHA\n        )\n        *\n        base_rgb.astype(\n            np.float32\n        )\n        +\n        OVERLAY_ALPHA\n        *\n        heatmap_rgb.astype(\n            np.float32\n        )\n    )\n\n    return np.clip(\n        overlay,\n        0,\n        255,\n    ).astype(\n        np.uint8\n    )\n\n\ndef compute_attribution_statistics(\n    normalized_cam,\n    crop_resize_rgb,\n):\n\n    cam = np.asarray(\n        normalized_cam,\n        dtype=np.float64,\n    )\n\n    if cam.shape != (\n        TARGET_SIZE,\n        TARGET_SIZE,\n    ):\n\n        raise RuntimeError(\n            \"Unexpected Grad-CAM++ spatial shape.\"\n        )\n\n    total_attention = float(\n        cam.sum()\n    )\n\n    if total_attention <= HEATMAP_EPSILON:\n\n        raise RuntimeError(\n            \"Grad-CAM++ total attention is zero.\"\n        )\n\n    # The retinal mask is derived before CLAHE to avoid enhancement\n    # altering the padded-background threshold.\n    gray_image = cv2.cvtColor(\n        crop_resize_rgb,\n        cv2.COLOR_RGB2GRAY,\n    )\n\n    retinal_mask = (\n        gray_image\n        >\n        7\n    ).astype(\n        np.float64\n    )\n\n    border_width = int(\n        round(\n            0.10\n            *\n            TARGET_SIZE\n        )\n    )\n\n    border_mask = np.zeros_like(\n        cam,\n        dtype=np.float64,\n    )\n\n    border_mask[\n        :border_width,\n        :\n    ] = 1.0\n\n    border_mask[\n        -border_width:,\n        :\n    ] = 1.0\n\n    border_mask[\n        :,\n        :border_width\n    ] = 1.0\n\n    border_mask[\n        :,\n        -border_width:\n    ] = 1.0\n\n    retinal_attention_fraction = float(\n        (\n            cam\n            *\n            retinal_mask\n        ).sum()\n        /\n        total_attention\n    )\n\n    border_attention_fraction = float(\n        (\n            cam\n            *\n            border_mask\n        ).sum()\n        /\n        total_attention\n    )\n\n    y_coordinates, x_coordinates = np.indices(\n        cam.shape\n    )\n\n    center_of_mass_x = float(\n        (\n            cam\n            *\n            x_coordinates\n        ).sum()\n        /\n        total_attention\n        /\n        (\n            TARGET_SIZE\n            -\n            1\n        )\n    )\n\n    center_of_mass_y = float(\n        (\n            cam\n            *\n            y_coordinates\n        ).sum()\n        /\n        total_attention\n        /\n        (\n            TARGET_SIZE\n            -\n            1\n        )\n    )\n\n    probability_distribution = (\n        cam.ravel()\n        /\n        total_attention\n    )\n\n    positive_distribution = probability_distribution[\n        probability_distribution\n        >\n        0\n    ]\n\n    normalized_entropy = float(\n        (\n            -\n            np.sum(\n                positive_distribution\n                *\n                np.log(\n                    positive_distribution\n                )\n            )\n        )\n        /\n        math.log(\n            cam.size\n        )\n    )\n\n    return {\n        \"heatmap_minimum\": float(\n            cam.min()\n        ),\n\n        \"heatmap_maximum\": float(\n            cam.max()\n        ),\n\n        \"heatmap_mean\": float(\n            cam.mean()\n        ),\n\n        \"heatmap_standard_deviation\": float(\n            cam.std()\n        ),\n\n        \"heatmap_nonzero_fraction\": float(\n            np.mean(\n                cam\n                >\n                0\n            )\n        ),\n\n        \"retinal_attention_fraction\": (\n            retinal_attention_fraction\n        ),\n\n        \"outer_10_percent_border_attention_fraction\": (\n            border_attention_fraction\n        ),\n\n        \"attention_center_of_mass_x_normalized\": (\n            center_of_mass_x\n        ),\n\n        \"attention_center_of_mass_y_normalized\": (\n            center_of_mass_y\n        ),\n\n        \"normalized_spatial_entropy\": (\n            normalized_entropy\n        ),\n\n        \"heatmap_finite\": bool(\n            np.isfinite(\n                cam\n            ).all()\n        ),\n\n        \"heatmap_nondegenerate\": bool(\n            cam.max()\n            -\n            cam.min()\n            >\n            HEATMAP_EPSILON\n        ),\n    }\n\n\n# =============================================================================\n# 8. Completed-stage and partial-output protection\n# =============================================================================\n\nif STATE_PATH.exists():\n\n    existing_state = read_json(\n        STATE_PATH\n    )\n\n    if existing_state.get(\n        \"status\"\n    ) == \"complete\":\n\n        raise RuntimeError(\n            \"Step 13B-R is already complete. Do not rerun it.\"\n        )\n\n\nfor stage_directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    FIGURE_DIR,\n]:\n\n    if (\n        stage_directory.exists()\n        and\n        any(\n            path.is_file()\n            for path\n            in stage_directory.rglob(\n                \"*\"\n            )\n        )\n    ):\n\n        raise RuntimeError(\n            \"Partial Step 13B-R outputs already exist:\\n\"\n            f\"{stage_directory}\\n\"\n            \"Do not mix outputs from multiple runs.\"\n        )\n\n\n# =============================================================================\n# 9. Prerequisite verification\n# =============================================================================\n\nfor required_path in [\n    STEP10D_STATE_PATH,\n    STEP13A_STATE_PATH,\n    SELECTION_PATH,\n    FINAL_CHECKPOINT_PATH,\n    FAILED_STAGE_QA_PATH,\n]:\n\n    if not required_path.exists():\n\n        raise FileNotFoundError(\n            f\"Required Step 13B-R evidence missing: {required_path}\"\n        )\n\n\nstep10d_state = read_json(\n    STEP10D_STATE_PATH\n)\n\nstep13a_state = read_json(\n    STEP13A_STATE_PATH\n)\n\n\nif step10d_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 10D final-test stage is incomplete.\"\n    )\n\n\nif int(\n    step10d_state.get(\n        \"test_evaluation_count\",\n        -1,\n    )\n) != 1:\n\n    raise RuntimeError(\n        \"Final-test evaluation count is not exactly one.\"\n    )\n\n\nif step10d_state.get(\n    \"another_test_evaluation_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Final-test evaluation is not formally closed.\"\n    )\n\n\nif step10d_state.get(\n    \"model_change_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Frozen-model protection is not preserved.\"\n    )\n\n\nif step13a_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 13A case selection is incomplete.\"\n    )\n\n\nif step13a_state.get(\n    \"xai_case_selection_locked\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13A XAI selection is not locked.\"\n    )\n\n\nif step13a_state.get(\n    \"case_replacement_after_heatmap_review_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Case replacement is not formally prohibited.\"\n    )\n\n\nif str(\n    step13a_state.get(\n        \"selection_fingerprint_sha256\"\n    )\n) != EXPECTED_SELECTION_FINGERPRINT:\n\n    raise RuntimeError(\n        \"Step 13A selection fingerprint mismatch.\"\n    )\n\n\ncheckpoint_sha256 = sha256_file(\n    FINAL_CHECKPOINT_PATH\n)\n\n\nif checkpoint_sha256 != EXPECTED_CHECKPOINT_SHA256:\n\n    raise RuntimeError(\n        \"Final registered checkpoint SHA-256 mismatch.\"\n    )\n\n\n# =============================================================================\n# 10. Verify diagnostic evidence supporting the amended gate\n# =============================================================================\n\ndiagnostic_df = pd.read_csv(\n    FAILED_STAGE_QA_PATH\n)\n\n\nrequired_diagnostic_columns = {\n    \"selection_order\",\n    \"sample_id\",\n    \"amp_prediction_matches\",\n    \"fp32_prediction_matches\",\n    \"maximum_amp_probability_difference\",\n    \"maximum_fp32_probability_difference\",\n}\n\n\nmissing_diagnostic_columns = (\n    required_diagnostic_columns.difference(\n        diagnostic_df.columns\n    )\n)\n\n\nif missing_diagnostic_columns:\n\n    raise RuntimeError(\n        \"The previous diagnostic table is missing columns: \"\n        f\"{sorted(missing_diagnostic_columns)}\"\n    )\n\n\ndiagnostic_df[\n    \"amp_prediction_matches\"\n] = normalize_bool_series(\n    diagnostic_df[\n        \"amp_prediction_matches\"\n    ]\n)\n\n\ndiagnostic_df[\n    \"fp32_prediction_matches\"\n] = normalize_bool_series(\n    diagnostic_df[\n        \"fp32_prediction_matches\"\n    ]\n)\n\n\nif len(\n    diagnostic_df\n) != EXPECTED_SELECTED_CASES:\n\n    raise RuntimeError(\n        \"Previous diagnostic row count is not 10.\"\n    )\n\n\nif not diagnostic_df[\n    \"amp_prediction_matches\"\n].all():\n\n    raise RuntimeError(\n        \"The previous diagnostic did not reproduce all AMP argmax labels.\"\n    )\n\n\nif not diagnostic_df[\n    \"fp32_prediction_matches\"\n].all():\n\n    raise RuntimeError(\n        \"The previous diagnostic did not reproduce all FP32 argmax labels.\"\n    )\n\n\ndiagnostic_maximum_amp_difference = float(\n    diagnostic_df[\n        \"maximum_amp_probability_difference\"\n    ].max()\n)\n\n\ndiagnostic_maximum_fp32_difference = float(\n    diagnostic_df[\n        \"maximum_fp32_probability_difference\"\n    ].max()\n)\n\n\nif diagnostic_maximum_amp_difference > AMP_PROBABILITY_TOLERANCE:\n\n    raise RuntimeError(\n        \"The revised AMP tolerance does not cover the observed \"\n        \"diagnostic numerical variation.\"\n    )\n\n\nprevious_figure_files = []\n\nif FAILED_STAGE_FIGURE_DIR.exists():\n\n    previous_figure_files = [\n        path\n        for path\n        in FAILED_STAGE_FIGURE_DIR.rglob(\n            \"*\"\n        )\n        if path.is_file()\n    ]\n\n\nif previous_figure_files:\n\n    raise RuntimeError(\n        \"The failed Step 13B directory unexpectedly contains \"\n        \"partial figure files. Do not continue until reviewed.\"\n    )\n\n\n# =============================================================================\n# 11. Verify locked selection table\n# =============================================================================\n\nselected_cases_df = pd.read_csv(\n    SELECTION_PATH\n)\n\n\nrequired_selection_columns = {\n    \"selection_order\",\n    \"case_role\",\n    \"case_category\",\n    \"sample_id\",\n    \"true_grade\",\n    \"true_class\",\n    \"predicted_grade\",\n    \"predicted_class\",\n    \"image_path\",\n    \"selection_method\",\n    \"selected_predicted_confidence\",\n    \"probability_grade_0\",\n    \"probability_grade_1\",\n    \"probability_grade_2\",\n    \"probability_grade_3\",\n    \"probability_grade_4\",\n}\n\n\nmissing_selection_columns = (\n    required_selection_columns.difference(\n        selected_cases_df.columns\n    )\n)\n\n\nif missing_selection_columns:\n\n    raise RuntimeError(\n        \"Step 13A selection table is missing columns: \"\n        f\"{sorted(missing_selection_columns)}\"\n    )\n\n\nselected_cases_df = selected_cases_df.sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\nif len(\n    selected_cases_df\n) != EXPECTED_SELECTED_CASES:\n\n    raise RuntimeError(\n        \"Expected exactly 10 locked XAI cases.\"\n    )\n\n\nif int(\n    (\n        selected_cases_df[\n            \"case_role\"\n        ]\n        ==\n        \"correct\"\n    ).sum()\n) != EXPECTED_CORRECT_CASES:\n\n    raise RuntimeError(\n        \"Correct-case count mismatch.\"\n    )\n\n\nif int(\n    (\n        selected_cases_df[\n            \"case_role\"\n        ]\n        ==\n        \"error\"\n    ).sum()\n) != EXPECTED_ERROR_CASES:\n\n    raise RuntimeError(\n        \"Error-case count mismatch.\"\n    )\n\n\nif selected_cases_df[\n    \"sample_id\"\n].duplicated().any():\n\n    raise RuntimeError(\n        \"Duplicate selected samples detected.\"\n    )\n\n\nfor image_path in selected_cases_df[\n    \"image_path\"\n]:\n\n    if not Path(\n        str(\n            image_path\n        )\n    ).exists():\n\n        raise FileNotFoundError(\n            f\"Selected image is missing: {image_path}\"\n        )\n\n\nobserved_selection_fingerprint = dataframe_fingerprint(\n    selected_cases_df[\n        [\n            \"selection_order\",\n            \"case_category\",\n            \"sample_id\",\n            \"true_grade\",\n            \"predicted_grade\",\n            \"selection_method\",\n            \"image_path\",\n        ]\n    ]\n)\n\n\nif observed_selection_fingerprint != EXPECTED_SELECTION_FINGERPRINT:\n\n    raise RuntimeError(\n        \"Current selection CSV does not reproduce \"\n        \"the locked Step 13A fingerprint.\"\n    )\n\n\n# =============================================================================\n# 12. Load frozen EMA checkpoint\n# =============================================================================\n\ntry:\n\n    checkpoint = torch.load(\n        FINAL_CHECKPOINT_PATH,\n        map_location=\"cpu\",\n        weights_only=False,\n        mmap=True,\n    )\n\nexcept Exception:\n\n    checkpoint = torch.load(\n        FINAL_CHECKPOINT_PATH,\n        map_location=\"cpu\",\n        weights_only=False,\n    )\n\n\ncheckpoint_top_level_keys = (\n    sorted(\n        str(\n            key\n        )\n        for key\n        in checkpoint.keys()\n    )\n    if isinstance(\n        checkpoint,\n        dict,\n    )\n    else\n    []\n)\n\n\nema_source_key, raw_ema_state_dict = find_ema_state_dict(\n    checkpoint\n)\n\n\n(\n    model,\n    state_dict_variant,\n    loaded_state_tensor_count,\n    rejected_load_variants,\n) = load_registered_ema_model(\n    raw_ema_state_dict\n)\n\n\nmodel_parameter_count = int(\n    sum(\n        parameter.numel()\n        for parameter\n        in model.parameters()\n    )\n)\n\n\nif model_parameter_count != EXPECTED_PARAMETER_COUNT:\n\n    raise RuntimeError(\n        \"Registered model parameter-count mismatch: \"\n        f\"{model_parameter_count:,}\"\n    )\n\n\ndevice = torch.device(\n    \"cuda\"\n    if torch.cuda.is_available()\n    else\n    \"cpu\"\n)\n\n\nmodel = model.to(\n    device\n)\n\nmodel.eval()\n\n\nfor parameter in model.parameters():\n\n    parameter.requires_grad_(\n        False\n    )\n\n\ntarget_layer = model.features[\n    -1\n]\n\ntarget_layer_name = (\n    \"features.\"\n    +\n    str(\n        len(\n            model.features\n        )\n        -\n        1\n    )\n)\n\n\ndel raw_ema_state_dict\ndel checkpoint\n\ngc.collect()\n\n\n# =============================================================================\n# 13. Corrected prediction-reproduction gate\n# =============================================================================\n\nprediction_qa_records = []\n\nmaximum_amp_probability_difference = 0.0\nmaximum_fp32_probability_difference = 0.0\nmaximum_saved_row_sum_deviation = 0.0\n\namp_enabled = bool(\n    device.type\n    ==\n    \"cuda\"\n)\n\n\nfor _, row in selected_cases_df.iterrows():\n\n    prepared = prepare_selected_image(\n        Path(\n            str(\n                row[\n                    \"image_path\"\n                ]\n            )\n        )\n    )\n\n    input_batch = (\n        prepared[\n            \"image_tensor\"\n        ]\n        .unsqueeze(\n            0\n        )\n        .to(\n            device\n        )\n    )\n\n\n    saved_probabilities = np.array(\n        [\n            float(\n                row[\n                    f\"probability_grade_{class_index}\"\n                ]\n            )\n            for class_index\n            in range(\n                NUM_CLASSES\n            )\n        ],\n        dtype=np.float64,\n    )\n\n\n    if not np.isfinite(\n        saved_probabilities\n    ).all():\n\n        raise RuntimeError(\n            \"A saved probability vector contains non-finite values.\"\n        )\n\n\n    if (\n        saved_probabilities.min()\n        <\n        -1.0e-6\n        or\n        saved_probabilities.max()\n        >\n        1.0\n        +\n        1.0e-6\n    ):\n\n        raise RuntimeError(\n            \"A saved probability vector is outside [0, 1].\"\n        )\n\n\n    saved_row_sum_deviation = float(\n        abs(\n            saved_probabilities.sum()\n            -\n            1.0\n        )\n    )\n\n\n    maximum_saved_row_sum_deviation = max(\n        maximum_saved_row_sum_deviation,\n        saved_row_sum_deviation,\n    )\n\n\n    if saved_row_sum_deviation > SAVED_ROW_SUM_TOLERANCE:\n\n        raise RuntimeError(\n            \"A saved probability vector does not sum sufficiently \"\n            \"close to one.\"\n        )\n\n\n    saved_prediction = int(\n        row[\n            \"predicted_grade\"\n        ]\n    )\n\n\n    saved_probability_argmax = int(\n        np.argmax(\n            saved_probabilities\n        )\n    )\n\n\n    if saved_probability_argmax != saved_prediction:\n\n        raise RuntimeError(\n            \"Saved probability argmax does not reproduce \"\n            \"the locked predicted grade.\"\n        )\n\n\n    with torch.no_grad():\n\n        if amp_enabled:\n\n            with torch.autocast(\n                device_type=\"cuda\",\n                dtype=torch.float16,\n                enabled=True,\n            ):\n\n                amp_logits = model(\n                    input_batch\n                )\n\n        else:\n\n            amp_logits = model(\n                input_batch\n            )\n\n\n        amp_probabilities = (\n            torch.softmax(\n                amp_logits.float(),\n                dim=1,\n            )[\n                0\n            ]\n            .detach()\n            .cpu()\n            .numpy()\n            .astype(\n                np.float64\n            )\n        )\n\n\n        fp32_logits = model(\n            input_batch.float()\n        )\n\n\n        fp32_probabilities = (\n            torch.softmax(\n                fp32_logits.float(),\n                dim=1,\n            )[\n                0\n            ]\n            .detach()\n            .cpu()\n            .numpy()\n            .astype(\n                np.float64\n            )\n        )\n\n\n    amp_prediction = int(\n        np.argmax(\n            amp_probabilities\n        )\n    )\n\n\n    fp32_prediction = int(\n        np.argmax(\n            fp32_probabilities\n        )\n    )\n\n\n    amp_probability_difference = float(\n        np.max(\n            np.abs(\n                amp_probabilities\n                -\n                saved_probabilities\n            )\n        )\n    )\n\n\n    fp32_probability_difference = float(\n        np.max(\n            np.abs(\n                fp32_probabilities\n                -\n                saved_probabilities\n            )\n        )\n    )\n\n\n    maximum_amp_probability_difference = max(\n        maximum_amp_probability_difference,\n        amp_probability_difference,\n    )\n\n\n    maximum_fp32_probability_difference = max(\n        maximum_fp32_probability_difference,\n        fp32_probability_difference,\n    )\n\n\n    amp_prediction_matches = bool(\n        amp_prediction\n        ==\n        saved_prediction\n    )\n\n\n    fp32_prediction_matches = bool(\n        fp32_prediction\n        ==\n        saved_prediction\n    )\n\n\n    amp_probability_match_passed = bool(\n        amp_probability_difference\n        <=\n        AMP_PROBABILITY_TOLERANCE\n    )\n\n\n    prediction_qa_records.append({\n        \"selection_order\": int(\n            row[\n                \"selection_order\"\n            ]\n        ),\n\n        \"sample_id\": str(\n            row[\n                \"sample_id\"\n            ]\n        ),\n\n        \"true_grade\": int(\n            row[\n                \"true_grade\"\n            ]\n        ),\n\n        \"saved_predicted_grade\": (\n            saved_prediction\n        ),\n\n        \"saved_probability_argmax\": (\n            saved_probability_argmax\n        ),\n\n        \"amp_reproduced_grade\": (\n            amp_prediction\n        ),\n\n        \"fp32_reproduced_grade\": (\n            fp32_prediction\n        ),\n\n        \"saved_argmax_matches_locked_prediction\": (\n            True\n        ),\n\n        \"amp_prediction_matches\": (\n            amp_prediction_matches\n        ),\n\n        \"fp32_prediction_matches\": (\n            fp32_prediction_matches\n        ),\n\n        \"maximum_amp_probability_difference\": (\n            amp_probability_difference\n        ),\n\n        \"maximum_fp32_probability_difference\": (\n            fp32_probability_difference\n        ),\n\n        \"amp_probability_tolerance\": (\n            AMP_PROBABILITY_TOLERANCE\n        ),\n\n        \"amp_probability_match_passed\": (\n            amp_probability_match_passed\n        ),\n\n        \"saved_probability_row_sum\": float(\n            saved_probabilities.sum()\n        ),\n\n        \"saved_probability_row_sum_deviation\": (\n            saved_row_sum_deviation\n        ),\n\n        \"amp_probability_row_sum\": float(\n            amp_probabilities.sum()\n        ),\n\n        \"fp32_probability_row_sum\": float(\n            fp32_probabilities.sum()\n        ),\n\n        \"crop_status\": (\n            prepared[\n                \"crop_status\"\n            ]\n        ),\n    })\n\n\n    del prepared\n    del input_batch\n    del amp_logits\n    del fp32_logits\n\n    gc.collect()\n\n    if device.type == \"cuda\":\n\n        torch.cuda.empty_cache()\n\n\nprediction_qa_df = pd.DataFrame(\n    prediction_qa_records\n)\n\n\nprediction_gate_passed = bool(\n    prediction_qa_df[\n        \"saved_argmax_matches_locked_prediction\"\n    ].all()\n    and\n    prediction_qa_df[\n        \"amp_prediction_matches\"\n    ].all()\n    and\n    prediction_qa_df[\n        \"fp32_prediction_matches\"\n    ].all()\n    and\n    prediction_qa_df[\n        \"amp_probability_match_passed\"\n    ].all()\n)\n\n\nif not prediction_gate_passed:\n\n    print(\n        \"\\nCORRECTED PREDICTION GATE FAILED\"\n    )\n\n    print(\n        prediction_qa_df.to_string(\n            index=False\n        )\n    )\n\n    raise RuntimeError(\n        \"Step 13B-R prediction-reproduction gate failed. \"\n        \"No output directories or Grad-CAM++ maps were created.\"\n    )\n\n\n# Output folders are created only after the corrected gate passes.\n\nfor directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    FIGURE_DIR,\n]:\n\n    directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\natomic_csv_save(\n    prediction_qa_df,\n    PREDICTION_QA_PATH,\n)\n\n\n# =============================================================================\n# 14. Save gate amendment and model/preprocessing evidence\n# =============================================================================\n\ngate_amendment_record = {\n    \"step\": (\n        \"STEP_13B_R_NUMERICAL_EQUIVALENCE_GATE_AMENDMENT\"\n    ),\n\n    \"status\": (\n        \"locked\"\n    ),\n\n    \"created_utc\": (\n        utc_now()\n    ),\n\n    \"previous_gate\": {\n        \"amp_probability_tolerance\": (\n            0.0005\n        ),\n\n        \"result\": (\n            \"failed_due_to_numerical_probability_variation\"\n        ),\n\n        \"amp_argmax_matches\": (\n            10\n        ),\n\n        \"fp32_argmax_matches\": (\n            10\n        ),\n\n        \"maximum_observed_amp_probability_difference\": (\n            diagnostic_maximum_amp_difference\n        ),\n\n        \"maximum_observed_fp32_probability_difference\": (\n            diagnostic_maximum_fp32_difference\n        ),\n\n        \"partial_heatmaps_generated\": (\n            False\n        ),\n    },\n\n    \"corrected_gate\": {\n        \"amp_probability_tolerance\": (\n            AMP_PROBABILITY_TOLERANCE\n        ),\n\n        \"amp_argmax_match_required\": (\n            True\n        ),\n\n        \"fp32_argmax_match_required\": (\n            True\n        ),\n\n        \"saved_probability_argmax_match_required\": (\n            True\n        ),\n\n        \"saved_probability_validity_required\": (\n            True\n        ),\n\n        \"fp32_probability_difference_used_as_failure_gate\": (\n            False\n        ),\n\n        \"rationale\": (\n            \"The saved predictions were produced under AMP, \"\n            \"whereas Grad-CAM++ requires an FP32 gradient pass. \"\n            \"All AMP and FP32 class decisions were identical. \"\n            \"The revised threshold only accommodates observed \"\n            \"sub-percentage numerical variation and is not used \"\n            \"for training, calibration, threshold optimization \"\n            \"or model selection.\"\n        ),\n    },\n\n    \"current_gate_result\": {\n        \"passed\": (\n            prediction_gate_passed\n        ),\n\n        \"maximum_amp_probability_difference\": (\n            maximum_amp_probability_difference\n        ),\n\n        \"maximum_fp32_probability_difference\": (\n            maximum_fp32_probability_difference\n        ),\n\n        \"maximum_saved_probability_row_sum_deviation\": (\n            maximum_saved_row_sum_deviation\n        ),\n    },\n}\n\n\natomic_json_save(\n    gate_amendment_record,\n    DIAGNOSTIC_AMENDMENT_PATH,\n)\n\n\nmodel_load_evidence = {\n    \"step\": (\n        \"STEP_13B_R_FROZEN_MODEL_RESTORATION\"\n    ),\n\n    \"loaded_utc\": (\n        utc_now()\n    ),\n\n    \"checkpoint_path\": str(\n        FINAL_CHECKPOINT_PATH\n    ),\n\n    \"checkpoint_sha256\": (\n        checkpoint_sha256\n    ),\n\n    \"checkpoint_size_bytes\": int(\n        FINAL_CHECKPOINT_PATH.stat().st_size\n    ),\n\n    \"checkpoint_top_level_keys\": (\n        checkpoint_top_level_keys\n    ),\n\n    \"ema_source_key\": (\n        ema_source_key\n    ),\n\n    \"state_dict_normalization_variant\": (\n        state_dict_variant\n    ),\n\n    \"loaded_state_tensor_count\": (\n        loaded_state_tensor_count\n    ),\n\n    \"strict_state_dict_loading_passed\": (\n        True\n    ),\n\n    \"registered_model\": (\n        \"torchvision EfficientNet-B0\"\n    ),\n\n    \"num_classes\": (\n        NUM_CLASSES\n    ),\n\n    \"parameter_count\": (\n        model_parameter_count\n    ),\n\n    \"target_layer_name\": (\n        target_layer_name\n    ),\n\n    \"target_layer_type\": (\n        type(\n            target_layer\n        ).__name__\n    ),\n\n    \"device\": str(\n        device\n    ),\n\n    \"ema_weights_used\": (\n        True\n    ),\n\n    \"model_training_performed\": (\n        False\n    ),\n\n    \"model_parameters_modified\": (\n        False\n    ),\n}\n\n\natomic_json_save(\n    model_load_evidence,\n    MODEL_LOAD_EVIDENCE_PATH,\n)\n\n\npreprocessing_evidence = {\n    \"step\": (\n        \"STEP_13B_R_EXACT_LOCKED_PREPROCESSING\"\n    ),\n\n    \"variant\": (\n        \"always_clahe\"\n    ),\n\n    \"input_size\": (\n        TARGET_SIZE\n    ),\n\n    \"retinal_foreground_threshold\": (\n        7\n    ),\n\n    \"morphological_kernel\": (\n        \"9x9 elliptical close\"\n    ),\n\n    \"minimum_bounding_area_ratio\": (\n        0.10\n    ),\n\n    \"crop_margin_fraction\": (\n        0.02\n    ),\n\n    \"square_padding_value\": (\n        0\n    ),\n\n    \"downsampling_interpolation\": (\n        \"cv2.INTER_AREA\"\n    ),\n\n    \"upsampling_interpolation\": (\n        \"cv2.INTER_CUBIC\"\n    ),\n\n    \"clahe_colour_space\": (\n        \"LAB lightness channel\"\n    ),\n\n    \"clahe_clip_limit\": (\n        1.5\n    ),\n\n    \"clahe_tile_grid_size\": [\n        8,\n        8,\n    ],\n\n    \"normalization_mean\": (\n        IMAGENET_MEAN.tolist()\n    ),\n\n    \"normalization_std\": (\n        IMAGENET_STD.tolist()\n    ),\n\n    \"preprocessing_tuning_performed\": (\n        False\n    ),\n}\n\n\natomic_json_save(\n    preprocessing_evidence,\n    PREPROCESSING_EVIDENCE_PATH,\n)\n\n\n# =============================================================================\n# 15. Generate locked Grad-CAM++ maps\n# =============================================================================\n\ncapture = ActivationGradientCapture(\n    target_layer\n)\n\nattribution_records = []\ncase_output_records = []\ngenerated_figure_files = []\n\n\nfor _, row in selected_cases_df.iterrows():\n\n    selection_order = int(\n        row[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        row[\n            \"sample_id\"\n        ]\n    )\n\n    true_grade = int(\n        row[\n            \"true_grade\"\n        ]\n    )\n\n    predicted_grade = int(\n        row[\n            \"predicted_grade\"\n        ]\n    )\n\n    case_role = str(\n        row[\n            \"case_role\"\n        ]\n    )\n\n\n    case_directory = (\n        FIGURE_DIR\n        /\n        (\n            f\"{selection_order:02d}_\"\n            f\"{sample_id}\"\n        )\n    )\n\n    case_directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\n    prepared = prepare_selected_image(\n        Path(\n            str(\n                row[\n                    \"image_path\"\n                ]\n            )\n        )\n    )\n\n\n    original_preview_path = (\n        case_directory\n        /\n        \"01_original_preview.png\"\n    )\n\n    crop_resize_path = (\n        case_directory\n        /\n        \"02_retinal_crop_resize.png\"\n    )\n\n    model_input_path = (\n        case_directory\n        /\n        \"03_always_clahe_model_input.png\"\n    )\n\n\n    save_rgb_png(\n        prepared[\n            \"original_preview\"\n        ],\n        original_preview_path,\n    )\n\n    save_rgb_png(\n        prepared[\n            \"crop_resize_rgb\"\n        ],\n        crop_resize_path,\n    )\n\n    save_rgb_png(\n        prepared[\n            \"model_input_rgb\"\n        ],\n        model_input_path,\n    )\n\n\n    generated_figure_files.extend([\n        original_preview_path,\n        crop_resize_path,\n        model_input_path,\n    ])\n\n\n    target_specs = [\n        (\n            \"predicted_class\",\n            predicted_grade,\n        ),\n    ]\n\n\n    if case_role == \"error\":\n\n        target_specs.append(\n            (\n                \"reference_class\",\n                true_grade,\n            )\n        )\n\n\n    predicted_overlay_path = \"\"\n    reference_overlay_path = \"\"\n\n\n    for target_type, target_grade in target_specs:\n\n        result = generate_gradcam_plus_plus(\n            model=model,\n            capture=capture,\n            image_tensor=prepared[\n                \"image_tensor\"\n            ],\n            target_grade=target_grade,\n            device=device,\n        )\n\n\n        if result[\n            \"predicted_grade\"\n        ] != predicted_grade:\n\n            raise RuntimeError(\n                \"The FP32 Grad-CAM++ forward prediction changed \"\n                f\"for selected sample {sample_id}.\"\n            )\n\n\n        cam = result[\n            \"cam\"\n        ]\n\n\n        heatmap_rgb = colourize_heatmap(\n            cam\n        )\n\n\n        overlay_rgb = create_overlay(\n            prepared[\n                \"model_input_rgb\"\n            ],\n            heatmap_rgb,\n        )\n\n\n        if target_type == \"predicted_class\":\n\n            cam_array_path = (\n                case_directory\n                /\n                \"04_predicted_class_gradcampp.npy\"\n            )\n\n            heatmap_path = (\n                case_directory\n                /\n                \"05_predicted_class_heatmap.png\"\n            )\n\n            overlay_path = (\n                case_directory\n                /\n                \"06_predicted_class_overlay.png\"\n            )\n\n            predicted_overlay_path = str(\n                overlay_path\n            )\n\n        else:\n\n            cam_array_path = (\n                case_directory\n                /\n                \"07_reference_class_gradcampp.npy\"\n            )\n\n            heatmap_path = (\n                case_directory\n                /\n                \"08_reference_class_heatmap.png\"\n            )\n\n            overlay_path = (\n                case_directory\n                /\n                \"09_reference_class_overlay.png\"\n            )\n\n            reference_overlay_path = str(\n                overlay_path\n            )\n\n\n        atomic_numpy_save(\n            cam,\n            cam_array_path,\n        )\n\n        save_rgb_png(\n            heatmap_rgb,\n            heatmap_path,\n        )\n\n        save_rgb_png(\n            overlay_rgb,\n            overlay_path,\n        )\n\n\n        generated_figure_files.extend([\n            cam_array_path,\n            heatmap_path,\n            overlay_path,\n        ])\n\n\n        attribution_statistics = compute_attribution_statistics(\n            cam,\n            prepared[\n                \"crop_resize_rgb\"\n            ],\n        )\n\n\n        attribution_records.append({\n            \"selection_order\": (\n                selection_order\n            ),\n\n            \"sample_id\": (\n                sample_id\n            ),\n\n            \"case_role\": (\n                case_role\n            ),\n\n            \"case_category\": str(\n                row[\n                    \"case_category\"\n                ]\n            ),\n\n            \"true_grade\": (\n                true_grade\n            ),\n\n            \"true_class\": (\n                CLASS_NAMES[\n                    true_grade\n                ]\n            ),\n\n            \"locked_predicted_grade\": (\n                predicted_grade\n            ),\n\n            \"locked_predicted_class\": (\n                CLASS_NAMES[\n                    predicted_grade\n                ]\n            ),\n\n            \"target_type\": (\n                target_type\n            ),\n\n            \"target_grade\": (\n                target_grade\n            ),\n\n            \"target_class\": (\n                CLASS_NAMES[\n                    target_grade\n                ]\n            ),\n\n            \"target_logit\": float(\n                result[\n                    \"logits\"\n                ][\n                    target_grade\n                ]\n            ),\n\n            \"target_probability_fp32\": float(\n                result[\n                    \"probabilities\"\n                ][\n                    target_grade\n                ]\n            ),\n\n            \"fp32_argmax_grade\": int(\n                result[\n                    \"predicted_grade\"\n                ]\n            ),\n\n            \"target_layer\": (\n                target_layer_name\n            ),\n\n            \"activation_shape\": (\n                \"x\".join(\n                    str(\n                        value\n                    )\n                    for value\n                    in result[\n                        \"activation_shape\"\n                    ]\n                )\n            ),\n\n            \"gradient_shape\": (\n                \"x\".join(\n                    str(\n                        value\n                    )\n                    for value\n                    in result[\n                        \"gradient_shape\"\n                    ]\n                )\n            ),\n\n            \"cam_array_path\": str(\n                cam_array_path\n            ),\n\n            \"heatmap_path\": str(\n                heatmap_path\n            ),\n\n            \"overlay_path\": str(\n                overlay_path\n            ),\n\n            **attribution_statistics,\n        })\n\n\n        del result\n        del cam\n        del heatmap_rgb\n        del overlay_rgb\n\n        gc.collect()\n\n        if device.type == \"cuda\":\n\n            torch.cuda.empty_cache()\n\n\n    case_output_records.append({\n        \"selection_order\": (\n            selection_order\n        ),\n\n        \"sample_id\": (\n            sample_id\n        ),\n\n        \"case_role\": (\n            case_role\n        ),\n\n        \"case_category\": str(\n            row[\n                \"case_category\"\n            ]\n        ),\n\n        \"true_class\": (\n            CLASS_NAMES[\n                true_grade\n            ]\n        ),\n\n        \"predicted_class\": (\n            CLASS_NAMES[\n                predicted_grade\n            ]\n        ),\n\n        \"crop_status\": (\n            prepared[\n                \"crop_status\"\n            ]\n        ),\n\n        \"original_preview_path\": str(\n            original_preview_path\n        ),\n\n        \"retinal_crop_resize_path\": str(\n            crop_resize_path\n        ),\n\n        \"model_input_path\": str(\n            model_input_path\n        ),\n\n        \"predicted_class_overlay_path\": (\n            predicted_overlay_path\n        ),\n\n        \"reference_class_overlay_path\": (\n            reference_overlay_path\n        ),\n\n        \"map_count\": (\n            2\n            if case_role == \"error\"\n            else\n            1\n        ),\n\n        \"selected_after_model_lock\": (\n            True\n        ),\n\n        \"used_for_model_selection\": (\n            False\n        ),\n    })\n\n\n    del prepared\n\n    gc.collect()\n\n    if device.type == \"cuda\":\n\n        torch.cuda.empty_cache()\n\n\ncapture.remove()\n\n\nattribution_df = pd.DataFrame(\n    attribution_records\n).sort_values(\n    [\n        \"selection_order\",\n        \"target_type\",\n    ]\n).reset_index(\n    drop=True\n)\n\n\ncase_output_index_df = pd.DataFrame(\n    case_output_records\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\npredicted_map_count = int(\n    (\n        attribution_df[\n            \"target_type\"\n        ]\n        ==\n        \"predicted_class\"\n    ).sum()\n)\n\n\nreference_map_count = int(\n    (\n        attribution_df[\n            \"target_type\"\n        ]\n        ==\n        \"reference_class\"\n    ).sum()\n)\n\n\ntotal_map_count = int(\n    len(\n        attribution_df\n    )\n)\n\n\nif predicted_map_count != EXPECTED_PREDICTED_MAPS:\n\n    raise RuntimeError(\n        \"Predicted-class Grad-CAM++ map count mismatch.\"\n    )\n\n\nif reference_map_count != EXPECTED_REFERENCE_MAPS:\n\n    raise RuntimeError(\n        \"Reference-class Grad-CAM++ map count mismatch.\"\n    )\n\n\nif total_map_count != EXPECTED_TOTAL_MAPS:\n\n    raise RuntimeError(\n        \"Total Grad-CAM++ map count mismatch.\"\n    )\n\n\nall_heatmaps_finite = bool(\n    attribution_df[\n        \"heatmap_finite\"\n    ].all()\n)\n\n\nall_heatmaps_nondegenerate = bool(\n    attribution_df[\n        \"heatmap_nondegenerate\"\n    ].all()\n)\n\n\nif not all_heatmaps_finite:\n\n    raise RuntimeError(\n        \"One or more Grad-CAM++ maps contain non-finite values.\"\n    )\n\n\nif not all_heatmaps_nondegenerate:\n\n    raise RuntimeError(\n        \"One or more Grad-CAM++ maps are degenerate.\"\n    )\n\n\natomic_csv_save(\n    attribution_df,\n    ATTRIBUTION_METRICS_PATH,\n)\n\n\natomic_csv_save(\n    case_output_index_df,\n    CASE_OUTPUT_INDEX_PATH,\n)\n\n\n# =============================================================================\n# 16. Publication-ready case table\n# =============================================================================\n\npredicted_maps_df = attribution_df[\n    attribution_df[\n        \"target_type\"\n    ]\n    ==\n    \"predicted_class\"\n].copy()\n\n\npublication_case_df = selected_cases_df[\n    [\n        \"selection_order\",\n        \"case_role\",\n        \"case_category\",\n        \"sample_id\",\n        \"true_class\",\n        \"predicted_class\",\n        \"selected_predicted_confidence\",\n    ]\n].merge(\n    predicted_maps_df[\n        [\n            \"selection_order\",\n            \"target_probability_fp32\",\n            \"retinal_attention_fraction\",\n            \"outer_10_percent_border_attention_fraction\",\n            \"normalized_spatial_entropy\",\n            \"overlay_path\",\n        ]\n    ],\n    on=\"selection_order\",\n    how=\"left\",\n)\n\n\npublication_case_df.columns = [\n    \"Order\",\n    \"Case role\",\n    \"Case category\",\n    \"Image ID\",\n    \"Reference grade\",\n    \"Locked predicted grade\",\n    \"Saved predicted confidence\",\n    \"FP32 predicted-class probability\",\n    \"Retinal attention fraction\",\n    \"Outer-border attention fraction\",\n    \"Normalized spatial entropy\",\n    \"Predicted-class overlay\",\n]\n\n\natomic_csv_save(\n    publication_case_df,\n    PUBLICATION_CASE_TABLE_PATH,\n)\n\n\n# =============================================================================\n# 17. Formal summary\n# =============================================================================\n\nmean_retinal_attention = float(\n    predicted_maps_df[\n        \"retinal_attention_fraction\"\n    ].mean()\n)\n\n\nmean_border_attention = float(\n    predicted_maps_df[\n        \"outer_10_percent_border_attention_fraction\"\n    ].mean()\n)\n\n\nmaximum_border_index = (\n    predicted_maps_df[\n        \"outer_10_percent_border_attention_fraction\"\n    ].idxmax()\n)\n\n\nmaximum_border_attention = float(\n    predicted_maps_df.loc[\n        maximum_border_index,\n        \"outer_10_percent_border_attention_fraction\",\n    ]\n)\n\n\nmaximum_border_case = str(\n    predicted_maps_df.loc[\n        maximum_border_index,\n        \"sample_id\",\n    ]\n)\n\n\nsummary_record = {\n    \"step\": (\n        \"STEP_13B_R_CORRECTED_LOCKED_GRADCAM_PLUS_PLUS\"\n    ),\n\n    \"status\": (\n        \"completed\"\n    ),\n\n    \"completed_utc\": (\n        utc_now()\n    ),\n\n    \"checkpoint\": {\n        \"path\": str(\n            FINAL_CHECKPOINT_PATH\n        ),\n\n        \"sha256\": (\n            checkpoint_sha256\n        ),\n\n        \"ema_source_key\": (\n            ema_source_key\n        ),\n\n        \"state_dict_variant\": (\n            state_dict_variant\n        ),\n\n        \"parameter_count\": (\n            model_parameter_count\n        ),\n    },\n\n    \"selection\": {\n        \"selected_case_count\": (\n            EXPECTED_SELECTED_CASES\n        ),\n\n        \"selection_fingerprint_sha256\": (\n            observed_selection_fingerprint\n        ),\n\n        \"case_replacement_after_review_allowed\": (\n            False\n        ),\n    },\n\n    \"prediction_reproduction\": {\n        \"gate_passed\": (\n            prediction_gate_passed\n        ),\n\n        \"amp_argmax_matches\": int(\n            prediction_qa_df[\n                \"amp_prediction_matches\"\n            ].sum()\n        ),\n\n        \"fp32_argmax_matches\": int(\n            prediction_qa_df[\n                \"fp32_prediction_matches\"\n            ].sum()\n        ),\n\n        \"maximum_amp_probability_difference\": (\n            maximum_amp_probability_difference\n        ),\n\n        \"amp_probability_tolerance\": (\n            AMP_PROBABILITY_TOLERANCE\n        ),\n\n        \"maximum_fp32_probability_difference\": (\n            maximum_fp32_probability_difference\n        ),\n\n        \"maximum_saved_probability_row_sum_deviation\": (\n            maximum_saved_row_sum_deviation\n        ),\n    },\n\n    \"gradcam_plus_plus\": {\n        \"target_layer\": (\n            target_layer_name\n        ),\n\n        \"predicted_class_map_count\": (\n            predicted_map_count\n        ),\n\n        \"reference_class_map_count\": (\n            reference_map_count\n        ),\n\n        \"total_map_count\": (\n            total_map_count\n        ),\n\n        \"all_heatmaps_finite\": (\n            all_heatmaps_finite\n        ),\n\n        \"all_heatmaps_nondegenerate\": (\n            all_heatmaps_nondegenerate\n        ),\n\n        \"mean_predicted_map_retinal_attention_fraction\": (\n            mean_retinal_attention\n        ),\n\n        \"mean_predicted_map_border_attention_fraction\": (\n            mean_border_attention\n        ),\n\n        \"maximum_predicted_map_border_attention_fraction\": (\n            maximum_border_attention\n        ),\n\n        \"maximum_border_attention_sample_id\": (\n            maximum_border_case\n        ),\n    },\n\n    \"safety\": {\n        \"selected_images_decoded\": (\n            EXPECTED_SELECTED_CASES\n        ),\n\n        \"new_training_performed\": (\n            False\n        ),\n\n        \"optimizer_created\": (\n            False\n        ),\n\n        \"full_validation_evaluation_performed\": (\n            False\n        ),\n\n        \"full_final_test_evaluation_performed\": (\n            False\n        ),\n\n        \"performance_metrics_recomputed\": (\n            False\n        ),\n\n        \"threshold_tuning_performed\": (\n            False\n        ),\n\n        \"model_change_performed\": (\n            False\n        ),\n    },\n\n    \"interpretation_constraint\": (\n        \"Grad-CAM++ localizes image regions contributing \"\n        \"to a selected class score. It does not establish \"\n        \"that highlighted regions are clinically verified \"\n        \"lesions or provide lesion segmentation.\"\n    ),\n\n    \"next_stage\": (\n        \"STEP_13C_XAI_VISUAL_QA_AND_PUBLICATION_FIGURE\"\n    ),\n}\n\n\natomic_json_save(\n    summary_record,\n    SUMMARY_PATH,\n)\n\n\n# =============================================================================\n# 18. Manuscript method note\n# =============================================================================\n\nmanuscript_note = f\"\"\"\nSTEP 13B-R — CORRECTED GRAD-CAM++ EXPLANATION GENERATION\n\nGrad-CAM++ explanations were generated from the frozen exponential\nmoving-average weights of the registered EfficientNet-B0 model. The\ncheckpoint SHA-256 was:\n\n{checkpoint_sha256}\n\nThe final EfficientNet feature block ({target_layer_name}) was used as\nthe attribution layer. The exact locked Always-CLAHE preprocessing\npipeline was restored, including retinal-field cropping, square\npadding, 384 x 384 resizing, mild LAB-CLAHE enhancement and ImageNet\nnormalization.\n\nBefore attribution generation, all ten locked predictions were\nreproduced under both AMP and FP32 inference. AMP and FP32 argmax\npredictions matched the saved locked decisions in 10 of 10 cases.\nThe maximum AMP probability difference was\n{maximum_amp_probability_difference:.8f}, below the documented\nnumerical-equivalence limit of {AMP_PROBABILITY_TOLERANCE:.4f}.\n\nThe numerical-equivalence amendment only accommodated small\nprecision-related differences between saved AMP probabilities,\nsingle-case AMP inference and the FP32 gradient computation required\nfor Grad-CAM++. It did not alter any class prediction, model weight,\nthreshold, calibration rule, model-selection decision or performance\nmetric.\n\nA predicted-class Grad-CAM++ map was generated for each of the ten\npreselected cases. For the five misclassified cases, an additional\nreference-class map was generated. In total, {total_map_count}\nattribution maps were produced.\n\nThe cases were locked before attribution review. No case replacement,\nnew training, threshold optimization, full-validation evaluation,\nfull-test evaluation or model modification was performed. Grad-CAM++\nis interpreted as a model-attribution method and not as validated\nlesion segmentation.\n\"\"\".strip()\n\n\natomic_text_save(\n    manuscript_note,\n    MANUSCRIPT_NOTE_PATH,\n)\n\n\n# =============================================================================\n# 19. Formal state\n# =============================================================================\n\nstate_record = {\n    \"step\": (\n        \"STEP_13B_R_CORRECTED_LOCKED_GRADCAM_PLUS_PLUS\"\n    ),\n\n    \"status\": (\n        \"complete\"\n    ),\n\n    \"updated_utc\": (\n        utc_now()\n    ),\n\n    \"checkpoint_sha256\": (\n        checkpoint_sha256\n    ),\n\n    \"ema_weights_used\": (\n        True\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        observed_selection_fingerprint\n    ),\n\n    \"selected_case_count\": (\n        EXPECTED_SELECTED_CASES\n    ),\n\n    \"selected_images_decoded\": (\n        EXPECTED_SELECTED_CASES\n    ),\n\n    \"prediction_reproduction_gate_passed\": (\n        prediction_gate_passed\n    ),\n\n    \"corrected_amp_probability_tolerance\": (\n        AMP_PROBABILITY_TOLERANCE\n    ),\n\n    \"gradcam_plus_plus_generated\": (\n        True\n    ),\n\n    \"predicted_class_map_count\": (\n        predicted_map_count\n    ),\n\n    \"reference_class_map_count\": (\n        reference_map_count\n    ),\n\n    \"total_map_count\": (\n        total_map_count\n    ),\n\n    \"all_heatmaps_finite\": (\n        all_heatmaps_finite\n    ),\n\n    \"all_heatmaps_nondegenerate\": (\n        all_heatmaps_nondegenerate\n    ),\n\n    \"case_replacement_after_heatmap_review_allowed\": (\n        False\n    ),\n\n    \"new_training_performed\": (\n        False\n    ),\n\n    \"optimizer_created\": (\n        False\n    ),\n\n    \"full_validation_evaluation_performed\": (\n        False\n    ),\n\n    \"full_final_test_evaluation_performed\": (\n        False\n    ),\n\n    \"performance_metrics_recomputed\": (\n        False\n    ),\n\n    \"model_change_allowed\": (\n        False\n    ),\n\n    \"another_validation_evaluation_allowed\": (\n        False\n    ),\n\n    \"another_test_evaluation_allowed\": (\n        False\n    ),\n\n    \"authoritative_attribution_table\": str(\n        ATTRIBUTION_METRICS_PATH\n    ),\n\n    \"authoritative_case_output_index\": str(\n        CASE_OUTPUT_INDEX_PATH\n    ),\n\n    \"figure_directory\": str(\n        FIGURE_DIR\n    ),\n\n    \"next_stage\": (\n        \"STEP_13C_XAI_VISUAL_QA_AND_PUBLICATION_FIGURE\"\n    ),\n}\n\n\natomic_json_save(\n    state_record,\n    STATE_PATH,\n)\n\n\n# =============================================================================\n# 20. Manifest and verified backup\n# =============================================================================\n\nfigure_files = sorted(\n    path\n    for path\n    in FIGURE_DIR.rglob(\n        \"*\"\n    )\n    if path.is_file()\n)\n\n\nmanifest_sources = [\n    STEP10D_STATE_PATH,\n    STEP13A_STATE_PATH,\n    SELECTION_PATH,\n    FAILED_STAGE_QA_PATH,\n    PREDICTION_QA_PATH,\n    ATTRIBUTION_METRICS_PATH,\n    CASE_OUTPUT_INDEX_PATH,\n    DIAGNOSTIC_AMENDMENT_PATH,\n    MODEL_LOAD_EVIDENCE_PATH,\n    PREPROCESSING_EVIDENCE_PATH,\n    PUBLICATION_CASE_TABLE_PATH,\n    SUMMARY_PATH,\n    MANUSCRIPT_NOTE_PATH,\n    STATE_PATH,\n    *figure_files,\n]\n\n\nmanifest_records = []\n\n\nfor source_path in manifest_sources:\n\n    if not source_path.exists():\n\n        raise FileNotFoundError(\n            f\"Step 13B-R manifest source missing: {source_path}\"\n        )\n\n    manifest_records.append({\n        \"relative_path\": str(\n            source_path.relative_to(\n                PROJECT\n            )\n        ),\n\n        \"size_bytes\": int(\n            source_path.stat().st_size\n        ),\n\n        \"sha256\": sha256_file(\n            source_path\n        ),\n    })\n\n\natomic_csv_save(\n    pd.DataFrame(\n        manifest_records\n    ),\n    MANIFEST_PATH,\n)\n\n\nbackup_members = create_verified_zip(\n    BACKUP_PATH,\n    [\n        PREDICTION_QA_PATH,\n        ATTRIBUTION_METRICS_PATH,\n        CASE_OUTPUT_INDEX_PATH,\n        DIAGNOSTIC_AMENDMENT_PATH,\n        MODEL_LOAD_EVIDENCE_PATH,\n        PREPROCESSING_EVIDENCE_PATH,\n        PUBLICATION_CASE_TABLE_PATH,\n        SUMMARY_PATH,\n        MANUSCRIPT_NOTE_PATH,\n        STATE_PATH,\n        MANIFEST_PATH,\n        *figure_files,\n    ],\n)\n\n\n# =============================================================================\n# 21. Release GPU memory\n# =============================================================================\n\nmodel.zero_grad(\n    set_to_none=True\n)\n\nmodel = model.cpu()\n\ndel model\ndel target_layer\n\ngc.collect()\n\nif torch.cuda.is_available():\n\n    torch.cuda.empty_cache()\n\n\n# =============================================================================\n# 22. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"STEP 13B-R — CORRECTED LOCKED \"\n    \"GRAD-CAM++ GENERATION COMPLETED\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nEVIDENCE AND MODEL SAFETY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"New training performed                :\",\n    False\n)\n\nprint(\n    \"Optimizer created                     :\",\n    False\n)\n\nprint(\n    \"Selected-case XAI inference performed :\",\n    True\n)\n\nprint(\n    \"Selected images decoded               :\",\n    EXPECTED_SELECTED_CASES\n)\n\nprint(\n    \"Full validation evaluation performed  :\",\n    False\n)\n\nprint(\n    \"Full final-test evaluation performed  :\",\n    False\n)\n\nprint(\n    \"Performance metrics recomputed        :\",\n    False\n)\n\nprint(\n    \"Frozen model changed                  :\",\n    False\n)\n\nprint(\n    \"Case replacement after review allowed :\",\n    False\n)\n\n\nprint(\n    \"\\nMODEL RESTORATION\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Checkpoint SHA-256                    :\",\n    checkpoint_sha256\n)\n\nprint(\n    \"EMA source key                        :\",\n    ema_source_key\n)\n\nprint(\n    \"State-dict normalization              :\",\n    state_dict_variant\n)\n\nprint(\n    \"Strict state-dict loading passed      :\",\n    True\n)\n\nprint(\n    \"Registered parameters                 :\",\n    f\"{model_parameter_count:,}\"\n)\n\nprint(\n    \"Grad-CAM++ target layer               :\",\n    target_layer_name\n)\n\nprint(\n    \"Device                                :\",\n    device\n)\n\n\nprint(\n    \"\\nCORRECTED PREDICTION-REPRODUCTION GATE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Locked selected cases                 :\",\n    len(\n        selected_cases_df\n    )\n)\n\nprint(\n    \"Saved probability argmax matches      :\",\n    int(\n        prediction_qa_df[\n            \"saved_argmax_matches_locked_prediction\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        prediction_qa_df\n    )\n)\n\nprint(\n    \"AMP argmax matches                    :\",\n    int(\n        prediction_qa_df[\n            \"amp_prediction_matches\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        prediction_qa_df\n    )\n)\n\nprint(\n    \"FP32 argmax matches                   :\",\n    int(\n        prediction_qa_df[\n            \"fp32_prediction_matches\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        prediction_qa_df\n    )\n)\n\nprint(\n    \"AMP numerical-equivalence limit       :\",\n    f\"{AMP_PROBABILITY_TOLERANCE:.6f}\"\n)\n\nprint(\n    \"Maximum AMP probability difference    :\",\n    f\"{maximum_amp_probability_difference:.8f}\"\n)\n\nprint(\n    \"Maximum FP32 probability difference   :\",\n    f\"{maximum_fp32_probability_difference:.8f}\"\n)\n\nprint(\n    \"Prediction-reproduction gate passed   :\",\n    prediction_gate_passed\n)\n\n\nprint(\n    \"\\nGRAD-CAM++ OUTPUT\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Predicted-class maps                  :\",\n    predicted_map_count\n)\n\nprint(\n    \"Reference-class error maps            :\",\n    reference_map_count\n)\n\nprint(\n    \"Total attribution maps                :\",\n    total_map_count\n)\n\nprint(\n    \"All heatmaps finite                   :\",\n    all_heatmaps_finite\n)\n\nprint(\n    \"All heatmaps nondegenerate            :\",\n    all_heatmaps_nondegenerate\n)\n\nprint(\n    \"Mean retinal attention fraction       :\",\n    f\"{mean_retinal_attention:.6f}\"\n)\n\nprint(\n    \"Mean outer-border attention fraction  :\",\n    f\"{mean_border_attention:.6f}\"\n)\n\nprint(\n    \"Maximum border-attention case         :\",\n    maximum_border_case\n)\n\nprint(\n    \"Maximum border-attention fraction     :\",\n    f\"{maximum_border_attention:.6f}\"\n)\n\n\nprint(\n    \"\\nATTRIBUTION SUMMARY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_attribution = attribution_df[\n    [\n        \"selection_order\",\n        \"sample_id\",\n        \"case_role\",\n        \"target_type\",\n        \"target_class\",\n        \"target_probability_fp32\",\n        \"retinal_attention_fraction\",\n        \"outer_10_percent_border_attention_fraction\",\n        \"normalized_spatial_entropy\",\n    ]\n].copy()\n\n\nfor column in [\n    \"target_probability_fp32\",\n    \"retinal_attention_fraction\",\n    \"outer_10_percent_border_attention_fraction\",\n    \"normalized_spatial_entropy\",\n]:\n\n    display_attribution[\n        column\n    ] = display_attribution[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_attribution.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nSAVED OUTPUTS\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Figure directory                     :\",\n    FIGURE_DIR\n)\n\nprint(\n    \"Prediction QA                         :\",\n    PREDICTION_QA_PATH\n)\n\nprint(\n    \"Attribution metrics                   :\",\n    ATTRIBUTION_METRICS_PATH\n)\n\nprint(\n    \"Case output index                     :\",\n    CASE_OUTPUT_INDEX_PATH\n)\n\nprint(\n    \"Publication case table                :\",\n    PUBLICATION_CASE_TABLE_PATH\n)\n\n\nprint(\n    \"\\nBACKUP\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup path                           :\",\n    BACKUP_PATH\n)\n\nprint(\n    \"Backup members                        :\",\n    len(\n        backup_members\n    )\n)\n\nprint(\n    \"ZIP integrity passed                  :\",\n    True\n)\n\n\nprint(\n    \"\\nNEXT STAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"READY FOR STEP 13C — XAI VISUAL QA \"\n    \"AND PUBLICATION FIGURE\"\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T15:59:08.169939Z","iopub.execute_input":"2026-07-18T15:59:08.170557Z","iopub.status.idle":"2026-07-18T15:59:23.133052Z","shell.execute_reply.started":"2026-07-18T15:59:08.170523Z","shell.execute_reply":"2026-07-18T15:59:23.132341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 13C — XAI VISUAL QA AND PUBLICATION FIGURE GENERATION\n#\n# Uses only the already-saved Step 13B-R outputs.\n#\n# Analyses:\n#   1. File-integrity and visual-quality QA\n#   2. Heatmap normalization and non-degeneracy verification\n#   3. Predicted-class versus reference-class CAM similarity\n#   4. Correct-case versus error-case descriptive attribution comparison\n#   5. Publication composite figures:\n#        Figure A — five correctly classified cases\n#        Figure B — five representative error cases\n#\n# Outputs:\n#   - 600-dpi PNG\n#   - PDF\n#   - SVG\n#   - QA tables\n#   - Publication captions\n#   - Verified backup\n#\n# Safety:\n#   - No training\n#   - No optimizer\n#   - No model loading\n#   - No model inference\n#   - No raw APTOS image loading\n#   - No validation/test evaluation\n#   - No prediction regeneration\n#   - No case replacement\n#\n# Run this new cell only. Do not use Run All.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\n\nimport gc\nimport hashlib\nimport json\nimport math\nimport os\nimport zipfile\n\nimport matplotlib\nmatplotlib.use(\"Agg\")\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\n\n# =============================================================================\n# 1. Project paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nSTEP13A_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13a_xai_case_selection_state.json\"\n)\n\nSTEP13B_R_STATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13b_r_gradcam_plus_plus_state.json\"\n)\n\nSELECTION_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13a_xai_case_selection\"\n    / \"step_13a_selected_xai_cases.csv\"\n)\n\nATTRIBUTION_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13b_r_gradcam_plus_plus\"\n    / \"step_13b_r_gradcampp_attribution_metrics.csv\"\n)\n\nCASE_INDEX_PATH = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13b_r_gradcam_plus_plus\"\n    / \"step_13b_r_case_output_index.csv\"\n)\n\nSTEP13B_R_SUMMARY_PATH = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13b_r_gradcam_plus_plus\"\n    / \"step_13b_r_gradcam_plus_plus_summary.json\"\n)\n\nSOURCE_FIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_13b_r_gradcam_plus_plus\"\n)\n\n\nOUTPUT_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_13c_xai_visual_qa\"\n)\n\nOUTPUT_EVIDENCE_DIR = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_13c_xai_visual_qa\"\n)\n\nOUTPUT_FIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_13c_xai_publication\"\n)\n\n\nVISUAL_QA_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13c_visual_file_qa.csv\"\n)\n\nCASE_QA_SUMMARY_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13c_case_level_visual_qa.csv\"\n)\n\nERROR_CAM_SIMILARITY_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13c_error_predicted_reference_cam_similarity.csv\"\n)\n\nCORRECT_ERROR_COMPARISON_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13c_correct_vs_error_attribution_comparison.csv\"\n)\n\nFIGURE_INDEX_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_13c_publication_figure_index.csv\"\n)\n\nPUBLICATION_XAI_TABLE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_publication_xai_summary_table.csv\"\n)\n\nSOURCE_VERIFICATION_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_source_verification.json\"\n)\n\nFIGURE_CAPTION_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_publication_figure_captions.txt\"\n)\n\nMANUSCRIPT_NOTE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_xai_results_interpretation.txt\"\n)\n\nSUMMARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_xai_visual_qa_summary.json\"\n)\n\nMANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_13c_xai_visual_qa_manifest.csv\"\n)\n\nSTATE_PATH = (\n    PROJECT\n    / \"00_state\"\n    / \"step_13c_xai_visual_qa_state.json\"\n)\n\nBACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_13c_xai_visual_qa_backup.zip\"\n)\n\n\nCORRECT_FIGURE_PNG = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_A_correct_cases_gradcampp_600dpi.png\"\n)\n\nCORRECT_FIGURE_PDF = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_A_correct_cases_gradcampp.pdf\"\n)\n\nCORRECT_FIGURE_SVG = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_A_correct_cases_gradcampp.svg\"\n)\n\n\nERROR_FIGURE_PNG = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_B_error_cases_gradcampp_600dpi.png\"\n)\n\nERROR_FIGURE_PDF = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_B_error_cases_gradcampp.pdf\"\n)\n\nERROR_FIGURE_SVG = (\n    OUTPUT_FIGURE_DIR\n    / \"figure_13c_B_error_cases_gradcampp.svg\"\n)\n\n\n# =============================================================================\n# 2. Locked expectations\n# =============================================================================\n\nEXPECTED_SELECTION_FINGERPRINT = (\n    \"da2b47d1c8cce21ab4231edf96e27f227\"\n    \"bfb24fca7fdc8bb508b47ceb5a34f89\"\n)\n\nEXPECTED_SELECTED_CASES = 10\nEXPECTED_CORRECT_CASES = 5\nEXPECTED_ERROR_CASES = 5\n\nEXPECTED_PREDICTED_MAPS = 10\nEXPECTED_REFERENCE_MAPS = 5\nEXPECTED_TOTAL_MAPS = 15\n\nEXPECTED_PNG_FILES = 60\nEXPECTED_NPY_FILES = 15\nEXPECTED_VISUAL_FILES = 75\n\nEXPECTED_IMAGE_SIZE = (\n    384,\n    384,\n)\n\nCAM_EPSILON = 1.0e-12\nTOP_ATTENTION_FRACTION = 0.20\n\n\nDISPLAY_NAMES = {\n    \"No_DR\": \"No DR\",\n    \"Mild\": \"Mild DR\",\n    \"Moderate\": \"Moderate DR\",\n    \"Severe\": \"Severe DR\",\n    \"Proliferative_DR\": \"Proliferative DR\",\n}\n\n\n# =============================================================================\n# 3. Utility functions\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef read_json(path):\n\n    with open(\n        path,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        return json.load(\n            file\n        )\n\n\ndef atomic_json_save(\n    record,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        json.dump(\n            record,\n            file,\n            indent=2,\n            ensure_ascii=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_csv_save(\n    dataframe,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    dataframe.to_csv(\n        temporary_path,\n        index=False,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_text_save(\n    text,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        file.write(\n            text\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef sha256_file(path):\n\n    digest = hashlib.sha256()\n\n    with open(\n        path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef dataframe_fingerprint(\n    dataframe,\n):\n\n    canonical = dataframe.copy()\n\n    canonical = canonical.sort_values(\n        list(\n            canonical.columns\n        )\n    ).reset_index(\n        drop=True\n    )\n\n    csv_bytes = canonical.to_csv(\n        index=False,\n        float_format=\"%.10f\",\n        lineterminator=\"\\n\",\n    ).encode(\n        \"utf-8\"\n    )\n\n    return hashlib.sha256(\n        csv_bytes\n    ).hexdigest()\n\n\ndef load_rgb_png(path):\n\n    with Image.open(\n        path\n    ) as image:\n\n        image.verify()\n\n\n    with Image.open(\n        path\n    ) as image:\n\n        rgb_array = np.asarray(\n            image.convert(\n                \"RGB\"\n            ),\n            dtype=np.uint8,\n        ).copy()\n\n\n    return rgb_array\n\n\ndef load_cam_array(path):\n\n    cam = np.load(\n        path,\n        allow_pickle=False,\n    )\n\n    cam = np.asarray(\n        cam,\n        dtype=np.float32,\n    )\n\n    return cam\n\n\ndef normalized_center_of_mass(\n    cam,\n):\n\n    cam_float = np.asarray(\n        cam,\n        dtype=np.float64,\n    )\n\n    total = float(\n        cam_float.sum()\n    )\n\n\n    if total <= CAM_EPSILON:\n\n        return (\n            np.nan,\n            np.nan,\n        )\n\n\n    y_grid, x_grid = np.indices(\n        cam_float.shape\n    )\n\n\n    x_coordinate = float(\n        (\n            cam_float\n            *\n            x_grid\n        ).sum()\n        /\n        total\n        /\n        (\n            cam_float.shape[\n                1\n            ]\n            -\n            1\n        )\n    )\n\n\n    y_coordinate = float(\n        (\n            cam_float\n            *\n            y_grid\n        ).sum()\n        /\n        total\n        /\n        (\n            cam_float.shape[\n                0\n            ]\n            -\n            1\n        )\n    )\n\n\n    return (\n        x_coordinate,\n        y_coordinate,\n    )\n\n\ndef pearson_similarity(\n    first,\n    second,\n):\n\n    first_flat = np.asarray(\n        first,\n        dtype=np.float64,\n    ).ravel()\n\n    second_flat = np.asarray(\n        second,\n        dtype=np.float64,\n    ).ravel()\n\n\n    first_centered = (\n        first_flat\n        -\n        first_flat.mean()\n    )\n\n    second_centered = (\n        second_flat\n        -\n        second_flat.mean()\n    )\n\n\n    denominator = float(\n        np.linalg.norm(\n            first_centered\n        )\n        *\n        np.linalg.norm(\n            second_centered\n        )\n    )\n\n\n    if denominator <= CAM_EPSILON:\n\n        return np.nan\n\n\n    return float(\n        np.dot(\n            first_centered,\n            second_centered,\n        )\n        /\n        denominator\n    )\n\n\ndef cosine_similarity(\n    first,\n    second,\n):\n\n    first_flat = np.asarray(\n        first,\n        dtype=np.float64,\n    ).ravel()\n\n    second_flat = np.asarray(\n        second,\n        dtype=np.float64,\n    ).ravel()\n\n\n    denominator = float(\n        np.linalg.norm(\n            first_flat\n        )\n        *\n        np.linalg.norm(\n            second_flat\n        )\n    )\n\n\n    if denominator <= CAM_EPSILON:\n\n        return np.nan\n\n\n    return float(\n        np.dot(\n            first_flat,\n            second_flat,\n        )\n        /\n        denominator\n    )\n\n\ndef top_fraction_mask(\n    cam,\n    top_fraction,\n):\n\n    threshold = float(\n        np.quantile(\n            cam,\n            1.0\n            -\n            top_fraction,\n        )\n    )\n\n    return (\n        cam\n        >=\n        threshold\n    )\n\n\ndef mask_iou(\n    first_mask,\n    second_mask,\n):\n\n    intersection = int(\n        np.logical_and(\n            first_mask,\n            second_mask,\n        ).sum()\n    )\n\n    union = int(\n        np.logical_or(\n            first_mask,\n            second_mask,\n        ).sum()\n    )\n\n\n    if union == 0:\n\n        return np.nan\n\n\n    return float(\n        intersection\n        /\n        union\n    )\n\n\ndef save_publication_figure(\n    figure,\n    png_path,\n    pdf_path,\n    svg_path,\n):\n\n    figure.savefig(\n        png_path,\n        dpi=600,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.04,\n    )\n\n    figure.savefig(\n        pdf_path,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.04,\n    )\n\n    figure.savefig(\n        svg_path,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.04,\n    )\n\n    plt.close(\n        figure\n    )\n\n    gc.collect()\n\n\ndef create_verified_zip(\n    zip_path,\n    source_files,\n):\n\n    temporary_path = zip_path.with_suffix(\n        zip_path.suffix + \".tmp\"\n    )\n\n\n    if temporary_path.exists():\n\n        temporary_path.unlink()\n\n\n    unique_files = []\n\n\n    for source_file in source_files:\n\n        source_file = Path(\n            source_file\n        )\n\n\n        if (\n            source_file.exists()\n            and\n            source_file.is_file()\n            and\n            source_file not in unique_files\n        ):\n\n            unique_files.append(\n                source_file\n            )\n\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"w\",\n        compression=zipfile.ZIP_DEFLATED,\n        compresslevel=6,\n    ) as archive:\n\n        for source_file in unique_files:\n\n            archive.write(\n                source_file,\n                arcname=str(\n                    source_file.relative_to(\n                        PROJECT\n                    )\n                ),\n            )\n\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"r\",\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            \"Step 13C backup ZIP is damaged at: \"\n            f\"{damaged_member}\"\n        )\n\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            \"Duplicate members detected in Step 13C backup.\"\n        )\n\n\n    os.replace(\n        temporary_path,\n        zip_path,\n    )\n\n\n    return members\n\n\n# =============================================================================\n# 4. Completion and partial-output protection\n# =============================================================================\n\nif STATE_PATH.exists():\n\n    existing_state = read_json(\n        STATE_PATH\n    )\n\n\n    if existing_state.get(\n        \"status\"\n    ) == \"complete\":\n\n        raise RuntimeError(\n            \"Step 13C is already complete. Do not rerun it.\"\n        )\n\n\nfor output_directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    OUTPUT_FIGURE_DIR,\n]:\n\n    if (\n        output_directory.exists()\n        and\n        any(\n            path.is_file()\n            for path\n            in output_directory.rglob(\n                \"*\"\n            )\n        )\n    ):\n\n        raise RuntimeError(\n            \"Partial Step 13C outputs already exist:\\n\"\n            f\"{output_directory}\\n\"\n            \"Do not mix outputs from multiple runs.\"\n        )\n\n\n# =============================================================================\n# 5. Required-source verification\n# =============================================================================\n\nrequired_paths = [\n    STEP13A_STATE_PATH,\n    STEP13B_R_STATE_PATH,\n    SELECTION_PATH,\n    ATTRIBUTION_PATH,\n    CASE_INDEX_PATH,\n    STEP13B_R_SUMMARY_PATH,\n    SOURCE_FIGURE_DIR,\n]\n\n\nfor required_path in required_paths:\n\n    if not required_path.exists():\n\n        raise FileNotFoundError(\n            f\"Required Step 13C source is missing: {required_path}\"\n        )\n\n\nstep13a_state = read_json(\n    STEP13A_STATE_PATH\n)\n\nstep13b_state = read_json(\n    STEP13B_R_STATE_PATH\n)\n\nstep13b_summary = read_json(\n    STEP13B_R_SUMMARY_PATH\n)\n\n\nif step13a_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 13A is incomplete.\"\n    )\n\n\nif step13a_state.get(\n    \"xai_case_selection_locked\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13A selection is not locked.\"\n    )\n\n\nif step13a_state.get(\n    \"case_replacement_after_heatmap_review_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Case replacement is not prohibited.\"\n    )\n\n\nif step13b_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 13B-R is incomplete.\"\n    )\n\n\nif step13b_state.get(\n    \"prediction_reproduction_gate_passed\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13B-R prediction reproduction did not pass.\"\n    )\n\n\nif step13b_state.get(\n    \"gradcam_plus_plus_generated\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13B-R attribution generation is incomplete.\"\n    )\n\n\nif step13b_state.get(\n    \"all_heatmaps_finite\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13B-R contains non-finite heatmaps.\"\n    )\n\n\nif step13b_state.get(\n    \"all_heatmaps_nondegenerate\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13B-R contains degenerate heatmaps.\"\n    )\n\n\nif step13b_state.get(\n    \"case_replacement_after_heatmap_review_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Step 13B-R does not preserve the locked-case policy.\"\n    )\n\n\nselection_fingerprint = str(\n    step13b_state.get(\n        \"selection_fingerprint_sha256\"\n    )\n)\n\n\nif selection_fingerprint != EXPECTED_SELECTION_FINGERPRINT:\n\n    raise RuntimeError(\n        \"Step 13B-R selection fingerprint mismatch.\"\n    )\n\n\n# =============================================================================\n# 6. Load source evidence tables\n# =============================================================================\n\nselection_df = pd.read_csv(\n    SELECTION_PATH\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\nattribution_df = pd.read_csv(\n    ATTRIBUTION_PATH\n).sort_values(\n    [\n        \"selection_order\",\n        \"target_type\",\n    ]\n).reset_index(\n    drop=True\n)\n\n\ncase_index_df = pd.read_csv(\n    CASE_INDEX_PATH\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\nif len(\n    selection_df\n) != EXPECTED_SELECTED_CASES:\n\n    raise RuntimeError(\n        \"Step 13C expected 10 locked cases.\"\n    )\n\n\nif int(\n    (\n        selection_df[\n            \"case_role\"\n        ]\n        ==\n        \"correct\"\n    ).sum()\n) != EXPECTED_CORRECT_CASES:\n\n    raise RuntimeError(\n        \"Correct-case count mismatch.\"\n    )\n\n\nif int(\n    (\n        selection_df[\n            \"case_role\"\n        ]\n        ==\n        \"error\"\n    ).sum()\n) != EXPECTED_ERROR_CASES:\n\n    raise RuntimeError(\n        \"Error-case count mismatch.\"\n    )\n\n\npredicted_attribution_df = attribution_df[\n    attribution_df[\n        \"target_type\"\n    ]\n    ==\n    \"predicted_class\"\n].copy()\n\n\nreference_attribution_df = attribution_df[\n    attribution_df[\n        \"target_type\"\n    ]\n    ==\n    \"reference_class\"\n].copy()\n\n\nif len(\n    predicted_attribution_df\n) != EXPECTED_PREDICTED_MAPS:\n\n    raise RuntimeError(\n        \"Predicted-class attribution count mismatch.\"\n    )\n\n\nif len(\n    reference_attribution_df\n) != EXPECTED_REFERENCE_MAPS:\n\n    raise RuntimeError(\n        \"Reference-class attribution count mismatch.\"\n    )\n\n\nif len(\n    attribution_df\n) != EXPECTED_TOTAL_MAPS:\n\n    raise RuntimeError(\n        \"Total attribution count mismatch.\"\n    )\n\n\nobserved_fingerprint = dataframe_fingerprint(\n    selection_df[\n        [\n            \"selection_order\",\n            \"case_category\",\n            \"sample_id\",\n            \"true_grade\",\n            \"predicted_grade\",\n            \"selection_method\",\n            \"image_path\",\n        ]\n    ]\n)\n\n\nif observed_fingerprint != EXPECTED_SELECTION_FINGERPRINT:\n\n    raise RuntimeError(\n        \"Current selection table does not reproduce \"\n        \"the locked fingerprint.\"\n    )\n\n\n# =============================================================================\n# 7. Build exact expected visual-file inventory\n# =============================================================================\n\nvisual_file_records = []\n\n\nfor _, row in selection_df.iterrows():\n\n    order = int(\n        row[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        row[\n            \"sample_id\"\n        ]\n    )\n\n    role = str(\n        row[\n            \"case_role\"\n        ]\n    )\n\n\n    case_directory = (\n        SOURCE_FIGURE_DIR\n        /\n        f\"{order:02d}_{sample_id}\"\n    )\n\n\n    expected_files = [\n        (\n            \"original_preview\",\n            \"png\",\n            case_directory\n            /\n            \"01_original_preview.png\",\n        ),\n\n        (\n            \"retinal_crop_resize\",\n            \"png\",\n            case_directory\n            /\n            \"02_retinal_crop_resize.png\",\n        ),\n\n        (\n            \"model_input\",\n            \"png\",\n            case_directory\n            /\n            \"03_always_clahe_model_input.png\",\n        ),\n\n        (\n            \"predicted_cam\",\n            \"npy\",\n            case_directory\n            /\n            \"04_predicted_class_gradcampp.npy\",\n        ),\n\n        (\n            \"predicted_heatmap\",\n            \"png\",\n            case_directory\n            /\n            \"05_predicted_class_heatmap.png\",\n        ),\n\n        (\n            \"predicted_overlay\",\n            \"png\",\n            case_directory\n            /\n            \"06_predicted_class_overlay.png\",\n        ),\n    ]\n\n\n    if role == \"error\":\n\n        expected_files.extend([\n            (\n                \"reference_cam\",\n                \"npy\",\n                case_directory\n                /\n                \"07_reference_class_gradcampp.npy\",\n            ),\n\n            (\n                \"reference_heatmap\",\n                \"png\",\n                case_directory\n                /\n                \"08_reference_class_heatmap.png\",\n            ),\n\n            (\n                \"reference_overlay\",\n                \"png\",\n                case_directory\n                /\n                \"09_reference_class_overlay.png\",\n            ),\n        ])\n\n\n    for artifact_role, file_type, file_path in expected_files:\n\n        visual_file_records.append({\n            \"selection_order\": (\n                order\n            ),\n\n            \"sample_id\": (\n                sample_id\n            ),\n\n            \"case_role\": (\n                role\n            ),\n\n            \"case_category\": str(\n                row[\n                    \"case_category\"\n                ]\n            ),\n\n            \"artifact_role\": (\n                artifact_role\n            ),\n\n            \"file_type\": (\n                file_type\n            ),\n\n            \"file_path\": str(\n                file_path\n            ),\n\n            \"file_exists\": bool(\n                file_path.exists()\n            ),\n        })\n\n\nvisual_inventory_df = pd.DataFrame(\n    visual_file_records\n)\n\n\nif not visual_inventory_df[\n    \"file_exists\"\n].all():\n\n    missing_files = visual_inventory_df[\n        ~visual_inventory_df[\n            \"file_exists\"\n        ]\n    ][\n        \"file_path\"\n    ].tolist()\n\n    raise FileNotFoundError(\n        \"Missing Step 13B-R visual files:\\n\"\n        +\n        \"\\n\".join(\n            missing_files\n        )\n    )\n\n\npng_count = int(\n    (\n        visual_inventory_df[\n            \"file_type\"\n        ]\n        ==\n        \"png\"\n    ).sum()\n)\n\n\nnpy_count = int(\n    (\n        visual_inventory_df[\n            \"file_type\"\n        ]\n        ==\n        \"npy\"\n    ).sum()\n)\n\n\nif png_count != EXPECTED_PNG_FILES:\n\n    raise RuntimeError(\n        \"Expected 60 Step 13B-R PNG files.\"\n    )\n\n\nif npy_count != EXPECTED_NPY_FILES:\n\n    raise RuntimeError(\n        \"Expected 15 Step 13B-R CAM arrays.\"\n    )\n\n\nif len(\n    visual_inventory_df\n) != EXPECTED_VISUAL_FILES:\n\n    raise RuntimeError(\n        \"Expected 75 Step 13B-R visual artifacts.\"\n    )\n\n\n# =============================================================================\n# 8. File-level visual QA\n# =============================================================================\n\nvisual_qa_records = []\n\n\nfor _, artifact in visual_inventory_df.iterrows():\n\n    file_path = Path(\n        artifact[\n            \"file_path\"\n        ]\n    )\n\n    file_type = str(\n        artifact[\n            \"file_type\"\n        ]\n    )\n\n    artifact_role = str(\n        artifact[\n            \"artifact_role\"\n        ]\n    )\n\n\n    if file_type == \"png\":\n\n        image = load_rgb_png(\n            file_path\n        )\n\n        height, width, channels = image.shape\n\n        pixel_minimum = int(\n            image.min()\n        )\n\n        pixel_maximum = int(\n            image.max()\n        )\n\n        pixel_range = int(\n            pixel_maximum\n            -\n            pixel_minimum\n        )\n\n        pixel_mean = float(\n            image.mean()\n        )\n\n        pixel_standard_deviation = float(\n            image.std()\n        )\n\n        finite_passed = bool(\n            np.isfinite(\n                image\n            ).all()\n        )\n\n        shape_passed = bool(\n            (\n                width,\n                height,\n            )\n            ==\n            EXPECTED_IMAGE_SIZE\n            and\n            channels\n            ==\n            3\n        )\n\n        nonblank_passed = bool(\n            pixel_range\n            >\n            10\n            and\n            pixel_standard_deviation\n            >\n            2.0\n        )\n\n        normalized_range_passed = (\n            np.nan\n        )\n\n        cam_minimum = (\n            np.nan\n        )\n\n        cam_maximum = (\n            np.nan\n        )\n\n        cam_range = (\n            np.nan\n        )\n\n        cam_standard_deviation = (\n            np.nan\n        )\n\n        del image\n\n\n    else:\n\n        cam = load_cam_array(\n            file_path\n        )\n\n        height, width = cam.shape\n\n        channels = (\n            np.nan\n        )\n\n        cam_minimum = float(\n            cam.min()\n        )\n\n        cam_maximum = float(\n            cam.max()\n        )\n\n        cam_range = float(\n            cam_maximum\n            -\n            cam_minimum\n        )\n\n        cam_standard_deviation = float(\n            cam.std()\n        )\n\n        finite_passed = bool(\n            np.isfinite(\n                cam\n            ).all()\n        )\n\n        shape_passed = bool(\n            (\n                width,\n                height,\n            )\n            ==\n            EXPECTED_IMAGE_SIZE\n        )\n\n        normalized_range_passed = bool(\n            cam_minimum\n            >=\n            -1.0e-6\n            and\n            cam_maximum\n            <=\n            1.0\n            +\n            1.0e-6\n        )\n\n        nonblank_passed = bool(\n            cam_range\n            >\n            0.50\n            and\n            cam_standard_deviation\n            >\n            1.0e-4\n        )\n\n        pixel_minimum = (\n            np.nan\n        )\n\n        pixel_maximum = (\n            np.nan\n        )\n\n        pixel_range = (\n            np.nan\n        )\n\n        pixel_mean = (\n            np.nan\n        )\n\n        pixel_standard_deviation = (\n            np.nan\n        )\n\n        del cam\n\n\n    artifact_qa_passed = bool(\n        finite_passed\n        and\n        shape_passed\n        and\n        nonblank_passed\n        and\n        (\n            normalized_range_passed\n            if file_type\n            ==\n            \"npy\"\n            else\n            True\n        )\n    )\n\n\n    visual_qa_records.append({\n        **artifact.to_dict(),\n\n        \"size_bytes\": int(\n            file_path.stat().st_size\n        ),\n\n        \"sha256\": sha256_file(\n            file_path\n        ),\n\n        \"width\": (\n            width\n        ),\n\n        \"height\": (\n            height\n        ),\n\n        \"channels\": (\n            channels\n        ),\n\n        \"pixel_minimum\": (\n            pixel_minimum\n        ),\n\n        \"pixel_maximum\": (\n            pixel_maximum\n        ),\n\n        \"pixel_range\": (\n            pixel_range\n        ),\n\n        \"pixel_mean\": (\n            pixel_mean\n        ),\n\n        \"pixel_standard_deviation\": (\n            pixel_standard_deviation\n        ),\n\n        \"cam_minimum\": (\n            cam_minimum\n        ),\n\n        \"cam_maximum\": (\n            cam_maximum\n        ),\n\n        \"cam_range\": (\n            cam_range\n        ),\n\n        \"cam_standard_deviation\": (\n            cam_standard_deviation\n        ),\n\n        \"finite_passed\": (\n            finite_passed\n        ),\n\n        \"shape_passed\": (\n            shape_passed\n        ),\n\n        \"normalized_range_passed\": (\n            normalized_range_passed\n        ),\n\n        \"nonblank_passed\": (\n            nonblank_passed\n        ),\n\n        \"artifact_qa_passed\": (\n            artifact_qa_passed\n        ),\n    })\n\n\nvisual_qa_df = pd.DataFrame(\n    visual_qa_records\n)\n\n\nif not visual_qa_df[\n    \"artifact_qa_passed\"\n].all():\n\n    failed_artifacts = visual_qa_df[\n        ~visual_qa_df[\n            \"artifact_qa_passed\"\n        ]\n    ][\n        [\n            \"selection_order\",\n            \"sample_id\",\n            \"artifact_role\",\n            \"file_path\",\n            \"finite_passed\",\n            \"shape_passed\",\n            \"normalized_range_passed\",\n            \"nonblank_passed\",\n        ]\n    ]\n\n    print(\n        \"\\nFAILED VISUAL ARTIFACTS\"\n    )\n\n    print(\n        failed_artifacts.to_string(\n            index=False\n        )\n    )\n\n    raise RuntimeError(\n        \"Step 13C visual artifact QA failed.\"\n    )\n\n\n# =============================================================================\n# 9. Overlay-integrity QA\n# =============================================================================\n\ncase_qa_records = []\n\n\nfor _, row in selection_df.iterrows():\n\n    order = int(\n        row[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        row[\n            \"sample_id\"\n        ]\n    )\n\n    role = str(\n        row[\n            \"case_role\"\n        ]\n    )\n\n\n    case_directory = (\n        SOURCE_FIGURE_DIR\n        /\n        f\"{order:02d}_{sample_id}\"\n    )\n\n\n    model_input = load_rgb_png(\n        case_directory\n        /\n        \"03_always_clahe_model_input.png\"\n    ).astype(\n        np.float32\n    )\n\n\n    predicted_heatmap = load_rgb_png(\n        case_directory\n        /\n        \"05_predicted_class_heatmap.png\"\n    ).astype(\n        np.float32\n    )\n\n\n    predicted_overlay = load_rgb_png(\n        case_directory\n        /\n        \"06_predicted_class_overlay.png\"\n    ).astype(\n        np.float32\n    )\n\n\n    predicted_overlay_difference = float(\n        np.mean(\n            np.abs(\n                predicted_overlay\n                -\n                model_input\n            )\n        )\n    )\n\n\n    predicted_heatmap_dynamic_range = float(\n        predicted_heatmap.max()\n        -\n        predicted_heatmap.min()\n    )\n\n\n    predicted_overlay_passed = bool(\n        predicted_overlay_difference\n        >\n        1.0\n        and\n        predicted_heatmap_dynamic_range\n        >\n        20.0\n    )\n\n\n    reference_overlay_difference = (\n        np.nan\n    )\n\n    reference_heatmap_dynamic_range = (\n        np.nan\n    )\n\n    reference_overlay_passed = (\n        True\n    )\n\n\n    if role == \"error\":\n\n        reference_heatmap = load_rgb_png(\n            case_directory\n            /\n            \"08_reference_class_heatmap.png\"\n        ).astype(\n            np.float32\n        )\n\n\n        reference_overlay = load_rgb_png(\n            case_directory\n            /\n            \"09_reference_class_overlay.png\"\n        ).astype(\n            np.float32\n        )\n\n\n        reference_overlay_difference = float(\n            np.mean(\n                np.abs(\n                    reference_overlay\n                    -\n                    model_input\n                )\n            )\n        )\n\n\n        reference_heatmap_dynamic_range = float(\n            reference_heatmap.max()\n            -\n            reference_heatmap.min()\n        )\n\n\n        reference_overlay_passed = bool(\n            reference_overlay_difference\n            >\n            1.0\n            and\n            reference_heatmap_dynamic_range\n            >\n            20.0\n        )\n\n\n        del reference_heatmap\n        del reference_overlay\n\n\n    case_qa_passed = bool(\n        predicted_overlay_passed\n        and\n        reference_overlay_passed\n    )\n\n\n    case_qa_records.append({\n        \"selection_order\": (\n            order\n        ),\n\n        \"sample_id\": (\n            sample_id\n        ),\n\n        \"case_role\": (\n            role\n        ),\n\n        \"case_category\": str(\n            row[\n                \"case_category\"\n            ]\n        ),\n\n        \"predicted_overlay_mean_absolute_difference\": (\n            predicted_overlay_difference\n        ),\n\n        \"predicted_heatmap_dynamic_range\": (\n            predicted_heatmap_dynamic_range\n        ),\n\n        \"predicted_overlay_passed\": (\n            predicted_overlay_passed\n        ),\n\n        \"reference_overlay_mean_absolute_difference\": (\n            reference_overlay_difference\n        ),\n\n        \"reference_heatmap_dynamic_range\": (\n            reference_heatmap_dynamic_range\n        ),\n\n        \"reference_overlay_passed\": (\n            reference_overlay_passed\n        ),\n\n        \"case_visual_qa_passed\": (\n            case_qa_passed\n        ),\n    })\n\n\n    del model_input\n    del predicted_heatmap\n    del predicted_overlay\n\n\ncase_qa_df = pd.DataFrame(\n    case_qa_records\n)\n\n\nif not case_qa_df[\n    \"case_visual_qa_passed\"\n].all():\n\n    raise RuntimeError(\n        \"One or more case-level overlay QA checks failed.\"\n    )\n\n\n# =============================================================================\n# 10. Predicted versus reference CAM similarity in error cases\n# =============================================================================\n\nsimilarity_records = []\n\n\nerror_cases_df = selection_df[\n    selection_df[\n        \"case_role\"\n    ]\n    ==\n    \"error\"\n].copy()\n\n\nfor _, row in error_cases_df.iterrows():\n\n    order = int(\n        row[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        row[\n            \"sample_id\"\n        ]\n    )\n\n\n    case_directory = (\n        SOURCE_FIGURE_DIR\n        /\n        f\"{order:02d}_{sample_id}\"\n    )\n\n\n    predicted_cam = load_cam_array(\n        case_directory\n        /\n        \"04_predicted_class_gradcampp.npy\"\n    )\n\n\n    reference_cam = load_cam_array(\n        case_directory\n        /\n        \"07_reference_class_gradcampp.npy\"\n    )\n\n\n    predicted_mask = top_fraction_mask(\n        predicted_cam,\n        TOP_ATTENTION_FRACTION,\n    )\n\n\n    reference_mask = top_fraction_mask(\n        reference_cam,\n        TOP_ATTENTION_FRACTION,\n    )\n\n\n    predicted_com_x, predicted_com_y = normalized_center_of_mass(\n        predicted_cam\n    )\n\n\n    reference_com_x, reference_com_y = normalized_center_of_mass(\n        reference_cam\n    )\n\n\n    center_distance = float(\n        math.sqrt(\n            (\n                predicted_com_x\n                -\n                reference_com_x\n            )\n            **\n            2\n            +\n            (\n                predicted_com_y\n                -\n                reference_com_y\n            )\n            **\n            2\n        )\n    )\n\n\n    similarity_records.append({\n        \"selection_order\": (\n            order\n        ),\n\n        \"sample_id\": (\n            sample_id\n        ),\n\n        \"case_category\": str(\n            row[\n                \"case_category\"\n            ]\n        ),\n\n        \"true_class\": str(\n            row[\n                \"true_class\"\n            ]\n        ),\n\n        \"predicted_class\": str(\n            row[\n                \"predicted_class\"\n            ]\n        ),\n\n        \"pearson_similarity\": pearson_similarity(\n            predicted_cam,\n            reference_cam,\n        ),\n\n        \"cosine_similarity\": cosine_similarity(\n            predicted_cam,\n            reference_cam,\n        ),\n\n        \"top_20_percent_attention_iou\": mask_iou(\n            predicted_mask,\n            reference_mask,\n        ),\n\n        \"mean_absolute_cam_difference\": float(\n            np.mean(\n                np.abs(\n                    predicted_cam\n                    -\n                    reference_cam\n                )\n            )\n        ),\n\n        \"predicted_center_x\": (\n            predicted_com_x\n        ),\n\n        \"predicted_center_y\": (\n            predicted_com_y\n        ),\n\n        \"reference_center_x\": (\n            reference_com_x\n        ),\n\n        \"reference_center_y\": (\n            reference_com_y\n        ),\n\n        \"normalized_center_distance\": (\n            center_distance\n        ),\n\n        \"analysis_scope\": (\n            \"Descriptive comparison only\"\n        ),\n    })\n\n\n    del predicted_cam\n    del reference_cam\n    del predicted_mask\n    del reference_mask\n\n\nsimilarity_df = pd.DataFrame(\n    similarity_records\n).sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\n# =============================================================================\n# 11. Correct versus error attribution comparison\n# =============================================================================\n\npredicted_metrics_df = attribution_df[\n    attribution_df[\n        \"target_type\"\n    ]\n    ==\n    \"predicted_class\"\n].copy()\n\n\nselected_role_df = selection_df[\n    [\n        \"selection_order\",\n        \"case_role\",\n        \"case_category\",\n    ]\n].copy()\n\n\npredicted_metrics_df = predicted_metrics_df.merge(\n    selected_role_df,\n    on=[\n        \"selection_order\",\n        \"case_role\",\n        \"case_category\",\n    ],\n    how=\"inner\",\n)\n\n\ncomparison_metrics = [\n    \"retinal_attention_fraction\",\n    \"outer_10_percent_border_attention_fraction\",\n    \"normalized_spatial_entropy\",\n    \"heatmap_mean\",\n    \"heatmap_standard_deviation\",\n]\n\n\ncomparison_records = []\n\n\nfor metric_name in comparison_metrics:\n\n    correct_values = predicted_metrics_df.loc[\n        predicted_metrics_df[\n            \"case_role\"\n        ]\n        ==\n        \"correct\",\n        metric_name,\n    ].to_numpy(\n        dtype=np.float64\n    )\n\n\n    error_values = predicted_metrics_df.loc[\n        predicted_metrics_df[\n            \"case_role\"\n        ]\n        ==\n        \"error\",\n        metric_name,\n    ].to_numpy(\n        dtype=np.float64\n    )\n\n\n    if (\n        len(\n            correct_values\n        )\n        !=\n        EXPECTED_CORRECT_CASES\n        or\n        len(\n            error_values\n        )\n        !=\n        EXPECTED_ERROR_CASES\n    ):\n\n        raise RuntimeError(\n            \"Correct-versus-error attribution group-size mismatch.\"\n        )\n\n\n    comparison_records.append({\n        \"metric\": (\n            metric_name\n        ),\n\n        \"correct_n\": int(\n            len(\n                correct_values\n            )\n        ),\n\n        \"correct_mean\": float(\n            np.mean(\n                correct_values\n            )\n        ),\n\n        \"correct_median\": float(\n            np.median(\n                correct_values\n            )\n        ),\n\n        \"correct_minimum\": float(\n            np.min(\n                correct_values\n            )\n        ),\n\n        \"correct_maximum\": float(\n            np.max(\n                correct_values\n            )\n        ),\n\n        \"error_n\": int(\n            len(\n                error_values\n            )\n        ),\n\n        \"error_mean\": float(\n            np.mean(\n                error_values\n            )\n        ),\n\n        \"error_median\": float(\n            np.median(\n                error_values\n            )\n        ),\n\n        \"error_minimum\": float(\n            np.min(\n                error_values\n            )\n        ),\n\n        \"error_maximum\": float(\n            np.max(\n                error_values\n            )\n        ),\n\n        \"error_minus_correct_mean\": float(\n            np.mean(\n                error_values\n            )\n            -\n            np.mean(\n                correct_values\n            )\n        ),\n\n        \"error_minus_correct_median\": float(\n            np.median(\n                error_values\n            )\n            -\n            np.median(\n                correct_values\n            )\n        ),\n\n        \"statistical_test_performed\": (\n            False\n        ),\n\n        \"interpretation_scope\": (\n            \"Descriptive; five cases per group\"\n        ),\n    })\n\n\ncorrect_error_comparison_df = pd.DataFrame(\n    comparison_records\n)\n\n\n# =============================================================================\n# 12. Create output directories only after QA passes\n# =============================================================================\n\nfor directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    OUTPUT_FIGURE_DIR,\n]:\n\n    directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\natomic_csv_save(\n    visual_qa_df,\n    VISUAL_QA_PATH,\n)\n\n\natomic_csv_save(\n    case_qa_df,\n    CASE_QA_SUMMARY_PATH,\n)\n\n\natomic_csv_save(\n    similarity_df,\n    ERROR_CAM_SIMILARITY_PATH,\n)\n\n\natomic_csv_save(\n    correct_error_comparison_df,\n    CORRECT_ERROR_COMPARISON_PATH,\n)\n\n\n# =============================================================================\n# 13. Publication Figure A — correctly classified cases\n# =============================================================================\n\ncorrect_cases_df = selection_df[\n    selection_df[\n        \"case_role\"\n    ]\n    ==\n    \"correct\"\n].sort_values(\n    \"selection_order\"\n).reset_index(\n    drop=True\n)\n\n\nfigure_a, axes_a = plt.subplots(\n    nrows=5,\n    ncols=2,\n    figsize=(\n        7.2,\n        10.0,\n    ),\n    squeeze=False,\n)\n\n\nfor row_index, case in correct_cases_df.iterrows():\n\n    order = int(\n        case[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        case[\n            \"sample_id\"\n        ]\n    )\n\n    case_directory = (\n        SOURCE_FIGURE_DIR\n        /\n        f\"{order:02d}_{sample_id}\"\n    )\n\n\n    model_input = load_rgb_png(\n        case_directory\n        /\n        \"03_always_clahe_model_input.png\"\n    )\n\n\n    predicted_overlay = load_rgb_png(\n        case_directory\n        /\n        \"06_predicted_class_overlay.png\"\n    )\n\n\n    class_display = DISPLAY_NAMES.get(\n        str(\n            case[\n                \"true_class\"\n            ]\n        ),\n        str(\n            case[\n                \"true_class\"\n            ]\n        ),\n    )\n\n\n    saved_confidence = float(\n        case[\n            \"selected_predicted_confidence\"\n        ]\n    )\n\n\n    axes_a[\n        row_index,\n        0\n    ].imshow(\n        model_input\n    )\n\n    axes_a[\n        row_index,\n        1\n    ].imshow(\n        predicted_overlay\n    )\n\n\n    axes_a[\n        row_index,\n        0\n    ].set_title(\n        (\n            f\"{chr(65 + row_index)}1  Model input\\n\"\n            f\"Reference: {class_display}\"\n        ),\n        fontsize=7,\n    )\n\n\n    axes_a[\n        row_index,\n        1\n    ].set_title(\n        (\n            f\"{chr(65 + row_index)}2  Predicted-class Grad-CAM++\\n\"\n            f\"Prediction: {class_display}; \"\n            f\"p={saved_confidence:.3f}\"\n        ),\n        fontsize=7,\n    )\n\n\n    for column_index in range(\n        2\n    ):\n\n        axes_a[\n            row_index,\n            column_index\n        ].axis(\n            \"off\"\n        )\n\n\nfigure_a.suptitle(\n    \"Grad-CAM++ attribution in correctly classified diabetic-retinopathy cases\",\n    fontsize=10,\n    y=0.998,\n)\n\n\nfigure_a.text(\n    0.5,\n    0.004,\n    (\n        \"Warm colours indicate stronger relative attribution within each map. \"\n        \"Attribution maps are normalized independently and are not lesion segmentations.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6,\n)\n\n\nfigure_a.tight_layout(\n    rect=(\n        0,\n        0.025,\n        1,\n        0.985,\n    ),\n    h_pad=0.8,\n    w_pad=0.5,\n)\n\n\nsave_publication_figure(\n    figure_a,\n    CORRECT_FIGURE_PNG,\n    CORRECT_FIGURE_PDF,\n    CORRECT_FIGURE_SVG,\n)\n\n\n# =============================================================================\n# 14. Publication Figure B — representative errors\n# =============================================================================\n\nfigure_b, axes_b = plt.subplots(\n    nrows=5,\n    ncols=3,\n    figsize=(\n        10.5,\n        10.0,\n    ),\n    squeeze=False,\n)\n\n\nfor row_index, case in error_cases_df.reset_index(\n    drop=True\n).iterrows():\n\n    order = int(\n        case[\n            \"selection_order\"\n        ]\n    )\n\n    sample_id = str(\n        case[\n            \"sample_id\"\n        ]\n    )\n\n    case_directory = (\n        SOURCE_FIGURE_DIR\n        /\n        f\"{order:02d}_{sample_id}\"\n    )\n\n\n    model_input = load_rgb_png(\n        case_directory\n        /\n        \"03_always_clahe_model_input.png\"\n    )\n\n\n    predicted_overlay = load_rgb_png(\n        case_directory\n        /\n        \"06_predicted_class_overlay.png\"\n    )\n\n\n    reference_overlay = load_rgb_png(\n        case_directory\n        /\n        \"09_reference_class_overlay.png\"\n    )\n\n\n    true_display = DISPLAY_NAMES.get(\n        str(\n            case[\n                \"true_class\"\n            ]\n        ),\n        str(\n            case[\n                \"true_class\"\n            ]\n        ),\n    )\n\n\n    predicted_display = DISPLAY_NAMES.get(\n        str(\n            case[\n                \"predicted_class\"\n            ]\n        ),\n        str(\n            case[\n                \"predicted_class\"\n            ]\n        ),\n    )\n\n\n    saved_confidence = float(\n        case[\n            \"selected_predicted_confidence\"\n        ]\n    )\n\n\n    axes_b[\n        row_index,\n        0\n    ].imshow(\n        model_input\n    )\n\n    axes_b[\n        row_index,\n        1\n    ].imshow(\n        predicted_overlay\n    )\n\n    axes_b[\n        row_index,\n        2\n    ].imshow(\n        reference_overlay\n    )\n\n\n    panel_letter = chr(\n        65\n        +\n        row_index\n    )\n\n\n    axes_b[\n        row_index,\n        0\n    ].set_title(\n        (\n            f\"{panel_letter}1  Model input\\n\"\n            f\"Reference: {true_display}\"\n        ),\n        fontsize=7,\n    )\n\n\n    axes_b[\n        row_index,\n        1\n    ].set_title(\n        (\n            f\"{panel_letter}2  Predicted-class map\\n\"\n            f\"{predicted_display}; p={saved_confidence:.3f}\"\n        ),\n        fontsize=7,\n    )\n\n\n    axes_b[\n        row_index,\n        2\n    ].set_title(\n        (\n            f\"{panel_letter}3  Reference-class map\\n\"\n            f\"{true_display}\"\n        ),\n        fontsize=7,\n    )\n\n\n    for column_index in range(\n        3\n    ):\n\n        axes_b[\n            row_index,\n            column_index\n        ].axis(\n            \"off\"\n        )\n\n\nfigure_b.suptitle(\n    \"Predicted-class and reference-class Grad-CAM++ maps for representative errors\",\n    fontsize=10,\n    y=0.998,\n)\n\n\nfigure_b.text(\n    0.5,\n    0.004,\n    (\n        \"Cases were locked before attribution review. \"\n        \"Maps are normalized independently; cross-panel colour intensity \"\n        \"must not be interpreted as an absolute attribution comparison.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6,\n)\n\n\nfigure_b.tight_layout(\n    rect=(\n        0,\n        0.025,\n        1,\n        0.985,\n    ),\n    h_pad=0.8,\n    w_pad=0.5,\n)\n\n\nsave_publication_figure(\n    figure_b,\n    ERROR_FIGURE_PNG,\n    ERROR_FIGURE_PDF,\n    ERROR_FIGURE_SVG,\n)\n\n\n# =============================================================================\n# 15. Verify publication figures\n# =============================================================================\n\npublication_figure_paths = [\n    CORRECT_FIGURE_PNG,\n    CORRECT_FIGURE_PDF,\n    CORRECT_FIGURE_SVG,\n    ERROR_FIGURE_PNG,\n    ERROR_FIGURE_PDF,\n    ERROR_FIGURE_SVG,\n]\n\n\nfigure_index_records = []\n\n\nfor figure_path in publication_figure_paths:\n\n    if not figure_path.exists():\n\n        raise FileNotFoundError(\n            f\"Publication figure was not created: {figure_path}\"\n        )\n\n\n    suffix = figure_path.suffix.lower()\n\n\n    width_pixels = (\n        np.nan\n    )\n\n    height_pixels = (\n        np.nan\n    )\n\n\n    if suffix == \".png\":\n\n        with Image.open(\n            figure_path\n        ) as image:\n\n            image.verify()\n\n\n        with Image.open(\n            figure_path\n        ) as image:\n\n            width_pixels, height_pixels = image.size\n\n\n        if (\n            width_pixels\n            <\n            3000\n            or\n            height_pixels\n            <\n            3000\n        ):\n\n            raise RuntimeError(\n                \"Publication PNG resolution is unexpectedly low.\"\n            )\n\n\n    figure_index_records.append({\n        \"figure_name\": (\n            figure_path.stem\n        ),\n\n        \"file_path\": str(\n            figure_path\n        ),\n\n        \"file_format\": (\n            suffix.replace(\n                \".\",\n                \"\"\n            ).upper()\n        ),\n\n        \"size_bytes\": int(\n            figure_path.stat().st_size\n        ),\n\n        \"width_pixels\": (\n            width_pixels\n        ),\n\n        \"height_pixels\": (\n            height_pixels\n        ),\n\n        \"sha256\": sha256_file(\n            figure_path\n        ),\n\n        \"integrity_passed\": (\n            True\n        ),\n    })\n\n\nfigure_index_df = pd.DataFrame(\n    figure_index_records\n)\n\n\natomic_csv_save(\n    figure_index_df,\n    FIGURE_INDEX_PATH,\n)\n\n\n# =============================================================================\n# 16. Publication XAI summary table\n# =============================================================================\n\npublication_xai_df = selection_df[\n    [\n        \"selection_order\",\n        \"case_role\",\n        \"case_category\",\n        \"sample_id\",\n        \"true_class\",\n        \"predicted_class\",\n        \"selected_predicted_confidence\",\n    ]\n].merge(\n    predicted_attribution_df[\n        [\n            \"selection_order\",\n            \"target_probability_fp32\",\n            \"retinal_attention_fraction\",\n            \"outer_10_percent_border_attention_fraction\",\n            \"normalized_spatial_entropy\",\n        ]\n    ],\n    on=\"selection_order\",\n    how=\"left\",\n)\n\n\npublication_xai_df.columns = [\n    \"Order\",\n    \"Case role\",\n    \"Case category\",\n    \"Image ID\",\n    \"Reference grade\",\n    \"Predicted grade\",\n    \"Saved predicted confidence\",\n    \"FP32 predicted-class probability\",\n    \"Retinal attention fraction\",\n    \"Outer-border attention fraction\",\n    \"Normalized spatial entropy\",\n]\n\n\natomic_csv_save(\n    publication_xai_df,\n    PUBLICATION_XAI_TABLE_PATH,\n)\n\n\n# =============================================================================\n# 17. Source verification\n# =============================================================================\n\nsource_verification_record = {\n    \"step\": (\n        \"STEP_13C_XAI_VISUAL_QA_AND_PUBLICATION_FIGURES\"\n    ),\n\n    \"verified_utc\": (\n        utc_now()\n    ),\n\n    \"step_13a_state_path\": str(\n        STEP13A_STATE_PATH\n    ),\n\n    \"step_13a_state_sha256\": sha256_file(\n        STEP13A_STATE_PATH\n    ),\n\n    \"step_13b_r_state_path\": str(\n        STEP13B_R_STATE_PATH\n    ),\n\n    \"step_13b_r_state_sha256\": sha256_file(\n        STEP13B_R_STATE_PATH\n    ),\n\n    \"selection_path\": str(\n        SELECTION_PATH\n    ),\n\n    \"selection_sha256\": sha256_file(\n        SELECTION_PATH\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        observed_fingerprint\n    ),\n\n    \"attribution_table_path\": str(\n        ATTRIBUTION_PATH\n    ),\n\n    \"attribution_table_sha256\": sha256_file(\n        ATTRIBUTION_PATH\n    ),\n\n    \"case_index_path\": str(\n        CASE_INDEX_PATH\n    ),\n\n    \"case_index_sha256\": sha256_file(\n        CASE_INDEX_PATH\n    ),\n\n    \"source_visual_file_count\": int(\n        len(\n            visual_inventory_df\n        )\n    ),\n\n    \"source_png_count\": (\n        png_count\n    ),\n\n    \"source_cam_array_count\": (\n        npy_count\n    ),\n\n    \"all_visual_artifacts_passed\": bool(\n        visual_qa_df[\n            \"artifact_qa_passed\"\n        ].all()\n    ),\n\n    \"all_case_overlay_checks_passed\": bool(\n        case_qa_df[\n            \"case_visual_qa_passed\"\n        ].all()\n    ),\n\n    \"case_replacement_performed\": (\n        False\n    ),\n\n    \"model_loaded\": (\n        False\n    ),\n\n    \"model_inference_performed\": (\n        False\n    ),\n\n    \"raw_aptos_images_loaded\": (\n        False\n    ),\n}\n\n\natomic_json_save(\n    source_verification_record,\n    SOURCE_VERIFICATION_PATH,\n)\n\n\n# =============================================================================\n# 18. Figure captions\n# =============================================================================\n\nfigure_captions = \"\"\"\nFigure 13C-A. Grad-CAM++ attribution maps for five correctly classified\nfinal-test cases, comprising one preselected example from each diabetic-\nretinopathy grade. The left column shows the exact Always-CLAHE model\ninput, and the right column shows the predicted-class attribution overlay.\nCases were selected deterministically before visualization. Warm colours\nrepresent stronger relative attribution within each independently\nnormalized map. The maps indicate model-attributed retinal regions and\nmust not be interpreted as clinically validated lesion segmentations.\n\nFigure 13C-B. Predicted-class and reference-class Grad-CAM++ maps for five\npreselected final-test errors. Rows represent Moderate-to-Mild,\nSevere-to-Moderate, Proliferative-DR-to-Moderate,\nProliferative-DR-to-Mild, and Moderate-to-Severe predictions,\nrespectively. The first column shows the exact model input, the second\ncolumn shows attribution for the locked predicted class, and the third\ncolumn shows attribution for the reference class. Because every CAM was\nnormalized independently, colour intensity must not be compared as an\nabsolute attribution magnitude across panels.\n\"\"\".strip()\n\n\natomic_text_save(\n    figure_captions,\n    FIGURE_CAPTION_PATH,\n)\n\n\n# =============================================================================\n# 19. Formal interpretation\n# =============================================================================\n\nmean_predicted_reference_pearson = float(\n    similarity_df[\n        \"pearson_similarity\"\n    ].mean()\n)\n\n\nmedian_predicted_reference_pearson = float(\n    similarity_df[\n        \"pearson_similarity\"\n    ].median()\n)\n\n\nmean_top_attention_iou = float(\n    similarity_df[\n        \"top_20_percent_attention_iou\"\n    ].mean()\n)\n\n\nmean_center_distance = float(\n    similarity_df[\n        \"normalized_center_distance\"\n    ].mean()\n)\n\n\npredicted_maps_mean_retinal_attention = float(\n    predicted_attribution_df[\n        \"retinal_attention_fraction\"\n    ].mean()\n)\n\n\npredicted_maps_mean_border_attention = float(\n    predicted_attribution_df[\n        \"outer_10_percent_border_attention_fraction\"\n    ].mean()\n)\n\n\nmaximum_border_row = predicted_attribution_df.loc[\n    predicted_attribution_df[\n        \"outer_10_percent_border_attention_fraction\"\n    ].idxmax()\n]\n\n\nmaximum_border_sample = str(\n    maximum_border_row[\n        \"sample_id\"\n    ]\n)\n\n\nmaximum_border_fraction = float(\n    maximum_border_row[\n        \"outer_10_percent_border_attention_fraction\"\n    ]\n)\n\n\ninterpretation_note = f\"\"\"\nSTEP 13C — XAI VISUAL QA AND RESULTS INTERPRETATION\n\nAll {EXPECTED_VISUAL_FILES} expected Step 13B-R visual artifacts passed\nfile-integrity, dimensionality, finite-value and non-degeneracy checks.\nThis included {EXPECTED_PNG_FILES} PNG images and\n{EXPECTED_NPY_FILES} raw Grad-CAM++ arrays. All ten predicted-class\noverlays and all five reference-class error overlays differed\nmeaningfully from their underlying model inputs, confirming that no\nblank or unchanged attribution panels were included.\n\nThe mean proportion of predicted-class attribution located within the\nretinal field was {predicted_maps_mean_retinal_attention:.3f}, whereas\nthe mean attribution within the outer 10% image border was\n{predicted_maps_mean_border_attention:.3f}. The largest peripheral\nattention fraction occurred in image {maximum_border_sample}\n({maximum_border_fraction:.3f}). This case was retained because case\nreplacement after attribution review was prohibited.\n\nAcross the five error cases, the mean Pearson similarity between the\npredicted-class and reference-class CAMs was\n{mean_predicted_reference_pearson:.3f}, with a median of\n{median_predicted_reference_pearson:.3f}. The mean intersection-over-\nunion between their top 20% attribution regions was\n{mean_top_attention_iou:.3f}, and the mean normalized distance between\ntheir attention centres was {mean_center_distance:.3f}. These quantities\nprovide descriptive evidence about whether the alternative class scores\nwere supported by overlapping or spatially distinct image regions. They\nare not lesion-localization accuracy measures.\n\nCorrect-versus-error comparisons were restricted to descriptive\nsummaries because only five locked cases were present in each group.\nNo inferential statistical test, attribution-based case replacement or\npost hoc model modification was performed.\n\nThe publication figures use independently normalized attribution maps.\nTherefore, colour intensity can be interpreted within an individual map\nbut must not be treated as an absolute cross-image comparison. Grad-CAM++\nidentifies image regions contributing to a selected class score; it does\nnot verify the presence, type or boundaries of diabetic-retinopathy\nlesions.\n\"\"\".strip()\n\n\natomic_text_save(\n    interpretation_note,\n    MANUSCRIPT_NOTE_PATH,\n)\n\n\n# =============================================================================\n# 20. Formal summary\n# =============================================================================\n\nsummary_record = {\n    \"step\": (\n        \"STEP_13C_XAI_VISUAL_QA_AND_PUBLICATION_FIGURES\"\n    ),\n\n    \"status\": (\n        \"completed\"\n    ),\n\n    \"completed_utc\": (\n        utc_now()\n    ),\n\n    \"source_integrity\": {\n        \"selection_fingerprint_sha256\": (\n            observed_fingerprint\n        ),\n\n        \"selected_case_count\": int(\n            len(\n                selection_df\n            )\n        ),\n\n        \"source_visual_file_count\": int(\n            len(\n                visual_qa_df\n            )\n        ),\n\n        \"source_png_count\": (\n            png_count\n        ),\n\n        \"source_cam_array_count\": (\n            npy_count\n        ),\n\n        \"all_visual_artifacts_passed\": bool(\n            visual_qa_df[\n                \"artifact_qa_passed\"\n            ].all()\n        ),\n\n        \"all_case_overlay_checks_passed\": bool(\n            case_qa_df[\n                \"case_visual_qa_passed\"\n            ].all()\n        ),\n    },\n\n    \"error_cam_similarity\": {\n        \"error_case_count\": int(\n            len(\n                similarity_df\n            )\n        ),\n\n        \"mean_pearson_similarity\": (\n            mean_predicted_reference_pearson\n        ),\n\n        \"median_pearson_similarity\": (\n            median_predicted_reference_pearson\n        ),\n\n        \"mean_cosine_similarity\": float(\n            similarity_df[\n                \"cosine_similarity\"\n            ].mean()\n        ),\n\n        \"mean_top_20_percent_attention_iou\": (\n            mean_top_attention_iou\n        ),\n\n        \"mean_absolute_cam_difference\": float(\n            similarity_df[\n                \"mean_absolute_cam_difference\"\n            ].mean()\n        ),\n\n        \"mean_normalized_center_distance\": (\n            mean_center_distance\n        ),\n    },\n\n    \"predicted_class_attribution\": {\n        \"mean_retinal_attention_fraction\": (\n            predicted_maps_mean_retinal_attention\n        ),\n\n        \"mean_outer_border_attention_fraction\": (\n            predicted_maps_mean_border_attention\n        ),\n\n        \"maximum_border_attention_sample_id\": (\n            maximum_border_sample\n        ),\n\n        \"maximum_border_attention_fraction\": (\n            maximum_border_fraction\n        ),\n    },\n\n    \"publication_figures\": {\n        \"correct_cases\": {\n            \"png\": str(\n                CORRECT_FIGURE_PNG\n            ),\n\n            \"pdf\": str(\n                CORRECT_FIGURE_PDF\n            ),\n\n            \"svg\": str(\n                CORRECT_FIGURE_SVG\n            ),\n        },\n\n        \"error_cases\": {\n            \"png\": str(\n                ERROR_FIGURE_PNG\n            ),\n\n            \"pdf\": str(\n                ERROR_FIGURE_PDF\n            ),\n\n            \"svg\": str(\n                ERROR_FIGURE_SVG\n            ),\n        },\n\n        \"png_dpi\": (\n            600\n        ),\n\n        \"publication_figure_integrity_passed\": (\n            True\n        ),\n    },\n\n    \"interpretation_constraints\": [\n        (\n            \"CAM intensity was normalized independently \"\n            \"within each attribution map.\"\n        ),\n\n        (\n            \"Grad-CAM++ maps are model attributions and \"\n            \"not validated lesion segmentations.\"\n        ),\n\n        (\n            \"Correct-versus-error comparisons are \"\n            \"descriptive because each group contains \"\n            \"five preselected cases.\"\n        ),\n    ],\n\n    \"safety\": {\n        \"new_training_performed\": (\n            False\n        ),\n\n        \"optimizer_created\": (\n            False\n        ),\n\n        \"model_loaded\": (\n            False\n        ),\n\n        \"model_inference_performed\": (\n            False\n        ),\n\n        \"raw_aptos_images_loaded\": (\n            False\n        ),\n\n        \"validation_evaluated\": (\n            False\n        ),\n\n        \"final_test_evaluated\": (\n            False\n        ),\n\n        \"predictions_regenerated\": (\n            False\n        ),\n\n        \"case_replacement_performed\": (\n            False\n        ),\n\n        \"model_change_performed\": (\n            False\n        ),\n    },\n\n    \"next_stage\": (\n        \"STEP_14_PUBLICATION_METRICS_TABLES_AND_FIGURES\"\n    ),\n}\n\n\natomic_json_save(\n    summary_record,\n    SUMMARY_PATH,\n)\n\n\n# =============================================================================\n# 21. Formal state\n# =============================================================================\n\nstate_record = {\n    \"step\": (\n        \"STEP_13C_XAI_VISUAL_QA_AND_PUBLICATION_FIGURES\"\n    ),\n\n    \"status\": (\n        \"complete\"\n    ),\n\n    \"updated_utc\": (\n        utc_now()\n    ),\n\n    \"xai_visual_qa_completed\": (\n        True\n    ),\n\n    \"all_visual_artifacts_passed\": bool(\n        visual_qa_df[\n            \"artifact_qa_passed\"\n        ].all()\n    ),\n\n    \"all_case_overlay_checks_passed\": bool(\n        case_qa_df[\n            \"case_visual_qa_passed\"\n        ].all()\n    ),\n\n    \"predicted_reference_cam_similarity_completed\": (\n        True\n    ),\n\n    \"correct_error_descriptive_comparison_completed\": (\n        True\n    ),\n\n    \"publication_figures_generated\": (\n        True\n    ),\n\n    \"publication_png_dpi\": (\n        600\n    ),\n\n    \"selection_fingerprint_sha256\": (\n        observed_fingerprint\n    ),\n\n    \"case_replacement_after_review_allowed\": (\n        False\n    ),\n\n    \"new_training_performed\": (\n        False\n    ),\n\n    \"model_loaded\": (\n        False\n    ),\n\n    \"model_inference_performed\": (\n        False\n    ),\n\n    \"raw_aptos_images_loaded\": (\n        False\n    ),\n\n    \"validation_evaluated\": (\n        False\n    ),\n\n    \"final_test_evaluated\": (\n        False\n    ),\n\n    \"predictions_regenerated\": (\n        False\n    ),\n\n    \"model_change_allowed\": (\n        False\n    ),\n\n    \"another_validation_evaluation_allowed\": (\n        False\n    ),\n\n    \"another_test_evaluation_allowed\": (\n        False\n    ),\n\n    \"correct_cases_figure_png\": str(\n        CORRECT_FIGURE_PNG\n    ),\n\n    \"error_cases_figure_png\": str(\n        ERROR_FIGURE_PNG\n    ),\n\n    \"next_stage\": (\n        \"STEP_14_PUBLICATION_METRICS_TABLES_AND_FIGURES\"\n    ),\n}\n\n\natomic_json_save(\n    state_record,\n    STATE_PATH,\n)\n\n\n# =============================================================================\n# 22. Manifest and verified backup\n# =============================================================================\n\nmanifest_sources = [\n    STEP13A_STATE_PATH,\n    STEP13B_R_STATE_PATH,\n    SELECTION_PATH,\n    ATTRIBUTION_PATH,\n    CASE_INDEX_PATH,\n    STEP13B_R_SUMMARY_PATH,\n    VISUAL_QA_PATH,\n    CASE_QA_SUMMARY_PATH,\n    ERROR_CAM_SIMILARITY_PATH,\n    CORRECT_ERROR_COMPARISON_PATH,\n    FIGURE_INDEX_PATH,\n    PUBLICATION_XAI_TABLE_PATH,\n    SOURCE_VERIFICATION_PATH,\n    FIGURE_CAPTION_PATH,\n    MANUSCRIPT_NOTE_PATH,\n    SUMMARY_PATH,\n    STATE_PATH,\n    CORRECT_FIGURE_PNG,\n    CORRECT_FIGURE_PDF,\n    CORRECT_FIGURE_SVG,\n    ERROR_FIGURE_PNG,\n    ERROR_FIGURE_PDF,\n    ERROR_FIGURE_SVG,\n]\n\n\nmanifest_records = []\n\n\nfor source_path in manifest_sources:\n\n    if not source_path.exists():\n\n        raise FileNotFoundError(\n            f\"Step 13C manifest source missing: {source_path}\"\n        )\n\n\n    manifest_records.append({\n        \"relative_path\": str(\n            source_path.relative_to(\n                PROJECT\n            )\n        ),\n\n        \"size_bytes\": int(\n            source_path.stat().st_size\n        ),\n\n        \"sha256\": sha256_file(\n            source_path\n        ),\n    })\n\n\natomic_csv_save(\n    pd.DataFrame(\n        manifest_records\n    ),\n    MANIFEST_PATH,\n)\n\n\nbackup_members = create_verified_zip(\n    BACKUP_PATH,\n    [\n        VISUAL_QA_PATH,\n        CASE_QA_SUMMARY_PATH,\n        ERROR_CAM_SIMILARITY_PATH,\n        CORRECT_ERROR_COMPARISON_PATH,\n        FIGURE_INDEX_PATH,\n        PUBLICATION_XAI_TABLE_PATH,\n        SOURCE_VERIFICATION_PATH,\n        FIGURE_CAPTION_PATH,\n        MANUSCRIPT_NOTE_PATH,\n        SUMMARY_PATH,\n        STATE_PATH,\n        MANIFEST_PATH,\n        CORRECT_FIGURE_PNG,\n        CORRECT_FIGURE_PDF,\n        CORRECT_FIGURE_SVG,\n        ERROR_FIGURE_PNG,\n        ERROR_FIGURE_PDF,\n        ERROR_FIGURE_SVG,\n    ],\n)\n\n\ngc.collect()\n\n\n# =============================================================================\n# 23. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"STEP 13C — XAI VISUAL QA AND \"\n    \"PUBLICATION FIGURES COMPLETED\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nEVIDENCE SAFETY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"New training performed                :\",\n    False\n)\n\nprint(\n    \"Model loaded                          :\",\n    False\n)\n\nprint(\n    \"Model inference performed             :\",\n    False\n)\n\nprint(\n    \"Raw APTOS images loaded               :\",\n    False\n)\n\nprint(\n    \"Validation evaluated                  :\",\n    False\n)\n\nprint(\n    \"Final test evaluated                  :\",\n    False\n)\n\nprint(\n    \"Predictions regenerated               :\",\n    False\n)\n\nprint(\n    \"Case replacement performed            :\",\n    False\n)\n\nprint(\n    \"Frozen model changed                  :\",\n    False\n)\n\n\nprint(\n    \"\\nVISUAL ARTIFACT QA\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Expected visual artifacts             :\",\n    EXPECTED_VISUAL_FILES\n)\n\nprint(\n    \"Observed visual artifacts             :\",\n    len(\n        visual_qa_df\n    )\n)\n\nprint(\n    \"PNG files                             :\",\n    png_count\n)\n\nprint(\n    \"Raw CAM arrays                        :\",\n    npy_count\n)\n\nprint(\n    \"Artifact integrity pass count         :\",\n    int(\n        visual_qa_df[\n            \"artifact_qa_passed\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        visual_qa_df\n    )\n)\n\nprint(\n    \"Case-level overlay pass count         :\",\n    int(\n        case_qa_df[\n            \"case_visual_qa_passed\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        case_qa_df\n    )\n)\n\nprint(\n    \"All visual QA passed                  :\",\n    bool(\n        visual_qa_df[\n            \"artifact_qa_passed\"\n        ].all()\n        and\n        case_qa_df[\n            \"case_visual_qa_passed\"\n        ].all()\n    )\n)\n\n\nprint(\n    \"\\nERROR-CASE PREDICTED VS REFERENCE CAM SIMILARITY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_similarity = similarity_df[\n    [\n        \"selection_order\",\n        \"sample_id\",\n        \"case_category\",\n        \"pearson_similarity\",\n        \"cosine_similarity\",\n        \"top_20_percent_attention_iou\",\n        \"mean_absolute_cam_difference\",\n        \"normalized_center_distance\",\n    ]\n].copy()\n\n\nfor column in [\n    \"pearson_similarity\",\n    \"cosine_similarity\",\n    \"top_20_percent_attention_iou\",\n    \"mean_absolute_cam_difference\",\n    \"normalized_center_distance\",\n]:\n\n    display_similarity[\n        column\n    ] = display_similarity[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_similarity.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nSIMILARITY SUMMARY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Mean Pearson similarity               :\",\n    f\"{mean_predicted_reference_pearson:.6f}\"\n)\n\nprint(\n    \"Median Pearson similarity             :\",\n    f\"{median_predicted_reference_pearson:.6f}\"\n)\n\nprint(\n    \"Mean top-20% attention IoU            :\",\n    f\"{mean_top_attention_iou:.6f}\"\n)\n\nprint(\n    \"Mean normalized centre distance       :\",\n    f\"{mean_center_distance:.6f}\"\n)\n\n\nprint(\n    \"\\nCORRECT VS ERROR ATTRIBUTION — DESCRIPTIVE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_comparison = correct_error_comparison_df[\n    [\n        \"metric\",\n        \"correct_mean\",\n        \"error_mean\",\n        \"error_minus_correct_mean\",\n        \"correct_median\",\n        \"error_median\",\n    ]\n].copy()\n\n\nfor column in [\n    \"correct_mean\",\n    \"error_mean\",\n    \"error_minus_correct_mean\",\n    \"correct_median\",\n    \"error_median\",\n]:\n\n    display_comparison[\n        column\n    ] = display_comparison[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_comparison.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nPUBLICATION FIGURES\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Correct-case 600-dpi PNG              :\",\n    CORRECT_FIGURE_PNG\n)\n\nprint(\n    \"Correct-case PDF                      :\",\n    CORRECT_FIGURE_PDF\n)\n\nprint(\n    \"Correct-case SVG                      :\",\n    CORRECT_FIGURE_SVG\n)\n\nprint(\n    \"Error-case 600-dpi PNG                :\",\n    ERROR_FIGURE_PNG\n)\n\nprint(\n    \"Error-case PDF                        :\",\n    ERROR_FIGURE_PDF\n)\n\nprint(\n    \"Error-case SVG                        :\",\n    ERROR_FIGURE_SVG\n)\n\nprint(\n    \"Publication figure integrity passed   :\",\n    bool(\n        figure_index_df[\n            \"integrity_passed\"\n        ].all()\n    )\n)\n\n\nprint(\n    \"\\nBACKUP\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup path                           :\",\n    BACKUP_PATH\n)\n\nprint(\n    \"Backup members                        :\",\n    len(\n        backup_members\n    )\n)\n\nprint(\n    \"ZIP integrity passed                  :\",\n    True\n)\n\n\nprint(\n    \"\\nNEXT STAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"READY FOR STEP 14 — PUBLICATION \"\n    \"METRICS TABLES AND FIGURES\"\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T16:11:10.443408Z","iopub.execute_input":"2026-07-18T16:11:10.444103Z","iopub.status.idle":"2026-07-18T16:11:25.883186Z","shell.execute_reply.started":"2026-07-18T16:11:10.444066Z","shell.execute_reply":"2026-07-18T16:11:25.882410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 14 — PUBLICATION METRICS TABLES AND FIGURES\n#\n# Uses preserved and locked evidence only.\n#\n# Generates:\n#   Tables\n#     A. Primary CV / validation / final-test performance\n#     B. Holdout confidence intervals\n#     C. Class-wise recall\n#     D. Generalization gaps\n#     E. Same-seed preprocessing ablation\n#     F. Repeated-seed preprocessing robustness\n#     G. Common-fold model-complexity comparison\n#     H. Bounded-revision screen\n#     I. Error-direction summary\n#     J. Binary severity-detection summary\n#     K. External mother-paper context (explicitly non-comparable)\n#\n#   Figures\n#     14A. Primary performance across evaluation scopes\n#     14B. Locked-validation normalized confusion matrix\n#     14C. Final-test normalized confusion matrix\n#     14D. Class-wise recall\n#     14E. Same-seed preprocessing ablation\n#     14F. Common-fold complexity–QWK trade-off\n#\n# Figure formats:\n#   - 600-dpi PNG\n#   - PDF\n#   - SVG\n#\n# Safety:\n#   - No training\n#   - No optimizer\n#   - No model loading\n#   - No inference\n#   - No raw image loading\n#   - No validation/test evaluation\n#   - No prediction regeneration\n#   - No tuning or model modification\n#\n# Run this new cell only. Do not use Run All.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\n\nimport gc\nimport hashlib\nimport json\nimport math\nimport os\nimport zipfile\n\nimport matplotlib\nmatplotlib.use(\"Agg\")\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\n\n# =============================================================================\n# 1. Project paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nSTATE_DIR = (\n    PROJECT\n    / \"00_state\"\n)\n\nSTEP10D_STATE_PATH = (\n    STATE_DIR\n    / \"step_10d_final_test_state.json\"\n)\n\nSTEP12A_STATE_PATH = (\n    STATE_DIR\n    / \"step_12a_generalization_gap_analysis_state.json\"\n)\n\nSTEP12B_STATE_PATH = (\n    STATE_DIR\n    / \"step_12b_locked_prediction_error_analysis_state.json\"\n)\n\nSTEP13C_STATE_PATH = (\n    STATE_DIR\n    / \"step_13c_xai_visual_qa_state.json\"\n)\n\n\nSTEP12A_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_12a_generalization_gap_analysis\"\n)\n\nSTEP12A_EVIDENCE_DIR = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_12a_generalization_gap_analysis\"\n)\n\n\nSTEP12B_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_12b_locked_prediction_error_analysis\"\n)\n\nCONFUSION_LONG_SOURCE_PATH = (\n    STEP12B_METRIC_DIR\n    / \"step_12b_confusion_matrix_long.csv\"\n)\n\nCLASS_ERROR_SOURCE_PATH = (\n    STEP12B_METRIC_DIR\n    / \"step_12b_class_specific_error_analysis.csv\"\n)\n\nERROR_DIRECTION_SOURCE_PATH = (\n    STEP12B_METRIC_DIR\n    / \"step_12b_undergrading_overgrading_summary.csv\"\n)\n\nBINARY_SEVERITY_SOURCE_PATH = (\n    STEP12B_METRIC_DIR\n    / \"step_12b_binary_severity_detection.csv\"\n)\n\n\nSTEP11A_R1_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_11a_r1_fair_reporting_backup.zip\"\n)\n\nSTEP11B_QA2_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_11b_qa2_parameter_reporting_correction_backup.zip\"\n)\n\nSTEP12A_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_12a_generalization_gap_analysis_backup.zip\"\n)\n\nSTEP12B_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_12b_locked_prediction_error_analysis_backup.zip\"\n)\n\nSTEP13C_BACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_13c_xai_visual_qa_backup.zip\"\n)\n\n\nOUTPUT_METRIC_DIR = (\n    PROJECT\n    / \"08_metrics\"\n    / \"step_14_publication_metrics_figures\"\n)\n\nOUTPUT_EVIDENCE_DIR = (\n    PROJECT\n    / \"12_paper_evidence\"\n    / \"step_14_publication_metrics_figures\"\n)\n\nOUTPUT_FIGURE_DIR = (\n    PROJECT\n    / \"09_figures\"\n    / \"step_14_publication_metrics_figures\"\n)\n\n\nTABLE_A_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14a_primary_performance.csv\"\n)\n\nTABLE_B_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14b_holdout_confidence_intervals.csv\"\n)\n\nTABLE_C_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14c_classwise_recall.csv\"\n)\n\nTABLE_D_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14d_generalization_gaps.csv\"\n)\n\nTABLE_E_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14e_preprocessing_ablation_same_seed.csv\"\n)\n\nTABLE_F_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14f_preprocessing_repeated_seed.csv\"\n)\n\nTABLE_G_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14g_model_complexity_common_folds.csv\"\n)\n\nTABLE_H_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14h_bounded_revision_screen.csv\"\n)\n\nTABLE_I_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14i_error_direction_summary.csv\"\n)\n\nTABLE_J_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14j_binary_severity_detection.csv\"\n)\n\nTABLE_K_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"table_14k_external_context_noncomparable.csv\"\n)\n\n\nCONFUSION_VALIDATION_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_14_locked_validation_confusion_matrix.csv\"\n)\n\nCONFUSION_TEST_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_14_final_test_confusion_matrix.csv\"\n)\n\nCONFUSION_VALIDATION_NORMALIZED_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_14_locked_validation_confusion_matrix_normalized.csv\"\n)\n\nCONFUSION_TEST_NORMALIZED_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_14_final_test_confusion_matrix_normalized.csv\"\n)\n\nSOURCE_VERIFICATION_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_source_verification.json\"\n)\n\nFIGURE_INDEX_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_publication_figure_index.csv\"\n)\n\nFIGURE_CAPTIONS_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_publication_figure_captions.txt\"\n)\n\nMANUSCRIPT_NOTE_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_publication_metrics_interpretation.txt\"\n)\n\nSUMMARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_publication_metrics_summary.json\"\n)\n\nMANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_14_publication_metrics_manifest.csv\"\n)\n\nSTATE_PATH = (\n    STATE_DIR\n    / \"step_14_publication_metrics_figures_state.json\"\n)\n\nBACKUP_PATH = (\n    PROJECT\n    / \"13_backups\"\n    / \"step_14_publication_metrics_figures_backup.zip\"\n)\n\n\n# =============================================================================\n# 2. Locked constants\n# =============================================================================\n\nCLASS_NAMES = [\n    \"No_DR\",\n    \"Mild\",\n    \"Moderate\",\n    \"Severe\",\n    \"Proliferative_DR\",\n]\n\nCLASS_DISPLAY_NAMES = [\n    \"No DR\",\n    \"Mild\",\n    \"Moderate\",\n    \"Severe\",\n    \"Proliferative DR\",\n]\n\nEVALUATION_SCOPES = [\n    \"Registered internal cross-validation\",\n    \"One-time locked validation\",\n    \"One-time final test\",\n]\n\nSHORT_SCOPE_NAMES = {\n    \"Registered internal cross-validation\": \"Internal CV\",\n    \"One-time locked validation\": \"Locked validation\",\n    \"One-time final test\": \"Final test\",\n}\n\nEXPECTED_ROWS = {\n    \"Registered internal cross-validation\": 2441,\n    \"One-time locked validation\": 524,\n    \"One-time final test\": 522,\n}\n\nEXPECTED_PRIMARY_METRICS = {\n    \"Registered internal cross-validation\": {\n        \"qwk\": 0.901608,\n        \"accuracy\": 0.836337,\n        \"balanced_accuracy\": 0.673770,\n        \"macro_f1\": 0.679991,\n    },\n    \"One-time locked validation\": {\n        \"qwk\": 0.783395,\n        \"accuracy\": 0.740458,\n        \"balanced_accuracy\": 0.586690,\n        \"macro_f1\": 0.565506,\n    },\n    \"One-time final test\": {\n        \"qwk\": 0.847063,\n        \"accuracy\": 0.770115,\n        \"balanced_accuracy\": 0.628933,\n        \"macro_f1\": 0.608571,\n    },\n}\n\nEXPECTED_HOLDOUT_CI = {\n    \"One-time locked validation\": {\n        \"qwk\": (\n            0.733764,\n            0.831016,\n        ),\n        \"accuracy\": (\n            0.701913,\n            0.778004,\n        ),\n        \"balanced_accuracy\": (\n            0.534153,\n            0.643336,\n        ),\n        \"macro_f1\": (\n            0.503419,\n            0.622154,\n        ),\n    },\n    \"One-time final test\": {\n        \"qwk\": (\n            0.810012,\n            0.879952,\n        ),\n        \"accuracy\": (\n            0.733205,\n            0.807692,\n        ),\n        \"balanced_accuracy\": (\n            0.570946,\n            0.684592,\n        ),\n        \"macro_f1\": (\n            0.544459,\n            0.663406,\n        ),\n    },\n}\n\nEXPECTED_CLASS_COUNTS = {\n    \"Locked validation\": {\n        \"No_DR\": 270,\n        \"Mild\": 51,\n        \"Moderate\": 138,\n        \"Severe\": 25,\n        \"Proliferative_DR\": 40,\n    },\n    \"Final test\": {\n        \"No_DR\": 269,\n        \"Mild\": 51,\n        \"Moderate\": 138,\n        \"Severe\": 25,\n        \"Proliferative_DR\": 39,\n    },\n}\n\nEXPECTED_CLASS_RECALL = {\n    \"Locked validation\": {\n        \"No_DR\": 0.977778,\n        \"Mild\": 0.843137,\n        \"Moderate\": 0.427536,\n        \"Severe\": 0.360000,\n        \"Proliferative_DR\": 0.325000,\n    },\n    \"Final test\": {\n        \"No_DR\": 0.977695,\n        \"Mild\": 0.882353,\n        \"Moderate\": 0.500000,\n        \"Severe\": 0.400000,\n        \"Proliferative_DR\": 0.384615,\n    },\n}\n\nEXPECTED_CONFUSION_ROW_TOTALS = {\n    \"Locked validation\": np.array(\n        [\n            270,\n            51,\n            138,\n            25,\n            40,\n        ],\n        dtype=np.int64,\n    ),\n    \"Final test\": np.array(\n        [\n            269,\n            51,\n            138,\n            25,\n            39,\n        ],\n        dtype=np.int64,\n    ),\n}\n\nMETRIC_LABELS = {\n    \"qwk\": \"Quadratic weighted κ\",\n    \"accuracy\": \"Accuracy\",\n    \"balanced_accuracy\": \"Balanced accuracy\",\n    \"macro_f1\": \"Macro F1\",\n}\n\nMETRIC_SHORT_LABELS = {\n    \"qwk\": \"QWK\",\n    \"accuracy\": \"Accuracy\",\n    \"balanced_accuracy\": \"Balanced accuracy\",\n    \"macro_f1\": \"Macro F1\",\n}\n\nTOLERANCE = 5.0e-6\n\n\n# =============================================================================\n# 3. Utility functions\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef read_json(\n    path,\n):\n\n    with open(\n        path,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        return json.load(\n            file\n        )\n\n\ndef atomic_json_save(\n    record,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        json.dump(\n            record,\n            file,\n            indent=2,\n            ensure_ascii=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_csv_save(\n    dataframe,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    dataframe.to_csv(\n        temporary_path,\n        index=False,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_text_save(\n    text,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        file.write(\n            text\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef sha256_file(\n    path,\n):\n\n    digest = hashlib.sha256()\n\n    with open(\n        path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef verify_zip(\n    path,\n):\n\n    if not path.exists():\n\n        raise FileNotFoundError(\n            f\"Required backup is missing: {path}\"\n        )\n\n    with zipfile.ZipFile(\n        path,\n        mode=\"r\",\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            f\"Damaged backup member detected: {damaged_member}\"\n        )\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            f\"Duplicate members detected in backup: {path}\"\n        )\n\n    return {\n        \"path\": str(\n            path\n        ),\n        \"sha256\": sha256_file(\n            path\n        ),\n        \"member_count\": len(\n            members\n        ),\n        \"integrity_passed\": True,\n    }\n\n\ndef normalized_columns(\n    dataframe,\n):\n\n    return {\n        str(\n            column\n        ).strip().lower(): column\n        for column\n        in dataframe.columns\n    }\n\n\ndef find_csv_with_columns(\n    search_directories,\n    required_columns,\n    preferred_tokens=None,\n):\n\n    preferred_tokens = [\n        str(\n            token\n        ).lower()\n        for token\n        in (\n            preferred_tokens\n            or\n            []\n        )\n    ]\n\n    candidate_records = []\n\n    for directory in search_directories:\n\n        if not directory.exists():\n\n            continue\n\n        for path in directory.rglob(\n            \"*.csv\"\n        ):\n\n            if (\n                not path.is_file()\n                or\n                path.stat().st_size\n                >\n                50 * 1024 * 1024\n            ):\n\n                continue\n\n            try:\n\n                header_df = pd.read_csv(\n                    path,\n                    nrows=3,\n                )\n\n            except Exception:\n\n                continue\n\n            column_map = normalized_columns(\n                header_df\n            )\n\n            if not set(\n                required_columns\n            ).issubset(\n                set(\n                    column_map.keys()\n                )\n            ):\n\n                continue\n\n            path_text = str(\n                path\n            ).lower()\n\n            score = sum(\n                10\n                for token\n                in preferred_tokens\n                if token\n                in path_text\n            )\n\n            score += len(\n                required_columns\n            )\n\n            candidate_records.append(\n                (\n                    score,\n                    path,\n                )\n            )\n\n    if not candidate_records:\n\n        return None\n\n    candidate_records.sort(\n        key=lambda item: (\n            -item[\n                0\n            ],\n            str(\n                item[\n                    1\n                ]\n            ),\n        )\n    )\n\n    selected_path = candidate_records[\n        0\n    ][\n        1\n    ]\n\n    selected_df = pd.read_csv(\n        selected_path\n    )\n\n    return {\n        \"path\": selected_path,\n        \"dataframe\": selected_df,\n    }\n\n\ndef create_verified_zip(\n    zip_path,\n    source_files,\n):\n\n    temporary_path = zip_path.with_suffix(\n        zip_path.suffix + \".tmp\"\n    )\n\n    if temporary_path.exists():\n\n        temporary_path.unlink()\n\n    unique_files = []\n\n    for source_file in source_files:\n\n        source_file = Path(\n            source_file\n        )\n\n        if (\n            source_file.exists()\n            and\n            source_file.is_file()\n            and\n            source_file not in unique_files\n        ):\n\n            unique_files.append(\n                source_file\n            )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"w\",\n        compression=zipfile.ZIP_DEFLATED,\n        compresslevel=6,\n    ) as archive:\n\n        for source_file in unique_files:\n\n            archive.write(\n                source_file,\n                arcname=str(\n                    source_file.relative_to(\n                        PROJECT\n                    )\n                ),\n            )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"r\",\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            \"Step 14 backup ZIP is damaged at: \"\n            f\"{damaged_member}\"\n        )\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            \"Duplicate members detected in Step 14 backup.\"\n        )\n\n    os.replace(\n        temporary_path,\n        zip_path,\n    )\n\n    return members\n\n\ndef save_figure_triplet(\n    figure,\n    stem,\n):\n\n    png_path = (\n        OUTPUT_FIGURE_DIR\n        /\n        f\"{stem}_600dpi.png\"\n    )\n\n    pdf_path = (\n        OUTPUT_FIGURE_DIR\n        /\n        f\"{stem}.pdf\"\n    )\n\n    svg_path = (\n        OUTPUT_FIGURE_DIR\n        /\n        f\"{stem}.svg\"\n    )\n\n    figure.savefig(\n        png_path,\n        dpi=600,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.05,\n    )\n\n    figure.savefig(\n        pdf_path,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.05,\n    )\n\n    figure.savefig(\n        svg_path,\n        facecolor=\"white\",\n        bbox_inches=\"tight\",\n        pad_inches=0.05,\n    )\n\n    plt.close(\n        figure\n    )\n\n    gc.collect()\n\n    return [\n        png_path,\n        pdf_path,\n        svg_path,\n    ]\n\n\ndef annotate_bars(\n    axis,\n    bars,\n    decimal_places=3,\n    y_offset=0.006,\n):\n\n    for bar in bars:\n\n        height = float(\n            bar.get_height()\n        )\n\n        axis.text(\n            bar.get_x()\n            +\n            bar.get_width()\n            /\n            2.0,\n            height\n            +\n            y_offset,\n            f\"{height:.{decimal_places}f}\",\n            ha=\"center\",\n            va=\"bottom\",\n            fontsize=6.5,\n            rotation=0,\n        )\n\n\ndef create_confusion_figure(\n    count_matrix,\n    normalized_matrix,\n    title,\n    stem,\n):\n\n    figure, axis = plt.subplots(\n        figsize=(\n            6.8,\n            6.2,\n        )\n    )\n\n    image = axis.imshow(\n        normalized_matrix,\n        cmap=\"Greys\",\n        vmin=0.0,\n        vmax=1.0,\n        interpolation=\"nearest\",\n    )\n\n    colour_bar = figure.colorbar(\n        image,\n        ax=axis,\n        fraction=0.046,\n        pad=0.04,\n    )\n\n    colour_bar.set_label(\n        \"Row-normalized proportion\",\n        fontsize=8,\n    )\n\n    axis.set_xticks(\n        np.arange(\n            len(\n                CLASS_DISPLAY_NAMES\n            )\n        )\n    )\n\n    axis.set_yticks(\n        np.arange(\n            len(\n                CLASS_DISPLAY_NAMES\n            )\n        )\n    )\n\n    axis.set_xticklabels(\n        CLASS_DISPLAY_NAMES,\n        rotation=35,\n        ha=\"right\",\n        fontsize=7,\n    )\n\n    axis.set_yticklabels(\n        CLASS_DISPLAY_NAMES,\n        fontsize=7,\n    )\n\n    axis.set_xlabel(\n        \"Predicted grade\",\n        fontsize=9,\n    )\n\n    axis.set_ylabel(\n        \"Reference grade\",\n        fontsize=9,\n    )\n\n    axis.set_title(\n        title,\n        fontsize=10,\n        pad=10,\n    )\n\n    for row_index in range(\n        count_matrix.shape[\n            0\n        ]\n    ):\n\n        for column_index in range(\n            count_matrix.shape[\n                1\n            ]\n        ):\n\n            proportion = float(\n                normalized_matrix[\n                    row_index,\n                    column_index,\n                ]\n            )\n\n            count = int(\n                count_matrix[\n                    row_index,\n                    column_index,\n                ]\n            )\n\n            text_colour = (\n                \"white\"\n                if proportion\n                >=\n                0.50\n                else\n                \"black\"\n            )\n\n            axis.text(\n                column_index,\n                row_index,\n                (\n                    f\"{count}\\n\"\n                    f\"({100.0 * proportion:.1f}%)\"\n                ),\n                ha=\"center\",\n                va=\"center\",\n                fontsize=6.5,\n                color=text_colour,\n            )\n\n    axis.set_ylim(\n        len(\n            CLASS_DISPLAY_NAMES\n        )\n        -\n        0.5,\n        -0.5,\n    )\n\n    figure.tight_layout()\n\n    return save_figure_triplet(\n        figure,\n        stem,\n    )\n\n\n# =============================================================================\n# 4. Completion and partial-output protection\n# =============================================================================\n\nif STATE_PATH.exists():\n\n    existing_state = read_json(\n        STATE_PATH\n    )\n\n    if existing_state.get(\n        \"status\"\n    ) == \"complete\":\n\n        raise RuntimeError(\n            \"Step 14 is already complete. Do not rerun it.\"\n        )\n\n\nfor output_directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    OUTPUT_FIGURE_DIR,\n]:\n\n    if (\n        output_directory.exists()\n        and\n        any(\n            path.is_file()\n            for path\n            in output_directory.rglob(\n                \"*\"\n            )\n        )\n    ):\n\n        raise RuntimeError(\n            \"Partial Step 14 outputs already exist:\\n\"\n            f\"{output_directory}\\n\"\n            \"Do not mix outputs from multiple runs.\"\n        )\n\n\n# =============================================================================\n# 5. Prerequisite verification\n# =============================================================================\n\nrequired_paths = [\n    STEP10D_STATE_PATH,\n    STEP12A_STATE_PATH,\n    STEP12B_STATE_PATH,\n    STEP13C_STATE_PATH,\n    CONFUSION_LONG_SOURCE_PATH,\n    CLASS_ERROR_SOURCE_PATH,\n    ERROR_DIRECTION_SOURCE_PATH,\n    BINARY_SEVERITY_SOURCE_PATH,\n    STEP11A_R1_BACKUP_PATH,\n    STEP11B_QA2_BACKUP_PATH,\n    STEP12A_BACKUP_PATH,\n    STEP12B_BACKUP_PATH,\n    STEP13C_BACKUP_PATH,\n]\n\n\nfor required_path in required_paths:\n\n    if not required_path.exists():\n\n        raise FileNotFoundError(\n            f\"Required Step 14 evidence is missing: {required_path}\"\n        )\n\n\nstep10d_state = read_json(\n    STEP10D_STATE_PATH\n)\n\nstep12a_state = read_json(\n    STEP12A_STATE_PATH\n)\n\nstep12b_state = read_json(\n    STEP12B_STATE_PATH\n)\n\nstep13c_state = read_json(\n    STEP13C_STATE_PATH\n)\n\n\nif step10d_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 10D final-test stage is incomplete.\"\n    )\n\n\nif int(\n    step10d_state.get(\n        \"test_evaluation_count\",\n        -1,\n    )\n) != 1:\n\n    raise RuntimeError(\n        \"Final-test evaluation count is not exactly one.\"\n    )\n\n\nif step10d_state.get(\n    \"another_test_evaluation_allowed\"\n) is not False:\n\n    raise RuntimeError(\n        \"Final-test evaluation is not formally closed.\"\n    )\n\n\nif step12a_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 12A is incomplete.\"\n    )\n\n\nif step12a_state.get(\n    \"generalization_gap_analysis_completed\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 12A generalization analysis is incomplete.\"\n    )\n\n\nif step12b_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 12B is incomplete.\"\n    )\n\n\nif step12b_state.get(\n    \"locked_metrics_reproduced_from_saved_predictions\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 12B metric reproduction did not pass.\"\n    )\n\n\nif step13c_state.get(\n    \"status\"\n) != \"complete\":\n\n    raise RuntimeError(\n        \"Step 13C is incomplete.\"\n    )\n\n\nif step13c_state.get(\n    \"publication_figures_generated\"\n) is not True:\n\n    raise RuntimeError(\n        \"Step 13C publication XAI figures are incomplete.\"\n    )\n\n\nbackup_verification = {\n    \"step_11a_r1\": verify_zip(\n        STEP11A_R1_BACKUP_PATH\n    ),\n    \"step_11b_qa2\": verify_zip(\n        STEP11B_QA2_BACKUP_PATH\n    ),\n    \"step_12a\": verify_zip(\n        STEP12A_BACKUP_PATH\n    ),\n    \"step_12b\": verify_zip(\n        STEP12B_BACKUP_PATH\n    ),\n    \"step_13c\": verify_zip(\n        STEP13C_BACKUP_PATH\n    ),\n}\n\n\n# =============================================================================\n# 6. Recover and verify Step 12A primary performance evidence\n# =============================================================================\n\nperformance_source = find_csv_with_columns(\n    search_directories=[\n        STEP12A_METRIC_DIR,\n        STEP12A_EVIDENCE_DIR,\n    ],\n    required_columns=[\n        \"evaluation_scope\",\n        \"rows\",\n        \"qwk\",\n        \"accuracy\",\n        \"balanced_accuracy\",\n        \"macro_f1\",\n    ],\n    preferred_tokens=[\n        \"performance\",\n        \"evaluation\",\n        \"scope\",\n    ],\n)\n\n\nexpected_performance_records = []\n\n\nfor evaluation_scope in EVALUATION_SCOPES:\n\n    expected_performance_records.append({\n        \"evaluation_scope\": (\n            evaluation_scope\n        ),\n        \"rows\": (\n            EXPECTED_ROWS[\n                evaluation_scope\n            ]\n        ),\n        \"qwk\": (\n            EXPECTED_PRIMARY_METRICS[\n                evaluation_scope\n            ][\n                \"qwk\"\n            ]\n        ),\n        \"accuracy\": (\n            EXPECTED_PRIMARY_METRICS[\n                evaluation_scope\n            ][\n                \"accuracy\"\n            ]\n        ),\n        \"balanced_accuracy\": (\n            EXPECTED_PRIMARY_METRICS[\n                evaluation_scope\n            ][\n                \"balanced_accuracy\"\n            ]\n        ),\n        \"macro_f1\": (\n            EXPECTED_PRIMARY_METRICS[\n                evaluation_scope\n            ][\n                \"macro_f1\"\n            ]\n        ),\n    })\n\n\nperformance_df = pd.DataFrame(\n    expected_performance_records\n)\n\n\nperformance_source_mode = (\n    \"locked_constants_verified_against_step_12a_output\"\n)\n\n\nif performance_source is not None:\n\n    source_df = performance_source[\n        \"dataframe\"\n    ].copy()\n\n    for numeric_column in [\n        \"rows\",\n        \"qwk\",\n        \"accuracy\",\n        \"balanced_accuracy\",\n        \"macro_f1\",\n    ]:\n\n        source_df[\n            numeric_column\n        ] = pd.to_numeric(\n            source_df[\n                numeric_column\n            ],\n            errors=\"raise\",\n        )\n\n    source_df = source_df[\n        source_df[\n            \"evaluation_scope\"\n        ].isin(\n            EVALUATION_SCOPES\n        )\n    ].copy()\n\n    if len(\n        source_df\n    ) != 3:\n\n        raise RuntimeError(\n            \"Step 12A performance table does not contain \"\n            \"the three expected evaluation scopes.\"\n        )\n\n    source_df = source_df.set_index(\n        \"evaluation_scope\"\n    )\n\n    for _, expected_row in performance_df.iterrows():\n\n        evaluation_scope = expected_row[\n            \"evaluation_scope\"\n        ]\n\n        if evaluation_scope not in source_df.index:\n\n            raise RuntimeError(\n                \"Step 12A performance source is missing scope: \"\n                f\"{evaluation_scope}\"\n            )\n\n        observed_row = source_df.loc[\n            evaluation_scope\n        ]\n\n        if int(\n            observed_row[\n                \"rows\"\n            ]\n        ) != int(\n            expected_row[\n                \"rows\"\n            ]\n        ):\n\n            raise RuntimeError(\n                \"Performance row-count verification failed.\"\n            )\n\n        for metric_name in [\n            \"qwk\",\n            \"accuracy\",\n            \"balanced_accuracy\",\n            \"macro_f1\",\n        ]:\n\n            difference = abs(\n                float(\n                    observed_row[\n                        metric_name\n                    ]\n                )\n                -\n                float(\n                    expected_row[\n                        metric_name\n                    ]\n                )\n            )\n\n            if difference > TOLERANCE:\n\n                raise RuntimeError(\n                    \"Step 12A performance verification failed for \"\n                    f\"{evaluation_scope}, metric={metric_name}.\"\n                )\n\nelse:\n\n    performance_source_mode = (\n        \"locked_constants_fallback_step_12a_state_and_backup_verified\"\n    )\n\n\n# =============================================================================\n# 7. Primary publication performance tables\n# =============================================================================\n\ntable_a_df = performance_df.copy()\n\ntable_a_df[\n    \"Evaluation scope\"\n] = table_a_df[\n    \"evaluation_scope\"\n].map(\n    SHORT_SCOPE_NAMES\n)\n\ntable_a_df[\n    \"Samples\"\n] = table_a_df[\n    \"rows\"\n].astype(\n    int\n)\n\ntable_a_df[\n    \"QWK\"\n] = table_a_df[\n    \"qwk\"\n]\n\ntable_a_df[\n    \"Accuracy\"\n] = table_a_df[\n    \"accuracy\"\n]\n\ntable_a_df[\n    \"Balanced accuracy\"\n] = table_a_df[\n    \"balanced_accuracy\"\n]\n\ntable_a_df[\n    \"Macro F1\"\n] = table_a_df[\n    \"macro_f1\"\n]\n\ntable_a_df[\n    \"Evaluation status\"\n] = [\n    \"Registered group-aware internal cross-validation\",\n    \"One-time locked holdout\",\n    \"One-time final untouched holdout\",\n]\n\ntable_a_df = table_a_df[\n    [\n        \"Evaluation scope\",\n        \"Samples\",\n        \"QWK\",\n        \"Accuracy\",\n        \"Balanced accuracy\",\n        \"Macro F1\",\n        \"Evaluation status\",\n    ]\n]\n\n\nholdout_ci_records = []\n\n\nfor evaluation_scope in [\n    \"One-time locked validation\",\n    \"One-time final test\",\n]:\n\n    for metric_name in [\n        \"qwk\",\n        \"accuracy\",\n        \"balanced_accuracy\",\n        \"macro_f1\",\n    ]:\n\n        lower, upper = EXPECTED_HOLDOUT_CI[\n            evaluation_scope\n        ][\n            metric_name\n        ]\n\n        holdout_ci_records.append({\n            \"Evaluation scope\": (\n                SHORT_SCOPE_NAMES[\n                    evaluation_scope\n                ]\n            ),\n            \"Metric\": (\n                METRIC_LABELS[\n                    metric_name\n                ]\n            ),\n            \"Point estimate\": (\n                EXPECTED_PRIMARY_METRICS[\n                    evaluation_scope\n                ][\n                    metric_name\n                ]\n            ),\n            \"95% CI lower\": (\n                lower\n            ),\n            \"95% CI upper\": (\n                upper\n            ),\n            \"95% CI\": (\n                f\"{lower:.6f}–{upper:.6f}\"\n            ),\n            \"CI interpretation\": (\n                \"Bootstrap confidence interval\"\n            ),\n        })\n\n\ntable_b_df = pd.DataFrame(\n    holdout_ci_records\n)\n\n\n# =============================================================================\n# 8. Generalization-gap table\n# =============================================================================\n\ngeneralization_records = []\n\n\ncomparison_definitions = [\n    (\n        \"Locked validation minus internal CV\",\n        \"One-time locked validation\",\n        \"Registered internal cross-validation\",\n    ),\n    (\n        \"Final test minus internal CV\",\n        \"One-time final test\",\n        \"Registered internal cross-validation\",\n    ),\n    (\n        \"Final test minus locked validation\",\n        \"One-time final test\",\n        \"One-time locked validation\",\n    ),\n]\n\n\nperformance_indexed = performance_df.set_index(\n    \"evaluation_scope\"\n)\n\n\nfor metric_name in [\n    \"qwk\",\n    \"accuracy\",\n    \"balanced_accuracy\",\n    \"macro_f1\",\n]:\n\n    for (\n        comparison_name,\n        minuend_scope,\n        subtrahend_scope,\n    ) in comparison_definitions:\n\n        minuend_value = float(\n            performance_indexed.loc[\n                minuend_scope,\n                metric_name,\n            ]\n        )\n\n        subtrahend_value = float(\n            performance_indexed.loc[\n                subtrahend_scope,\n                metric_name,\n            ]\n        )\n\n        absolute_gap = (\n            minuend_value\n            -\n            subtrahend_value\n        )\n\n        relative_gap_percent = (\n            absolute_gap\n            /\n            subtrahend_value\n            *\n            100.0\n        )\n\n        direction = (\n            \"Improvement\"\n            if absolute_gap\n            >\n            0\n            else\n            \"Decline\"\n            if absolute_gap\n            <\n            0\n            else\n            \"No change\"\n        )\n\n        generalization_records.append({\n            \"Metric\": (\n                METRIC_LABELS[\n                    metric_name\n                ]\n            ),\n            \"Metric code\": (\n                metric_name\n            ),\n            \"Comparison\": (\n                comparison_name\n            ),\n            \"Absolute gap\": (\n                absolute_gap\n            ),\n            \"Relative gap (%)\": (\n                relative_gap_percent\n            ),\n            \"Direction\": (\n                direction\n            ),\n        })\n\n\ntable_d_df = pd.DataFrame(\n    generalization_records\n)\n\n\n# =============================================================================\n# 9. Class-wise recall and confusion matrices\n# =============================================================================\n\nclass_error_df = pd.read_csv(\n    CLASS_ERROR_SOURCE_PATH\n)\n\n\nrequired_class_columns = {\n    \"evaluation_scope\",\n    \"class_name\",\n    \"true_count\",\n    \"exact_grade_recall\",\n}\n\n\nif not required_class_columns.issubset(\n    class_error_df.columns\n):\n\n    raise RuntimeError(\n        \"Step 12B class-error table has unexpected columns.\"\n    )\n\n\nclass_error_df = class_error_df[\n    class_error_df[\n        \"evaluation_scope\"\n    ].isin(\n        [\n            \"Locked validation\",\n            \"Final test\",\n        ]\n    )\n].copy()\n\n\nclass_publication_records = []\n\n\nfor evaluation_scope in [\n    \"Locked validation\",\n    \"Final test\",\n]:\n\n    scope_df = class_error_df[\n        class_error_df[\n            \"evaluation_scope\"\n        ]\n        ==\n        evaluation_scope\n    ].copy()\n\n    if len(\n        scope_df\n    ) != 5:\n\n        raise RuntimeError(\n            \"Class-wise recall table does not contain five classes \"\n            f\"for {evaluation_scope}.\"\n        )\n\n    scope_df = scope_df.set_index(\n        \"class_name\"\n    )\n\n    for class_name, display_name in zip(\n        CLASS_NAMES,\n        CLASS_DISPLAY_NAMES,\n    ):\n\n        if class_name not in scope_df.index:\n\n            raise RuntimeError(\n                f\"Missing class in Step 12B evidence: {class_name}\"\n            )\n\n        observed_count = int(\n            scope_df.loc[\n                class_name,\n                \"true_count\",\n            ]\n        )\n\n        observed_recall = float(\n            scope_df.loc[\n                class_name,\n                \"exact_grade_recall\",\n            ]\n        )\n\n        expected_count = (\n            EXPECTED_CLASS_COUNTS[\n                evaluation_scope\n            ][\n                class_name\n            ]\n        )\n\n        expected_recall = (\n            EXPECTED_CLASS_RECALL[\n                evaluation_scope\n            ][\n                class_name\n            ]\n        )\n\n        if observed_count != expected_count:\n\n            raise RuntimeError(\n                \"Class-count verification failed for \"\n                f\"{evaluation_scope}, {class_name}.\"\n            )\n\n        if abs(\n            observed_recall\n            -\n            expected_recall\n        ) > TOLERANCE:\n\n            raise RuntimeError(\n                \"Class-recall verification failed for \"\n                f\"{evaluation_scope}, {class_name}.\"\n            )\n\n        class_publication_records.append({\n            \"Evaluation scope\": (\n                evaluation_scope\n            ),\n            \"Class\": (\n                display_name\n            ),\n            \"Class code\": (\n                class_name\n            ),\n            \"Samples\": (\n                observed_count\n            ),\n            \"Recall\": (\n                observed_recall\n            ),\n        })\n\n\ntable_c_df = pd.DataFrame(\n    class_publication_records\n)\n\n\nconfusion_long_df = pd.read_csv(\n    CONFUSION_LONG_SOURCE_PATH\n)\n\n\nrequired_confusion_columns = {\n    \"evaluation_scope\",\n    \"true_class\",\n    \"predicted_class\",\n    \"count\",\n}\n\n\nif not required_confusion_columns.issubset(\n    confusion_long_df.columns\n):\n\n    raise RuntimeError(\n        \"Step 12B confusion table has unexpected columns.\"\n    )\n\n\nconfusion_matrices = {}\nnormalized_confusion_matrices = {}\n\n\nfor evaluation_scope in [\n    \"Locked validation\",\n    \"Final test\",\n]:\n\n    scope_df = confusion_long_df[\n        confusion_long_df[\n            \"evaluation_scope\"\n        ]\n        ==\n        evaluation_scope\n    ].copy()\n\n    pivot_df = scope_df.pivot(\n        index=\"true_class\",\n        columns=\"predicted_class\",\n        values=\"count\",\n    )\n\n    pivot_df = pivot_df.reindex(\n        index=CLASS_NAMES,\n        columns=CLASS_NAMES,\n        fill_value=0,\n    )\n\n    count_matrix = pivot_df.to_numpy(\n        dtype=np.int64\n    )\n\n    observed_row_totals = count_matrix.sum(\n        axis=1\n    )\n\n    expected_row_totals = (\n        EXPECTED_CONFUSION_ROW_TOTALS[\n            evaluation_scope\n        ]\n    )\n\n    if not np.array_equal(\n        observed_row_totals,\n        expected_row_totals,\n    ):\n\n        raise RuntimeError(\n            \"Confusion-matrix row totals do not match \"\n            f\"{evaluation_scope} class counts.\"\n        )\n\n    normalized_matrix = np.divide(\n        count_matrix,\n        observed_row_totals[\n            :,\n            None,\n        ],\n        out=np.zeros_like(\n            count_matrix,\n            dtype=np.float64,\n        ),\n        where=(\n            observed_row_totals[\n                :,\n                None,\n            ]\n            !=\n            0\n        ),\n    )\n\n    confusion_matrices[\n        evaluation_scope\n    ] = count_matrix\n\n    normalized_confusion_matrices[\n        evaluation_scope\n    ] = normalized_matrix\n\n\nvalidation_confusion_df = pd.DataFrame(\n    confusion_matrices[\n        \"Locked validation\"\n    ],\n    index=CLASS_DISPLAY_NAMES,\n    columns=CLASS_DISPLAY_NAMES,\n)\n\nvalidation_confusion_df.insert(\n    0,\n    \"Reference grade\",\n    validation_confusion_df.index,\n)\n\nvalidation_confusion_df = validation_confusion_df.reset_index(\n    drop=True\n)\n\n\ntest_confusion_df = pd.DataFrame(\n    confusion_matrices[\n        \"Final test\"\n    ],\n    index=CLASS_DISPLAY_NAMES,\n    columns=CLASS_DISPLAY_NAMES,\n)\n\ntest_confusion_df.insert(\n    0,\n    \"Reference grade\",\n    test_confusion_df.index,\n)\n\ntest_confusion_df = test_confusion_df.reset_index(\n    drop=True\n)\n\n\nvalidation_confusion_normalized_df = pd.DataFrame(\n    normalized_confusion_matrices[\n        \"Locked validation\"\n    ],\n    index=CLASS_DISPLAY_NAMES,\n    columns=CLASS_DISPLAY_NAMES,\n)\n\nvalidation_confusion_normalized_df.insert(\n    0,\n    \"Reference grade\",\n    validation_confusion_normalized_df.index,\n)\n\nvalidation_confusion_normalized_df = (\n    validation_confusion_normalized_df.reset_index(\n        drop=True\n    )\n)\n\n\ntest_confusion_normalized_df = pd.DataFrame(\n    normalized_confusion_matrices[\n        \"Final test\"\n    ],\n    index=CLASS_DISPLAY_NAMES,\n    columns=CLASS_DISPLAY_NAMES,\n)\n\ntest_confusion_normalized_df.insert(\n    0,\n    \"Reference grade\",\n    test_confusion_normalized_df.index,\n)\n\ntest_confusion_normalized_df = (\n    test_confusion_normalized_df.reset_index(\n        drop=True\n    )\n)\n\n\n# =============================================================================\n# 10. Preprocessing ablation tables\n# =============================================================================\n\ntable_e_df = pd.DataFrame([\n    {\n        \"Preprocessing variant\": (\n            \"Retinal crop only\"\n        ),\n        \"Seed\": 42,\n        \"QWK\": 0.892082,\n        \"Accuracy\": 0.820156,\n        \"Balanced accuracy\": 0.661619,\n        \"Macro F1\": 0.659862,\n        \"Comparison status\": (\n            \"Same seed and same group-aware folds\"\n        ),\n    },\n    {\n        \"Preprocessing variant\": (\n            \"Always mild LAB-CLAHE\"\n        ),\n        \"Seed\": 42,\n        \"QWK\": 0.898596,\n        \"Accuracy\": 0.832446,\n        \"Balanced accuracy\": 0.667986,\n        \"Macro F1\": 0.674525,\n        \"Comparison status\": (\n            \"Same seed and same group-aware folds\"\n        ),\n    },\n    {\n        \"Preprocessing variant\": (\n            \"Stochastic mild LAB-CLAHE\"\n        ),\n        \"Seed\": 42,\n        \"QWK\": 0.899112,\n        \"Accuracy\": 0.842278,\n        \"Balanced accuracy\": 0.681953,\n        \"Macro F1\": 0.691066,\n        \"Comparison status\": (\n            \"Same seed and same group-aware folds\"\n        ),\n    },\n])\n\n\ntable_f_df = pd.DataFrame([\n    {\n        \"Preprocessing variant\": (\n            \"Always mild LAB-CLAHE\"\n        ),\n        \"Seed count\": 2,\n        \"Mean QWK\": 0.902228,\n        \"QWK SD\": 0.005138,\n        \"Mean accuracy\": 0.836338,\n        \"Mean balanced accuracy\": 0.673618,\n        \"Mean macro F1\": 0.681278,\n        \"Formal selection\": (\n            \"Selected by pre-specified balanced-accuracy tie-break\"\n        ),\n    },\n    {\n        \"Preprocessing variant\": (\n            \"Stochastic mild LAB-CLAHE\"\n        ),\n        \"Seed count\": 2,\n        \"Mean QWK\": 0.902873,\n        \"QWK SD\": 0.005318,\n        \"Mean accuracy\": 0.835518,\n        \"Mean balanced accuracy\": 0.669081,\n        \"Mean macro F1\": 0.678123,\n        \"Formal selection\": (\n            \"Not selected; no significant QWK superiority\"\n        ),\n    },\n])\n\n\n# =============================================================================\n# 11. Model-complexity and bounded-revision tables\n# =============================================================================\n\ntable_g_df = pd.DataFrame([\n    {\n        \"Model\": (\n            \"Registered EfficientNet-B0 baseline\"\n        ),\n        \"Common evaluation folds\": (\n            \"Folds 2–3\"\n        ),\n        \"Parameters\": 4_013_953,\n        \"Parameters (million)\": (\n            4_013_953\n            /\n            1_000_000.0\n        ),\n        \"Mean QWK\": 0.906681,\n        \"Mean balanced accuracy\": 0.674669,\n        \"Mean macro F1\": 0.680174,\n        \"QWK per million parameters\": (\n            0.906681\n            /\n            (\n                4_013_953\n                /\n                1_000_000.0\n            )\n        ),\n        \"Formal outcome\": (\n            \"Selected compact architecture\"\n        ),\n    },\n    {\n        \"Model\": (\n            \"Original OLG-DRNet\"\n        ),\n        \"Common evaluation folds\": (\n            \"Folds 2–3\"\n        ),\n        \"Parameters\": 5_773_319,\n        \"Parameters (million)\": (\n            5_773_319\n            /\n            1_000_000.0\n        ),\n        \"Mean QWK\": 0.900731,\n        \"Mean balanced accuracy\": 0.643946,\n        \"Mean macro F1\": 0.655466,\n        \"QWK per million parameters\": (\n            0.900731\n            /\n            (\n                5_773_319\n                /\n                1_000_000.0\n            )\n        ),\n        \"Formal outcome\": (\n            \"No demonstrated value-add\"\n        ),\n    },\n    {\n        \"Model\": (\n            \"EfficientNet-B4 + Swin-Tiny comparator\"\n        ),\n        \"Common evaluation folds\": (\n            \"Folds 2–3\"\n        ),\n        \"Parameters\": 46_385_865,\n        \"Parameters (million)\": (\n            46_385_865\n            /\n            1_000_000.0\n        ),\n        \"Mean QWK\": 0.901207,\n        \"Mean balanced accuracy\": 0.633002,\n        \"Mean macro F1\": 0.654726,\n        \"QWK per million parameters\": (\n            0.901207\n            /\n            (\n                46_385_865\n                /\n                1_000_000.0\n            )\n        ),\n        \"Formal outcome\": (\n            \"No demonstrated value-add\"\n        ),\n    },\n])\n\n\ntable_h_df = pd.DataFrame([\n    {\n        \"Model\": (\n            \"Registered EfficientNet-B0 baseline\"\n        ),\n        \"Evaluation scope\": (\n            \"Fold 1 bounded-revision screen\"\n        ),\n        \"Parameters\": 4_013_953,\n        \"QWK\": 0.891461,\n        \"Balanced accuracy\": 0.671972,\n        \"Macro F1\": 0.679624,\n        \"Severe recall\": np.nan,\n        \"PDR recall\": np.nan,\n        \"Formal gate outcome\": (\n            \"Reference comparator\"\n        ),\n    },\n    {\n        \"Model\": (\n            \"R1 residual lesion-adapter revision\"\n        ),\n        \"Evaluation scope\": (\n            \"Fold 1 bounded-revision screen\"\n        ),\n        \"Parameters\": 4_222_276,\n        \"QWK\": 0.871755,\n        \"Balanced accuracy\": 0.683776,\n        \"Macro F1\": 0.647215,\n        \"Severe recall\": 0.705882,\n        \"PDR recall\": 0.346154,\n        \"Formal gate outcome\": (\n            \"Rejected: failed QWK and macro-F1 gates\"\n        ),\n    },\n])\n\n\n# =============================================================================\n# 12. Error-direction and severity tables\n# =============================================================================\n\nerror_direction_df = pd.read_csv(\n    ERROR_DIRECTION_SOURCE_PATH\n)\n\n\nrequired_error_columns = {\n    \"evaluation_scope\",\n    \"rows\",\n    \"exact_correct\",\n    \"total_errors\",\n    \"undergraded_count\",\n    \"overgraded_count\",\n    \"undergraded_fraction_among_errors\",\n    \"adjacent_fraction_among_errors\",\n    \"large_error_count\",\n    \"large_error_fraction_all_rows\",\n}\n\n\nif not required_error_columns.issubset(\n    error_direction_df.columns\n):\n\n    raise RuntimeError(\n        \"Step 12B error-direction table has unexpected columns.\"\n    )\n\n\ntable_i_df = error_direction_df[\n    [\n        \"evaluation_scope\",\n        \"rows\",\n        \"exact_correct\",\n        \"total_errors\",\n        \"undergraded_count\",\n        \"overgraded_count\",\n        \"undergraded_fraction_among_errors\",\n        \"adjacent_error_count\",\n        \"adjacent_fraction_among_errors\",\n        \"large_error_count\",\n        \"large_error_fraction_all_rows\",\n    ]\n].copy()\n\n\ntable_i_df.columns = [\n    \"Evaluation scope\",\n    \"Samples\",\n    \"Exact-grade correct\",\n    \"Exact-grade errors\",\n    \"Undergraded errors\",\n    \"Overgraded errors\",\n    \"Undergrading fraction among errors\",\n    \"Adjacent errors\",\n    \"Adjacent fraction among errors\",\n    \"Errors >1 grade\",\n    \"Errors >1 grade fraction of all samples\",\n]\n\n\nbinary_severity_df = pd.read_csv(\n    BINARY_SEVERITY_SOURCE_PATH\n)\n\n\nrequired_binary_columns = {\n    \"evaluation_scope\",\n    \"analysis_name\",\n    \"true_positive\",\n    \"false_negative\",\n    \"true_negative\",\n    \"false_positive\",\n    \"sensitivity\",\n    \"specificity\",\n    \"precision\",\n    \"negative_predictive_value\",\n    \"accuracy\",\n}\n\n\nif not required_binary_columns.issubset(\n    binary_severity_df.columns\n):\n\n    raise RuntimeError(\n        \"Step 12B binary-severity table has unexpected columns.\"\n    )\n\n\ntable_j_df = binary_severity_df[\n    [\n        \"evaluation_scope\",\n        \"analysis_name\",\n        \"true_positive\",\n        \"false_negative\",\n        \"true_negative\",\n        \"false_positive\",\n        \"sensitivity\",\n        \"specificity\",\n        \"precision\",\n        \"negative_predictive_value\",\n        \"accuracy\",\n    ]\n].copy()\n\n\ntable_j_df.columns = [\n    \"Evaluation scope\",\n    \"Descriptive threshold analysis\",\n    \"True positive\",\n    \"False negative\",\n    \"True negative\",\n    \"False positive\",\n    \"Sensitivity\",\n    \"Specificity\",\n    \"Precision\",\n    \"Negative predictive value\",\n    \"Accuracy\",\n]\n\n\n# =============================================================================\n# 13. External context table — explicitly non-comparable\n# =============================================================================\n\ntable_k_df = pd.DataFrame([\n    {\n        \"Study/model\": (\n            \"Mother-paper hybrid CNN–Transformer\"\n        ),\n        \"Evaluation description\": (\n            \"Paper-reported APTOS evaluation\"\n        ),\n        \"QWK\": 0.863900,\n        \"Accuracy\": 0.805500,\n        \"Balanced accuracy\": 0.670000,\n        \"Macro F1\": 0.657700,\n        \"Formal same-split comparison\": (\n            False\n        ),\n        \"Interpretation\": (\n            \"External numerical context only; split and \"\n            \"experimental protocol are not identical\"\n        ),\n    },\n    {\n        \"Study/model\": (\n            \"Registered EfficientNet-B0 final model\"\n        ),\n        \"Evaluation description\": (\n            \"One-time locked untouched final test\"\n        ),\n        \"QWK\": 0.847063,\n        \"Accuracy\": 0.770115,\n        \"Balanced accuracy\": 0.628933,\n        \"Macro F1\": 0.608571,\n        \"Formal same-split comparison\": (\n            False\n        ),\n        \"Interpretation\": (\n            \"Leakage-controlled internal result; no claim of \"\n            \"formal superiority or equivalence\"\n        ),\n    },\n])\n\n\n# =============================================================================\n# 14. Create output directories after all source checks pass\n# =============================================================================\n\nfor directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n    OUTPUT_FIGURE_DIR,\n]:\n\n    directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\n# =============================================================================\n# 15. Save all publication tables\n# =============================================================================\n\natomic_csv_save(\n    table_a_df,\n    TABLE_A_PATH,\n)\n\natomic_csv_save(\n    table_b_df,\n    TABLE_B_PATH,\n)\n\natomic_csv_save(\n    table_c_df,\n    TABLE_C_PATH,\n)\n\natomic_csv_save(\n    table_d_df,\n    TABLE_D_PATH,\n)\n\natomic_csv_save(\n    table_e_df,\n    TABLE_E_PATH,\n)\n\natomic_csv_save(\n    table_f_df,\n    TABLE_F_PATH,\n)\n\natomic_csv_save(\n    table_g_df,\n    TABLE_G_PATH,\n)\n\natomic_csv_save(\n    table_h_df,\n    TABLE_H_PATH,\n)\n\natomic_csv_save(\n    table_i_df,\n    TABLE_I_PATH,\n)\n\natomic_csv_save(\n    table_j_df,\n    TABLE_J_PATH,\n)\n\natomic_csv_save(\n    table_k_df,\n    TABLE_K_PATH,\n)\n\natomic_csv_save(\n    validation_confusion_df,\n    CONFUSION_VALIDATION_PATH,\n)\n\natomic_csv_save(\n    test_confusion_df,\n    CONFUSION_TEST_PATH,\n)\n\natomic_csv_save(\n    validation_confusion_normalized_df,\n    CONFUSION_VALIDATION_NORMALIZED_PATH,\n)\n\natomic_csv_save(\n    test_confusion_normalized_df,\n    CONFUSION_TEST_NORMALIZED_PATH,\n)\n\n\n# =============================================================================\n# 16. Publication figure style\n# =============================================================================\n\nplt.rcParams.update({\n    \"font.family\": \"serif\",\n    \"font.serif\": [\n        \"Times New Roman\",\n        \"DejaVu Serif\",\n    ],\n    \"font.size\": 8,\n    \"axes.titlesize\": 10,\n    \"axes.labelsize\": 9,\n    \"xtick.labelsize\": 7,\n    \"ytick.labelsize\": 7,\n    \"legend.fontsize\": 7,\n    \"axes.linewidth\": 0.8,\n    \"figure.facecolor\": \"white\",\n    \"axes.facecolor\": \"white\",\n    \"savefig.facecolor\": \"white\",\n})\n\n\ngenerated_figure_paths = []\n\n\n# =============================================================================\n# 17. Figure 14A — primary performance\n# =============================================================================\n\nmetric_codes = [\n    \"qwk\",\n    \"accuracy\",\n    \"balanced_accuracy\",\n    \"macro_f1\",\n]\n\nmetric_display = [\n    \"QWK\",\n    \"Accuracy\",\n    \"Balanced\\naccuracy\",\n    \"Macro F1\",\n]\n\nx_positions = np.arange(\n    len(\n        metric_codes\n    )\n)\n\nbar_width = 0.23\n\nscope_styles = [\n    {\n        \"scope\": (\n            \"Registered internal cross-validation\"\n        ),\n        \"label\": (\n            \"Internal CV\"\n        ),\n        \"shade\": \"0.25\",\n        \"hatch\": \"\",\n    },\n    {\n        \"scope\": (\n            \"One-time locked validation\"\n        ),\n        \"label\": (\n            \"Locked validation\"\n        ),\n        \"shade\": \"0.55\",\n        \"hatch\": \"///\",\n    },\n    {\n        \"scope\": (\n            \"One-time final test\"\n        ),\n        \"label\": (\n            \"Final test\"\n        ),\n        \"shade\": \"0.82\",\n        \"hatch\": \"...\",\n    },\n]\n\n\nfigure_a, axis_a = plt.subplots(\n    figsize=(\n        8.0,\n        5.3,\n    )\n)\n\n\nfor scope_index, style in enumerate(\n    scope_styles\n):\n\n    values = [\n        EXPECTED_PRIMARY_METRICS[\n            style[\n                \"scope\"\n            ]\n        ][\n            metric_code\n        ]\n        for metric_code\n        in metric_codes\n    ]\n\n    bars = axis_a.bar(\n        x_positions\n        +\n        (\n            scope_index\n            -\n            1\n        )\n        *\n        bar_width,\n        values,\n        width=bar_width,\n        label=style[\n            \"label\"\n        ],\n        color=style[\n            \"shade\"\n        ],\n        edgecolor=\"black\",\n        linewidth=0.7,\n        hatch=style[\n            \"hatch\"\n        ],\n    )\n\n    annotate_bars(\n        axis_a,\n        bars,\n        decimal_places=3,\n        y_offset=0.008,\n    )\n\n\naxis_a.set_xticks(\n    x_positions\n)\n\naxis_a.set_xticklabels(\n    metric_display\n)\n\naxis_a.set_ylim(\n    0.50,\n    0.96,\n)\n\naxis_a.set_ylabel(\n    \"Performance\"\n)\n\naxis_a.set_title(\n    \"Registered internal CV and untouched holdout performance\"\n)\n\naxis_a.grid(\n    axis=\"y\",\n    linestyle=\"--\",\n    linewidth=0.5,\n    alpha=0.5,\n)\n\naxis_a.legend(\n    frameon=False,\n    loc=\"lower left\",\n)\n\nfigure_a.text(\n    0.5,\n    0.01,\n    (\n        \"Internal cross-validation was used for model development; \"\n        \"validation and final test were evaluated once after model locking.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6.5,\n)\n\nfigure_a.tight_layout(\n    rect=(\n        0,\n        0.04,\n        1,\n        1,\n    )\n)\n\n\ngenerated_figure_paths.extend(\n    save_figure_triplet(\n        figure_a,\n        \"figure_14A_primary_performance\",\n    )\n)\n\n\n# =============================================================================\n# 18. Figures 14B and 14C — normalized confusion matrices\n# =============================================================================\n\ngenerated_figure_paths.extend(\n    create_confusion_figure(\n        count_matrix=(\n            confusion_matrices[\n                \"Locked validation\"\n            ]\n        ),\n        normalized_matrix=(\n            normalized_confusion_matrices[\n                \"Locked validation\"\n            ]\n        ),\n        title=(\n            \"Locked-validation confusion matrix\"\n        ),\n        stem=(\n            \"figure_14B_locked_validation_confusion_matrix\"\n        ),\n    )\n)\n\n\ngenerated_figure_paths.extend(\n    create_confusion_figure(\n        count_matrix=(\n            confusion_matrices[\n                \"Final test\"\n            ]\n        ),\n        normalized_matrix=(\n            normalized_confusion_matrices[\n                \"Final test\"\n            ]\n        ),\n        title=(\n            \"Final-test confusion matrix\"\n        ),\n        stem=(\n            \"figure_14C_final_test_confusion_matrix\"\n        ),\n    )\n)\n\n\n# =============================================================================\n# 19. Figure 14D — class-wise recall\n# =============================================================================\n\nvalidation_recall = [\n    EXPECTED_CLASS_RECALL[\n        \"Locked validation\"\n    ][\n        class_name\n    ]\n    for class_name\n    in CLASS_NAMES\n]\n\ntest_recall = [\n    EXPECTED_CLASS_RECALL[\n        \"Final test\"\n    ][\n        class_name\n    ]\n    for class_name\n    in CLASS_NAMES\n]\n\n\nfigure_d, axis_d = plt.subplots(\n    figsize=(\n        8.2,\n        5.4,\n    )\n)\n\n\nx_positions = np.arange(\n    len(\n        CLASS_DISPLAY_NAMES\n    )\n)\n\nbar_width = 0.34\n\n\nvalidation_bars = axis_d.bar(\n    x_positions\n    -\n    bar_width\n    /\n    2.0,\n    validation_recall,\n    width=bar_width,\n    label=\"Locked validation\",\n    color=\"0.55\",\n    edgecolor=\"black\",\n    linewidth=0.7,\n    hatch=\"///\",\n)\n\n\ntest_bars = axis_d.bar(\n    x_positions\n    +\n    bar_width\n    /\n    2.0,\n    test_recall,\n    width=bar_width,\n    label=\"Final test\",\n    color=\"0.82\",\n    edgecolor=\"black\",\n    linewidth=0.7,\n    hatch=\"...\",\n)\n\n\nannotate_bars(\n    axis_d,\n    validation_bars,\n    decimal_places=3,\n    y_offset=0.012,\n)\n\nannotate_bars(\n    axis_d,\n    test_bars,\n    decimal_places=3,\n    y_offset=0.012,\n)\n\n\naxis_d.set_xticks(\n    x_positions\n)\n\naxis_d.set_xticklabels(\n    CLASS_DISPLAY_NAMES,\n    rotation=25,\n    ha=\"right\",\n)\n\naxis_d.set_ylim(\n    0.0,\n    1.10,\n)\n\naxis_d.set_ylabel(\n    \"Exact-grade recall\"\n)\n\naxis_d.set_title(\n    \"Class-wise recall on locked validation and final test\"\n)\n\naxis_d.grid(\n    axis=\"y\",\n    linestyle=\"--\",\n    linewidth=0.5,\n    alpha=0.5,\n)\n\naxis_d.legend(\n    frameon=False,\n    loc=\"upper right\",\n)\n\nfigure_d.text(\n    0.5,\n    0.005,\n    (\n        \"No-DR and Mild cases were identified more reliably than \"\n        \"Moderate, Severe and Proliferative DR.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6.5,\n)\n\nfigure_d.tight_layout(\n    rect=(\n        0,\n        0.04,\n        1,\n        1,\n    )\n)\n\n\ngenerated_figure_paths.extend(\n    save_figure_triplet(\n        figure_d,\n        \"figure_14D_classwise_recall\",\n    )\n)\n\n\n# =============================================================================\n# 20. Figure 14E — same-seed preprocessing ablation\n# =============================================================================\n\nablation_metric_columns = [\n    \"QWK\",\n    \"Accuracy\",\n    \"Balanced accuracy\",\n    \"Macro F1\",\n]\n\nablation_metric_labels = [\n    \"QWK\",\n    \"Accuracy\",\n    \"Balanced\\naccuracy\",\n    \"Macro F1\",\n]\n\nvariant_styles = [\n    {\n        \"variant\": (\n            \"Retinal crop only\"\n        ),\n        \"label\": (\n            \"Crop only\"\n        ),\n        \"shade\": \"0.25\",\n        \"hatch\": \"\",\n    },\n    {\n        \"variant\": (\n            \"Always mild LAB-CLAHE\"\n        ),\n        \"label\": (\n            \"Always CLAHE\"\n        ),\n        \"shade\": \"0.55\",\n        \"hatch\": \"///\",\n    },\n    {\n        \"variant\": (\n            \"Stochastic mild LAB-CLAHE\"\n        ),\n        \"label\": (\n            \"Stochastic CLAHE\"\n        ),\n        \"shade\": \"0.82\",\n        \"hatch\": \"...\",\n    },\n]\n\n\nfigure_e, axis_e = plt.subplots(\n    figsize=(\n        8.0,\n        5.3,\n    )\n)\n\n\nx_positions = np.arange(\n    len(\n        ablation_metric_columns\n    )\n)\n\nbar_width = 0.23\n\n\nfor variant_index, style in enumerate(\n    variant_styles\n):\n\n    variant_row = table_e_df[\n        table_e_df[\n            \"Preprocessing variant\"\n        ]\n        ==\n        style[\n            \"variant\"\n        ]\n    ].iloc[\n        0\n    ]\n\n    values = [\n        float(\n            variant_row[\n                column\n            ]\n        )\n        for column\n        in ablation_metric_columns\n    ]\n\n    bars = axis_e.bar(\n        x_positions\n        +\n        (\n            variant_index\n            -\n            1\n        )\n        *\n        bar_width,\n        values,\n        width=bar_width,\n        label=style[\n            \"label\"\n        ],\n        color=style[\n            \"shade\"\n        ],\n        edgecolor=\"black\",\n        linewidth=0.7,\n        hatch=style[\n            \"hatch\"\n        ],\n    )\n\n    annotate_bars(\n        axis_e,\n        bars,\n        decimal_places=3,\n        y_offset=0.006,\n    )\n\n\naxis_e.set_xticks(\n    x_positions\n)\n\naxis_e.set_xticklabels(\n    ablation_metric_labels\n)\n\naxis_e.set_ylim(\n    0.62,\n    0.93,\n)\n\naxis_e.set_ylabel(\n    \"Three-fold OOF performance\"\n)\n\naxis_e.set_title(\n    \"Preprocessing ablation under the same seed and folds\"\n)\n\naxis_e.grid(\n    axis=\"y\",\n    linestyle=\"--\",\n    linewidth=0.5,\n    alpha=0.5,\n)\n\naxis_e.legend(\n    frameon=False,\n    loc=\"lower left\",\n)\n\nfigure_e.text(\n    0.5,\n    0.01,\n    (\n        \"Seed 42 only. The repeated-seed selection used a \"\n        \"pre-specified balanced-accuracy tie-break.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6.5,\n)\n\nfigure_e.tight_layout(\n    rect=(\n        0,\n        0.04,\n        1,\n        1,\n    )\n)\n\n\ngenerated_figure_paths.extend(\n    save_figure_triplet(\n        figure_e,\n        \"figure_14E_preprocessing_ablation_same_seed\",\n    )\n)\n\n\n# =============================================================================\n# 21. Figure 14F — complexity versus common-fold QWK\n# =============================================================================\n\nfigure_f, axis_f = plt.subplots(\n    figsize=(\n        7.6,\n        5.5,\n    )\n)\n\n\nmodel_markers = [\n    \"o\",\n    \"s\",\n    \"^\",\n]\n\nmodel_shades = [\n    \"0.15\",\n    \"0.50\",\n    \"0.80\",\n]\n\n\nfor row_index, row in table_g_df.iterrows():\n\n    parameter_million = float(\n        row[\n            \"Parameters (million)\"\n        ]\n    )\n\n    qwk = float(\n        row[\n            \"Mean QWK\"\n        ]\n    )\n\n    axis_f.scatter(\n        parameter_million,\n        qwk,\n        s=95,\n        marker=model_markers[\n            row_index\n        ],\n        facecolor=model_shades[\n            row_index\n        ],\n        edgecolor=\"black\",\n        linewidth=0.8,\n        zorder=3,\n    )\n\n    label = (\n        \"EfficientNet-B0\"\n        if row_index\n        ==\n        0\n        else\n        \"OLG-DRNet\"\n        if row_index\n        ==\n        1\n        else\n        \"B4 + Swin-Tiny\"\n    )\n\n    x_offset = (\n        0.35\n        if row_index\n        <\n        2\n        else\n        -12.0\n    )\n\n    y_offset = (\n        0.0015\n        if row_index\n        ==\n        0\n        else\n        -0.0035\n        if row_index\n        ==\n        1\n        else\n        0.0015\n    )\n\n    axis_f.annotate(\n        (\n            f\"{label}\\n\"\n            f\"{parameter_million:.2f}M; \"\n            f\"QWK={qwk:.3f}\"\n        ),\n        xy=(\n            parameter_million,\n            qwk,\n        ),\n        xytext=(\n            parameter_million\n            +\n            x_offset,\n            qwk\n            +\n            y_offset,\n        ),\n        fontsize=7,\n        arrowprops={\n            \"arrowstyle\": \"-\",\n            \"linewidth\": 0.6,\n            \"color\": \"black\",\n        },\n    )\n\n\naxis_f.set_xscale(\n    \"log\"\n)\n\naxis_f.set_xlim(\n    3.2,\n    65.0,\n)\n\naxis_f.set_ylim(\n    0.890,\n    0.912,\n)\n\naxis_f.set_xlabel(\n    \"Trainable parameters (million; logarithmic scale)\"\n)\n\naxis_f.set_ylabel(\n    \"Mean QWK on common folds 2–3\"\n)\n\naxis_f.set_title(\n    \"Model-complexity and common-fold performance trade-off\"\n)\n\naxis_f.grid(\n    linestyle=\"--\",\n    linewidth=0.5,\n    alpha=0.5,\n)\n\nfigure_f.text(\n    0.5,\n    0.01,\n    (\n        \"Only models evaluated on the same folds are shown. \"\n        \"The compact baseline achieved the strongest common-fold QWK.\"\n    ),\n    ha=\"center\",\n    va=\"bottom\",\n    fontsize=6.5,\n)\n\nfigure_f.tight_layout(\n    rect=(\n        0,\n        0.04,\n        1,\n        1,\n    )\n)\n\n\ngenerated_figure_paths.extend(\n    save_figure_triplet(\n        figure_f,\n        \"figure_14F_model_complexity_tradeoff\",\n    )\n)\n\n\n# =============================================================================\n# 22. Publication-figure integrity\n# =============================================================================\n\nfigure_index_records = []\n\n\nfor figure_path in generated_figure_paths:\n\n    if not figure_path.exists():\n\n        raise FileNotFoundError(\n            f\"Step 14 figure is missing: {figure_path}\"\n        )\n\n    suffix = figure_path.suffix.lower()\n\n    width_pixels = np.nan\n    height_pixels = np.nan\n\n    if suffix == \".png\":\n\n        with Image.open(\n            figure_path\n        ) as image:\n\n            image.verify()\n\n        with Image.open(\n            figure_path\n        ) as image:\n\n            width_pixels, height_pixels = image.size\n\n        if (\n            width_pixels\n            <\n            2800\n            or\n            height_pixels\n            <\n            2600\n        ):\n\n            raise RuntimeError(\n                \"A Step 14 publication PNG has unexpectedly \"\n                \"low pixel dimensions.\"\n            )\n\n    figure_index_records.append({\n        \"Figure\": (\n            figure_path.stem\n        ),\n        \"Path\": str(\n            figure_path\n        ),\n        \"Format\": (\n            suffix.replace(\n                \".\",\n                \"\"\n            ).upper()\n        ),\n        \"Size bytes\": int(\n            figure_path.stat().st_size\n        ),\n        \"Width pixels\": (\n            width_pixels\n        ),\n        \"Height pixels\": (\n            height_pixels\n        ),\n        \"SHA-256\": sha256_file(\n            figure_path\n        ),\n        \"Integrity passed\": (\n            True\n        ),\n    })\n\n\nfigure_index_df = pd.DataFrame(\n    figure_index_records\n)\n\n\nif len(\n    figure_index_df\n) != 18:\n\n    raise RuntimeError(\n        \"Expected 18 Step 14 publication figure files.\"\n    )\n\n\natomic_csv_save(\n    figure_index_df,\n    FIGURE_INDEX_PATH,\n)\n\n\n# =============================================================================\n# 23. Figure captions\n# =============================================================================\n\nfigure_captions = \"\"\"\nFigure 14A. Registered group-aware internal cross-validation performance\nand one-time locked validation and final-test performance for the selected\nEfficientNet-B0 model. Cross-validation was used during development,\nwhereas both holdouts remained unavailable for model fitting. The decline\nfrom cross-validation to untouched holdouts indicates measurable\ngeneralization optimism.\n\nFigure 14B. Row-normalized confusion matrix for the one-time locked\nvalidation set. Cell labels report the number and row-wise percentage of\nsamples. Most No-DR and Mild cases were identified correctly, whereas\nModerate, Severe and Proliferative DR showed lower exact-grade\nsensitivity.\n\nFigure 14C. Row-normalized confusion matrix for the one-time final test.\nThe dominant error direction was undergrading, particularly\nModerate-to-Mild and advanced-disease-to-lower-grade predictions.\n\nFigure 14D. Exact-grade recall by diabetic-retinopathy class on locked\nvalidation and final test. Recall was high for No-DR and Mild cases but\nsubstantially lower for Moderate, Severe and Proliferative DR.\n\nFigure 14E. Same-seed preprocessing ablation under identical group-aware\nfolds. The figure provides a fair within-seed comparison of retinal crop,\nalways-applied mild LAB-CLAHE and stochastic mild LAB-CLAHE. Formal\npreprocessing selection additionally considered repeated-seed balanced\naccuracy according to the pre-specified tie-break rule.\n\nFigure 14F. Parameter count and mean QWK for models evaluated on common\nfolds 2–3. The compact EfficientNet-B0 baseline achieved higher common-\nfold QWK than the original OLG-DRNet and the substantially larger\nEfficientNet-B4 plus Swin-Tiny comparator. The figure is a same-fold\narchitectural comparison and does not include the separately screened R1\nrevision.\n\"\"\".strip()\n\n\natomic_text_save(\n    figure_captions,\n    FIGURE_CAPTIONS_PATH,\n)\n\n\n# =============================================================================\n# 24. Source verification\n# =============================================================================\n\nsource_verification_record = {\n    \"step\": (\n        \"STEP_14_PUBLICATION_METRICS_TABLES_AND_FIGURES\"\n    ),\n    \"verified_utc\": (\n        utc_now()\n    ),\n    \"state_sources\": {\n        \"step_10d\": {\n            \"path\": str(\n                STEP10D_STATE_PATH\n            ),\n            \"sha256\": sha256_file(\n                STEP10D_STATE_PATH\n            ),\n            \"status\": (\n                step10d_state.get(\n                    \"status\"\n                )\n            ),\n        },\n        \"step_12a\": {\n            \"path\": str(\n                STEP12A_STATE_PATH\n            ),\n            \"sha256\": sha256_file(\n                STEP12A_STATE_PATH\n            ),\n            \"status\": (\n                step12a_state.get(\n                    \"status\"\n                )\n            ),\n        },\n        \"step_12b\": {\n            \"path\": str(\n                STEP12B_STATE_PATH\n            ),\n            \"sha256\": sha256_file(\n                STEP12B_STATE_PATH\n            ),\n            \"status\": (\n                step12b_state.get(\n                    \"status\"\n                )\n            ),\n        },\n        \"step_13c\": {\n            \"path\": str(\n                STEP13C_STATE_PATH\n            ),\n            \"sha256\": sha256_file(\n                STEP13C_STATE_PATH\n            ),\n            \"status\": (\n                step13c_state.get(\n                    \"status\"\n                )\n            ),\n        },\n    },\n    \"performance_source\": {\n        \"mode\": (\n            performance_source_mode\n        ),\n        \"discovered_path\": (\n            str(\n                performance_source[\n                    \"path\"\n                ]\n            )\n            if performance_source\n            is not None\n            else\n            \"\"\n        ),\n        \"locked_values_verified\": (\n            True\n        ),\n    },\n    \"step_12b_sources\": {\n        \"confusion_matrix\": {\n            \"path\": str(\n                CONFUSION_LONG_SOURCE_PATH\n            ),\n            \"sha256\": sha256_file(\n                CONFUSION_LONG_SOURCE_PATH\n            ),\n        },\n        \"class_specific_error\": {\n            \"path\": str(\n                CLASS_ERROR_SOURCE_PATH\n            ),\n            \"sha256\": sha256_file(\n                CLASS_ERROR_SOURCE_PATH\n            ),\n        },\n        \"error_direction\": {\n            \"path\": str(\n                ERROR_DIRECTION_SOURCE_PATH\n            ),\n            \"sha256\": sha256_file(\n                ERROR_DIRECTION_SOURCE_PATH\n            ),\n        },\n        \"binary_severity\": {\n            \"path\": str(\n                BINARY_SEVERITY_SOURCE_PATH\n            ),\n            \"sha256\": sha256_file(\n                BINARY_SEVERITY_SOURCE_PATH\n            ),\n        },\n    },\n    \"backup_verification\": (\n        backup_verification\n    ),\n    \"locked_constants_used\": {\n        \"preprocessing_ablation\": (\n            \"Step 11A-R1 verified output\"\n        ),\n        \"model_complexity\": (\n            \"Step 11B-QA2 corrected parameter evidence\"\n        ),\n        \"holdout_confidence_intervals\": (\n            \"Step 10C and Step 10D locked outputs\"\n        ),\n        \"external_context\": (\n            \"Mother-paper reported APTOS values; \"\n            \"marked non-comparable\"\n        ),\n    },\n    \"safety\": {\n        \"new_training_performed\": (\n            False\n        ),\n        \"optimizer_created\": (\n            False\n        ),\n        \"model_loaded\": (\n            False\n        ),\n        \"model_inference_performed\": (\n            False\n        ),\n        \"raw_images_loaded\": (\n            False\n        ),\n        \"validation_evaluated\": (\n            False\n        ),\n        \"final_test_evaluated\": (\n            False\n        ),\n        \"predictions_regenerated\": (\n            False\n        ),\n        \"model_change_performed\": (\n            False\n        ),\n    },\n}\n\n\natomic_json_save(\n    source_verification_record,\n    SOURCE_VERIFICATION_PATH,\n)\n\n\n# =============================================================================\n# 25. Manuscript interpretation note\n# =============================================================================\n\nmanuscript_note = \"\"\"\nSTEP 14 — PUBLICATION METRICS INTERPRETATION\n\nThe registered EfficientNet-B0 model achieved an internal group-aware\ncross-validation QWK of 0.9016. Performance declined to 0.7834 on the\none-time locked validation set and 0.8471 on the one-time final test,\ndemonstrating that internal cross-validation was optimistic relative to\nboth untouched holdouts. The final-test point estimates were higher than\nthe validation estimates, but their bootstrap confidence intervals\noverlapped for all primary metrics; the difference should therefore be\ndescribed as sampling variability rather than a statistically established\nimprovement.\n\nClass-wise analysis showed high recall for No-DR and Mild cases but lower\nexact-grade recall for Moderate, Severe and Proliferative DR. The\nprediction-level audit established that 100 of 120 final-test errors were\nundergradings. Moderate-to-Mild was the most frequent transition, while\nProliferative DR had the lowest exact-grade recall. Despite these\nlimitations, 94.6% of final-test predictions remained within one ordinal\ngrade of the reference.\n\nThe same-seed preprocessing ablation showed modest performance differences\namong crop-only, always-CLAHE and stochastic-CLAHE variants. The repeated-\nseed analysis did not establish QWK superiority of stochastic CLAHE.\nAlways-applied mild LAB-CLAHE was selected using the pre-specified\nbalanced-accuracy tie-break and remained locked before final training.\n\nOn common folds 2–3, the compact EfficientNet-B0 baseline achieved higher\nQWK, balanced accuracy and macro F1 than both the original OLG-DRNet and\nthe substantially larger EfficientNet-B4 plus Swin-Tiny comparator. The\nbounded R1 revision improved Severe recall on its screening fold but\nfailed the pre-specified QWK and macro-F1 gates and was rejected.\n\nThe mother-paper metrics are included only as external numerical context.\nThey were obtained under a different experimental protocol and cannot\nsupport a formal superiority, inferiority or equivalence claim. The\nstrongest defensible contribution is a leakage-controlled, reproducible\nevaluation showing that a compact architecture can remain competitive\nwhile substantially reducing parameter count, together with transparent\nnegative architecture findings, ablation evidence, holdout uncertainty,\nerror-direction analysis and locked XAI evaluation.\n\"\"\".strip()\n\n\natomic_text_save(\n    manuscript_note,\n    MANUSCRIPT_NOTE_PATH,\n)\n\n\n# =============================================================================\n# 26. Formal summary\n# =============================================================================\n\nfinal_test_row = performance_df[\n    performance_df[\n        \"evaluation_scope\"\n    ]\n    ==\n    \"One-time final test\"\n].iloc[\n    0\n]\n\n\nsummary_record = {\n    \"step\": (\n        \"STEP_14_PUBLICATION_METRICS_TABLES_AND_FIGURES\"\n    ),\n    \"status\": (\n        \"completed\"\n    ),\n    \"completed_utc\": (\n        utc_now()\n    ),\n    \"publication_tables\": {\n        \"table_count\": 11,\n        \"primary_performance\": str(\n            TABLE_A_PATH\n        ),\n        \"holdout_confidence_intervals\": str(\n            TABLE_B_PATH\n        ),\n        \"classwise_recall\": str(\n            TABLE_C_PATH\n        ),\n        \"generalization_gaps\": str(\n            TABLE_D_PATH\n        ),\n        \"preprocessing_ablation_same_seed\": str(\n            TABLE_E_PATH\n        ),\n        \"preprocessing_repeated_seed\": str(\n            TABLE_F_PATH\n        ),\n        \"model_complexity_common_folds\": str(\n            TABLE_G_PATH\n        ),\n        \"bounded_revision_screen\": str(\n            TABLE_H_PATH\n        ),\n        \"error_direction\": str(\n            TABLE_I_PATH\n        ),\n        \"binary_severity\": str(\n            TABLE_J_PATH\n        ),\n        \"external_context_noncomparable\": str(\n            TABLE_K_PATH\n        ),\n    },\n    \"publication_figures\": {\n        \"figure_family_count\": 6,\n        \"file_count\": int(\n            len(\n                figure_index_df\n            )\n        ),\n        \"formats\": [\n            \"PNG 600 dpi\",\n            \"PDF\",\n            \"SVG\",\n        ],\n        \"all_integrity_checks_passed\": bool(\n            figure_index_df[\n                \"Integrity passed\"\n            ].all()\n        ),\n    },\n    \"final_test_primary_metrics\": {\n        \"qwk\": float(\n            final_test_row[\n                \"qwk\"\n            ]\n        ),\n        \"accuracy\": float(\n            final_test_row[\n                \"accuracy\"\n            ]\n        ),\n        \"balanced_accuracy\": float(\n            final_test_row[\n                \"balanced_accuracy\"\n            ]\n        ),\n        \"macro_f1\": float(\n            final_test_row[\n                \"macro_f1\"\n            ]\n        ),\n    },\n    \"formal_findings\": {\n        \"cv_minus_final_test_qwk_optimism\": (\n            0.901608\n            -\n            0.847063\n        ),\n        \"final_test_lowest_recall_class\": (\n            \"Proliferative DR\"\n        ),\n        \"final_test_lowest_recall\": (\n            0.384615\n        ),\n        \"final_test_undergraded_errors\": (\n            100\n        ),\n        \"final_test_total_errors\": (\n            120\n        ),\n        \"final_test_within_one_grade_accuracy\": (\n            0.946360\n        ),\n        \"selected_architecture\": (\n            \"Registered EfficientNet-B0 baseline\"\n        ),\n        \"selected_parameter_count\": (\n            4_013_953\n        ),\n    },\n    \"claim_constraints\": [\n        (\n            \"External mother-paper values are not a \"\n            \"formal same-split comparison.\"\n        ),\n        (\n            \"Holdout point-estimate differences must not \"\n            \"be described as statistically significant \"\n            \"based only on overlapping confidence intervals.\"\n        ),\n        (\n            \"Binary severity analyses are post hoc \"\n            \"descriptive summaries.\"\n        ),\n        (\n            \"XAI maps are model attributions and not \"\n            \"validated lesion segmentations.\"\n        ),\n    ],\n    \"safety\": {\n        \"new_training_performed\": (\n            False\n        ),\n        \"optimizer_created\": (\n            False\n        ),\n        \"model_loaded\": (\n            False\n        ),\n        \"model_inference_performed\": (\n            False\n        ),\n        \"raw_images_loaded\": (\n            False\n        ),\n        \"validation_evaluated\": (\n            False\n        ),\n        \"final_test_evaluated\": (\n            False\n        ),\n        \"predictions_regenerated\": (\n            False\n        ),\n        \"model_change_performed\": (\n            False\n        ),\n    },\n    \"next_stage\": (\n        \"STEP_15_FINAL_REPRODUCIBILITY_AUDIT_AND_RESULTS_PACKAGE\"\n    ),\n}\n\n\natomic_json_save(\n    summary_record,\n    SUMMARY_PATH,\n)\n\n\n# =============================================================================\n# 27. Formal state\n# =============================================================================\n\nstate_record = {\n    \"step\": (\n        \"STEP_14_PUBLICATION_METRICS_TABLES_AND_FIGURES\"\n    ),\n    \"status\": (\n        \"complete\"\n    ),\n    \"updated_utc\": (\n        utc_now()\n    ),\n    \"publication_metrics_tables_completed\": (\n        True\n    ),\n    \"publication_figure_generation_completed\": (\n        True\n    ),\n    \"publication_table_count\": (\n        11\n    ),\n    \"publication_figure_family_count\": (\n        6\n    ),\n    \"publication_figure_file_count\": int(\n        len(\n            figure_index_df\n        )\n    ),\n    \"all_publication_figure_integrity_checks_passed\": bool(\n        figure_index_df[\n            \"Integrity passed\"\n        ].all()\n    ),\n    \"preprocessing_ablation_included\": (\n        True\n    ),\n    \"model_architecture_comparison_included\": (\n        True\n    ),\n    \"bounded_revision_screen_included\": (\n        True\n    ),\n    \"generalization_gap_included\": (\n        True\n    ),\n    \"classwise_performance_included\": (\n        True\n    ),\n    \"error_direction_analysis_included\": (\n        True\n    ),\n    \"external_context_marked_noncomparable\": (\n        True\n    ),\n    \"new_training_performed\": (\n        False\n    ),\n    \"model_loaded\": (\n        False\n    ),\n    \"model_inference_performed\": (\n        False\n    ),\n    \"raw_images_loaded\": (\n        False\n    ),\n    \"validation_evaluated\": (\n        False\n    ),\n    \"final_test_evaluated\": (\n        False\n    ),\n    \"predictions_regenerated\": (\n        False\n    ),\n    \"model_change_allowed\": (\n        False\n    ),\n    \"another_validation_evaluation_allowed\": (\n        False\n    ),\n    \"another_test_evaluation_allowed\": (\n        False\n    ),\n    \"figure_directory\": str(\n        OUTPUT_FIGURE_DIR\n    ),\n    \"table_directory\": str(\n        OUTPUT_EVIDENCE_DIR\n    ),\n    \"next_stage\": (\n        \"STEP_15_FINAL_REPRODUCIBILITY_AUDIT_AND_RESULTS_PACKAGE\"\n    ),\n}\n\n\natomic_json_save(\n    state_record,\n    STATE_PATH,\n)\n\n\n# =============================================================================\n# 28. Manifest and verified backup\n# =============================================================================\n\ntable_paths = [\n    TABLE_A_PATH,\n    TABLE_B_PATH,\n    TABLE_C_PATH,\n    TABLE_D_PATH,\n    TABLE_E_PATH,\n    TABLE_F_PATH,\n    TABLE_G_PATH,\n    TABLE_H_PATH,\n    TABLE_I_PATH,\n    TABLE_J_PATH,\n    TABLE_K_PATH,\n]\n\nmetric_paths = [\n    CONFUSION_VALIDATION_PATH,\n    CONFUSION_TEST_PATH,\n    CONFUSION_VALIDATION_NORMALIZED_PATH,\n    CONFUSION_TEST_NORMALIZED_PATH,\n]\n\nevidence_paths = [\n    SOURCE_VERIFICATION_PATH,\n    FIGURE_INDEX_PATH,\n    FIGURE_CAPTIONS_PATH,\n    MANUSCRIPT_NOTE_PATH,\n    SUMMARY_PATH,\n    STATE_PATH,\n]\n\nmanifest_sources = [\n    STEP10D_STATE_PATH,\n    STEP12A_STATE_PATH,\n    STEP12B_STATE_PATH,\n    STEP13C_STATE_PATH,\n    CONFUSION_LONG_SOURCE_PATH,\n    CLASS_ERROR_SOURCE_PATH,\n    ERROR_DIRECTION_SOURCE_PATH,\n    BINARY_SEVERITY_SOURCE_PATH,\n    *table_paths,\n    *metric_paths,\n    *evidence_paths,\n    *generated_figure_paths,\n]\n\n\nmanifest_records = []\n\n\nfor source_path in manifest_sources:\n\n    if not source_path.exists():\n\n        raise FileNotFoundError(\n            f\"Step 14 manifest source is missing: {source_path}\"\n        )\n\n    manifest_records.append({\n        \"relative_path\": str(\n            source_path.relative_to(\n                PROJECT\n            )\n        ),\n        \"size_bytes\": int(\n            source_path.stat().st_size\n        ),\n        \"sha256\": sha256_file(\n            source_path\n        ),\n    })\n\n\natomic_csv_save(\n    pd.DataFrame(\n        manifest_records\n    ),\n    MANIFEST_PATH,\n)\n\n\nbackup_members = create_verified_zip(\n    BACKUP_PATH,\n    [\n        *table_paths,\n        *metric_paths,\n        SOURCE_VERIFICATION_PATH,\n        FIGURE_INDEX_PATH,\n        FIGURE_CAPTIONS_PATH,\n        MANUSCRIPT_NOTE_PATH,\n        SUMMARY_PATH,\n        STATE_PATH,\n        MANIFEST_PATH,\n        *generated_figure_paths,\n    ],\n)\n\n\ngc.collect()\n\n\n# =============================================================================\n# 29. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"STEP 14 — PUBLICATION METRICS TABLES \"\n    \"AND FIGURES COMPLETED\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nEVIDENCE SAFETY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"New training performed                :\",\n    False\n)\n\nprint(\n    \"Optimizer created                     :\",\n    False\n)\n\nprint(\n    \"Model loaded                          :\",\n    False\n)\n\nprint(\n    \"Model inference performed             :\",\n    False\n)\n\nprint(\n    \"Raw images loaded                     :\",\n    False\n)\n\nprint(\n    \"Validation evaluated                  :\",\n    False\n)\n\nprint(\n    \"Final test evaluated                  :\",\n    False\n)\n\nprint(\n    \"Predictions regenerated               :\",\n    False\n)\n\nprint(\n    \"Frozen model changed                  :\",\n    False\n)\n\n\nprint(\n    \"\\nSOURCE VERIFICATION\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Step 10D state complete               :\",\n    step10d_state.get(\n        \"status\"\n    )\n    ==\n    \"complete\"\n)\n\nprint(\n    \"Step 12A state complete               :\",\n    step12a_state.get(\n        \"status\"\n    )\n    ==\n    \"complete\"\n)\n\nprint(\n    \"Step 12B state complete               :\",\n    step12b_state.get(\n        \"status\"\n    )\n    ==\n    \"complete\"\n)\n\nprint(\n    \"Step 13C state complete               :\",\n    step13c_state.get(\n        \"status\"\n    )\n    ==\n    \"complete\"\n)\n\nprint(\n    \"Locked performance values verified    :\",\n    True\n)\n\nprint(\n    \"Backup integrity checks passed        :\",\n    all(\n        record[\n            \"integrity_passed\"\n        ]\n        for record\n        in backup_verification.values()\n    )\n)\n\n\nprint(\n    \"\\nPRIMARY PERFORMANCE TABLE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_primary = table_a_df[\n    [\n        \"Evaluation scope\",\n        \"Samples\",\n        \"QWK\",\n        \"Accuracy\",\n        \"Balanced accuracy\",\n        \"Macro F1\",\n    ]\n].copy()\n\n\nfor column in [\n    \"QWK\",\n    \"Accuracy\",\n    \"Balanced accuracy\",\n    \"Macro F1\",\n]:\n\n    display_primary[\n        column\n    ] = display_primary[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_primary.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nFINAL-TEST CLASS-WISE RECALL\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_recall = table_c_df[\n    table_c_df[\n        \"Evaluation scope\"\n    ]\n    ==\n    \"Final test\"\n][\n    [\n        \"Class\",\n        \"Samples\",\n        \"Recall\",\n    ]\n].copy()\n\n\ndisplay_recall[\n    \"Recall\"\n] = display_recall[\n    \"Recall\"\n].map(\n    lambda value: f\"{float(value):.6f}\"\n)\n\n\nprint(\n    display_recall.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nPREPROCESSING ABLATION — SAME SEED\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_ablation = table_e_df[\n    [\n        \"Preprocessing variant\",\n        \"QWK\",\n        \"Accuracy\",\n        \"Balanced accuracy\",\n        \"Macro F1\",\n    ]\n].copy()\n\n\nfor column in [\n    \"QWK\",\n    \"Accuracy\",\n    \"Balanced accuracy\",\n    \"Macro F1\",\n]:\n\n    display_ablation[\n        column\n    ] = display_ablation[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_ablation.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nCOMMON-FOLD MODEL COMPLEXITY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\ndisplay_complexity = table_g_df[\n    [\n        \"Model\",\n        \"Parameters\",\n        \"Mean QWK\",\n        \"Mean balanced accuracy\",\n        \"Mean macro F1\",\n        \"QWK per million parameters\",\n        \"Formal outcome\",\n    ]\n].copy()\n\n\nfor column in [\n    \"Mean QWK\",\n    \"Mean balanced accuracy\",\n    \"Mean macro F1\",\n    \"QWK per million parameters\",\n]:\n\n    display_complexity[\n        column\n    ] = display_complexity[\n        column\n    ].map(\n        lambda value: f\"{float(value):.6f}\"\n    )\n\n\nprint(\n    display_complexity.to_string(\n        index=False\n    )\n)\n\n\nprint(\n    \"\\nPUBLICATION OUTPUTS\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Publication tables                    :\",\n    11\n)\n\nprint(\n    \"Figure families                       :\",\n    6\n)\n\nprint(\n    \"Figure files                          :\",\n    len(\n        figure_index_df\n    )\n)\n\nprint(\n    \"600-dpi PNG files                     :\",\n    int(\n        (\n            figure_index_df[\n                \"Format\"\n            ]\n            ==\n            \"PNG\"\n        ).sum()\n    )\n)\n\nprint(\n    \"PDF files                             :\",\n    int(\n        (\n            figure_index_df[\n                \"Format\"\n            ]\n            ==\n            \"PDF\"\n        ).sum()\n    )\n)\n\nprint(\n    \"SVG files                             :\",\n    int(\n        (\n            figure_index_df[\n                \"Format\"\n            ]\n            ==\n            \"SVG\"\n        ).sum()\n    )\n)\n\nprint(\n    \"All figure integrity checks passed    :\",\n    bool(\n        figure_index_df[\n            \"Integrity passed\"\n        ].all()\n    )\n)\n\nprint(\n    \"Table directory                       :\",\n    OUTPUT_EVIDENCE_DIR\n)\n\nprint(\n    \"Figure directory                      :\",\n    OUTPUT_FIGURE_DIR\n)\n\n\nprint(\n    \"\\nBACKUP\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup path                           :\",\n    BACKUP_PATH\n)\n\nprint(\n    \"Backup members                        :\",\n    len(\n        backup_members\n    )\n)\n\nprint(\n    \"ZIP integrity passed                  :\",\n    True\n)\n\n\nprint(\n    \"\\nNEXT STAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"READY FOR STEP 15 — FINAL REPRODUCIBILITY \"\n    \"AUDIT AND RESULTS PACKAGE\"\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T16:25:15.797802Z","iopub.execute_input":"2026-07-18T16:25:15.798645Z","iopub.status.idle":"2026-07-18T16:25:27.654826Z","shell.execute_reply.started":"2026-07-18T16:25:15.798613Z","shell.execute_reply":"2026-07-18T16:25:27.653909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 15 — FINAL REPRODUCIBILITY AUDIT AND RESULTS PACKAGING\n#\n# Creates:\n#   1. Manuscript Results Package\n#   2. Reproducibility Evidence Package\n#   3. Final audit report\n#   4. Environment record\n#   5. Claim-boundary document\n#   6. Package manifests and hashes\n#\n# Safety:\n#   - No training\n#   - No optimizer\n#   - No model loading\n#   - No model inference\n#   - No image decoding\n#   - No validation/test evaluation\n#   - No prediction regeneration\n#   - No model or threshold modification\n#\n# Run this new cell only. Do not use Run All.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\n\nimport hashlib\nimport json\nimport os\nimport platform\nimport sys\nimport zipfile\n\nimport cv2\nimport matplotlib\nimport numpy as np\nimport pandas as pd\nimport PIL\nimport sklearn\nimport torch\nimport torchvision\n\n\n# =============================================================================\n# 1. Project paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nSTATE_DIR = (\n    PROJECT\n    / \"00_state\"\n)\n\nSPLIT_DIR = (\n    PROJECT\n    / \"03_splits\"\n)\n\nMETRIC_ROOT = (\n    PROJECT\n    / \"08_metrics\"\n)\n\nFIGURE_ROOT = (\n    PROJECT\n    / \"09_figures\"\n)\n\nEVIDENCE_ROOT = (\n    PROJECT\n    / \"12_paper_evidence\"\n)\n\nBACKUP_ROOT = (\n    PROJECT\n    / \"13_backups\"\n)\n\n\nSTEP10D_STATE_PATH = (\n    STATE_DIR\n    / \"step_10d_final_test_state.json\"\n)\n\nSTEP12A_STATE_PATH = (\n    STATE_DIR\n    / \"step_12a_generalization_gap_analysis_state.json\"\n)\n\nSTEP12B_STATE_PATH = (\n    STATE_DIR\n    / \"step_12b_locked_prediction_error_analysis_state.json\"\n)\n\nSTEP13A_STATE_PATH = (\n    STATE_DIR\n    / \"step_13a_xai_case_selection_state.json\"\n)\n\nSTEP13B_R_STATE_PATH = (\n    STATE_DIR\n    / \"step_13b_r_gradcam_plus_plus_state.json\"\n)\n\nSTEP13C_STATE_PATH = (\n    STATE_DIR\n    / \"step_13c_xai_visual_qa_state.json\"\n)\n\nSTEP14_STATE_PATH = (\n    STATE_DIR\n    / \"step_14_publication_metrics_figures_state.json\"\n)\n\n\nSTEP14_SUMMARY_PATH = (\n    EVIDENCE_ROOT\n    / \"step_14_publication_metrics_figures\"\n    / \"step_14_publication_metrics_summary.json\"\n)\n\nSTEP14_TABLE_A_PATH = (\n    EVIDENCE_ROOT\n    / \"step_14_publication_metrics_figures\"\n    / \"table_14a_primary_performance.csv\"\n)\n\nSTEP14_FIGURE_INDEX_PATH = (\n    EVIDENCE_ROOT\n    / \"step_14_publication_metrics_figures\"\n    / \"step_14_publication_figure_index.csv\"\n)\n\nSTEP13C_SUMMARY_PATH = (\n    EVIDENCE_ROOT\n    / \"step_13c_xai_visual_qa\"\n    / \"step_13c_xai_visual_qa_summary.json\"\n)\n\nSTEP13B_R_SUMMARY_PATH = (\n    EVIDENCE_ROOT\n    / \"step_13b_r_gradcam_plus_plus\"\n    / \"step_13b_r_gradcam_plus_plus_summary.json\"\n)\n\nFINAL_CHECKPOINT_PATH = (\n    PROJECT\n    / \"06_checkpoints\"\n    / \"step_10b_registered_baseline_final_training\"\n    / \"step_10b_final_training_checkpoint.pt\"\n)\n\nFAILED_STEP13B_FIGURE_DIR = (\n    FIGURE_ROOT\n    / \"step_13b_gradcam_plus_plus\"\n)\n\n\nOUTPUT_METRIC_DIR = (\n    METRIC_ROOT\n    / \"step_15_final_reproducibility_audit\"\n)\n\nOUTPUT_EVIDENCE_DIR = (\n    EVIDENCE_ROOT\n    / \"step_15_final_reproducibility_audit\"\n)\n\n\nAUDIT_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_15_final_reproducibility_audit.csv\"\n)\n\nBACKUP_AUDIT_PATH = (\n    OUTPUT_METRIC_DIR\n    / \"step_15_required_backup_integrity.csv\"\n)\n\nENVIRONMENT_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_software_environment.json\"\n)\n\nREADME_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"README_FINAL_RESULTS_PACKAGE.txt\"\n)\n\nCLAIM_BOUNDARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_claim_boundaries.txt\"\n)\n\nRESULTS_MANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_manuscript_results_package_manifest.csv\"\n)\n\nREPRO_MANIFEST_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_reproducibility_package_manifest.csv\"\n)\n\nPACKAGE_INDEX_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_package_index.csv\"\n)\n\nSUMMARY_PATH = (\n    OUTPUT_EVIDENCE_DIR\n    / \"step_15_final_reproducibility_summary.json\"\n)\n\nSTATE_PATH = (\n    STATE_DIR\n    / \"step_15_final_reproducibility_audit_state.json\"\n)\n\n\nMANUSCRIPT_PACKAGE_PATH = (\n    BACKUP_ROOT\n    / \"DR_PUBLICATION_2026_MANUSCRIPT_RESULTS_PACKAGE.zip\"\n)\n\nREPRODUCIBILITY_PACKAGE_PATH = (\n    BACKUP_ROOT\n    / \"DR_PUBLICATION_2026_REPRODUCIBILITY_EVIDENCE_PACKAGE.zip\"\n)\n\nSTEP15_BACKUP_PATH = (\n    BACKUP_ROOT\n    / \"step_15_final_reproducibility_audit_backup.zip\"\n)\n\n\n# =============================================================================\n# 2. Locked expectations\n# =============================================================================\n\nEXPECTED_CHECKPOINT_SHA256 = (\n    \"7b46f5562e2a5884027a16c20e2b15de\"\n    \"a2f185195cbdde4e9169528fc7c14f71\"\n)\n\nEXPECTED_SELECTION_FINGERPRINT = (\n    \"da2b47d1c8cce21ab4231edf96e27f227\"\n    \"bfb24fca7fdc8bb508b47ceb5a34f89\"\n)\n\nEXPECTED_PARAMETER_COUNT = 4_013_953\n\nEXPECTED_FINAL_METRICS = {\n    \"QWK\": 0.847063,\n    \"Accuracy\": 0.770115,\n    \"Balanced accuracy\": 0.628933,\n    \"Macro F1\": 0.608571,\n}\n\nEXPECTED_STEP14_TABLE_COUNT = 11\nEXPECTED_STEP14_FIGURE_FILES = 18\nEXPECTED_STEP14_FIGURE_FAMILIES = 6\n\nEXPECTED_SELECTED_XAI_CASES = 10\nEXPECTED_PREDICTED_CLASS_MAPS = 10\nEXPECTED_REFERENCE_CLASS_MAPS = 5\nEXPECTED_TOTAL_XAI_MAPS = 15\n\nEXPECTED_STEP13C_VISUAL_ARTIFACTS = 75\nEXPECTED_STEP13C_PUBLICATION_FIGURE_FILES = 6\n\nTOLERANCE = 5.0e-6\n\n\nREQUIRED_BACKUP_NAMES = [\n    \"step_06eE_formal_preprocessing_lock_backup.zip\",\n    \"step_10c_locked_validation_backup.zip\",\n    \"step_10d_final_test_backup.zip\",\n    \"step_11a_r1_fair_reporting_backup.zip\",\n    \"step_11b_qa2_parameter_reporting_correction_backup.zip\",\n    \"step_12a_generalization_gap_analysis_backup.zip\",\n    \"step_12b_locked_prediction_error_analysis_backup.zip\",\n    \"step_13a_xai_case_selection_backup.zip\",\n    \"step_13b_r_gradcam_plus_plus_backup.zip\",\n    \"step_13c_xai_visual_qa_backup.zip\",\n    \"step_14_publication_metrics_figures_backup.zip\",\n]\n\n\n# =============================================================================\n# 3. Utility functions\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef read_json(\n    path,\n):\n\n    with open(\n        path,\n        \"r\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        return json.load(\n            file\n        )\n\n\ndef atomic_json_save(\n    record,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        json.dump(\n            record,\n            file,\n            indent=2,\n            ensure_ascii=False,\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_csv_save(\n    dataframe,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    dataframe.to_csv(\n        temporary_path,\n        index=False,\n    )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef atomic_text_save(\n    text,\n    path,\n):\n\n    temporary_path = path.with_suffix(\n        path.suffix + \".tmp\"\n    )\n\n    with open(\n        temporary_path,\n        \"w\",\n        encoding=\"utf-8\",\n    ) as file:\n\n        file.write(\n            text\n        )\n\n    os.replace(\n        temporary_path,\n        path,\n    )\n\n\ndef sha256_file(\n    path,\n):\n\n    digest = hashlib.sha256()\n\n    with open(\n        path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef format_bytes(\n    size_bytes,\n):\n\n    size = float(\n        size_bytes\n    )\n\n    units = [\n        \"B\",\n        \"KB\",\n        \"MB\",\n        \"GB\",\n    ]\n\n    for unit in units:\n\n        if size < 1024.0:\n\n            return f\"{size:.2f} {unit}\"\n\n        size /= 1024.0\n\n    return f\"{size:.2f} TB\"\n\n\ndef collect_files(\n    paths,\n):\n\n    collected = []\n\n    for path in paths:\n\n        path = Path(\n            path\n        )\n\n        if not path.exists():\n\n            continue\n\n        if path.is_file():\n\n            collected.append(\n                path\n            )\n\n        elif path.is_dir():\n\n            collected.extend(\n                sorted(\n                    candidate\n                    for candidate\n                    in path.rglob(\n                        \"*\"\n                    )\n                    if (\n                        candidate.is_file()\n                        and\n                        not candidate.name.endswith(\n                            \".tmp\"\n                        )\n                    )\n                )\n            )\n\n    unique_files = []\n\n    seen = set()\n\n    for path in collected:\n\n        resolved = path.resolve()\n\n        if resolved in seen:\n\n            continue\n\n        seen.add(\n            resolved\n        )\n\n        unique_files.append(\n            path\n        )\n\n    return unique_files\n\n\ndef create_verified_zip(\n    zip_path,\n    source_files,\n):\n\n    temporary_path = zip_path.with_suffix(\n        zip_path.suffix + \".tmp\"\n    )\n\n    if temporary_path.exists():\n\n        temporary_path.unlink()\n\n    unique_files = []\n\n    seen = set()\n\n    for source_file in source_files:\n\n        source_file = Path(\n            source_file\n        )\n\n        if (\n            not source_file.exists()\n            or\n            not source_file.is_file()\n        ):\n\n            continue\n\n        resolved = source_file.resolve()\n\n        if resolved in seen:\n\n            continue\n\n        seen.add(\n            resolved\n        )\n\n        unique_files.append(\n            source_file\n        )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"w\",\n        compression=zipfile.ZIP_DEFLATED,\n        compresslevel=6,\n        allowZip64=True,\n    ) as archive:\n\n        for source_file in unique_files:\n\n            archive.write(\n                source_file,\n                arcname=str(\n                    source_file.relative_to(\n                        PROJECT\n                    )\n                ),\n            )\n\n    with zipfile.ZipFile(\n        temporary_path,\n        mode=\"r\",\n        allowZip64=True,\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            f\"Damaged ZIP member detected: {damaged_member}\"\n        )\n\n    if len(\n        members\n    ) != len(\n        set(\n            members\n        )\n    ):\n\n        raise RuntimeError(\n            f\"Duplicate members detected in ZIP: {zip_path}\"\n        )\n\n    os.replace(\n        temporary_path,\n        zip_path,\n    )\n\n    return members\n\n\ndef verify_existing_zip(\n    zip_path,\n):\n\n    zip_path = Path(\n        zip_path\n    )\n\n    if not zip_path.exists():\n\n        return {\n            \"backup_name\": (\n                zip_path.name\n            ),\n            \"backup_path\": str(\n                zip_path\n            ),\n            \"exists\": False,\n            \"size_bytes\": 0,\n            \"member_count\": 0,\n            \"zip_integrity_passed\": False,\n            \"sha256\": \"\",\n        }\n\n    with zipfile.ZipFile(\n        zip_path,\n        mode=\"r\",\n        allowZip64=True,\n    ) as archive:\n\n        members = archive.namelist()\n        damaged_member = archive.testzip()\n\n    return {\n        \"backup_name\": (\n            zip_path.name\n        ),\n        \"backup_path\": str(\n            zip_path\n        ),\n        \"exists\": True,\n        \"size_bytes\": int(\n            zip_path.stat().st_size\n        ),\n        \"member_count\": int(\n            len(\n                members\n            )\n        ),\n        \"zip_integrity_passed\": bool(\n            damaged_member is None\n            and\n            len(\n                members\n            )\n            ==\n            len(\n                set(\n                    members\n                )\n            )\n        ),\n        \"sha256\": sha256_file(\n            zip_path\n        ),\n    }\n\n\ndef package_manifest(\n    file_paths,\n    package_role,\n):\n\n    records = []\n\n    for file_path in sorted(\n        file_paths,\n        key=lambda path: str(\n            path\n        ),\n    ):\n\n        records.append({\n            \"package_role\": (\n                package_role\n            ),\n            \"relative_path\": str(\n                file_path.relative_to(\n                    PROJECT\n                )\n            ),\n            \"size_bytes\": int(\n                file_path.stat().st_size\n            ),\n            \"sha256\": sha256_file(\n                file_path\n            ),\n        })\n\n    return pd.DataFrame(\n        records\n    )\n\n\n# =============================================================================\n# 4. Completion and partial-output protection\n# =============================================================================\n\nif STATE_PATH.exists():\n\n    existing_state = read_json(\n        STATE_PATH\n    )\n\n    if existing_state.get(\n        \"status\"\n    ) == \"complete\":\n\n        raise RuntimeError(\n            \"Step 15 is already complete. Do not rerun it.\"\n        )\n\n\nfor output_directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n]:\n\n    if (\n        output_directory.exists()\n        and\n        any(\n            path.is_file()\n            for path\n            in output_directory.rglob(\n                \"*\"\n            )\n        )\n    ):\n\n        raise RuntimeError(\n            \"Partial Step 15 outputs already exist:\\n\"\n            f\"{output_directory}\\n\"\n            \"Do not mix outputs from multiple runs.\"\n        )\n\n\nfor package_path in [\n    MANUSCRIPT_PACKAGE_PATH,\n    REPRODUCIBILITY_PACKAGE_PATH,\n    STEP15_BACKUP_PATH,\n]:\n\n    if package_path.exists():\n\n        raise RuntimeError(\n            \"A Step 15 package already exists:\\n\"\n            f\"{package_path}\\n\"\n            \"Do not overwrite an existing final package.\"\n        )\n\n\n# =============================================================================\n# 5. Required evidence verification\n# =============================================================================\n\nrequired_paths = [\n    STEP10D_STATE_PATH,\n    STEP12A_STATE_PATH,\n    STEP12B_STATE_PATH,\n    STEP13A_STATE_PATH,\n    STEP13B_R_STATE_PATH,\n    STEP13C_STATE_PATH,\n    STEP14_STATE_PATH,\n    STEP14_SUMMARY_PATH,\n    STEP14_TABLE_A_PATH,\n    STEP14_FIGURE_INDEX_PATH,\n    STEP13C_SUMMARY_PATH,\n    STEP13B_R_SUMMARY_PATH,\n    FINAL_CHECKPOINT_PATH,\n]\n\n\nfor required_path in required_paths:\n\n    if not required_path.exists():\n\n        raise FileNotFoundError(\n            f\"Required Step 15 evidence is missing: {required_path}\"\n        )\n\n\nstep10d_state = read_json(\n    STEP10D_STATE_PATH\n)\n\nstep12a_state = read_json(\n    STEP12A_STATE_PATH\n)\n\nstep12b_state = read_json(\n    STEP12B_STATE_PATH\n)\n\nstep13a_state = read_json(\n    STEP13A_STATE_PATH\n)\n\nstep13b_r_state = read_json(\n    STEP13B_R_STATE_PATH\n)\n\nstep13c_state = read_json(\n    STEP13C_STATE_PATH\n)\n\nstep14_state = read_json(\n    STEP14_STATE_PATH\n)\n\nstep14_summary = read_json(\n    STEP14_SUMMARY_PATH\n)\n\nstep13c_summary = read_json(\n    STEP13C_SUMMARY_PATH\n)\n\nstep13b_r_summary = read_json(\n    STEP13B_R_SUMMARY_PATH\n)\n\n\n# =============================================================================\n# 6. Locked metric verification\n# =============================================================================\n\nprimary_table_df = pd.read_csv(\n    STEP14_TABLE_A_PATH\n)\n\n\nfinal_test_rows = primary_table_df[\n    primary_table_df[\n        \"Evaluation scope\"\n    ]\n    ==\n    \"Final test\"\n]\n\n\nif len(\n    final_test_rows\n) != 1:\n\n    raise RuntimeError(\n        \"Step 14 primary table does not contain exactly \"\n        \"one final-test row.\"\n    )\n\n\nfinal_test_row = final_test_rows.iloc[\n    0\n]\n\n\nmetric_verification = {}\n\n\nfor metric_name, expected_value in EXPECTED_FINAL_METRICS.items():\n\n    observed_value = float(\n        final_test_row[\n            metric_name\n        ]\n    )\n\n    absolute_difference = abs(\n        observed_value\n        -\n        expected_value\n    )\n\n    metric_verification[\n        metric_name\n    ] = {\n        \"observed\": (\n            observed_value\n        ),\n        \"expected\": (\n            expected_value\n        ),\n        \"absolute_difference\": (\n            absolute_difference\n        ),\n        \"passed\": bool(\n            absolute_difference\n            <=\n            TOLERANCE\n        ),\n    }\n\n\nif not all(\n    record[\n        \"passed\"\n    ]\n    for record\n    in metric_verification.values()\n):\n\n    raise RuntimeError(\n        \"Final-test metric verification failed.\"\n    )\n\n\n# =============================================================================\n# 7. Figure and backup integrity\n# =============================================================================\n\nstep14_figure_index_df = pd.read_csv(\n    STEP14_FIGURE_INDEX_PATH\n)\n\n\nif len(\n    step14_figure_index_df\n) != EXPECTED_STEP14_FIGURE_FILES:\n\n    raise RuntimeError(\n        \"Step 14 publication figure-file count mismatch.\"\n    )\n\n\nif not step14_figure_index_df[\n    \"Integrity passed\"\n].astype(\n    bool\n).all():\n\n    raise RuntimeError(\n        \"One or more Step 14 figure integrity checks failed.\"\n    )\n\n\nbackup_records = []\n\n\nfor backup_name in REQUIRED_BACKUP_NAMES:\n\n    backup_records.append(\n        verify_existing_zip(\n            BACKUP_ROOT\n            /\n            backup_name\n        )\n    )\n\n\nbackup_audit_df = pd.DataFrame(\n    backup_records\n)\n\n\nrequired_backups_passed = bool(\n    backup_audit_df[\n        \"exists\"\n    ].all()\n    and\n    backup_audit_df[\n        \"zip_integrity_passed\"\n    ].all()\n)\n\n\nif not required_backups_passed:\n\n    failed_backups = backup_audit_df[\n        (\n            backup_audit_df[\n                \"exists\"\n            ]\n            !=\n            True\n        )\n        |\n        (\n            backup_audit_df[\n                \"zip_integrity_passed\"\n            ]\n            !=\n            True\n        )\n    ]\n\n    print(\n        failed_backups.to_string(\n            index=False\n        )\n    )\n\n    raise RuntimeError(\n        \"One or more required project backups are missing \"\n        \"or damaged.\"\n    )\n\n\nfailed_step13b_files = []\n\n\nif FAILED_STEP13B_FIGURE_DIR.exists():\n\n    failed_step13b_files = [\n        path\n        for path\n        in FAILED_STEP13B_FIGURE_DIR.rglob(\n            \"*\"\n        )\n        if path.is_file()\n    ]\n\n\n# =============================================================================\n# 8. Build formal reproducibility audit\n# =============================================================================\n\naudit_records = []\n\n\ndef add_audit_check(\n    category,\n    check_name,\n    observed,\n    expected,\n    passed,\n    critical=True,\n):\n\n    audit_records.append({\n        \"category\": (\n            category\n        ),\n        \"check_name\": (\n            check_name\n        ),\n        \"observed\": str(\n            observed\n        ),\n        \"expected\": str(\n            expected\n        ),\n        \"passed\": bool(\n            passed\n        ),\n        \"critical\": bool(\n            critical\n        ),\n    })\n\n\nfor stage_name, state_record in [\n    (\n        \"Step 10D\",\n        step10d_state,\n    ),\n    (\n        \"Step 12A\",\n        step12a_state,\n    ),\n    (\n        \"Step 12B\",\n        step12b_state,\n    ),\n    (\n        \"Step 13A\",\n        step13a_state,\n    ),\n    (\n        \"Step 13B-R\",\n        step13b_r_state,\n    ),\n    (\n        \"Step 13C\",\n        step13c_state,\n    ),\n    (\n        \"Step 14\",\n        step14_state,\n    ),\n]:\n\n    add_audit_check(\n        category=\"Stage completion\",\n        check_name=(\n            f\"{stage_name} status\"\n        ),\n        observed=(\n            state_record.get(\n                \"status\"\n            )\n        ),\n        expected=\"complete\",\n        passed=(\n            state_record.get(\n                \"status\"\n            )\n            ==\n            \"complete\"\n        ),\n    )\n\n\nadd_audit_check(\n    category=\"Final-test protection\",\n    check_name=\"Final-test evaluation count\",\n    observed=(\n        step10d_state.get(\n            \"test_evaluation_count\"\n        )\n    ),\n    expected=1,\n    passed=(\n        int(\n            step10d_state.get(\n                \"test_evaluation_count\",\n                -1,\n            )\n        )\n        ==\n        1\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Final-test protection\",\n    check_name=\"Additional final-test evaluation allowed\",\n    observed=(\n        step10d_state.get(\n            \"another_test_evaluation_allowed\"\n        )\n    ),\n    expected=False,\n    passed=(\n        step10d_state.get(\n            \"another_test_evaluation_allowed\"\n        )\n        is False\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Model protection\",\n    check_name=\"Model change allowed after final evaluation\",\n    observed=(\n        step10d_state.get(\n            \"model_change_allowed\"\n        )\n    ),\n    expected=False,\n    passed=(\n        step10d_state.get(\n            \"model_change_allowed\"\n        )\n        is False\n    ),\n)\n\n\ncheckpoint_sha256 = sha256_file(\n    FINAL_CHECKPOINT_PATH\n)\n\n\nadd_audit_check(\n    category=\"Checkpoint\",\n    check_name=\"Final checkpoint SHA-256\",\n    observed=(\n        checkpoint_sha256\n    ),\n    expected=(\n        EXPECTED_CHECKPOINT_SHA256\n    ),\n    passed=(\n        checkpoint_sha256\n        ==\n        EXPECTED_CHECKPOINT_SHA256\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Checkpoint\",\n    check_name=\"Registered parameter count\",\n    observed=(\n        step13b_r_summary[\n            \"checkpoint\"\n        ][\n            \"parameter_count\"\n        ]\n    ),\n    expected=(\n        EXPECTED_PARAMETER_COUNT\n    ),\n    passed=(\n        int(\n            step13b_r_summary[\n                \"checkpoint\"\n            ][\n                \"parameter_count\"\n            ]\n        )\n        ==\n        EXPECTED_PARAMETER_COUNT\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI selection\",\n    check_name=\"Locked selection fingerprint\",\n    observed=(\n        step13a_state.get(\n            \"selection_fingerprint_sha256\"\n        )\n    ),\n    expected=(\n        EXPECTED_SELECTION_FINGERPRINT\n    ),\n    passed=(\n        step13a_state.get(\n            \"selection_fingerprint_sha256\"\n        )\n        ==\n        EXPECTED_SELECTION_FINGERPRINT\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI selection\",\n    check_name=\"Selected XAI cases\",\n    observed=(\n        step13a_state.get(\n            \"selected_case_count\"\n        )\n    ),\n    expected=(\n        EXPECTED_SELECTED_XAI_CASES\n    ),\n    passed=(\n        int(\n            step13a_state.get(\n                \"selected_case_count\",\n                -1,\n            )\n        )\n        ==\n        EXPECTED_SELECTED_XAI_CASES\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI selection\",\n    check_name=\"Case replacement after review allowed\",\n    observed=(\n        step13a_state.get(\n            \"case_replacement_after_heatmap_review_allowed\"\n        )\n    ),\n    expected=False,\n    passed=(\n        step13a_state.get(\n            \"case_replacement_after_heatmap_review_allowed\"\n        )\n        is False\n    ),\n)\n\n\nfor map_name, observed_value, expected_value in [\n    (\n        \"Predicted-class Grad-CAM++ maps\",\n        step13b_r_state.get(\n            \"predicted_class_map_count\"\n        ),\n        EXPECTED_PREDICTED_CLASS_MAPS,\n    ),\n    (\n        \"Reference-class Grad-CAM++ maps\",\n        step13b_r_state.get(\n            \"reference_class_map_count\"\n        ),\n        EXPECTED_REFERENCE_CLASS_MAPS,\n    ),\n    (\n        \"Total Grad-CAM++ maps\",\n        step13b_r_state.get(\n            \"total_map_count\"\n        ),\n        EXPECTED_TOTAL_XAI_MAPS,\n    ),\n]:\n\n    add_audit_check(\n        category=\"XAI generation\",\n        check_name=(\n            map_name\n        ),\n        observed=(\n            observed_value\n        ),\n        expected=(\n            expected_value\n        ),\n        passed=(\n            int(\n                observed_value\n            )\n            ==\n            expected_value\n        ),\n    )\n\n\nadd_audit_check(\n    category=\"XAI generation\",\n    check_name=\"All heatmaps finite\",\n    observed=(\n        step13b_r_state.get(\n            \"all_heatmaps_finite\"\n        )\n    ),\n    expected=True,\n    passed=(\n        step13b_r_state.get(\n            \"all_heatmaps_finite\"\n        )\n        is True\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI generation\",\n    check_name=\"All heatmaps nondegenerate\",\n    observed=(\n        step13b_r_state.get(\n            \"all_heatmaps_nondegenerate\"\n        )\n    ),\n    expected=True,\n    passed=(\n        step13b_r_state.get(\n            \"all_heatmaps_nondegenerate\"\n        )\n        is True\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI visual QA\",\n    check_name=\"Visual artifacts\",\n    observed=(\n        step13c_summary[\n            \"source_integrity\"\n        ][\n            \"source_visual_file_count\"\n        ]\n    ),\n    expected=(\n        EXPECTED_STEP13C_VISUAL_ARTIFACTS\n    ),\n    passed=(\n        int(\n            step13c_summary[\n                \"source_integrity\"\n            ][\n                \"source_visual_file_count\"\n            ]\n        )\n        ==\n        EXPECTED_STEP13C_VISUAL_ARTIFACTS\n    ),\n)\n\n\nadd_audit_check(\n    category=\"XAI visual QA\",\n    check_name=\"All visual artifacts passed\",\n    observed=(\n        step13c_state.get(\n            \"all_visual_artifacts_passed\"\n        )\n    ),\n    expected=True,\n    passed=(\n        step13c_state.get(\n            \"all_visual_artifacts_passed\"\n        )\n        is True\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Publication outputs\",\n    check_name=\"Publication table count\",\n    observed=(\n        step14_state.get(\n            \"publication_table_count\"\n        )\n    ),\n    expected=(\n        EXPECTED_STEP14_TABLE_COUNT\n    ),\n    passed=(\n        int(\n            step14_state.get(\n                \"publication_table_count\",\n                -1,\n            )\n        )\n        ==\n        EXPECTED_STEP14_TABLE_COUNT\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Publication outputs\",\n    check_name=\"Publication figure families\",\n    observed=(\n        step14_state.get(\n            \"publication_figure_family_count\"\n        )\n    ),\n    expected=(\n        EXPECTED_STEP14_FIGURE_FAMILIES\n    ),\n    passed=(\n        int(\n            step14_state.get(\n                \"publication_figure_family_count\",\n                -1,\n            )\n        )\n        ==\n        EXPECTED_STEP14_FIGURE_FAMILIES\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Publication outputs\",\n    check_name=\"Publication figure files\",\n    observed=(\n        step14_state.get(\n            \"publication_figure_file_count\"\n        )\n    ),\n    expected=(\n        EXPECTED_STEP14_FIGURE_FILES\n    ),\n    passed=(\n        int(\n            step14_state.get(\n                \"publication_figure_file_count\",\n                -1,\n            )\n        )\n        ==\n        EXPECTED_STEP14_FIGURE_FILES\n    ),\n)\n\n\nfor metric_name, verification_record in metric_verification.items():\n\n    add_audit_check(\n        category=\"Final-test metrics\",\n        check_name=(\n            metric_name\n        ),\n        observed=(\n            f\"{verification_record['observed']:.6f}\"\n        ),\n        expected=(\n            f\"{verification_record['expected']:.6f}\"\n        ),\n        passed=(\n            verification_record[\n                \"passed\"\n            ]\n        ),\n    )\n\n\nadd_audit_check(\n    category=\"Failed-run isolation\",\n    check_name=\"Partial figures from failed Step 13B\",\n    observed=(\n        len(\n            failed_step13b_files\n        )\n    ),\n    expected=0,\n    passed=(\n        len(\n            failed_step13b_files\n        )\n        ==\n        0\n    ),\n)\n\n\nadd_audit_check(\n    category=\"Backups\",\n    check_name=\"Required verified backups\",\n    observed=(\n        int(\n            backup_audit_df[\n                \"zip_integrity_passed\"\n            ].sum()\n        )\n    ),\n    expected=(\n        len(\n            REQUIRED_BACKUP_NAMES\n        )\n    ),\n    passed=(\n        required_backups_passed\n    ),\n)\n\n\naudit_df = pd.DataFrame(\n    audit_records\n)\n\n\ncritical_audit_passed = bool(\n    audit_df.loc[\n        audit_df[\n            \"critical\"\n        ],\n        \"passed\",\n    ].all()\n)\n\n\nif not critical_audit_passed:\n\n    failed_checks = audit_df[\n        (\n            audit_df[\n                \"critical\"\n            ]\n            ==\n            True\n        )\n        &\n        (\n            audit_df[\n                \"passed\"\n            ]\n            !=\n            True\n        )\n    ]\n\n    print(\n        failed_checks.to_string(\n            index=False\n        )\n    )\n\n    raise RuntimeError(\n        \"The final reproducibility audit failed.\"\n    )\n\n\n# =============================================================================\n# 9. Create final audit directories and evidence\n# =============================================================================\n\nfor directory in [\n    OUTPUT_METRIC_DIR,\n    OUTPUT_EVIDENCE_DIR,\n]:\n\n    directory.mkdir(\n        parents=True,\n        exist_ok=True,\n    )\n\n\natomic_csv_save(\n    audit_df,\n    AUDIT_PATH,\n)\n\n\natomic_csv_save(\n    backup_audit_df,\n    BACKUP_AUDIT_PATH,\n)\n\n\n# =============================================================================\n# 10. Software environment record\n# =============================================================================\n\nenvironment_record = {\n    \"recorded_utc\": (\n        utc_now()\n    ),\n    \"platform\": (\n        platform.platform()\n    ),\n    \"python_version\": (\n        sys.version\n    ),\n    \"python_executable\": (\n        sys.executable\n    ),\n    \"packages\": {\n        \"torch\": (\n            torch.__version__\n        ),\n        \"torchvision\": (\n            torchvision.__version__\n        ),\n        \"numpy\": (\n            np.__version__\n        ),\n        \"pandas\": (\n            pd.__version__\n        ),\n        \"scikit_learn\": (\n            sklearn.__version__\n        ),\n        \"opencv\": (\n            cv2.__version__\n        ),\n        \"matplotlib\": (\n            matplotlib.__version__\n        ),\n        \"pillow\": (\n            PIL.__version__\n        ),\n    },\n    \"cuda\": {\n        \"available\": bool(\n            torch.cuda.is_available()\n        ),\n        \"torch_cuda_version\": (\n            torch.version.cuda\n        ),\n        \"device_count\": int(\n            torch.cuda.device_count()\n        ),\n        \"device_name\": (\n            torch.cuda.get_device_name(\n                0\n            )\n            if torch.cuda.is_available()\n            else\n            \"\"\n        ),\n    },\n    \"final_checkpoint\": {\n        \"path\": str(\n            FINAL_CHECKPOINT_PATH\n        ),\n        \"size_bytes\": int(\n            FINAL_CHECKPOINT_PATH.stat().st_size\n        ),\n        \"sha256\": (\n            checkpoint_sha256\n        ),\n        \"parameter_count\": (\n            EXPECTED_PARAMETER_COUNT\n        ),\n    },\n}\n\n\natomic_json_save(\n    environment_record,\n    ENVIRONMENT_PATH,\n)\n\n\n# =============================================================================\n# 11. Final README and claim boundaries\n# =============================================================================\n\nreadme_text = f\"\"\"\nDIABETIC RETINOPATHY PUBLICATION PROJECT — FINAL RESULTS PACKAGE\n\nProject root\n------------\n{PROJECT}\n\nSelected final model\n--------------------\nRegistered EfficientNet-B0 baseline\nParameters: {EXPECTED_PARAMETER_COUNT:,}\nInput size: 384 x 384\nPreprocessing: retinal crop plus always-applied mild LAB-CLAHE\nFinal weights: frozen EMA checkpoint\nCheckpoint SHA-256:\n{checkpoint_sha256}\n\nLocked final-test performance\n-----------------------------\nQWK:               {EXPECTED_FINAL_METRICS['QWK']:.6f}\nAccuracy:          {EXPECTED_FINAL_METRICS['Accuracy']:.6f}\nBalanced accuracy: {EXPECTED_FINAL_METRICS['Balanced accuracy']:.6f}\nMacro F1:          {EXPECTED_FINAL_METRICS['Macro F1']:.6f}\n\nCore reproducibility protections\n--------------------------------\nThe split was group-aware and permanently locked.\nValidation and final test were not used for model fitting.\nThe final test was evaluated exactly once.\nNo post-test model, threshold or preprocessing modification was permitted.\nXAI cases were selected deterministically before attribution review.\nCase replacement after viewing Grad-CAM++ maps was prohibited.\nAll key tables, figures, states and backups were SHA-256 indexed.\n\nImportant interpretation\n------------------------\nInternal cross-validation was optimistic relative to untouched holdouts.\nThe model performed strongly for No-DR and Mild cases but showed lower\nexact-grade sensitivity for Moderate, Severe and Proliferative DR.\nUndergrading accounted for most final-test errors.\nMost predictions nevertheless remained within one ordinal grade.\nGrad-CAM++ maps are model-attribution visualizations, not clinically\nvalidated lesion segmentations.\n\nPackage use\n-----------\nThe manuscript-results package contains publication tables, figures,\ncaptions and interpretation notes.\n\nThe reproducibility-evidence package contains the manuscript package,\nstate files, locked split evidence, final checkpoint, selected metric\ntables and verified stage backups.\n\"\"\".strip()\n\n\natomic_text_save(\n    readme_text,\n    README_PATH,\n)\n\n\nclaim_boundary_text = \"\"\"\nFINAL CLAIM BOUNDARIES\n\nAllowed claims\n--------------\n1. The selected compact EfficientNet-B0 model was evaluated under a\n   leakage-controlled, group-aware and permanently locked workflow.\n\n2. The final model achieved a final-test QWK of 0.8471, accuracy of\n   0.7701, balanced accuracy of 0.6289 and macro F1 of 0.6086.\n\n3. Internal cross-validation overestimated performance relative to the\n   untouched holdouts.\n\n4. No-DR and Mild recall were substantially higher than Moderate,\n   Severe and Proliferative-DR recall.\n\n5. Undergrading was the dominant final-test error direction.\n\n6. On common internal folds, the compact baseline outperformed the\n   evaluated OLG-DRNet and B4 plus Swin-Tiny comparator while using\n   substantially fewer parameters.\n\n7. XAI cases were locked before visualization, and attribution maps\n   predominantly remained within the retinal field.\n\nProhibited or unsupported claims\n--------------------------------\n1. Do not claim that the mother paper was formally beaten. Its reported\n   values came from a different experimental protocol and split.\n\n2. Do not claim statistical superiority or equivalence to the mother\n   paper.\n\n3. Do not describe overlapping validation and test confidence intervals\n   as proof that the two performances are equal.\n\n4. Do not claim that the model is clinically deployment-ready.\n\n5. Do not claim reliable advanced-DR screening; advanced-grade\n   sensitivity remained limited.\n\n6. Do not interpret Grad-CAM++ maps as verified lesion segmentation or\n   proof that clinically specific lesions were identified.\n\n7. Do not hide the failed complex architectures or the rejected bounded\n   revision. These are part of the transparent model-selection evidence.\n\n8. Do not use the final test for further tuning, calibration, threshold\n   selection or model replacement.\n\"\"\".strip()\n\n\natomic_text_save(\n    claim_boundary_text,\n    CLAIM_BOUNDARY_PATH,\n)\n\n\n# =============================================================================\n# 12. Collect manuscript-package sources\n# =============================================================================\n\nevidence_stage_tokens = [\n    \"step_11a\",\n    \"step_11b\",\n    \"step_12a\",\n    \"step_12b\",\n    \"step_13a\",\n    \"step_13b_r\",\n    \"step_13c\",\n    \"step_14\",\n]\n\n\nselected_evidence_directories = []\n\n\nif EVIDENCE_ROOT.exists():\n\n    for child in EVIDENCE_ROOT.iterdir():\n\n        if (\n            child.is_dir()\n            and\n            any(\n                token\n                in\n                child.name.lower()\n                for token\n                in evidence_stage_tokens\n            )\n        ):\n\n            selected_evidence_directories.append(\n                child\n            )\n\n\nmanuscript_source_paths = [\n    *selected_evidence_directories,\n    FIGURE_ROOT\n    / \"step_13c_xai_publication\",\n    FIGURE_ROOT\n    / \"step_14_publication_metrics_figures\",\n    README_PATH,\n    CLAIM_BOUNDARY_PATH,\n    AUDIT_PATH,\n    BACKUP_AUDIT_PATH,\n    ENVIRONMENT_PATH,\n]\n\n\nmanuscript_files = collect_files(\n    manuscript_source_paths\n)\n\n\nif not manuscript_files:\n\n    raise RuntimeError(\n        \"No manuscript-package files were collected.\"\n    )\n\n\nmanuscript_manifest_df = package_manifest(\n    manuscript_files,\n    package_role=(\n        \"manuscript_results\"\n    ),\n)\n\n\natomic_csv_save(\n    manuscript_manifest_df,\n    RESULTS_MANIFEST_PATH,\n)\n\n\nmanuscript_files_with_manifest = collect_files([\n    *manuscript_files,\n    RESULTS_MANIFEST_PATH,\n])\n\n\nmanuscript_members = create_verified_zip(\n    MANUSCRIPT_PACKAGE_PATH,\n    manuscript_files_with_manifest,\n)\n\n\n# =============================================================================\n# 13. Collect reproducibility-package sources\n# =============================================================================\n\nstate_files = collect_files([\n    STATE_DIR,\n])\n\n\nsplit_files = collect_files([\n    SPLIT_DIR,\n])\n\n\nselected_metric_directories = [\n    METRIC_ROOT\n    / \"step_12a_generalization_gap_analysis\",\n    METRIC_ROOT\n    / \"step_12b_locked_prediction_error_analysis\",\n    METRIC_ROOT\n    / \"step_13a_xai_case_selection\",\n    METRIC_ROOT\n    / \"step_13b_r_gradcam_plus_plus\",\n    METRIC_ROOT\n    / \"step_13c_xai_visual_qa\",\n    METRIC_ROOT\n    / \"step_14_publication_metrics_figures\",\n]\n\n\nselected_metric_files = collect_files(\n    selected_metric_directories\n)\n\n\nrequired_backup_files = [\n    BACKUP_ROOT\n    /\n    backup_name\n    for backup_name\n    in REQUIRED_BACKUP_NAMES\n]\n\n\noptional_code_directories = [\n    PROJECT\n    / \"02_scripts\",\n    PROJECT\n    / \"10_code\",\n    PROJECT\n    / \"11_code\",\n    PROJECT\n    / \"10_notebooks\",\n    PROJECT\n    / \"11_notebooks\",\n]\n\n\noptional_code_files = collect_files(\n    optional_code_directories\n)\n\n\nreproducibility_source_paths = [\n    MANUSCRIPT_PACKAGE_PATH,\n    FINAL_CHECKPOINT_PATH,\n    README_PATH,\n    CLAIM_BOUNDARY_PATH,\n    AUDIT_PATH,\n    BACKUP_AUDIT_PATH,\n    ENVIRONMENT_PATH,\n    RESULTS_MANIFEST_PATH,\n    *state_files,\n    *split_files,\n    *selected_metric_files,\n    *required_backup_files,\n    *optional_code_files,\n]\n\n\nreproducibility_files = collect_files(\n    reproducibility_source_paths\n)\n\n\nif not reproducibility_files:\n\n    raise RuntimeError(\n        \"No reproducibility-package files were collected.\"\n    )\n\n\nreproducibility_manifest_df = package_manifest(\n    reproducibility_files,\n    package_role=(\n        \"reproducibility_evidence\"\n    ),\n)\n\n\natomic_csv_save(\n    reproducibility_manifest_df,\n    REPRO_MANIFEST_PATH,\n)\n\n\nreproducibility_files_with_manifest = collect_files([\n    *reproducibility_files,\n    REPRO_MANIFEST_PATH,\n])\n\n\nreproducibility_members = create_verified_zip(\n    REPRODUCIBILITY_PACKAGE_PATH,\n    reproducibility_files_with_manifest,\n)\n\n\n# =============================================================================\n# 14. Package index\n# =============================================================================\n\npackage_index_df = pd.DataFrame([\n    {\n        \"package_name\": (\n            MANUSCRIPT_PACKAGE_PATH.name\n        ),\n        \"package_role\": (\n            \"Publication tables, figures, captions and \"\n            \"manuscript interpretation evidence\"\n        ),\n        \"package_path\": str(\n            MANUSCRIPT_PACKAGE_PATH\n        ),\n        \"size_bytes\": int(\n            MANUSCRIPT_PACKAGE_PATH.stat().st_size\n        ),\n        \"size_readable\": format_bytes(\n            MANUSCRIPT_PACKAGE_PATH.stat().st_size\n        ),\n        \"member_count\": int(\n            len(\n                manuscript_members\n            )\n        ),\n        \"sha256\": sha256_file(\n            MANUSCRIPT_PACKAGE_PATH\n        ),\n        \"zip_integrity_passed\": (\n            True\n        ),\n    },\n    {\n        \"package_name\": (\n            REPRODUCIBILITY_PACKAGE_PATH.name\n        ),\n        \"package_role\": (\n            \"States, split evidence, checkpoint, metrics, \"\n            \"verified backups and manuscript package\"\n        ),\n        \"package_path\": str(\n            REPRODUCIBILITY_PACKAGE_PATH\n        ),\n        \"size_bytes\": int(\n            REPRODUCIBILITY_PACKAGE_PATH.stat().st_size\n        ),\n        \"size_readable\": format_bytes(\n            REPRODUCIBILITY_PACKAGE_PATH.stat().st_size\n        ),\n        \"member_count\": int(\n            len(\n                reproducibility_members\n            )\n        ),\n        \"sha256\": sha256_file(\n            REPRODUCIBILITY_PACKAGE_PATH\n        ),\n        \"zip_integrity_passed\": (\n            True\n        ),\n    },\n])\n\n\natomic_csv_save(\n    package_index_df,\n    PACKAGE_INDEX_PATH,\n)\n\n\n# =============================================================================\n# 15. Final summary and state\n# =============================================================================\n\nsummary_record = {\n    \"step\": (\n        \"STEP_15_FINAL_REPRODUCIBILITY_AUDIT_AND_RESULTS_PACKAGE\"\n    ),\n    \"status\": (\n        \"completed\"\n    ),\n    \"completed_utc\": (\n        utc_now()\n    ),\n    \"audit\": {\n        \"total_checks\": int(\n            len(\n                audit_df\n            )\n        ),\n        \"critical_checks\": int(\n            audit_df[\n                \"critical\"\n            ].sum()\n        ),\n        \"passed_checks\": int(\n            audit_df[\n                \"passed\"\n            ].sum()\n        ),\n        \"critical_audit_passed\": (\n            critical_audit_passed\n        ),\n        \"required_backup_count\": int(\n            len(\n                REQUIRED_BACKUP_NAMES\n            )\n        ),\n        \"required_backups_passed\": (\n            required_backups_passed\n        ),\n    },\n    \"final_model\": {\n        \"architecture\": (\n            \"Registered EfficientNet-B0 baseline\"\n        ),\n        \"parameter_count\": (\n            EXPECTED_PARAMETER_COUNT\n        ),\n        \"checkpoint_path\": str(\n            FINAL_CHECKPOINT_PATH\n        ),\n        \"checkpoint_sha256\": (\n            checkpoint_sha256\n        ),\n    },\n    \"final_test\": {\n        \"evaluation_count\": 1,\n        \"qwk\": (\n            EXPECTED_FINAL_METRICS[\n                \"QWK\"\n            ]\n        ),\n        \"accuracy\": (\n            EXPECTED_FINAL_METRICS[\n                \"Accuracy\"\n            ]\n        ),\n        \"balanced_accuracy\": (\n            EXPECTED_FINAL_METRICS[\n                \"Balanced accuracy\"\n            ]\n        ),\n        \"macro_f1\": (\n            EXPECTED_FINAL_METRICS[\n                \"Macro F1\"\n            ]\n        ),\n        \"further_test_evaluation_allowed\": (\n            False\n        ),\n        \"model_change_allowed\": (\n            False\n        ),\n    },\n    \"packages\": {\n        \"manuscript_results\": {\n            \"path\": str(\n                MANUSCRIPT_PACKAGE_PATH\n            ),\n            \"member_count\": int(\n                len(\n                    manuscript_members\n                )\n            ),\n            \"size_bytes\": int(\n                MANUSCRIPT_PACKAGE_PATH.stat().st_size\n            ),\n            \"sha256\": sha256_file(\n                MANUSCRIPT_PACKAGE_PATH\n            ),\n        },\n        \"reproducibility_evidence\": {\n            \"path\": str(\n                REPRODUCIBILITY_PACKAGE_PATH\n            ),\n            \"member_count\": int(\n                len(\n                    reproducibility_members\n                )\n            ),\n            \"size_bytes\": int(\n                REPRODUCIBILITY_PACKAGE_PATH.stat().st_size\n            ),\n            \"sha256\": sha256_file(\n                REPRODUCIBILITY_PACKAGE_PATH\n            ),\n        },\n    },\n    \"safety\": {\n        \"new_training_performed\": (\n            False\n        ),\n        \"optimizer_created\": (\n            False\n        ),\n        \"model_loaded\": (\n            False\n        ),\n        \"model_inference_performed\": (\n            False\n        ),\n        \"raw_images_loaded\": (\n            False\n        ),\n        \"validation_evaluated\": (\n            False\n        ),\n        \"final_test_evaluated\": (\n            False\n        ),\n        \"predictions_regenerated\": (\n            False\n        ),\n        \"model_change_performed\": (\n            False\n        ),\n    },\n    \"next_stage\": (\n        \"STEP_16_MANUSCRIPT_METHODS_RESULTS_AND_DISCUSSION_DRAFTING\"\n    ),\n}\n\n\natomic_json_save(\n    summary_record,\n    SUMMARY_PATH,\n)\n\n\nstate_record = {\n    \"step\": (\n        \"STEP_15_FINAL_REPRODUCIBILITY_AUDIT_AND_RESULTS_PACKAGE\"\n    ),\n    \"status\": (\n        \"complete\"\n    ),\n    \"updated_utc\": (\n        utc_now()\n    ),\n    \"final_reproducibility_audit_completed\": (\n        True\n    ),\n    \"critical_audit_passed\": (\n        critical_audit_passed\n    ),\n    \"all_required_backups_verified\": (\n        required_backups_passed\n    ),\n    \"manuscript_results_package_created\": (\n        True\n    ),\n    \"reproducibility_evidence_package_created\": (\n        True\n    ),\n    \"manuscript_package_path\": str(\n        MANUSCRIPT_PACKAGE_PATH\n    ),\n    \"reproducibility_package_path\": str(\n        REPRODUCIBILITY_PACKAGE_PATH\n    ),\n    \"final_checkpoint_sha256\": (\n        checkpoint_sha256\n    ),\n    \"final_test_evaluation_count\": (\n        1\n    ),\n    \"another_test_evaluation_allowed\": (\n        False\n    ),\n    \"model_change_allowed\": (\n        False\n    ),\n    \"new_training_performed\": (\n        False\n    ),\n    \"model_loaded\": (\n        False\n    ),\n    \"model_inference_performed\": (\n        False\n    ),\n    \"raw_images_loaded\": (\n        False\n    ),\n    \"validation_evaluated\": (\n        False\n    ),\n    \"final_test_evaluated\": (\n        False\n    ),\n    \"predictions_regenerated\": (\n        False\n    ),\n    \"next_stage\": (\n        \"STEP_16_MANUSCRIPT_METHODS_RESULTS_AND_DISCUSSION_DRAFTING\"\n    ),\n}\n\n\natomic_json_save(\n    state_record,\n    STATE_PATH,\n)\n\n\n# =============================================================================\n# 16. Step 15 verified backup\n# =============================================================================\n\nstep15_backup_members = create_verified_zip(\n    STEP15_BACKUP_PATH,\n    [\n        AUDIT_PATH,\n        BACKUP_AUDIT_PATH,\n        ENVIRONMENT_PATH,\n        README_PATH,\n        CLAIM_BOUNDARY_PATH,\n        RESULTS_MANIFEST_PATH,\n        REPRO_MANIFEST_PATH,\n        PACKAGE_INDEX_PATH,\n        SUMMARY_PATH,\n        STATE_PATH,\n    ],\n)\n\n\n# =============================================================================\n# 17. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"STEP 15 — FINAL REPRODUCIBILITY AUDIT \"\n    \"AND RESULTS PACKAGING COMPLETED\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nEVIDENCE SAFETY\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"New training performed                :\",\n    False\n)\n\nprint(\n    \"Optimizer created                     :\",\n    False\n)\n\nprint(\n    \"Model loaded                          :\",\n    False\n)\n\nprint(\n    \"Model inference performed             :\",\n    False\n)\n\nprint(\n    \"Raw images loaded                     :\",\n    False\n)\n\nprint(\n    \"Validation evaluated                  :\",\n    False\n)\n\nprint(\n    \"Final test evaluated                  :\",\n    False\n)\n\nprint(\n    \"Predictions regenerated               :\",\n    False\n)\n\nprint(\n    \"Frozen model changed                  :\",\n    False\n)\n\n\nprint(\n    \"\\nFINAL REPRODUCIBILITY AUDIT\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Total audit checks                    :\",\n    len(\n        audit_df\n    )\n)\n\nprint(\n    \"Passed audit checks                   :\",\n    int(\n        audit_df[\n            \"passed\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        audit_df\n    )\n)\n\nprint(\n    \"Critical audit passed                 :\",\n    critical_audit_passed\n)\n\nprint(\n    \"Required backups verified             :\",\n    int(\n        backup_audit_df[\n            \"zip_integrity_passed\"\n        ].sum()\n    ),\n    \"/\",\n    len(\n        backup_audit_df\n    )\n)\n\nprint(\n    \"Failed Step 13B partial figures       :\",\n    len(\n        failed_step13b_files\n    )\n)\n\nprint(\n    \"Final-test evaluation count           :\",\n    step10d_state.get(\n        \"test_evaluation_count\"\n    )\n)\n\nprint(\n    \"Additional test evaluation allowed    :\",\n    step10d_state.get(\n        \"another_test_evaluation_allowed\"\n    )\n)\n\nprint(\n    \"Model change allowed                  :\",\n    step10d_state.get(\n        \"model_change_allowed\"\n    )\n)\n\n\nprint(\n    \"\\nFINAL MODEL AND METRICS\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Selected architecture                 :\",\n    \"Registered EfficientNet-B0 baseline\"\n)\n\nprint(\n    \"Registered parameters                 :\",\n    f\"{EXPECTED_PARAMETER_COUNT:,}\"\n)\n\nprint(\n    \"Checkpoint SHA-256                    :\",\n    checkpoint_sha256\n)\n\nprint(\n    \"Final-test QWK                        :\",\n    f\"{EXPECTED_FINAL_METRICS['QWK']:.6f}\"\n)\n\nprint(\n    \"Final-test accuracy                   :\",\n    f\"{EXPECTED_FINAL_METRICS['Accuracy']:.6f}\"\n)\n\nprint(\n    \"Final-test balanced accuracy          :\",\n    f\"{EXPECTED_FINAL_METRICS['Balanced accuracy']:.6f}\"\n)\n\nprint(\n    \"Final-test macro F1                   :\",\n    f\"{EXPECTED_FINAL_METRICS['Macro F1']:.6f}\"\n)\n\n\nprint(\n    \"\\nFINAL PACKAGES\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nfor _, package_row in package_index_df.iterrows():\n\n    print(\n        \"Package name                          :\",\n        package_row[\n            \"package_name\"\n        ]\n    )\n\n    print(\n        \"Package path                          :\",\n        package_row[\n            \"package_path\"\n        ]\n    )\n\n    print(\n        \"Package size                          :\",\n        package_row[\n            \"size_readable\"\n        ]\n    )\n\n    print(\n        \"Package members                       :\",\n        int(\n            package_row[\n                \"member_count\"\n            ]\n        )\n    )\n\n    print(\n        \"Package SHA-256                       :\",\n        package_row[\n            \"sha256\"\n        ]\n    )\n\n    print(\n        \"ZIP integrity passed                  :\",\n        bool(\n            package_row[\n                \"zip_integrity_passed\"\n            ]\n        )\n    )\n\n    print(\n        \"-\" * 126\n    )\n\n\nprint(\n    \"\\nSTEP 15 BACKUP\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup path                           :\",\n    STEP15_BACKUP_PATH\n)\n\nprint(\n    \"Backup members                        :\",\n    len(\n        step15_backup_members\n    )\n)\n\nprint(\n    \"ZIP integrity passed                  :\",\n    True\n)\n\n\nprint(\n    \"\\nNEXT STAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"READY FOR STEP 16 — MANUSCRIPT METHODS, \"\n    \"RESULTS AND DISCUSSION DRAFTING\"\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T16:35:49.928309Z","iopub.execute_input":"2026-07-18T16:35:49.928932Z","iopub.status.idle":"2026-07-18T16:35:55.973610Z","shell.execute_reply.started":"2026-07-18T16:35:49.928899Z","shell.execute_reply":"2026-07-18T16:35:55.972842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CORRECTED MASTER BACKUP PACKAGE\n#\n# Creates one Master ZIP containing all existing ZIP files from 13_backups.\n#\n# Important behavior:\n#   - Existing source ZIP files are NOT extracted\n#   - Existing source ZIP files are NOT renamed\n#   - Existing source ZIP files are NOT modified\n#   - Duplicate filenames inside legacy ZIPs are recorded as warnings\n#   - Duplicate nested filenames do NOT cause failure\n#   - Actual CRC corruption DOES cause failure\n#   - Every nested ZIP is hash-verified after being added to the Master ZIP\n#\n# Output:\n#   DR_PUBLICATION_2026_ALL_BACKUPS_MASTER.zip\n#\n# Run this new cell only. Do not use Run All.\n# =============================================================================\n\nfrom pathlib import Path\nfrom datetime import datetime, timezone\nfrom collections import Counter\n\nimport csv\nimport hashlib\nimport io\nimport json\nimport os\nimport zipfile\n\n\n# =============================================================================\n# 1. Fixed paths\n# =============================================================================\n\nPROJECT = Path(\n    \"/kaggle/working/DR_PUBLICATION_2026\"\n)\n\nBACKUP_DIR = (\n    PROJECT\n    / \"13_backups\"\n)\n\nMASTER_ZIP_PATH = (\n    BACKUP_DIR\n    / \"DR_PUBLICATION_2026_ALL_BACKUPS_MASTER.zip\"\n)\n\nMASTER_TEMP_PATH = (\n    BACKUP_DIR\n    / \"DR_PUBLICATION_2026_ALL_BACKUPS_MASTER.zip.tmp\"\n)\n\n\n# =============================================================================\n# 2. Metadata filenames inside the Master ZIP\n# =============================================================================\n\nINDEX_FILENAME = (\n    \"MASTER_BACKUP_INDEX.csv\"\n)\n\nDUPLICATE_REPORT_FILENAME = (\n    \"LEGACY_DUPLICATE_MEMBER_REPORT.csv\"\n)\n\nSUMMARY_FILENAME = (\n    \"MASTER_BACKUP_SUMMARY.json\"\n)\n\nREADME_FILENAME = (\n    \"README_MASTER_BACKUP.txt\"\n)\n\n\n# =============================================================================\n# 3. Utility functions\n# =============================================================================\n\ndef utc_now():\n\n    return datetime.now(\n        timezone.utc\n    ).isoformat()\n\n\ndef sha256_file(\n    file_path,\n):\n\n    digest = hashlib.sha256()\n\n    with open(\n        file_path,\n        \"rb\",\n    ) as file:\n\n        for chunk in iter(\n            lambda: file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef sha256_zip_member(\n    archive,\n    member_name,\n):\n\n    digest = hashlib.sha256()\n\n    with archive.open(\n        member_name,\n        mode=\"r\",\n    ) as member_file:\n\n        for chunk in iter(\n            lambda: member_file.read(\n                1024 * 1024\n            ),\n            b\"\",\n        ):\n\n            digest.update(\n                chunk\n            )\n\n    return digest.hexdigest()\n\n\ndef format_bytes(\n    number_of_bytes,\n):\n\n    size = float(\n        number_of_bytes\n    )\n\n    units = [\n        \"B\",\n        \"KB\",\n        \"MB\",\n        \"GB\",\n        \"TB\",\n    ]\n\n    for unit in units:\n\n        if size < 1024.0:\n\n            return f\"{size:.2f} {unit}\"\n\n        size /= 1024.0\n\n    return f\"{size:.2f} PB\"\n\n\ndef make_csv_text(\n    records,\n    fieldnames,\n):\n\n    buffer = io.StringIO()\n\n    writer = csv.DictWriter(\n        buffer,\n        fieldnames=fieldnames,\n        lineterminator=\"\\n\",\n    )\n\n    writer.writeheader()\n\n    writer.writerows(\n        records\n    )\n\n    return buffer.getvalue()\n\n\n# =============================================================================\n# 4. Basic safety checks\n# =============================================================================\n\nif not BACKUP_DIR.exists():\n\n    raise FileNotFoundError(\n        \"Backup directory does not exist:\\n\"\n        f\"{BACKUP_DIR}\"\n    )\n\n\nif MASTER_ZIP_PATH.exists():\n\n    raise RuntimeError(\n        \"The final Master ZIP already exists. \"\n        \"Do not overwrite it:\\n\"\n        f\"{MASTER_ZIP_PATH}\"\n    )\n\n\n# Remove only an unfinished temporary file from the previous failed attempt.\nif MASTER_TEMP_PATH.exists():\n\n    MASTER_TEMP_PATH.unlink()\n\n\n# =============================================================================\n# 5. Discover every existing source ZIP\n# =============================================================================\n\nsource_zip_files = sorted(\n    path\n    for path\n    in BACKUP_DIR.glob(\n        \"*.zip\"\n    )\n    if (\n        path.is_file()\n        and\n        path.name\n        !=\n        MASTER_ZIP_PATH.name\n    )\n)\n\n\nif not source_zip_files:\n\n    raise RuntimeError(\n        \"No source ZIP files were found in:\\n\"\n        f\"{BACKUP_DIR}\"\n    )\n\n\nsource_zip_names = [\n    source_zip.name\n    for source_zip\n    in source_zip_files\n]\n\n\nif len(\n    source_zip_names\n) != len(\n    set(\n        source_zip_names\n    )\n):\n\n    raise RuntimeError(\n        \"Duplicate source ZIP filenames were found \"\n        \"in the backup directory.\"\n    )\n\n\n# =============================================================================\n# 6. Verify source ZIPs\n#\n# Actual corruption:\n#   archive.testzip() returns a damaged member\n#\n# Legacy duplicate nested paths:\n#   allowed, but documented as warnings\n# =============================================================================\n\nsource_records = []\nduplicate_member_records = []\n\ntotal_source_size_bytes = 0\ntotal_nested_members = 0\ntotal_duplicate_names = 0\ntotal_extra_duplicate_entries = 0\nsource_zips_with_duplicate_members = 0\n\n\nfor source_zip in source_zip_files:\n\n    try:\n\n        with zipfile.ZipFile(\n            source_zip,\n            mode=\"r\",\n            allowZip64=True,\n        ) as archive:\n\n            zip_info_list = archive.infolist()\n\n            nested_member_names = [\n                zip_info.filename\n                for zip_info\n                in zip_info_list\n            ]\n\n            damaged_member = archive.testzip()\n\n\n    except zipfile.BadZipFile as error:\n\n        raise RuntimeError(\n            \"A source file is not a readable ZIP archive:\\n\"\n            f\"{source_zip}\\n\"\n            f\"Error: {error}\"\n        )\n\n\n    except Exception as error:\n\n        raise RuntimeError(\n            \"A source ZIP could not be verified:\\n\"\n            f\"{source_zip}\\n\"\n            f\"Error: {error}\"\n        )\n\n\n    # A non-None result indicates actual CRC corruption.\n    if damaged_member is not None:\n\n        raise RuntimeError(\n            \"A source ZIP contains an actually damaged CRC member:\\n\"\n            f\"{source_zip}\\n\"\n            f\"Damaged member: {damaged_member}\"\n        )\n\n\n    nested_name_counts = Counter(\n        nested_member_names\n    )\n\n\n    duplicated_nested_names = sorted(\n        member_name\n\n        for member_name, occurrence_count\n        in nested_name_counts.items()\n\n        if occurrence_count > 1\n    )\n\n\n    duplicate_name_count = len(\n        duplicated_nested_names\n    )\n\n\n    extra_duplicate_entry_count = sum(\n        nested_name_counts[\n            member_name\n        ]\n        -\n        1\n\n        for member_name\n        in duplicated_nested_names\n    )\n\n\n    if duplicate_name_count > 0:\n\n        source_zips_with_duplicate_members += 1\n\n\n    total_duplicate_names += duplicate_name_count\n\n    total_extra_duplicate_entries += (\n        extra_duplicate_entry_count\n    )\n\n\n    for duplicated_member_name in duplicated_nested_names:\n\n        occurrence_count = int(\n            nested_name_counts[\n                duplicated_member_name\n            ]\n        )\n\n        duplicate_member_records.append({\n            \"source_zip_filename\": (\n                source_zip.name\n            ),\n\n            \"duplicated_nested_member\": (\n                duplicated_member_name\n            ),\n\n            \"occurrence_count\": (\n                occurrence_count\n            ),\n\n            \"extra_duplicate_entries\": (\n                occurrence_count\n                -\n                1\n            ),\n\n            \"crc_corruption_detected\": (\n                False\n            ),\n\n            \"handling\": (\n                \"Preserved unchanged inside source ZIP; \"\n                \"reported as legacy warning\"\n            ),\n        })\n\n\n    source_size_bytes = int(\n        source_zip.stat().st_size\n    )\n\n    source_sha256 = sha256_file(\n        source_zip\n    )\n\n\n    total_source_size_bytes += source_size_bytes\n\n    total_nested_members += len(\n        nested_member_names\n    )\n\n\n    source_records.append({\n        \"zip_filename\": (\n            source_zip.name\n        ),\n\n        \"source_path\": str(\n            source_zip\n        ),\n\n        \"size_bytes\": (\n            source_size_bytes\n        ),\n\n        \"size_readable\": format_bytes(\n            source_size_bytes\n        ),\n\n        \"nested_member_count\": int(\n            len(\n                nested_member_names\n            )\n        ),\n\n        \"unique_nested_member_count\": int(\n            len(\n                nested_name_counts\n            )\n        ),\n\n        \"duplicate_nested_name_count\": int(\n            duplicate_name_count\n        ),\n\n        \"extra_duplicate_entry_count\": int(\n            extra_duplicate_entry_count\n        ),\n\n        \"legacy_duplicate_warning\": bool(\n            duplicate_name_count > 0\n        ),\n\n        \"crc_integrity_passed\": (\n            True\n        ),\n\n        \"source_sha256\": (\n            source_sha256\n        ),\n    })\n\n\n# =============================================================================\n# 7. Create metadata content\n# =============================================================================\n\nindex_fieldnames = [\n    \"zip_filename\",\n    \"source_path\",\n    \"size_bytes\",\n    \"size_readable\",\n    \"nested_member_count\",\n    \"unique_nested_member_count\",\n    \"duplicate_nested_name_count\",\n    \"extra_duplicate_entry_count\",\n    \"legacy_duplicate_warning\",\n    \"crc_integrity_passed\",\n    \"source_sha256\",\n]\n\n\nmaster_index_csv_text = make_csv_text(\n    source_records,\n    index_fieldnames,\n)\n\n\nduplicate_report_fieldnames = [\n    \"source_zip_filename\",\n    \"duplicated_nested_member\",\n    \"occurrence_count\",\n    \"extra_duplicate_entries\",\n    \"crc_corruption_detected\",\n    \"handling\",\n]\n\n\nlegacy_duplicate_report_csv_text = make_csv_text(\n    duplicate_member_records,\n    duplicate_report_fieldnames,\n)\n\n\nsummary_record = {\n    \"package_name\": (\n        MASTER_ZIP_PATH.name\n    ),\n\n    \"created_utc\": (\n        utc_now()\n    ),\n\n    \"project_root\": str(\n        PROJECT\n    ),\n\n    \"source_backup_directory\": str(\n        BACKUP_DIR\n    ),\n\n    \"source_zip_count\": int(\n        len(\n            source_zip_files\n        )\n    ),\n\n    \"total_source_zip_size_bytes\": int(\n        total_source_size_bytes\n    ),\n\n    \"total_source_zip_size_readable\": format_bytes(\n        total_source_size_bytes\n    ),\n\n    \"total_nested_member_entries\": int(\n        total_nested_members\n    ),\n\n    \"source_zips_with_legacy_duplicate_members\": int(\n        source_zips_with_duplicate_members\n    ),\n\n    \"total_duplicate_nested_names\": int(\n        total_duplicate_names\n    ),\n\n    \"total_extra_duplicate_entries\": int(\n        total_extra_duplicate_entries\n    ),\n\n    \"source_zip_crc_integrity_passed\": (\n        True\n    ),\n\n    \"legacy_duplicate_members_treated_as_corruption\": (\n        False\n    ),\n\n    \"source_zip_files_extracted\": (\n        False\n    ),\n\n    \"source_zip_files_renamed\": (\n        False\n    ),\n\n    \"source_zip_files_modified\": (\n        False\n    ),\n\n    \"packaging_compression\": (\n        \"ZIP_STORED\"\n    ),\n\n    \"reason_for_zip_stored\": (\n        \"Source files are already compressed ZIP archives; \"\n        \"storing avoids unnecessary recompression.\"\n    ),\n}\n\n\nsummary_json_text = json.dumps(\n    summary_record,\n    indent=2,\n    ensure_ascii=False,\n)\n\n\nreadme_text = f\"\"\"\nDR_PUBLICATION_2026 — COMPLETE MASTER BACKUP\n\nCreated UTC\n-----------\n{summary_record['created_utc']}\n\nPurpose\n-------\nThis archive provides one downloadable file containing every existing\nZIP package found in:\n\n{BACKUP_DIR}\n\nSource ZIP count\n----------------\n{len(source_zip_files)}\n\nTotal size of source ZIP files\n------------------------------\n{format_bytes(total_source_size_bytes)}\n\nPackaging policy\n----------------\n1. Source ZIP files were not extracted.\n2. Source ZIP files were not renamed.\n3. Source ZIP files were not modified.\n4. Each source ZIP remains available at the root of this Master ZIP\n   under its original filename.\n5. Every source ZIP passed its CRC integrity test.\n6. Legacy duplicate nested filenames were retained unchanged and\n   documented rather than treated as file corruption.\n\nLegacy duplicate-member warning\n-------------------------------\nSource ZIPs containing duplicate nested names:\n{source_zips_with_duplicate_members}\n\nTotal duplicate nested filenames:\n{total_duplicate_names}\n\nTotal extra duplicate entries:\n{total_extra_duplicate_entries}\n\nA duplicate nested path does not necessarily mean that the ZIP is\ncorrupted. It commonly occurs when an older backup routine writes the\nsame path more than once. The original ZIP has therefore been preserved\nwithout alteration.\n\nIncluded metadata\n-----------------\n{INDEX_FILENAME}\n    Index of all source ZIPs, sizes, nested-member counts and SHA-256\n    fingerprints.\n\n{DUPLICATE_REPORT_FILENAME}\n    Detailed report of legacy duplicate nested-member names.\n\n{SUMMARY_FILENAME}\n    Machine-readable Master Backup summary.\n\n{README_FILENAME}\n    This document.\n\nHow to use\n----------\nDownload and extract this one Master ZIP on a local computer. All\nindividual ZIP packages will remain separate and retain their original\nfilenames.\n\nImportant\n---------\nSome source packages intentionally contain overlapping evidence. This\nduplication is preserved because the purpose of this archive is complete\none-download backup coverage, not storage minimization.\n\"\"\".strip()\n\n\n# =============================================================================\n# 8. Create the Master ZIP\n#\n# ZIP_STORED is intentional because nested ZIP files are already compressed.\n# =============================================================================\n\ntry:\n\n    with zipfile.ZipFile(\n        MASTER_TEMP_PATH,\n        mode=\"w\",\n        compression=zipfile.ZIP_STORED,\n        allowZip64=True,\n    ) as master_archive:\n\n        for source_zip in source_zip_files:\n\n            master_archive.write(\n                source_zip,\n                arcname=source_zip.name,\n            )\n\n\n        master_archive.writestr(\n            INDEX_FILENAME,\n            master_index_csv_text,\n        )\n\n\n        master_archive.writestr(\n            DUPLICATE_REPORT_FILENAME,\n            legacy_duplicate_report_csv_text,\n        )\n\n\n        master_archive.writestr(\n            SUMMARY_FILENAME,\n            summary_json_text,\n        )\n\n\n        master_archive.writestr(\n            README_FILENAME,\n            readme_text,\n        )\n\n\nexcept Exception:\n\n    if MASTER_TEMP_PATH.exists():\n\n        MASTER_TEMP_PATH.unlink()\n\n    raise\n\n\n# =============================================================================\n# 9. Verify Master ZIP structure, CRC and byte-identical nested ZIP hashes\n# =============================================================================\n\nexpected_metadata_members = [\n    INDEX_FILENAME,\n    DUPLICATE_REPORT_FILENAME,\n    SUMMARY_FILENAME,\n    README_FILENAME,\n]\n\n\nexpected_master_members = (\n    source_zip_names\n    +\n    expected_metadata_members\n)\n\n\nnested_hash_verification_records = []\n\n\ntry:\n\n    with zipfile.ZipFile(\n        MASTER_TEMP_PATH,\n        mode=\"r\",\n        allowZip64=True,\n    ) as master_archive:\n\n        master_members = master_archive.namelist()\n\n        damaged_master_member = master_archive.testzip()\n\n\n        if damaged_master_member is not None:\n\n            raise RuntimeError(\n                \"The Master ZIP contains a CRC-damaged member:\\n\"\n                f\"{damaged_master_member}\"\n            )\n\n\n        missing_master_members = sorted(\n            set(\n                expected_master_members\n            )\n            -\n            set(\n                master_members\n            )\n        )\n\n\n        unexpected_master_members = sorted(\n            set(\n                master_members\n            )\n            -\n            set(\n                expected_master_members\n            )\n        )\n\n\n        duplicate_outer_members = bool(\n            len(\n                master_members\n            )\n            !=\n            len(\n                set(\n                    master_members\n                )\n            )\n        )\n\n\n        if missing_master_members:\n\n            raise RuntimeError(\n                \"The Master ZIP is missing expected members:\\n\"\n                +\n                \"\\n\".join(\n                    missing_master_members\n                )\n            )\n\n\n        if unexpected_master_members:\n\n            raise RuntimeError(\n                \"The Master ZIP contains unexpected members:\\n\"\n                +\n                \"\\n\".join(\n                    unexpected_master_members\n                )\n            )\n\n\n        if duplicate_outer_members:\n\n            raise RuntimeError(\n                \"Duplicate filenames were detected at the \"\n                \"outer Master ZIP level.\"\n            )\n\n\n        source_record_by_name = {\n            record[\n                \"zip_filename\"\n            ]: record\n\n            for record\n            in source_records\n        }\n\n\n        for source_zip_name in source_zip_names:\n\n            master_member_info = master_archive.getinfo(\n                source_zip_name\n            )\n\n\n            expected_source_size = int(\n                source_record_by_name[\n                    source_zip_name\n                ][\n                    \"size_bytes\"\n                ]\n            )\n\n\n            observed_member_size = int(\n                master_member_info.file_size\n            )\n\n\n            size_matches = bool(\n                observed_member_size\n                ==\n                expected_source_size\n            )\n\n\n            nested_member_sha256 = sha256_zip_member(\n                master_archive,\n                source_zip_name,\n            )\n\n\n            expected_source_sha256 = str(\n                source_record_by_name[\n                    source_zip_name\n                ][\n                    \"source_sha256\"\n                ]\n            )\n\n\n            sha256_matches = bool(\n                nested_member_sha256\n                ==\n                expected_source_sha256\n            )\n\n\n            nested_hash_verification_records.append({\n                \"zip_filename\": (\n                    source_zip_name\n                ),\n\n                \"expected_size_bytes\": (\n                    expected_source_size\n                ),\n\n                \"master_member_size_bytes\": (\n                    observed_member_size\n                ),\n\n                \"size_matches\": (\n                    size_matches\n                ),\n\n                \"expected_sha256\": (\n                    expected_source_sha256\n                ),\n\n                \"master_member_sha256\": (\n                    nested_member_sha256\n                ),\n\n                \"sha256_matches\": (\n                    sha256_matches\n                ),\n            })\n\n\n            if not size_matches:\n\n                raise RuntimeError(\n                    \"A nested ZIP size changed during Master packaging:\\n\"\n                    f\"{source_zip_name}\"\n                )\n\n\n            if not sha256_matches:\n\n                raise RuntimeError(\n                    \"A nested ZIP hash changed during Master packaging:\\n\"\n                    f\"{source_zip_name}\"\n                )\n\n\nexcept Exception:\n\n    if MASTER_TEMP_PATH.exists():\n\n        MASTER_TEMP_PATH.unlink()\n\n    raise\n\n\nall_nested_zip_hashes_match = all(\n    record[\n        \"sha256_matches\"\n    ]\n\n    for record\n    in nested_hash_verification_records\n)\n\n\nall_nested_zip_sizes_match = all(\n    record[\n        \"size_matches\"\n    ]\n\n    for record\n    in nested_hash_verification_records\n)\n\n\nmaster_integrity_passed = bool(\n    damaged_master_member is None\n    and\n    len(\n        missing_master_members\n    )\n    ==\n    0\n    and\n    len(\n        unexpected_master_members\n    )\n    ==\n    0\n    and\n    not duplicate_outer_members\n    and\n    all_nested_zip_hashes_match\n    and\n    all_nested_zip_sizes_match\n)\n\n\nif not master_integrity_passed:\n\n    if MASTER_TEMP_PATH.exists():\n\n        MASTER_TEMP_PATH.unlink()\n\n    raise RuntimeError(\n        \"Final Master ZIP integrity verification failed.\"\n    )\n\n\n# =============================================================================\n# 10. Atomically finalize the Master ZIP\n# =============================================================================\n\nos.replace(\n    MASTER_TEMP_PATH,\n    MASTER_ZIP_PATH,\n)\n\n\nmaster_size_bytes = int(\n    MASTER_ZIP_PATH.stat().st_size\n)\n\nmaster_sha256 = sha256_file(\n    MASTER_ZIP_PATH\n)\n\n\n# =============================================================================\n# 11. Final post-rename verification\n# =============================================================================\n\nwith zipfile.ZipFile(\n    MASTER_ZIP_PATH,\n    mode=\"r\",\n    allowZip64=True,\n) as final_master_archive:\n\n    final_master_members = (\n        final_master_archive.namelist()\n    )\n\n    final_damaged_member = (\n        final_master_archive.testzip()\n    )\n\n\nfinal_integrity_passed = bool(\n    final_damaged_member is None\n    and\n    len(\n        final_master_members\n    )\n    ==\n    len(\n        expected_master_members\n    )\n    and\n    len(\n        final_master_members\n    )\n    ==\n    len(\n        set(\n            final_master_members\n        )\n    )\n)\n\n\nif not final_integrity_passed:\n\n    raise RuntimeError(\n        \"The finalized Master ZIP did not pass \"\n        \"the final post-rename verification.\"\n    )\n\n\n# =============================================================================\n# 12. Controlled output\n# =============================================================================\n\nprint(\n    \"\\n\"\n    +\n    \"=\" * 126\n)\n\nprint(\n    \"CORRECTED MASTER BACKUP PACKAGE CREATED SUCCESSFULLY\"\n)\n\nprint(\n    \"=\" * 126\n)\n\n\nprint(\n    \"\\nSOURCE ZIP VERIFICATION\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Backup directory                      :\",\n    BACKUP_DIR\n)\n\nprint(\n    \"Source ZIP files found                :\",\n    len(\n        source_zip_files\n    )\n)\n\nprint(\n    \"Source ZIP CRC integrity passed       :\",\n    len(\n        source_zip_files\n    ),\n    \"/\",\n    len(\n        source_zip_files\n    )\n)\n\nprint(\n    \"Source ZIPs with legacy duplicates    :\",\n    source_zips_with_duplicate_members\n)\n\nprint(\n    \"Duplicate nested filenames recorded  :\",\n    total_duplicate_names\n)\n\nprint(\n    \"Extra duplicate entries recorded     :\",\n    total_extra_duplicate_entries\n)\n\nprint(\n    \"Total source ZIP size                 :\",\n    format_bytes(\n        total_source_size_bytes\n    )\n)\n\n\nprint(\n    \"\\nLEGACY DUPLICATE WARNINGS\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nif duplicate_member_records:\n\n    duplicate_summary_by_zip = {}\n\n    for record in duplicate_member_records:\n\n        zip_name = record[\n            \"source_zip_filename\"\n        ]\n\n        duplicate_summary_by_zip.setdefault(\n            zip_name,\n            {\n                \"duplicate_names\": 0,\n                \"extra_entries\": 0,\n            },\n        )\n\n        duplicate_summary_by_zip[\n            zip_name\n        ][\n            \"duplicate_names\"\n        ] += 1\n\n        duplicate_summary_by_zip[\n            zip_name\n        ][\n            \"extra_entries\"\n        ] += int(\n            record[\n                \"extra_duplicate_entries\"\n            ]\n        )\n\n\n    for zip_name in sorted(\n        duplicate_summary_by_zip\n    ):\n\n        warning_record = (\n            duplicate_summary_by_zip[\n                zip_name\n            ]\n        )\n\n        print(\n            f\"{zip_name}: \"\n            f\"{warning_record['duplicate_names']} duplicate names, \"\n            f\"{warning_record['extra_entries']} extra entries\"\n        )\n\nelse:\n\n    print(\n        \"No legacy duplicate nested filenames detected.\"\n    )\n\n\nprint(\n    \"\\nMASTER ZIP VERIFICATION\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Original source ZIP filenames kept   :\",\n    True\n)\n\nprint(\n    \"Source ZIPs extracted                :\",\n    False\n)\n\nprint(\n    \"Source ZIPs renamed                  :\",\n    False\n)\n\nprint(\n    \"Source ZIPs modified                 :\",\n    False\n)\n\nprint(\n    \"Nested ZIP size matches              :\",\n    sum(\n        record[\n            \"size_matches\"\n        ]\n        for record\n        in nested_hash_verification_records\n    ),\n    \"/\",\n    len(\n        nested_hash_verification_records\n    )\n)\n\nprint(\n    \"Nested ZIP SHA-256 matches           :\",\n    sum(\n        record[\n            \"sha256_matches\"\n        ]\n        for record\n        in nested_hash_verification_records\n    ),\n    \"/\",\n    len(\n        nested_hash_verification_records\n    )\n)\n\nprint(\n    \"Metadata files included              :\",\n    len(\n        expected_metadata_members\n    )\n)\n\nprint(\n    \"Missing Master members              :\",\n    len(\n        missing_master_members\n    )\n)\n\nprint(\n    \"Unexpected Master members           :\",\n    len(\n        unexpected_master_members\n    )\n)\n\nprint(\n    \"Outer duplicate members             :\",\n    0\n)\n\nprint(\n    \"Master ZIP integrity passed         :\",\n    final_integrity_passed\n)\n\n\nprint(\n    \"\\nFINAL MASTER PACKAGE\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Master ZIP path                      :\",\n    MASTER_ZIP_PATH\n)\n\nprint(\n    \"Master ZIP size                      :\",\n    format_bytes(\n        master_size_bytes\n    )\n)\n\nprint(\n    \"Individual source ZIPs inside        :\",\n    len(\n        source_zip_files\n    )\n)\n\nprint(\n    \"Total Master ZIP members             :\",\n    len(\n        final_master_members\n    )\n)\n\nprint(\n    \"Master ZIP SHA-256                   :\",\n    master_sha256\n)\n\n\nprint(\n    \"\\nDOWNLOAD\"\n)\n\nprint(\n    \"-\" * 126\n)\n\nprint(\n    \"Download only this single file from the Kaggle Files panel:\"\n)\n\nprint(\n    MASTER_ZIP_PATH\n)\n\nprint(\n    \"=\" * 126\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T16:50:21.642256Z","iopub.execute_input":"2026-07-18T16:50:21.642789Z","iopub.status.idle":"2026-07-18T16:51:47.432217Z","shell.execute_reply.started":"2026-07-18T16:50:21.642755Z","shell.execute_reply":"2026-07-18T16:51:47.431514Z"}},"outputs":[],"execution_count":null}]}