{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-09T18:21:58.808896Z","iopub.execute_input":"2026-08-09T18:21:58.809259Z","iopub.status.idle":"2026-08-09T18:22:08.826649Z","shell.execute_reply.started":"2026-08-09T18:21:58.809234Z","shell.execute_reply":"2026-08-09T18:22:08.825809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(torch.cuda.is_available())\nprint(torch.cuda.get_device_name(0) if torch.cuda.is_available() else \"No GPU\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T18:23:03.303826Z","iopub.execute_input":"2026-08-09T18:23:03.304567Z","iopub.status.idle":"2026-08-09T18:23:08.095249Z","shell.execute_reply.started":"2026-08-09T18:23:03.304538Z","shell.execute_reply":"2026-08-09T18:23:08.094494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames[:3]:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T18:24:50.102038Z","iopub.execute_input":"2026-08-09T18:24:50.103011Z","iopub.status.idle":"2026-08-09T18:24:54.591944Z","shell.execute_reply.started":"2026-08-09T18:24:50.102979Z","shell.execute_reply":"2026-08-09T18:24:54.591027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm import tqdm\n\ncv2.setNumThreads(0)\n\n# ============================================================\n# 1. Load labels\n# ============================================================\ndf = pd.read_csv('/kaggle/input/competitions/aptos2019-blindness-detection/train.csv')\ndf['filepath'] = df['id_code'].apply(\n    lambda x: f'/kaggle/input/competitions/aptos2019-blindness-detection/train_images/{x}.png'\n)\nprint(f\"Total images to process: {len(df)}\")\n\n# ============================================================\n# 2. CLAHE + resize function\n# ============================================================\nIMG_SIZE = 380\n\ndef preprocess_and_save(filepath, save_path):\n    img = Image.open(filepath).convert('RGB')\n    img = img.resize((IMG_SIZE, IMG_SIZE))  # resize first, saves CLAHE compute time on smaller image\n    img_np = np.array(img)\n    lab = cv2.cvtColor(img_np, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    l_clahe = clahe.apply(l)\n    lab_clahe = cv2.merge((l_clahe, a, b))\n    result = cv2.cvtColor(lab_clahe, cv2.COLOR_LAB2RGB)\n    Image.fromarray(result).save(save_path)\n\n# ============================================================\n# 3. Process all images, save to /kaggle/working/clahe_images/\n# ============================================================\nOUTPUT_DIR = '/kaggle/working/clahe_images'\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\nnew_filepaths = []\nfor idx, row in tqdm(df.iterrows(), total=len(df)):\n    save_path = os.path.join(OUTPUT_DIR, f\"{row['id_code']}.png\")\n    if not os.path.exists(save_path):  # skip if already done (lets you resume if interrupted)\n        preprocess_and_save(row['filepath'], save_path)\n    new_filepaths.append(save_path)\n\ndf['clahe_filepath'] = new_filepaths\n\n# ============================================================\n# 4. Save the updated dataframe (with new paths) so training step can reload it\n# ============================================================\ndf.to_csv('/kaggle/working/train_with_clahe_paths.csv', index=False)\n\nprint(f\"\\nDone. Processed images saved to: {OUTPUT_DIR}\")\nprint(f\"Updated dataframe saved to: /kaggle/working/train_with_clahe_paths.csv\")\nprint(f\"Sample files: {os.listdir(OUTPUT_DIR)[:5]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T19:22:28.383441Z","iopub.execute_input":"2026-08-09T19:22:28.383713Z","iopub.status.idle":"2026-08-09T19:39:01.415469Z","shell.execute_reply.started":"2026-08-09T19:22:28.383685Z","shell.execute_reply":"2026-08-09T19:39:01.414404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.path.exists('/kaggle/working/clahe_images'))\nprint(len(os.listdir('/kaggle/working/clahe_images')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T19:43:32.479147Z","iopub.execute_input":"2026-08-09T19:43:32.479463Z","iopub.status.idle":"2026-08-09T19:43:32.496336Z","shell.execute_reply.started":"2026-08-09T19:43:32.479426Z","shell.execute_reply":"2026-08-09T19:43:32.495101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm import tqdm\n\ncv2.setNumThreads(0)\n\n# ============================================================\n# 1. Load labels\n# ============================================================\ndf = pd.read_csv('/kaggle/input/competitions/aptos2019-blindness-detection/train.csv')\ndf['filepath'] = df['id_code'].apply(\n    lambda x: f'/kaggle/input/competitions/aptos2019-blindness-detection/train_images/{x}.png'\n)\nprint(f\"Total images to process: {len(df)}\")\n\n# ============================================================\n# 2. CLAHE + resize function\n# ============================================================\nIMG_SIZE = 380\n\ndef preprocess_and_save(filepath, save_path):\n    img = Image.open(filepath).convert('RGB')\n    img = img.resize((IMG_SIZE, IMG_SIZE))  # resize first, saves CLAHE compute time on smaller image\n    img_np = np.array(img)\n    lab = cv2.cvtColor(img_np, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    l_clahe = clahe.apply(l)\n    lab_clahe = cv2.merge((l_clahe, a, b))\n    result = cv2.cvtColor(lab_clahe, cv2.COLOR_LAB2RGB)\n    Image.fromarray(result).save(save_path)\n\n# ============================================================\n# 3. Process all images, save to /kaggle/working/clahe_images/\n# ============================================================\nOUTPUT_DIR = '/kaggle/working/clahe_images'\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\nnew_filepaths = []\nfor idx, row in tqdm(df.iterrows(), total=len(df)):\n    save_path = os.path.join(OUTPUT_DIR, f\"{row['id_code']}.png\")\n    if not os.path.exists(save_path):  # skip if already done (lets you resume if interrupted)\n        preprocess_and_save(row['filepath'], save_path)\n    new_filepaths.append(save_path)\n\ndf['clahe_filepath'] = new_filepaths\n\n# ============================================================\n# 4. Save the updated dataframe (with new paths) so training step can reload it\n# ============================================================\ndf.to_csv('/kaggle/working/train_with_clahe_paths.csv', index=False)\n\nprint(f\"\\nDone. Processed images saved to: {OUTPUT_DIR}\")\nprint(f\"Updated dataframe saved to: /kaggle/working/train_with_clahe_paths.csv\")\nprint(f\"Sample files: {os.listdir(OUTPUT_DIR)[:5]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T19:45:50.091368Z","iopub.execute_input":"2026-08-09T19:45:50.092206Z","iopub.status.idle":"2026-08-09T20:00:05.474578Z","shell.execute_reply.started":"2026-08-09T19:45:50.092173Z","shell.execute_reply":"2026-08-09T20:00:05.473534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, classification_report\nfrom PIL import Image\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\nprint(f\"GPU count: {torch.cuda.device_count()}\")\n\nassert os.path.exists('/kaggle/working/clahe_images'), \"CLAHE images missing\"\nprint(f\"CLAHE images found: {len(os.listdir('/kaggle/working/clahe_images'))}\")\n\ndf = pd.read_csv('/kaggle/working/train_with_clahe_paths.csv')\nprint(f\"Total images: {len(df)}\")\n\nclass1_df = df[df['diagnosis'] == 1]\nclass1_trainval, class1_test = train_test_split(class1_df, test_size=0.40, random_state=42)\nclass1_train, class1_val = train_test_split(class1_trainval, test_size=0.20, random_state=42)\n\nother_df = df[df['diagnosis'] != 1]\nother_trainval, other_test = train_test_split(other_df, test_size=0.15, random_state=42, stratify=other_df['diagnosis'])\nother_train, other_val = train_test_split(other_trainval, test_size=0.176, random_state=42, stratify=other_trainval['diagnosis'])\n\ntrain_df = pd.concat([class1_train, other_train]).sample(frac=1, random_state=42).reset_index(drop=True)\nval_df   = pd.concat([class1_val, other_val]).sample(frac=1, random_state=42).reset_index(drop=True)\ntest_df  = pd.concat([class1_test, other_test]).sample(frac=1, random_state=42).reset_index(drop=True)\n\nprint(f\"Train: {len(train_df)}, Val: {len(val_df)}, Test: {len(test_df)}\")\nprint(f\"Mild NPDR in test set: {(test_df['diagnosis'] == 1).sum()}\")\n\nclass APTOSDatasetPreprocessed(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.transform = transform\n    def __len__(self):\n        return len(self.df)\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = Image.open(row['clahe_filepath']).convert('RGB')\n        if self.transform:\n            img = self.transform(img)\n        return img, int(row['diagnosis'])\n\ntrain_transform = transforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(20),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\ntrain_dataset = APTOSDatasetPreprocessed(train_df, transform=train_transform)\nval_dataset   = APTOSDatasetPreprocessed(val_df, transform=eval_transform)\ntest_dataset  = APTOSDatasetPreprocessed(test_df, transform=eval_transform)\n\nBATCH_SIZE = 32\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=2)\nval_loader   = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\ntest_loader  = DataLoader(test_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\n\nprint(f\"Train batches: {len(train_loader)}, Val batches: {len(val_loader)}, Test batches: {len(test_loader)}\")\n\nmodel = models.resnet50(weights=models.ResNet50_Weights.IMAGENET1K_V2)\nmodel.fc = nn.Linear(model.fc.in_features, 5)\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs via DataParallel\")\n    model = nn.DataParallel(model)\nmodel = model.to(device)\n\nclass_counts = train_df['diagnosis'].value_counts().sort_index().values\nclass_weights = 1.0 / class_counts\nclass_weights = class_weights / class_weights.sum() * len(class_counts)\nclass_weights = torch.tensor(class_weights, dtype=torch.float32).to(device)\nprint(f\"Class weights: {class_weights}\")\n\ncriterion = nn.CrossEntropyLoss(weight=class_weights)\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='max', factor=0.5, patience=2)\n\ndef evaluate(model, loader):\n    model.eval()\n    all_preds, all_labels = [], []\n    with torch.no_grad():\n        for imgs, labels in loader:\n            imgs = imgs.to(device)\n            outputs = model(imgs)\n            preds = torch.argmax(outputs, dim=1).cpu().numpy()\n            all_preds.extend(preds)\n            all_labels.extend(labels.numpy())\n    kappa = cohen_kappa_score(all_labels, all_preds, weights='quadratic')\n    acc = accuracy_score(all_labels, all_preds)\n    return kappa, acc, all_preds, all_labels\n\nNUM_EPOCHS = 25\nPATIENCE = 5\nbest_kappa = -1\nepochs_no_improve = 0\nCHECKPOINT_PATH = '/kaggle/working/best_resnet50_dr_clahe380.pth'\n\nfor epoch in range(NUM_EPOCHS):\n    model.train()\n    running_loss = 0.0\n    for imgs, labels in train_loader:\n        imgs, labels = imgs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(imgs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item() * imgs.size(0)\n\n    train_loss = running_loss / len(train_dataset)\n    val_kappa, val_acc, _, _ = evaluate(model, val_loader)\n    scheduler.step(val_kappa)\n    print(f\"Epoch {epoch+1}/{NUM_EPOCHS} | Train Loss: {train_loss:.4f} | Val Kappa: {val_kappa:.4f} | Val Acc: {val_acc:.4f}\")\n\n    if val_kappa > best_kappa:\n        best_kappa = val_kappa\n        epochs_no_improve = 0\n        save_state = model.module.state_dict() if isinstance(model, nn.DataParallel) else model.state_dict()\n        torch.save(save_state, CHECKPOINT_PATH)\n        print(f\"  -> New best model saved (kappa={val_kappa:.4f})\")\n    else:\n        epochs_no_improve += 1\n        if epochs_no_improve >= PATIENCE:\n            print(f\"Early stopping at epoch {epoch+1}\")\n            break\n\nprint(f\"\\nTraining complete. Best validation kappa: {best_kappa:.4f}\")\n\neval_model = models.resnet50(weights=None)\neval_model.fc = nn.Linear(eval_model.fc.in_features, 5)\neval_model.load_state_dict(torch.load(CHECKPOINT_PATH))\neval_model = eval_model.to(device)\n\ntest_kappa, test_acc, test_preds, test_labels = evaluate(eval_model, test_loader)\nprint(f\"\\n=== TEST SET RESULTS (CLAHE + 380px, dual T4) ===\")\nprint(f\"Quadratic Weighted Kappa: {test_kappa:.4f}\")\nprint(f\"Accuracy: {test_acc:.4f}\")\nprint(classification_report(test_labels, test_preds, target_names=['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative']))\nprint(f\"\\nCheckpoint saved at: {CHECKPOINT_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T20:03:39.643273Z","iopub.execute_input":"2026-08-09T20:03:39.643702Z","iopub.status.idle":"2026-08-09T20:16:01.410696Z","shell.execute_reply.started":"2026-08-09T20:03:39.643669Z","shell.execute_reply":"2026-08-09T20:16:01.409579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.path.exists('/kaggle/working/best_resnet50_dr_clahe380.pth'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T03:39:59.542453Z","iopub.execute_input":"2026-08-10T03:39:59.542954Z","iopub.status.idle":"2026-08-10T03:39:59.547741Z","shell.execute_reply.started":"2026-08-10T03:39:59.542922Z","shell.execute_reply":"2026-08-10T03:39:59.546974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install grad-cam captum -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T21:08:02.886198Z","iopub.execute_input":"2026-08-09T21:08:02.886596Z","iopub.status.idle":"2026-08-09T21:08:14.784285Z","shell.execute_reply.started":"2026-08-09T21:08:02.886566Z","shell.execute_reply":"2026-08-09T21:08:14.783156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nfrom torch.utils.data import DataLoader\nfrom torchvision import transforms, models\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# --- Load the improved model ---\nmodel = models.resnet50(weights=None)\nmodel.fc = nn.Linear(model.fc.in_features, 5)\nmodel.load_state_dict(torch.load('/kaggle/working/best_resnet50_dr_clahe380.pth'))\nmodel = model.to(device)\nmodel.eval()\nprint(\"Model loaded\")\n\n# --- Rebuild the same split using the CLAHE dataframe ---\ndf = pd.read_csv('/kaggle/working/train_with_clahe_paths.csv')\n\nclass1_df = df[df['diagnosis'] == 1]\nclass1_trainval, class1_test = train_test_split(class1_df, test_size=0.40, random_state=42)\nclass1_train, class1_val = train_test_split(class1_trainval, test_size=0.20, random_state=42)\n\nother_df = df[df['diagnosis'] != 1]\nother_trainval, other_test = train_test_split(other_df, test_size=0.15, random_state=42, stratify=other_df['diagnosis'])\nother_train, other_val = train_test_split(other_trainval, test_size=0.176, random_state=42, stratify=other_trainval['diagnosis'])\n\ntest_df = pd.concat([class1_test, other_test]).sample(frac=1, random_state=42).reset_index(drop=True)\nprint(f\"Test set size: {len(test_df)}\")\n\n# --- Get predictions on the full test set ---\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\ndef load_and_preprocess(filepath):\n    img = Image.open(filepath).convert('RGB')  # already 380x380, CLAHE'd\n    img_np = np.array(img) / 255.0\n    tensor = eval_transform(img).unsqueeze(0).to(device)\n    return img_np.astype(np.float32), tensor\n\nall_preds = []\nwith torch.no_grad():\n    for _, row in test_df.iterrows():\n        _, tensor = load_and_preprocess(row['clahe_filepath'])\n        output = model(tensor)\n        pred = torch.argmax(output, dim=1).item()\n        all_preds.append(pred)\n\ntest_df['pred'] = all_preds\n\ncorrect_mild = test_df[(test_df['diagnosis'] == 1) & (test_df['pred'] == 1)].reset_index(drop=True)\nprint(f\"Correctly-classified Mild NPDR images: {len(correct_mild)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T21:08:25.282622Z","iopub.execute_input":"2026-08-09T21:08:25.283451Z","iopub.status.idle":"2026-08-09T21:08:37.229914Z","shell.execute_reply.started":"2026-08-09T21:08:25.283361Z","shell.execute_reply":"2026-08-09T21:08:37.228784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_grad_cam import GradCAMPlusPlus\nfrom pytorch_grad_cam.utils.image import show_cam_on_image\nfrom captum.attr import IntegratedGradients\nimport matplotlib.pyplot as plt\n\n# ============================================================\n# Grad-CAM++ \n# ============================================================\ntarget_layers = [model.layer4[-1]]\ncam_pp = GradCAMPlusPlus(model=model, target_layers=target_layers)\n\nn_samples = 12\nsample_rows = correct_mild.sample(n=min(n_samples, len(correct_mild)), random_state=42).reset_index(drop=True)\n\nn_cols = 6\nn_rows = 2 * ((len(sample_rows) + n_cols - 1) // n_cols)\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(3*n_cols, 3*n_rows))\n\nfor idx, row in sample_rows.iterrows():\n    img_np, input_tensor = load_and_preprocess(row['clahe_filepath'])\n    grayscale_cam = cam_pp(input_tensor=input_tensor, targets=None)[0]\n    visualization = show_cam_on_image(img_np, grayscale_cam, use_rgb=True)\n\n    group = idx // n_cols\n    col = idx % n_cols\n    orig_row = group * 2\n    cam_row = group * 2 + 1\n\n    axes[orig_row, col].imshow(img_np)\n    axes[orig_row, col].set_title(row['id_code'], fontsize=7)\n    axes[orig_row, col].axis('off')\n    axes[cam_row, col].imshow(visualization)\n    axes[cam_row, col].axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/clahe380_gradcam_pp.png', dpi=150)\nplt.show()\nprint(\"Saved: clahe380_gradcam_pp.png\")\n\n# ============================================================\n# Integrated Gradients\n# ============================================================\nig = IntegratedGradients(model)\n\ndef get_ig_attribution(input_tensor, target_class):\n    input_tensor = input_tensor.clone().requires_grad_()\n    attributions = ig.attribute(input_tensor, target=target_class, n_steps=50)\n    attr_np = attributions.squeeze().cpu().detach().numpy()\n    attr_np = np.transpose(attr_np, (1, 2, 0))\n    attr_map = np.mean(np.abs(attr_np), axis=2)\n    attr_map = (attr_map - attr_map.min()) / (attr_map.max() - attr_map.min() + 1e-8)\n    return attr_map\n\nn_samples_ig = 6\nsample_rows_ig = correct_mild.sample(n=min(n_samples_ig, len(correct_mild)), random_state=7).reset_index(drop=True)\n\nfig, axes = plt.subplots(2, n_samples_ig, figsize=(4*n_samples_ig, 8))\nfor i, row in sample_rows_ig.iterrows():\n    img_np, input_tensor = load_and_preprocess(row['clahe_filepath'])\n    attr_map = get_ig_attribution(input_tensor, target_class=1)\n\n    axes[0, i].imshow(img_np)\n    axes[0, i].set_title(row['id_code'], fontsize=9)\n    axes[0, i].axis('off')\n\n    axes[1, i].imshow(img_np)\n    axes[1, i].imshow(attr_map, cmap='hot', alpha=0.6)\n    axes[1, i].axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/clahe380_ig.png', dpi=150)\nplt.show()\nprint(\"Saved: clahe380_ig.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T21:12:00.197991Z","iopub.execute_input":"2026-08-09T21:12:00.199056Z","iopub.status.idle":"2026-08-09T21:12:19.519122Z","shell.execute_reply.started":"2026-08-09T21:12:00.199011Z","shell.execute_reply":"2026-08-09T21:12:19.517762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\ncv2.setNumThreads(0)\n\n# ============================================================\n# 1. Brightness reduction function\n# ============================================================\ndef reduce_brightness(pil_img, factor):\n    img_np = np.array(pil_img).astype(np.float32)\n    img_dark = img_np * factor\n    img_dark = np.clip(img_dark, 0, 255).astype(np.uint8)\n    return Image.fromarray(img_dark)\n\n# ============================================================\n# 2. Severity levels — calibrated toward the real clinical\n# underexposure threshold (~30-70 mean brightness on 0-255 scale)\n# ============================================================\nSEVERITY_LEVELS = {\n    'clean':    1.00,\n    'mild':     0.75,\n    'moderate': 0.55,\n    'severe':   0.35,\n    'extreme':  0.20,\n}\n\n# ============================================================\n# 3. Verify where each level actually lands on the brightness scale\n# ============================================================\nsample_img = Image.open(correct_mild.iloc[0]['clahe_filepath']).convert('RGB')\nsample_np = np.array(sample_img)\nmask = sample_np.sum(axis=2) > 15  # exclude black background margins\n\nprint(\"Mean brightness (foreground only) at each severity level:\")\nfor name, factor in SEVERITY_LEVELS.items():\n    darkened = reduce_brightness(sample_img, factor)\n    darkened_np = np.array(darkened)\n    brightness = darkened_np[mask].mean()\n    print(f\"  {name} (factor={factor}): mean brightness = {brightness:.1f}\")\n\n# ============================================================\n# 4. Visual pilot check on 5 sample images\n# ============================================================\npilot_sample = correct_mild.sample(n=5, random_state=42).reset_index(drop=True)\n\nfig, axes = plt.subplots(len(pilot_sample), len(SEVERITY_LEVELS), figsize=(3*len(SEVERITY_LEVELS), 3*len(pilot_sample)))\n\nfor row_idx, row in pilot_sample.iterrows():\n    img = Image.open(row['clahe_filepath']).convert('RGB')\n    for col_idx, (name, factor) in enumerate(SEVERITY_LEVELS.items()):\n        darkened = reduce_brightness(img, factor)\n        axes[row_idx, col_idx].imshow(np.array(darkened))\n        if row_idx == 0:\n            axes[row_idx, col_idx].set_title(f\"{name}\\n(factor={factor})\", fontsize=9)\n        axes[row_idx, col_idx].axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/underexposure_pilot.png', dpi=150)\nplt.show()\nprint(\"\\nSaved: underexposure_pilot.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T21:49:24.476903Z","iopub.execute_input":"2026-08-09T21:49:24.477650Z","iopub.status.idle":"2026-08-09T21:49:29.558307Z","shell.execute_reply.started":"2026-08-09T21:49:24.477598Z","shell.execute_reply":"2026-08-09T21:49:29.557232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nfrom PIL import Image\nfrom torchvision import transforms\nimport pandas as pd\n\n# ============================================================\n# 1. Preprocessing for model input (matches training normalization)\n# ============================================================\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\ndef predict_on_image(pil_img):\n    tensor = eval_transform(pil_img).unsqueeze(0).to(device)\n    with torch.no_grad():\n        output = model(tensor)\n        pred = torch.argmax(output, dim=1).item()\n        probs = torch.softmax(output, dim=1).cpu().numpy()[0]\n    return pred, probs\n\n# ============================================================\n# 2. Run every Mild NPDR image through every severity level\n# ============================================================\nresults = []\n\nfor idx, row in correct_mild.iterrows():\n    img = Image.open(row['clahe_filepath']).convert('RGB')\n\n    for severity_name, factor in SEVERITY_LEVELS.items():\n        darkened = reduce_brightness(img, factor)\n        pred, probs = predict_on_image(darkened)\n\n        results.append({\n            'id_code': row['id_code'],\n            'true_label': row['diagnosis'],\n            'severity': severity_name,\n            'brightness_factor': factor,\n            'predicted_label': pred,\n            'confidence': probs[pred],\n            'prob_no_dr': probs[0],\n            'flipped_to_no_dr': (pred == 0),\n        })\n\nresults_df = pd.DataFrame(results)\nresults_df.to_csv('/kaggle/working/perturbation_results.csv', index=False)\n\nprint(f\"Total predictions recorded: {len(results_df)}\")\nprint(f\"({len(correct_mild)} images × {len(SEVERITY_LEVELS)} severity levels)\")\n\n# ============================================================\n# 3. Quick summary: misclassification-to-No-DR rate per severity level\n# ============================================================\nsummary = results_df.groupby('severity').agg(\n    n_images=('id_code', 'count'),\n    n_flipped_to_no_dr=('flipped_to_no_dr', 'sum'),\n    flip_rate=('flipped_to_no_dr', 'mean'),\n    mean_confidence=('confidence', 'mean'),\n).reindex(SEVERITY_LEVELS.keys())\n\nsummary['flip_rate_pct'] = (summary['flip_rate'] * 100).round(1)\nprint(\"\\n=== Misclassification-to-No-DR rate by severity ===\")\nprint(summary[['n_images', 'n_flipped_to_no_dr', 'flip_rate_pct', 'mean_confidence']])\n\nsummary.to_csv('/kaggle/working/perturbation_summary.csv')\nprint(\"\\nSaved: perturbation_results.csv (full per-image data)\")\nprint(\"Saved: perturbation_summary.csv (aggregated by severity)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T21:58:01.221578Z","iopub.execute_input":"2026-08-09T21:58:01.222101Z","iopub.status.idle":"2026-08-09T21:58:09.894350Z","shell.execute_reply.started":"2026-08-09T21:58:01.222068Z","shell.execute_reply":"2026-08-09T21:58:09.893500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nfrom torchvision import models\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nimport numpy as np\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# --- Reload the model from the committed checkpoint ---\nmodel = models.resnet50(weights=None)\nmodel.fc = nn.Linear(model.fc.in_features, 5)\nmodel.load_state_dict(torch.load('/kaggle/working/best_resnet50_dr_clahe380.pth'))\nmodel = model.to(device)\nmodel.eval()\nprint(\"Model reloaded\")\n\n# --- Reload the CLAHE dataframe and rebuild the split ---\ndf = pd.read_csv('/kaggle/working/train_with_clahe_paths.csv')\n\nclass1_df = df[df['diagnosis'] == 1]\nclass1_trainval, class1_test = train_test_split(class1_df, test_size=0.40, random_state=42)\nclass1_train, class1_val = train_test_split(class1_trainval, test_size=0.20, random_state=42)\n\nother_df = df[df['diagnosis'] != 1]\nother_trainval, other_test = train_test_split(other_df, test_size=0.15, random_state=42, stratify=other_df['diagnosis'])\nother_train, other_val = train_test_split(other_trainval, test_size=0.176, random_state=42, stratify=other_trainval['diagnosis'])\n\ntest_df = pd.concat([class1_test, other_test]).sample(frac=1, random_state=42).reset_index(drop=True)\n\n# --- Rebuild predictions and the correct_mild filter ---\nfrom torchvision import transforms\n\neval_transform_check = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\nall_preds = []\nwith torch.no_grad():\n    for _, row in test_df.iterrows():\n        img = Image.open(row['clahe_filepath']).convert('RGB')\n        tensor = eval_transform_check(img).unsqueeze(0).to(device)\n        output = model(tensor)\n        pred = torch.argmax(output, dim=1).item()\n        all_preds.append(pred)\n\ntest_df['pred'] = all_preds\ncorrect_mild = test_df[(test_df['diagnosis'] == 1) & (test_df['pred'] == 1)].reset_index(drop=True)\nprint(f\"Correctly-classified Mild NPDR images: {len(correct_mild)}\")\n\n# --- Redefine reduce_brightness and SEVERITY_LEVELS ---\ndef reduce_brightness(pil_img, factor):\n    img_np = np.array(pil_img).astype(np.float32)\n    img_dark = img_np * factor\n    img_dark = np.clip(img_dark, 0, 255).astype(np.uint8)\n    return Image.fromarray(img_dark)\n\nSEVERITY_LEVELS = {\n    'clean':    1.00,\n    'mild':     0.75,\n    'moderate': 0.55,\n    'severe':   0.35,\n    'extreme':  0.20,\n}\n\nprint(\"All dependencies rebuilt — ready for Phase 6.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:46:48.895530Z","iopub.execute_input":"2026-08-09T22:46:48.896058Z","iopub.status.idle":"2026-08-09T22:47:04.452150Z","shell.execute_reply.started":"2026-08-09T22:46:48.896023Z","shell.execute_reply":"2026-08-09T22:47:04.451233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport gc\nfrom PIL import Image\nfrom captum.attr import IntegratedGradients\nfrom torchvision import transforms\n\ncv2.setNumThreads(0)\n\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\nig = IntegratedGradients(model)\n\ndef detect_optic_disc(pil_img, img_size=380):\n    img_np = np.array(pil_img)\n    gray = cv2.cvtColor(img_np, cv2.COLOR_RGB2GRAY)\n    gray_blur = cv2.GaussianBlur(gray, (25, 25), 0)\n    _, max_val, _, max_loc = cv2.minMaxLoc(gray_blur)\n    disc_radius = int(0.12 * img_size)\n    return max_loc, disc_radius\n\ndef make_disc_mask(disc_center, disc_radius, img_size=380):\n    yy, xx = np.mgrid[0:img_size, 0:img_size]\n    dist = np.sqrt((xx - disc_center[0])**2 + (yy - disc_center[1])**2)\n    return dist <= disc_radius\n\ndef get_ig_attribution(input_tensor, target_class, n_steps=25):\n    input_tensor = input_tensor.clone().requires_grad_()\n    attributions = ig.attribute(input_tensor, target=target_class, n_steps=n_steps, internal_batch_size=5)\n    attr_np = attributions.squeeze().cpu().detach().numpy()\n    attr_np = np.transpose(attr_np, (1, 2, 0))\n    attr_map = np.mean(np.abs(attr_np), axis=2)\n    del attributions, input_tensor\n    return attr_map\n\ndef disc_concentration(attr_map, disc_mask):\n    total = attr_map.sum()\n    if total <= 0:\n        return 0.0\n    return attr_map[disc_mask].sum() / total\n\nOUTPUT_PATH = '/kaggle/working/attention_shift_results.csv'\n\nalready_done = set()\ntry:\n    existing = pd.read_csv(OUTPUT_PATH)\n    already_done = set(existing['id_code'].unique())\n    print(f\"Resuming — {len(already_done)} images already processed, skipping them.\")\nexcept FileNotFoundError:\n    print(\"Starting fresh — no existing results found.\")\n\nfor idx, row in correct_mild.iterrows():\n    if row['id_code'] in already_done:\n        continue\n\n    clean_img = Image.open(row['clahe_filepath']).convert('RGB')\n    disc_center, disc_radius = detect_optic_disc(clean_img)\n    disc_mask = make_disc_mask(disc_center, disc_radius)\n\n    row_results = []\n    for severity_name, factor in SEVERITY_LEVELS.items():\n        darkened = reduce_brightness(clean_img, factor)\n        tensor = eval_transform(darkened).unsqueeze(0).to(device)\n\n        attr_map = get_ig_attribution(tensor, target_class=1, n_steps=25)\n        concentration = disc_concentration(attr_map, disc_mask)\n\n        row_results.append({\n            'id_code': row['id_code'],\n            'severity': severity_name,\n            'brightness_factor': factor,\n            'disc_concentration': concentration,\n        })\n\n        del tensor, attr_map\n\n    file_exists = pd.io.common.file_exists(OUTPUT_PATH)\n    pd.DataFrame(row_results).to_csv(OUTPUT_PATH, mode='a', header=not file_exists, index=False)\n\n    torch.cuda.empty_cache()\n    gc.collect()\n\n    if (idx + 1) % 10 == 0:\n        print(f\"Processed {idx + 1}/{len(correct_mild)} images... (GPU mem: {torch.cuda.memory_allocated()/1e9:.2f} GB)\")\n\nprint(\"\\nAll images processed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:47:04.453706Z","iopub.execute_input":"2026-08-09T22:47:04.453994Z","iopub.status.idle":"2026-08-09T22:52:33.919583Z","shell.execute_reply.started":"2026-08-09T22:47:04.453970Z","shell.execute_reply":"2026-08-09T22:52:33.918694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"attn_df = pd.read_csv('/kaggle/working/attention_shift_results.csv')\n\nattn_summary = attn_df.groupby('severity')['disc_concentration'].agg(['mean', 'std', 'count']).reindex(SEVERITY_LEVELS.keys())\nprint(\"=== Mean optic-disc attribution concentration by severity ===\")\nprint(attn_summary)\n\nattn_summary.to_csv('/kaggle/working/attention_shift_summary.csv')\nprint(\"\\nSaved: attention_shift_summary.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:53:09.060116Z","iopub.execute_input":"2026-08-09T22:53:09.060993Z","iopub.status.idle":"2026-08-09T22:53:09.080816Z","shell.execute_reply.started":"2026-08-09T22:53:09.060957Z","shell.execute_reply":"2026-08-09T22:53:09.079898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import stats\n\nclean_vals = attn_df[attn_df['severity'] == 'clean'].set_index('id_code')['disc_concentration']\n\nprint(\"Paired comparison: clean vs. each severity level (same 113 images)\")\nfor severity in ['mild', 'moderate', 'severe', 'extreme']:\n    sev_vals = attn_df[attn_df['severity'] == severity].set_index('id_code')['disc_concentration']\n    # align by id_code to ensure proper pairing\n    paired = pd.concat([clean_vals, sev_vals], axis=1, keys=['clean', severity]).dropna()\n    stat, pval = stats.wilcoxon(paired['clean'], paired[severity])\n    diff = paired[severity].mean() - paired['clean'].mean()\n    print(f\"  {severity}: mean diff = {diff:+.4f}, Wilcoxon p = {pval:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:54:46.918528Z","iopub.execute_input":"2026-08-09T22:54:46.919606Z","iopub.status.idle":"2026-08-09T22:54:46.946453Z","shell.execute_reply.started":"2026-08-09T22:54:46.919559Z","shell.execute_reply":"2026-08-09T22:54:46.945465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge attribution results with misclassification results\nmerged = attn_df.merge(\n    results_df[['id_code', 'severity', 'flipped_to_no_dr']],\n    on=['id_code', 'severity']\n)\n\n# Compare disc concentration: flipped vs non-flipped, at each severity level (excluding 'clean', where nothing flips)\nfrom scipy import stats\n\nfor severity in ['mild', 'moderate', 'severe', 'extreme']:\n    subset = merged[merged['severity'] == severity]\n    flipped = subset[subset['flipped_to_no_dr'] == True]['disc_concentration']\n    not_flipped = subset[subset['flipped_to_no_dr'] == False]['disc_concentration']\n    if len(flipped) >= 3:  # need enough samples for a meaningful test\n        stat, pval = stats.mannwhitneyu(flipped, not_flipped, alternative='two-sided')\n        print(f\"{severity}: flipped n={len(flipped)} (mean={flipped.mean():.4f}), not-flipped n={len(not_flipped)} (mean={not_flipped.mean():.4f}), p={pval:.4f}\")\n    else:\n        print(f\"{severity}: too few flipped cases (n={len(flipped)}) for a reliable test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:55:55.678855Z","iopub.execute_input":"2026-08-09T22:55:55.679380Z","iopub.status.idle":"2026-08-09T22:55:55.690005Z","shell.execute_reply.started":"2026-08-09T22:55:55.679345Z","shell.execute_reply":"2026-08-09T22:55:55.688892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results_df = pd.read_csv('/kaggle/working/perturbation_results.csv')\nprint(f\"Loaded {len(results_df)} rows\")\nprint(results_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:56:20.335900Z","iopub.execute_input":"2026-08-09T22:56:20.336686Z","iopub.status.idle":"2026-08-09T22:56:20.349991Z","shell.execute_reply.started":"2026-08-09T22:56:20.336650Z","shell.execute_reply":"2026-08-09T22:56:20.348965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged = attn_df.merge(\n    results_df[['id_code', 'severity', 'flipped_to_no_dr']],\n    on=['id_code', 'severity']\n)\n\nfrom scipy import stats\n\nfor severity in ['mild', 'moderate', 'severe', 'extreme']:\n    subset = merged[merged['severity'] == severity]\n    flipped = subset[subset['flipped_to_no_dr'] == True]['disc_concentration']\n    not_flipped = subset[subset['flipped_to_no_dr'] == False]['disc_concentration']\n    if len(flipped) >= 3:\n        stat, pval = stats.mannwhitneyu(flipped, not_flipped, alternative='two-sided')\n        print(f\"{severity}: flipped n={len(flipped)} (mean={flipped.mean():.4f}), not-flipped n={len(not_flipped)} (mean={not_flipped.mean():.4f}), p={pval:.4f}\")\n    else:\n        print(f\"{severity}: too few flipped cases (n={len(flipped)}) for a reliable test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T22:56:39.029061Z","iopub.execute_input":"2026-08-09T22:56:39.029800Z","iopub.status.idle":"2026-08-09T22:56:39.055972Z","shell.execute_reply.started":"2026-08-09T22:56:39.029763Z","shell.execute_reply":"2026-08-09T22:56:39.054912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Check every expected file exists\nexpected_files = [\n    '/kaggle/working/best_resnet50_dr_clahe380.pth',\n    '/kaggle/working/train_with_clahe_paths.csv',\n    '/kaggle/working/perturbation_results.csv',\n    '/kaggle/working/perturbation_summary.csv',\n    '/kaggle/working/attention_shift_results.csv',\n    '/kaggle/working/attention_shift_summary.csv',\n    '/kaggle/working/underexposure_pilot.png',\n    '/kaggle/working/clahe380_gradcam_pp.png',\n    '/kaggle/working/clahe380_ig.png',\n]\n\nprint(\"=== File existence check ===\")\nfor f in expected_files:\n    exists = os.path.exists(f)\n    size = os.path.getsize(f) / 1024 if exists else 0\n    status = \"✓ FOUND\" if exists else \"✗ MISSING\"\n    print(f\"{status} | {f} ({size:.1f} KB)\")\n\n# Specifically verify the two most important CSVs actually load with correct content\nprint(\"\\n=== Content verification ===\")\ntry:\n    attn = pd.read_csv('/kaggle/working/attention_shift_results.csv')\n    print(f\"attention_shift_results.csv: {len(attn)} rows, columns: {list(attn.columns)}\")\nexcept Exception as e:\n    print(f\"attention_shift_results.csv FAILED TO LOAD: {e}\")\n\ntry:\n    pert = pd.read_csv('/kaggle/working/perturbation_results.csv')\n    print(f\"perturbation_results.csv: {len(pert)} rows, columns: {list(pert.columns)}\")\nexcept Exception as e:\n    print(f\"perturbation_results.csv FAILED TO LOAD: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T02:47:49.659258Z","iopub.execute_input":"2026-08-10T02:47:49.659510Z","iopub.status.idle":"2026-08-10T02:47:50.878210Z","shell.execute_reply.started":"2026-08-10T02:47:49.659479Z","shell.execute_reply":"2026-08-10T02:47:50.877332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# Check every expected file exists\nexpected_files = [\n    '/kaggle/working/best_resnet50_dr_clahe380.pth',\n    '/kaggle/working/train_with_clahe_paths.csv',\n    '/kaggle/working/perturbation_results.csv',\n    '/kaggle/working/perturbation_summary.csv',\n    '/kaggle/working/attention_shift_results.csv',\n    '/kaggle/working/attention_shift_summary.csv',\n    '/kaggle/working/underexposure_pilot.png',\n    '/kaggle/working/clahe380_gradcam_pp.png',\n    '/kaggle/working/clahe380_ig.png',\n]\n\nprint(\"=== File existence check ===\")\nfor f in expected_files:\n    exists = os.path.exists(f)\n    size = os.path.getsize(f) / 1024 if exists else 0\n    status = \"✓ FOUND\" if exists else \"✗ MISSING\"\n    print(f\"{status} | {f} ({size:.1f} KB)\")\n\n# Specifically verify the two most important CSVs actually load with correct content\nprint(\"\\n=== Content verification ===\")\ntry:\n    attn = pd.read_csv('/kaggle/working/attention_shift_results.csv')\n    print(f\"attention_shift_results.csv: {len(attn)} rows, columns: {list(attn.columns)}\")\nexcept Exception as e:\n    print(f\"attention_shift_results.csv FAILED TO LOAD: {e}\")\n\ntry:\n    pert = pd.read_csv('/kaggle/working/perturbation_results.csv')\n    print(f\"perturbation_results.csv: {len(pert)} rows, columns: {list(pert.columns)}\")\nexcept Exception as e:\n    print(f\"perturbation_results.csv FAILED TO LOAD: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T02:49:16.122702Z","iopub.execute_input":"2026-08-10T02:49:16.123314Z","iopub.status.idle":"2026-08-10T02:49:16.131758Z","shell.execute_reply.started":"2026-08-10T02:49:16.123281Z","shell.execute_reply":"2026-08-10T02:49:16.130862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input/notebooks'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T02:49:55.092204Z","iopub.execute_input":"2026-08-10T02:49:55.093006Z","iopub.status.idle":"2026-08-10T02:49:57.582344Z","shell.execute_reply.started":"2026-08-10T02:49:55.092967Z","shell.execute_reply":"2026-08-10T02:49:57.581573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ntarget_files = [\n    'best_resnet50_dr_clahe380.pth',\n    'train_with_clahe_paths.csv',\n    'perturbation_results.csv',\n    'perturbation_summary.csv',\n    'attention_shift_results.csv',\n    'attention_shift_summary.csv',\n    'underexposure_pilot.png',\n    'clahe380_gradcam_pp.png',\n    'clahe380_ig.png',\n]\n\nfound = {}\nfor dirname, _, filenames in os.walk('/kaggle/input/notebooks'):\n    for filename in filenames:\n        if filename in target_files:\n            found[filename] = os.path.join(dirname, filename)\n\nprint(\"=== Checking for key files ===\")\nfor f in target_files:\n    if f in found:\n        size = os.path.getsize(found[f]) / 1024\n        print(f\"✓ FOUND | {f} ({size:.1f} KB) at {found[f]}\")\n    else:\n        print(f\"✗ MISSING | {f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T02:51:11.249699Z","iopub.execute_input":"2026-08-10T02:51:11.250037Z","iopub.status.idle":"2026-08-10T02:51:12.748446Z","shell.execute_reply.started":"2026-08-10T02:51:11.250011Z","shell.execute_reply":"2026-08-10T02:51:12.747572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install captum grad-cam -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T00:57:30.463901Z","iopub.execute_input":"2026-08-11T00:57:30.464189Z","iopub.status.idle":"2026-08-11T00:57:41.053413Z","shell.execute_reply.started":"2026-08-11T00:57:30.464166Z","shell.execute_reply":"2026-08-11T00:57:41.052375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom torchvision import models, transforms\nfrom captum.attr import IntegratedGradients\n\ncv2.setNumThreads(0)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nBASE = '/kaggle/input/notebooks/priyanshprasad1/dr-research-new'\n\n# ============================================================\n# 1. Reload model\n# ============================================================\nmodel = models.resnet50(weights=None)\nmodel.fc = nn.Linear(model.fc.in_features, 5)\nmodel.load_state_dict(torch.load(f'{BASE}/best_resnet50_dr_clahe380.pth'))\nmodel = model.to(device)\nmodel.eval()\nprint(\"Model loaded\")\n\n# ============================================================\n# 2. Reload results, fix image paths to point at mounted input\n# ============================================================\nresults_df = pd.read_csv(f'{BASE}/perturbation_results.csv')\n\ndf = pd.read_csv(f'{BASE}/train_with_clahe_paths.csv')\ndf['clahe_filepath'] = df['id_code'].apply(lambda x: f'{BASE}/clahe_images/{x}.png')\n\n# ============================================================\n# 3. Pick matched examples at EXTREME severity: flipped vs not-flipped\n# ============================================================\nextreme_results = results_df[results_df['severity'] == 'extreme']\nflipped_ids = extreme_results[extreme_results['flipped_to_no_dr'] == True]['id_code'].tolist()\nnot_flipped_ids = extreme_results[extreme_results['flipped_to_no_dr'] == False]['id_code'].tolist()\n\nn_examples = 3\nselected_flipped = flipped_ids[:n_examples]\nselected_not_flipped = not_flipped_ids[:n_examples]\n\nprint(f\"Flipped examples: {selected_flipped}\")\nprint(f\"Not-flipped examples: {selected_not_flipped}\")\n\n# ============================================================\n# 4. Reduce brightness, disc detection, IG attribution (same as Phase 6)\n# ============================================================\nSEVERITY_LEVELS = {'clean': 1.00, 'mild': 0.75, 'moderate': 0.55, 'severe': 0.35, 'extreme': 0.20}\n\ndef reduce_brightness(pil_img, factor):\n    img_np = np.array(pil_img).astype(np.float32)\n    img_dark = np.clip(img_np * factor, 0, 255).astype(np.uint8)\n    return Image.fromarray(img_dark)\n\ndef detect_optic_disc(pil_img, img_size=380):\n    img_np = np.array(pil_img)\n    gray = cv2.cvtColor(img_np, cv2.COLOR_RGB2GRAY)\n    gray_blur = cv2.GaussianBlur(gray, (25, 25), 0)\n    _, _, _, max_loc = cv2.minMaxLoc(gray_blur)\n    return max_loc, int(0.12 * img_size)\n\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\nig = IntegratedGradients(model)\n\ndef get_ig_heatmap(pil_img, target_class=1, n_steps=25):\n    tensor = eval_transform(pil_img).unsqueeze(0).to(device).requires_grad_()\n    attributions = ig.attribute(tensor, target=target_class, n_steps=n_steps, internal_batch_size=5)\n    attr_np = attributions.squeeze().cpu().detach().numpy()\n    attr_np = np.transpose(attr_np, (1, 2, 0))\n    attr_map = np.mean(np.abs(attr_np), axis=2)\n    attr_map = (attr_map - attr_map.min()) / (attr_map.max() - attr_map.min() + 1e-8)\n    del attributions, tensor\n    return attr_map\n\n# ============================================================\n# 5. Build the comparison figure\n# ============================================================\nall_ids = selected_flipped + selected_not_flipped\nlabels = ['FLIPPED to No DR'] * len(selected_flipped) + ['Correctly classified'] * len(selected_not_flipped)\n\nfig, axes = plt.subplots(2, len(all_ids), figsize=(4*len(all_ids), 8))\n\nfor i, (id_code, label) in enumerate(zip(all_ids, labels)):\n    row = df[df['id_code'] == id_code].iloc[0]\n    clean_img = Image.open(row['clahe_filepath']).convert('RGB')\n    darkened = reduce_brightness(clean_img, SEVERITY_LEVELS['extreme'])\n\n    disc_center, disc_radius = detect_optic_disc(clean_img)\n    attr_map = get_ig_heatmap(darkened, target_class=1)\n\n    img_np = np.array(darkened) / 255.0\n\n    axes[0, i].imshow(img_np)\n    circle = plt.Circle(disc_center, disc_radius, color='cyan', fill=False, linewidth=2)\n    axes[0, i].add_patch(circle)\n    axes[0, i].set_title(f\"{label}\\n{id_code}\", fontsize=10,\n                          color='darkred' if label == 'FLIPPED to No DR' else 'darkgreen')\n    axes[0, i].axis('off')\n\n    axes[1, i].imshow(img_np)\n    axes[1, i].imshow(attr_map, cmap='hot', alpha=0.6)\n    circle2 = plt.Circle(disc_center, disc_radius, color='cyan', fill=False, linewidth=2)\n    axes[1, i].add_patch(circle2)\n    axes[1, i].axis('off')\n\n    torch.cuda.empty_cache()\n\nplt.suptitle('Extreme underexposure: Flipped-to-\"No DR\" vs. Correctly-classified\\n(cyan circle = detected optic disc region)', fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/qualitative_comparison_extreme.png', dpi=150)\nplt.show()\nprint(\"\\nSaved: qualitative_comparison_extreme.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T02:54:40.844862Z","iopub.execute_input":"2026-08-10T02:54:40.845444Z","iopub.status.idle":"2026-08-10T02:54:51.378025Z","shell.execute_reply.started":"2026-08-10T02:54:40.845406Z","shell.execute_reply":"2026-08-10T02:54:51.377002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nexpected_files = [\n    '/kaggle/working/best_resnet50_dr_clahe380.pth',\n    '/kaggle/working/train_with_clahe_paths.csv',\n    '/kaggle/working/perturbation_results.csv',\n    '/kaggle/working/perturbation_summary.csv',\n    '/kaggle/working/attention_shift_results.csv',\n    '/kaggle/working/attention_shift_summary.csv',\n    '/kaggle/working/underexposure_pilot.png',\n    '/kaggle/working/clahe380_gradcam_pp.png',\n    '/kaggle/working/clahe380_ig.png',\n    '/kaggle/working/qualitative_comparison_extreme.png',\n]\n\nprint(\"=== File existence check (current session) ===\")\nfor f in expected_files:\n    exists = os.path.exists(f)\n    size = os.path.getsize(f) / 1024 if exists else 0\n    status = \"✓ FOUND\" if exists else \"✗ MISSING\"\n    print(f\"{status} | {f} ({size:.1f} KB)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-10T04:27:50.884678Z","iopub.execute_input":"2026-08-10T04:27:50.885385Z","iopub.status.idle":"2026-08-10T04:27:51.180635Z","shell.execute_reply.started":"2026-08-10T04:27:50.885355Z","shell.execute_reply":"2026-08-10T04:27:51.179679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ntarget_files = [\n    'best_resnet50_dr_clahe380.pth',\n    'train_with_clahe_paths.csv',\n    'perturbation_results.csv',\n    'perturbation_summary.csv',\n    'attention_shift_results.csv',\n    'attention_shift_summary.csv',\n    'underexposure_pilot.png',\n    'clahe380_gradcam_pp.png',\n    'clahe380_ig.png',\n    'qualitative_comparison_extreme.png',\n]\n\nfound = {}\nfor dirname, _, filenames in os.walk('/kaggle/input/notebooks'):\n    for filename in filenames:\n        if filename in target_files:\n            found[filename] = os.path.join(dirname, filename)\n\nprint(\"=== Checking for key files ===\")\nfor f in target_files:\n    if f in found:\n        size = os.path.getsize(found[f]) / 1024\n        print(f\"✓ FOUND | {f} ({size:.1f} KB) at {found[f]}\")\n    else:\n        print(f\"✗ MISSING | {f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T00:56:34.372805Z","iopub.execute_input":"2026-08-11T00:56:34.373513Z","iopub.status.idle":"2026-08-11T00:56:39.894584Z","shell.execute_reply.started":"2026-08-11T00:56:34.373482Z","shell.execute_reply":"2026-08-11T00:56:39.893878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torchvision import models, transforms\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\n\ncv2.setNumThreads(0)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# --- Attach your committed notebook as input first if you haven't ---\n# (Add Input panel -> search \"dr-research-new\" -> Add)\nBASE = '/kaggle/input/notebooks/priyanshprasad1/dr-research-new'\n\n# --- Reload model ---\nmodel = models.resnet50(weights=None)\nmodel.fc = nn.Linear(model.fc.in_features, 5)\nmodel.load_state_dict(torch.load(f'{BASE}/best_resnet50_dr_clahe380.pth'))\nmodel = model.to(device)\nmodel.eval()\nprint(\"Model loaded\")\n\n# --- Reload data and rebuild correct_mild ---\ndf = pd.read_csv(f'{BASE}/train_with_clahe_paths.csv')\ndf['clahe_filepath'] = df['id_code'].apply(lambda x: f'{BASE}/clahe_images/{x}.png')\n\nclass1_df = df[df['diagnosis'] == 1]\nclass1_trainval, class1_test = train_test_split(class1_df, test_size=0.40, random_state=42)\nclass1_train, class1_val = train_test_split(class1_trainval, test_size=0.20, random_state=42)\n\nother_df = df[df['diagnosis'] != 1]\nother_trainval, other_test = train_test_split(other_df, test_size=0.15, random_state=42, stratify=other_df['diagnosis'])\nother_train, other_val = train_test_split(other_trainval, test_size=0.176, random_state=42, stratify=other_trainval['diagnosis'])\n\ntest_df = pd.concat([class1_test, other_test]).sample(frac=1, random_state=42).reset_index(drop=True)\n\neval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\nall_preds = []\nwith torch.no_grad():\n    for _, row in test_df.iterrows():\n        img = Image.open(row['clahe_filepath']).convert('RGB')\n        tensor = eval_transform(img).unsqueeze(0).to(device)\n        output = model(tensor)\n        pred = torch.argmax(output, dim=1).item()\n        all_preds.append(pred)\n\ntest_df['pred'] = all_preds\ncorrect_mild = test_df[(test_df['diagnosis'] == 1) & (test_df['pred'] == 1)].reset_index(drop=True)\nprint(f\"Correctly-classified Mild NPDR images: {len(correct_mild)}\")\n\n# --- Redefine helper functions ---\ndef reduce_brightness(pil_img, factor):\n    img_np = np.array(pil_img).astype(np.float32)\n    img_dark = np.clip(img_np * factor, 0, 255).astype(np.uint8)\n    return Image.fromarray(img_dark)\n\ndef detect_optic_disc(pil_img, img_size=380):\n    img_np = np.array(pil_img)\n    gray = cv2.cvtColor(img_np, cv2.COLOR_RGB2GRAY)\n    gray_blur = cv2.GaussianBlur(gray, (25, 25), 0)\n    _, _, _, max_loc = cv2.minMaxLoc(gray_blur)\n    return max_loc, int(0.12 * img_size)\n\nSEVERITY_LEVELS = {'clean': 1.00, 'mild': 0.75, 'moderate': 0.55, 'severe': 0.35, 'extreme': 0.20}\n\n# --- Reload the two CSVs items 1-4 depend on ---\nresults_df = pd.read_csv(f'{BASE}/perturbation_results.csv')\nattn_df = pd.read_csv(f'{BASE}/attention_shift_results.csv')\n\nprint(\"\\nAll dependencies rebuilt — ready for items 1-4.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T01:00:46.181530Z","iopub.execute_input":"2026-08-11T01:00:46.181865Z","iopub.status.idle":"2026-08-11T01:00:57.304595Z","shell.execute_reply.started":"2026-08-11T01:00:46.181838Z","shell.execute_reply":"2026-08-11T01:00:57.303832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom captum.attr import IntegratedGradients\n\n# ============================================================\n# Baseline documentation: Captum's IntegratedGradients defaults\n# to an all-zeros (black) image baseline unless a different\n# baseline is explicitly passed. We used the default.\n# ============================================================\nprint(\"IG baseline used: all-zeros (black) image (Captum default, not explicitly overridden)\")\n\n# ============================================================\n# Completeness axiom check: sum of attributions should equal\n# (model output for real image) - (model output for baseline image)\n# Check this at both n_steps=25 (what we used) and n_steps=50 (original baseline check)\n# ============================================================\nig = IntegratedGradients(model)\n\ndef completeness_check(input_tensor, target_class, n_steps):\n    baseline = torch.zeros_like(input_tensor)\n    attributions, delta = ig.attribute(\n        input_tensor, baselines=baseline, target=target_class,\n        n_steps=n_steps, internal_batch_size=5, return_convergence_delta=True\n    )\n    attr_sum = attributions.sum().item()\n    return attr_sum, delta.item()\n\n# Run on a small sample (5 images) at both step counts to compare\nsample_ids = correct_mild['id_code'].tolist()[:5]\nprint(f\"\\n{'Image':<15} {'n_steps':<10} {'Attr Sum':<12} {'Convergence Delta':<20}\")\nprint(\"-\" * 60)\n\nfor id_code in sample_ids:\n    row = correct_mild[correct_mild['id_code'] == id_code].iloc[0]\n    img = Image.open(row['clahe_filepath']).convert('RGB')\n    tensor = eval_transform(img).unsqueeze(0).to(device).requires_grad_()\n\n    for n_steps in [25, 50]:\n        attr_sum, delta = completeness_check(tensor, target_class=1, n_steps=n_steps)\n        print(f\"{id_code:<15} {n_steps:<10} {attr_sum:<12.4f} {delta:<20.6f}\")\n\n    del tensor\n    torch.cuda.empty_cache()\n\nprint(\"\\nInterpretation: 'Convergence Delta' close to 0 indicates the attribution\")\nprint(\"satisfies the completeness axiom well. Compare delta magnitude between\")\nprint(\"n_steps=25 and n_steps=50 -- similar magnitudes support using 25 steps.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T00:59:01.017338Z","iopub.execute_input":"2026-08-11T00:59:01.017808Z","iopub.status.idle":"2026-08-11T00:59:08.424716Z","shell.execute_reply.started":"2026-08-11T00:59:01.017777Z","shell.execute_reply":"2026-08-11T00:59:08.423958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport random\n\n# ============================================================\n# Visual validation sample: 25 images with detected disc circled,\n# for manual eyeball verification\n# ============================================================\nrandom.seed(42)\nvalidation_sample = correct_mild.sample(n=25, random_state=42).reset_index(drop=True)\n\nfig, axes = plt.subplots(5, 5, figsize=(15, 15))\naxes = axes.flatten()\n\nfor i, row in validation_sample.iterrows():\n    img = Image.open(row['clahe_filepath']).convert('RGB')\n    disc_center, disc_radius = detect_optic_disc(img)\n\n    axes[i].imshow(np.array(img))\n    circle = plt.Circle(disc_center, disc_radius, color='cyan', fill=False, linewidth=2)\n    axes[i].add_patch(circle)\n    axes[i].set_title(row['id_code'], fontsize=8)\n    axes[i].axis('off')\n\nplt.suptitle('Disc localization validation sample (n=25) — visually check circle placement', fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/disc_localization_validation.png', dpi=150)\nplt.show()\nprint(\"Saved: disc_localization_validation.png\")\nprint(\"\\nManually count: how many of the 25 circles correctly land on the optic disc?\")\nprint(\"Report as: 'Manual review of 25 randomly sampled images confirmed correct\")\nprint(\"disc localization in X/25 (X%) cases.'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T00:59:47.305150Z","iopub.execute_input":"2026-08-11T00:59:47.305673Z","iopub.status.idle":"2026-08-11T00:59:53.657899Z","shell.execute_reply.started":"2026-08-11T00:59:47.305643Z","shell.execute_reply":"2026-08-11T00:59:53.656971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# ============================================================\n# Merge confidence + disc concentration data\n# ============================================================\nmerged = results_df.merge(\n    attn_df[['id_code', 'severity', 'disc_concentration']],\n    on=['id_code', 'severity']\n)\n\n# Focus on severe + extreme, where real failures exist\ntest_severities = merged[merged['severity'].isin(['severe', 'extreme'])]\n\nCONFIDENCE_THRESHOLD = 0.85\n\nresults_summary = []\n\nfor threshold_T in [0.07, 0.075, 0.08, 0.085, 0.09]:\n    # Standard confidence-only flagging: would flag anything with confidence < 0.85\n    # (but we know confidence stays high even when wrong, so this catches ~nothing)\n    confidence_only_catches = test_severities[\n        (test_severities['flipped_to_no_dr'] == True) &\n        (test_severities['confidence'] < CONFIDENCE_THRESHOLD)\n    ]\n\n    # Our proposed rule: confidence HIGH (>0.85) BUT disc concentration ALSO high (>T)\n    # -> flag for human review as suspected low-light artifact\n    total_failures = test_severities[test_severities['flipped_to_no_dr'] == True]\n    quality_gate_catches = total_failures[total_failures['disc_concentration'] > threshold_T]\n\n    results_summary.append({\n        'threshold_T': threshold_T,\n        'total_actual_failures': len(total_failures),\n        'caught_by_confidence_only': len(confidence_only_catches),\n        'caught_by_quality_gate': len(quality_gate_catches),\n        'quality_gate_catch_rate': len(quality_gate_catches) / len(total_failures) if len(total_failures) > 0 else 0,\n    })\n\nsummary_df = pd.DataFrame(results_summary)\nprint(\"=== Quality Gate Performance (severe + extreme combined) ===\")\nprint(summary_df.to_string(index=False))\n\nsummary_df.to_csv('/kaggle/working/quality_gate_analysis.csv', index=False)\nprint(\"\\nSaved: quality_gate_analysis.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-11T01:00:18.326194Z","iopub.execute_input":"2026-08-11T01:00:18.326669Z","iopub.status.idle":"2026-08-11T01:00:18.352825Z","shell.execute_reply.started":"2026-08-11T01:00:18.326592Z","shell.execute_reply":"2026-08-11T01:00:18.351879Z"}},"outputs":[],"execution_count":null}]}