{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"databundleVersionId":875431,"isSourceIdPinned":false,"mountSlug":"competitions/aptos2019-blindness-detection","sourceId":14774,"sourceType":"competition"}],"dockerImageVersionId":31401,"isGpuEnabled":true,"isInternetEnabled":true,"language":"python","sourceType":"notebook"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"9a17260d-c5cb-441f-8f81-a803dde08f3a","cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n\nimport numpy as np\nimport pandas as pd\nimport os\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:05.776470Z","iopub.execute_input":"2026-06-06T03:35:05.777467Z","iopub.status.idle":"2026-06-06T03:35:09.978680Z","shell.execute_reply.started":"2026-06-06T03:35:05.777427Z","shell.execute_reply":"2026-06-06T03:35:09.977818Z"}},"outputs":[],"execution_count":null},{"id":"f975553a-6f8f-4d6f-bdea-ce6061a58d42","cell_type":"code","source":"# ================================================================\n# Cell 1: Imports\n# CHANGE: No TensorFlow added. ShuffleNetV2 ships with torchvision,\n#         so the entire pipeline remains pure PyTorch — identical\n#         framework as the original EfficientNetB0 notebook.\n# ================================================================\n\nimport os, random, warnings, time\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as T\nimport torchvision.models as models   # ShuffleNetV2 lives here\nfrom torch.optim.lr_scheduler import OneCycleLR\n\nimport joblib\n\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score,\n    f1_score, confusion_matrix, classification_report,\n    cohen_kappa_score\n)\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils.class_weight import compute_class_weight\n\nprint('All libraries imported!')\nprint(f'PyTorch version : {torch.__version__}')\nprint(f'torchvision     : {__import__(\"torchvision\").__version__}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:09.980153Z","iopub.execute_input":"2026-06-06T03:35:09.980792Z","iopub.status.idle":"2026-06-06T03:35:09.987732Z","shell.execute_reply.started":"2026-06-06T03:35:09.980765Z","shell.execute_reply":"2026-06-06T03:35:09.987007Z"}},"outputs":[],"execution_count":null},{"id":"0510bb9d-18b8-4602-bb96-bf5949f121d6","cell_type":"code","source":"# ================================================================\n# Cell 2: Configuration\n# CHANGE: MODEL_NAME string updated. All numeric hyper-parameters\n#         are identical to the original EfficientNetB0 notebook.\n# ================================================================\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\ntorch.cuda.manual_seed_all(SEED)\n\nDEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f'Device: {DEVICE}')\n\nIMG_SIZE        = 224\nBATCH_SIZE      = 64\nEPOCHS          = 30\nBACKBONE_LR     = 2e-4\nHEAD_LR         = 1e-3\nWEIGHT_DECAY    = 1e-4\nPATIENCE        = 8\nMIXUP_PROB      = 0.2\nLABEL_SMOOTHING = 0.1\n\nBASE        = '/kaggle/input/competitions/aptos2019-blindness-detection'\nCLASS_NAMES = ['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative']\nNUM_CLASSES = 5\nMEAN        = [0.485, 0.456, 0.406]\nSTD         = [0.229, 0.224, 0.225]\n\n# CHANGE: identifier used for saved files and printed headers\nMODEL_NAME  = 'ShuffleNetV2_x1_0'   # was 'EfficientNetB0'\nCKPT_PATH   = '/kaggle/working/shufflenetv2_best.pth'\n\nprint('Configuration loaded!')\nprint(f'Backbone : {MODEL_NAME}')\nprint(f'Epochs   : {EPOCHS}  |  Batch : {BATCH_SIZE}  |  ImgSize : {IMG_SIZE}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:09.988803Z","iopub.execute_input":"2026-06-06T03:35:09.989131Z","iopub.status.idle":"2026-06-06T03:35:10.011509Z","shell.execute_reply.started":"2026-06-06T03:35:09.989098Z","shell.execute_reply":"2026-06-06T03:35:10.010809Z"}},"outputs":[],"execution_count":null},{"id":"6527b13f-2086-4668-9cf9-8efd52fd57ee","cell_type":"code","source":"# ================================================================\n# Cell 3: CLAHE Preprocessing Helpers — UNCHANGED\n# apply_clahe_fast, find_image, FundusDataset, mixup_batch are\n# byte-for-byte identical to the original EfficientNetB0 notebook.\n# ================================================================\n\ndef apply_clahe_fast(image_path, img_size=IMG_SIZE, clip_limit=2.0):\n    \"\"\"Fast CLAHE - no sharpening overhead\"\"\"\n    try:\n        img = cv2.imread(image_path)\n        if img is None:\n            raise ValueError('Cannot read')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, (img_size, img_size))\n\n        lab  = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n        l, a, b = cv2.split(lab)\n        clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=(8, 8))\n        l_c   = clahe.apply(l)\n        result = cv2.cvtColor(cv2.merge([l_c, a, b]), cv2.COLOR_LAB2RGB)\n\n        return Image.fromarray(result)\n    except Exception:\n        return Image.open(image_path).convert('RGB').resize((img_size, img_size))\n\n\ndef find_image(id_code, folders):\n    for folder in folders:\n        for ext in ['.png', '.jpeg', '.jpg']:\n            p = os.path.join(BASE, folder, str(id_code) + ext)\n            if os.path.exists(p):\n                return p\n    return None\n\n\nclass FundusDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = apply_clahe_fast(row['filepath'], IMG_SIZE)\n        if self.transform:\n            img = self.transform(img)\n        return img, int(row['label'])\n\n\ndef mixup_batch(imgs, labels, alpha=0.2):\n    lam  = np.random.beta(alpha, alpha)\n    bs   = imgs.size(0)\n    idx  = torch.randperm(bs, device=imgs.device)\n    mixed = lam * imgs + (1 - lam) * imgs[idx]\n    return mixed, labels, labels[idx], lam\n\n\nprint('CLAHE helper functions ready!')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:10.013242Z","iopub.execute_input":"2026-06-06T03:35:10.013660Z","iopub.status.idle":"2026-06-06T03:35:10.029978Z","shell.execute_reply.started":"2026-06-06T03:35:10.013638Z","shell.execute_reply":"2026-06-06T03:35:10.029118Z"}},"outputs":[],"execution_count":null},{"id":"c3cd4d4c-b792-428a-bd2e-eca4bab11a66","cell_type":"code","source":"# ================================================================\n# Cell 4: ShuffleNetV2_DR — Lightweight Backbone\n#\n# WHY ShuffleNetV2?\n#   • ~2.3 MB model vs ~20 MB EfficientNetB0\n#   • ~146 MFLOPs vs ~390 MFLOPs  (~2.7× fewer compute ops)\n#   • Designed for real-device deployment (mobile / edge)\n#   • Channel-split + channel-shuffle replaces depthwise convs\n#\n# CHANGES vs EfficientNetB0_DR:\n#   1. Backbone   : efficientnet_b0  →  shufflenet_v2_x1_0\n#   2. Feature stages exposed as conv1/maxpool/stage2/stage3/\n#                  stage4/conv5  (mirrors model.features in EffNet)\n#   3. Feature dim: 1280-d  →  1024-d  (ShuffleNetV2 x1.0 conv5)\n#   4. Classifier head input adjusted 1280 → 1024; all other\n#      layer sizes, dropout rates, BN, and ReLU are IDENTICAL.\n#   5. extract_features() method preserved for SVM extraction.\n#   6. gradcam_target_layer property added for automatic Grad-CAM\n#      layer identification (returns last Conv2d in conv5).\n# ================================================================\n\nclass ShuffleNetV2_DR(nn.Module):\n    \"\"\"\n    ShuffleNetV2 x1.0 drop-in replacement for EfficientNetB0.\n    Produces 1024-d feature vectors for the downstream SVM.\n    \"\"\"\n    # CHANGE: output feature dimension (EfficientNetB0 was 1280)\n    FEATURE_DIM = 1024\n\n    def __init__(self, num_classes=NUM_CLASSES, dropout=0.4):\n        super().__init__()\n\n        # CHANGE: load ShuffleNetV2 x1.0 with ImageNet pretrained weights\n        base = models.shufflenet_v2_x1_0(\n            weights=models.ShuffleNet_V2_X1_0_Weights.IMAGENET1K_V1\n        )\n\n        # CHANGE: expose backbone stages individually\n        #   conv1   : 3  → 24 ch  (stride-2 entry)\n        #   maxpool : 24 → 24 ch  (spatial down-sample)\n        #   stage2  : 24 → 116 ch\n        #   stage3  : 116→ 232 ch\n        #   stage4  : 232→ 464 ch\n        #   conv5   : 464→1024 ch  ← final conv (Grad-CAM target)\n        self.conv1   = base.conv1\n        self.maxpool = base.maxpool\n        self.stage2  = base.stage2\n        self.stage3  = base.stage3\n        self.stage4  = base.stage4\n        self.conv5   = base.conv5    # 464→1024, last spatial feature map\n        self.avgpool = nn.AdaptiveAvgPool2d(1)\n\n        # CHANGE: head input 1024 (was 1280); everything else unchanged\n        self.classifier = nn.Sequential(\n            nn.Dropout(p=dropout),\n            nn.Linear(self.FEATURE_DIM, 512),\n            nn.BatchNorm1d(512),\n            nn.ReLU(inplace=True),\n            nn.Dropout(p=0.3),\n            nn.Linear(512, 256),\n            nn.BatchNorm1d(256),\n            nn.ReLU(inplace=True),\n            nn.Dropout(p=0.2),\n            nn.Linear(256, num_classes),\n        )\n\n    # ── internal forward through all feature stages ──────────────\n    def _backbone(self, x):\n        x = self.conv1(x)\n        x = self.maxpool(x)\n        x = self.stage2(x)\n        x = self.stage3(x)\n        x = self.stage4(x)\n        x = self.conv5(x)   # (N, 1024, H/32, W/32)\n        return x\n\n    def forward(self, x):\n        x = self._backbone(x)\n        x = self.avgpool(x)\n        x = x.flatten(1)    # (N, 1024)\n        return self.classifier(x)\n\n    def extract_features(self, x):\n        \"\"\"Return 1024-d feature vector — used by SVM extraction.\"\"\"\n        x = self._backbone(x)\n        x = self.avgpool(x)\n        return x.flatten(1)  # (N, 1024)\n\n    @property\n    def gradcam_target_layer(self):\n        \"\"\"\n        CHANGE: Automatically identify the last Conv2d inside conv5.\n        This is the direct analogue of model.features[-1] used in\n        the original EfficientNetB0 Grad-CAM cell.\n        \"\"\"\n        last_conv = None\n        for m in self.conv5.modules():\n            if isinstance(m, nn.Conv2d):\n                last_conv = m\n        return last_conv\n\n\nprint('ShuffleNetV2_DR defined!')\nprint(f'Feature dimension : {ShuffleNetV2_DR.FEATURE_DIM}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:10.030829Z","iopub.execute_input":"2026-06-06T03:35:10.031559Z","iopub.status.idle":"2026-06-06T03:35:10.053108Z","shell.execute_reply.started":"2026-06-06T03:35:10.031527Z","shell.execute_reply":"2026-06-06T03:35:10.052377Z"}},"outputs":[],"execution_count":null},{"id":"dbb1e8f2-bc4c-474e-8acf-f35ad7eb60bb","cell_type":"code","source":"# ================================================================\n# Cell 5: Transforms — UNCHANGED\n# ================================================================\n\ntrain_tfm = T.Compose([\n    T.RandomHorizontalFlip(p=0.5),\n    T.RandomVerticalFlip(p=0.3),\n    T.RandomRotation(30),\n    T.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.15, hue=0.02),\n    T.RandomAffine(degrees=0, translate=(0.1, 0.1), scale=(0.9, 1.1)),\n    T.ToTensor(),\n    T.Normalize(MEAN, STD),\n])\n\nval_tfm = T.Compose([\n    T.ToTensor(),\n    T.Normalize(MEAN, STD),\n])\n\nprint('Transforms defined!')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:10.054012Z","iopub.execute_input":"2026-06-06T03:35:10.054492Z","iopub.status.idle":"2026-06-06T03:35:10.074018Z","shell.execute_reply.started":"2026-06-06T03:35:10.054470Z","shell.execute_reply":"2026-06-06T03:35:10.073143Z"}},"outputs":[],"execution_count":null},{"id":"27bc9724-be9a-47f1-8a85-6854f954042c","cell_type":"code","source":"# ================================================================\n# Cell 6: Load Data (80-20 Split) — UNCHANGED\n# ================================================================\n\nTRAIN_FOLDER = ['train_images']\ndf_raw = pd.read_csv(f'{BASE}/train.csv')\ndf_raw['label']    = df_raw['diagnosis']\ndf_raw['filepath'] = df_raw['id_code'].apply(\n    lambda x: find_image(x, TRAIN_FOLDER))\ndf_raw = df_raw[df_raw['filepath'].notna()].reset_index(drop=True)\n\ndf_tr, df_te = train_test_split(\n    df_raw, test_size=0.20, stratify=df_raw['label'], random_state=SEED)\n\nprint(f'Train: {len(df_tr)} | Test: {len(df_te)}')\nprint('\\nClass distribution (train):')\nfor i, name in enumerate(CLASS_NAMES):\n    c = (df_tr['label'] == i).sum()\n    print(f'  {name:15s}: {c:4d}')\n\ntrain_dl = DataLoader(FundusDataset(df_tr, train_tfm),\n                      batch_size=BATCH_SIZE, shuffle=True,\n                      num_workers=2, pin_memory=True)\ntest_dl  = DataLoader(FundusDataset(df_te, val_tfm),\n                      batch_size=BATCH_SIZE, shuffle=False,\n                      num_workers=2, pin_memory=True)\n\ncw = compute_class_weight('balanced',\n                           classes=np.arange(NUM_CLASSES),\n                           y=df_tr['label'].values)\ncriterion = nn.CrossEntropyLoss(\n    weight=torch.tensor(cw, dtype=torch.float).to(DEVICE),\n    label_smoothing=LABEL_SMOOTHING\n)\n\nprint('\\nClass weights:', {n: f'{w:.3f}' for n, w in zip(CLASS_NAMES, cw)})\nprint('✅ Data ready!')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:10.075237Z","iopub.execute_input":"2026-06-06T03:35:10.075594Z","iopub.status.idle":"2026-06-06T03:35:10.579591Z","shell.execute_reply.started":"2026-06-06T03:35:10.075562Z","shell.execute_reply":"2026-06-06T03:35:10.578894Z"}},"outputs":[],"execution_count":null},{"id":"abf500fb-94be-4f6a-b39f-7d6698207eb1","cell_type":"code","source":"# ================================================================\n# CLAHE Visualisation — UNCHANGED\n# ================================================================\n\nsample_path = df_tr.iloc[0]['filepath']\n\noriginal = cv2.imread(sample_path)\noriginal = cv2.cvtColor(original, cv2.COLOR_BGR2RGB)\n\nenhanced = apply_clahe_fast(sample_path)\n\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.imshow(original)\nplt.title('Original Fundus Image')\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nplt.imshow(enhanced)\nplt.title('CLAHE Enhanced Image')\nplt.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:10.580571Z","iopub.execute_input":"2026-06-06T03:35:10.580879Z","iopub.status.idle":"2026-06-06T03:35:11.296561Z","shell.execute_reply.started":"2026-06-06T03:35:10.580855Z","shell.execute_reply":"2026-06-06T03:35:11.295663Z"}},"outputs":[],"execution_count":null},{"id":"e0d1d7da-ae4f-441d-918b-4e172a021870","cell_type":"code","source":"# ================================================================\n# Cell 8: Initialise ShuffleNetV2 Model\n#\n# CHANGE: Instantiate ShuffleNetV2_DR instead of EfficientNetB0_DR.\n# CHANGE: Optimizer backbone param-group references renamed stages\n#         (conv1/stage2/stage3/stage4/conv5) instead of model.features.\n#         HEAD_LR and BACKBONE_LR values are IDENTICAL to original.\n# ================================================================\n\nmodel = ShuffleNetV2_DR(NUM_CLASSES).to(DEVICE)\n\ntotal_params     = sum(p.numel() for p in model.parameters())\ntrainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(f'Total parameters    : {total_params:,}')\nprint(f'Trainable parameters: {trainable_params:,}')\n\n# CHANGE: backbone parameter group uses ShuffleNetV2 stage attributes\nbackbone_params = (\n    list(model.conv1.parameters())  +\n    list(model.stage2.parameters()) +\n    list(model.stage3.parameters()) +\n    list(model.stage4.parameters()) +\n    list(model.conv5.parameters())\n)\n\noptimizer = optim.AdamW([\n    {'params': backbone_params,                 'lr': BACKBONE_LR},\n    {'params': model.classifier.parameters(),   'lr': HEAD_LR},\n], weight_decay=WEIGHT_DECAY)\n\nscheduler = OneCycleLR(\n    optimizer,\n    max_lr=[BACKBONE_LR, HEAD_LR],\n    steps_per_epoch=len(train_dl),\n    epochs=EPOCHS,\n    pct_start=0.2,\n)\n\n# CHANGE: auto-identify Grad-CAM layer now so it's visible at init\ngcam_layer = model.gradcam_target_layer\nprint(f'Grad-CAM target layer: {gcam_layer}')\nprint('Model initialised!')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:11.297632Z","iopub.execute_input":"2026-06-06T03:35:11.298247Z","iopub.status.idle":"2026-06-06T03:35:11.628737Z","shell.execute_reply.started":"2026-06-06T03:35:11.298219Z","shell.execute_reply":"2026-06-06T03:35:11.627775Z"}},"outputs":[],"execution_count":null},{"id":"0d4a0b74-81ef-41fe-b145-a71b9d451b5f","cell_type":"code","source":"# ================================================================\n# Cell 9: Training Loop — logic UNCHANGED\n# CHANGE: checkpoint saved to CKPT_PATH = shufflenetv2_best.pth\n# ================================================================\n\nbest_train_acc = 0.0\ntrain_losses, train_accs = [], []\npatience_ctr = 0\n\nprint(f'\\nTraining {MODEL_NAME} for {EPOCHS} epochs...\\n')\nprint(f'{\"Epoch\":>6}  {\"Loss\":>8}  {\"Acc\":>8}  {\"Time\":>8}  Status')\nprint('-' * 54)\n\nfor epoch in range(1, EPOCHS + 1):\n    t0 = time.time()\n    model.train()\n    tr_loss, tr_correct, total_train = 0.0, 0, 0\n\n    for imgs, lbls in train_dl:\n        imgs, lbls = imgs.to(DEVICE), lbls.to(DEVICE)\n\n        if random.random() < MIXUP_PROB:\n            mixed, lbl_a, lbl_b, lam = mixup_batch(imgs, lbls)\n            optimizer.zero_grad()\n            out  = model(mixed)\n            loss = lam * criterion(out, lbl_a) + (1 - lam) * criterion(out, lbl_b)\n        else:\n            optimizer.zero_grad()\n            out  = model(imgs)\n            loss = criterion(out, lbls)\n\n        loss.backward()\n        optimizer.step()\n        scheduler.step()\n\n        tr_loss     += loss.item()\n        tr_correct  += (out.argmax(1) == lbls).sum().item()\n        total_train += lbls.size(0)\n\n    t_loss = tr_loss / len(train_dl)\n    t_acc  = tr_correct / total_train * 100\n    train_losses.append(t_loss)\n    train_accs.append(t_acc)\n\n    if t_acc > best_train_acc:\n        best_train_acc = t_acc\n        patience_ctr   = 0\n        torch.save(model.state_dict(), CKPT_PATH)\n        status = f'✓ saved (best {t_acc:.2f}%)'\n    else:\n        patience_ctr += 1\n        status = f'patience {patience_ctr}/{PATIENCE}'\n\n    print(f'{epoch:>6}  {t_loss:>8.4f}  {t_acc:>7.2f}%  {time.time()-t0:>7.1f}s  {status}')\n\n    if patience_ctr >= PATIENCE:\n        print(f'\\nEarly stopping at epoch {epoch}')\n        break\n\nprint(f'\\n✅ Best Training Accuracy: {best_train_acc:.2f}%')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T03:35:11.630876Z","iopub.execute_input":"2026-06-06T03:35:11.631642Z","iopub.status.idle":"2026-06-06T04:39:35.680353Z","shell.execute_reply.started":"2026-06-06T03:35:11.631611Z","shell.execute_reply":"2026-06-06T04:39:35.679085Z"}},"outputs":[],"execution_count":null},{"id":"06e1824b-eebc-49ed-a95a-64d1354ca836","cell_type":"code","source":"# ================================================================\n# Cell 10: Training Curves — UNCHANGED\n# CHANGE: save filename updated to shufflenetv2_training_curves.png\n# ================================================================\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\naxes[0].plot(train_losses, linewidth=2, color='steelblue')\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('Loss')\naxes[0].set_title(f'{MODEL_NAME} — Training Loss')\naxes[0].grid(True, alpha=0.3)\n\naxes[1].plot(train_accs, linewidth=2, color='darkorange')\naxes[1].axhline(y=best_train_acc, color='r', linestyle='--',\n                label=f'Best: {best_train_acc:.2f}%')\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Accuracy (%)')\naxes[1].set_title(f'{MODEL_NAME} — Training Accuracy')\naxes[1].legend()\naxes[1].grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/shufflenetv2_training_curves.png', dpi=150)\nplt.show()\nprint('Training curves saved.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:39:35.682201Z","iopub.execute_input":"2026-06-06T04:39:35.682562Z","iopub.status.idle":"2026-06-06T04:39:36.293631Z","shell.execute_reply.started":"2026-06-06T04:39:35.682529Z","shell.execute_reply":"2026-06-06T04:39:36.292921Z"}},"outputs":[],"execution_count":null},{"id":"f402d023-df67-41a2-913f-7c059049275d","cell_type":"code","source":"# ================================================================\n# Cell 11: Feature Extraction with ShuffleNetV2\n#\n# CHANGE: Load shufflenetv2_best.pth checkpoint.\n# CHANGE: extract_features() now returns (N, 1024) vectors\n#         instead of (N, 1280) — everything else identical.\n# ================================================================\n\nmodel.load_state_dict(torch.load(CKPT_PATH, map_location=DEVICE))\nmodel.eval()\nprint('Best checkpoint loaded.')\n\n@torch.no_grad()\ndef extract_features(model, dataloader, device):\n    all_features, all_labels = [], []\n    for images, labels in tqdm(dataloader):\n        images   = images.to(device)\n        features = model.extract_features(images).cpu().numpy()  # (N, 1024)\n        all_features.append(features)\n        all_labels.append(labels.numpy())\n    return np.vstack(all_features), np.concatenate(all_labels)\n\n\nprint('\\nExtracting training features...')\ntrain_features, train_labels = extract_features(model, train_dl, DEVICE)\n\nprint('Extracting test features...')\ntest_features, test_labels   = extract_features(model, test_dl,  DEVICE)\n\nprint(f'\\n✅ Done!  Train : {train_features.shape}  Test : {test_features.shape}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:39:36.294775Z","iopub.execute_input":"2026-06-06T04:39:36.295116Z","iopub.status.idle":"2026-06-06T04:42:48.982397Z","shell.execute_reply.started":"2026-06-06T04:39:36.295095Z","shell.execute_reply":"2026-06-06T04:42:48.981306Z"}},"outputs":[],"execution_count":null},{"id":"d92dec04-f514-4f9d-8f9f-47d87a24509e","cell_type":"code","source":"# ================================================================\n# Cell 12: SVM Training — UNCHANGED\n# Same kernel (rbf), same grid C/gamma, same CV=3, class_weight.\n# ================================================================\n\nprint('\\nTraining SVM...')\n\nscaler       = StandardScaler()\ntrain_scaled = scaler.fit_transform(train_features)\ntest_scaled  = scaler.transform(test_features)\n\nparam_grid = {\n    'C'           : [1, 5, 10, 50],\n    'gamma'       : ['scale', 0.01],\n    'kernel'      : ['rbf'],\n    'class_weight': ['balanced']\n}\n\ngrid_search = GridSearchCV(\n    SVC(random_state=SEED, probability=True),\n    param_grid,\n    cv=3,\n    scoring='accuracy',\n    n_jobs=-1,\n    verbose=1\n)\n\ngrid_search.fit(train_scaled, train_labels)\n\nprint(f'\\n✅ Best parameters : {grid_search.best_params_}')\nprint(f'Best CV accuracy  : {grid_search.best_score_:.4f}')\n\nbest_svm = SVC(\n    kernel='rbf',\n    C=grid_search.best_params_['C'],\n    gamma=grid_search.best_params_['gamma'],\n    class_weight='balanced',\n    random_state=SEED\n)\n\nbest_svm.fit(train_scaled, train_labels)\ntest_predictions = best_svm.predict(test_scaled)\nprint('SVM training complete.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:42:48.983933Z","iopub.execute_input":"2026-06-06T04:42:48.984389Z","iopub.status.idle":"2026-06-06T04:46:04.267891Z","shell.execute_reply.started":"2026-06-06T04:42:48.984356Z","shell.execute_reply":"2026-06-06T04:46:04.267180Z"}},"outputs":[],"execution_count":null},{"id":"3af63db3-eb51-41aa-a66c-2712951ceacf","cell_type":"code","source":"# ================================================================\n# Cell 13: Evaluation Metrics — UNCHANGED\n# Accuracy, Precision, Recall, F1, Kappa, Classification Report,\n# Confusion Matrix — all identical to original.\n# CHANGE: header and save filename updated to ShuffleNetV2.\n# ================================================================\n\nacc   = accuracy_score(test_labels, test_predictions)\nprec  = precision_score(test_labels, test_predictions, average='weighted')\nrec   = recall_score(test_labels, test_predictions, average='weighted')\nf1    = f1_score(test_labels, test_predictions, average='weighted')\nkappa = cohen_kappa_score(test_labels, test_predictions, weights='quadratic')\n\nprint('\\n' + '='*60)\nprint(f'  {MODEL_NAME} + SVM  —  5-Class DR')   # CHANGE\nprint('='*60)\nprint(f'  Test Accuracy      : {acc*100:.2f}%')\nprint(f'  Weighted Precision : {prec:.4f}')\nprint(f'  Weighted Recall    : {rec:.4f}')\nprint(f'  Weighted F1-Score  : {f1:.4f}')\nprint(f'  Quadratic Kappa    : {kappa:.4f}')\nprint('='*60)\n\nprint('\\n📊 Classification Report:')\nprint(classification_report(test_labels, test_predictions,\n                            target_names=CLASS_NAMES))\n\n# Confusion Matrix\ncm = confusion_matrix(test_labels, test_predictions)\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=CLASS_NAMES, yticklabels=CLASS_NAMES)\nplt.title(f'{MODEL_NAME}+SVM | Acc: {acc*100:.2f}% | Kappa: {kappa:.4f}')  # CHANGE\nplt.ylabel('True Label')\nplt.xlabel('Predicted Label')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.savefig('/kaggle/working/shufflenetv2_confusion_matrix.png', dpi=150)   # CHANGE\nplt.show()\n\nprint('\\n' + '='*60)\nprint(f'🎯 FINAL TEST ACCURACY: {acc*100:.2f}%')\nprint('='*60)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:46:04.269095Z","iopub.execute_input":"2026-06-06T04:46:04.269897Z","iopub.status.idle":"2026-06-06T04:46:04.772925Z","shell.execute_reply.started":"2026-06-06T04:46:04.269870Z","shell.execute_reply":"2026-06-06T04:46:04.772266Z"}},"outputs":[],"execution_count":null},{"id":"a7bae243-eeac-42c5-966f-1f02171929ca","cell_type":"code","source":"# ================================================================\n# Grad-CAM Visualisation — ShuffleNetV2\n#\n# CHANGE: target_layer is now model.gradcam_target_layer which\n#         automatically resolves to the last Conv2d inside conv5\n#         (the 464→1024 pointwise conv).\n#         This is the direct analogue of model.features[-1] used\n#         in the original EfficientNetB0 Grad-CAM cell.\n#\n# Hook registration, backprop, weighting, and overlay code are\n# IDENTICAL to the original notebook.\n# ================================================================\n\nmodel.eval()\n\n# CHANGE: auto-resolve last Conv2d in conv5\ntarget_layer = model.gradcam_target_layer\nprint(f'Grad-CAM target layer : {target_layer}')\n\nactivations = []\ngradients   = []\n\ndef forward_hook(module, inp, out):\n    activations.append(out)\n\ndef backward_hook(module, grad_in, grad_out):\n    gradients.append(grad_out[0])\n\nfh = target_layer.register_forward_hook(forward_hook)\nbh = target_layer.register_full_backward_hook(backward_hook)\n\n# ── visualise multiple test samples (one per DR class) ─────────\nn_samples = min(5, len(df_te))\nsample_indices = []\nfor cls in range(NUM_CLASSES):\n    idx_list = df_te.index[df_te['label'] == cls].tolist()\n    if idx_list:\n        # get positional index in df_te (reset_index was not called)\n        pos = df_te.reset_index().index[df_te.reset_index()['label'] == cls][0]\n        sample_indices.append(pos)\n\nfig, axes = plt.subplots(len(sample_indices), 3,\n                          figsize=(15, 5 * len(sample_indices)))\nif len(sample_indices) == 1:\n    axes = [axes]   # keep consistent list-of-rows format\n\nfor row_i, s_idx in enumerate(sample_indices):\n    # clear lists for each sample\n    activations.clear()\n    gradients.clear()\n\n    s_path   = df_te.iloc[s_idx]['filepath']\n    true_cls = int(df_te.iloc[s_idx]['label'])\n\n    img_pil  = Image.open(s_path).convert('RGB').resize((IMG_SIZE, IMG_SIZE))\n    img_np   = np.array(img_pil)\n\n    x      = val_tfm(img_pil).unsqueeze(0).to(DEVICE)\n    output = model(x)\n    pred_class = output.argmax(1).item()\n\n    model.zero_grad()\n    output[:, pred_class].backward()\n\n    grads   = gradients[0]                           # (1, C, h, w)\n    acts    = activations[0]                         # (1, C, h, w)\n    weights = grads.mean(dim=(2, 3), keepdim=True)   # (1, C, 1, 1)\n    cam     = (weights * acts).sum(dim=1).squeeze()  # (h, w)\n    cam     = torch.relu(cam)\n    cam     = cam.detach().cpu().numpy()\n\n    cam = cv2.resize(cam, (IMG_SIZE, IMG_SIZE))\n    cam = (cam - cam.min()) / (cam.max() - cam.min() + 1e-8)\n\n    heatmap = cv2.applyColorMap(np.uint8(cam * 255), cv2.COLORMAP_JET)\n    overlay = cv2.addWeighted(img_np, 0.6, heatmap, 0.4, 0)\n\n    axes[row_i][0].imshow(img_np)\n    axes[row_i][0].set_title(f'Original\\nTrue: {CLASS_NAMES[true_cls]}')\n    axes[row_i][0].axis('off')\n\n    axes[row_i][1].imshow(cam, cmap='jet')\n    axes[row_i][1].set_title(f'Grad-CAM Heatmap\\n(layer: conv5)')\n    axes[row_i][1].axis('off')\n\n    axes[row_i][2].imshow(cv2.cvtColor(overlay, cv2.COLOR_BGR2RGB))\n    axes[row_i][2].set_title(f'Overlay\\nPredicted: {CLASS_NAMES[pred_class]}')\n    axes[row_i][2].axis('off')\n\nplt.suptitle(f'ShuffleNetV2 Grad-CAM — One Sample per DR Class',\n             fontsize=14, fontweight='bold', y=1.01)\nplt.tight_layout()\nplt.savefig('/kaggle/working/shufflenetv2_gradcam.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Grad-CAM figure saved → /kaggle/working/shufflenetv2_gradcam.png')\n\nfh.remove()\nbh.remove()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:46:04.774076Z","iopub.execute_input":"2026-06-06T04:46:04.774410Z","iopub.status.idle":"2026-06-06T04:46:11.377305Z","shell.execute_reply.started":"2026-06-06T04:46:04.774388Z","shell.execute_reply":"2026-06-06T04:46:11.376268Z"}},"outputs":[],"execution_count":null},{"id":"98e92f4c-12e0-43a2-af8e-dc58d715eea2","cell_type":"code","source":"# ================================================================\n# Save Models for Deployment\n#\n# Saves THREE artefacts needed for inference:\n#   1. shufflenetv2_best.pth       — PyTorch state-dict (backbone+head)\n#   2. svm_model_shufflenetv2.pkl  — trained SVM classifier\n#   3. scaler_shufflenetv2.pkl     — fitted StandardScaler\n#\n# Also exports a TorchScript-traced version for production serving\n# (no Python source required at inference time).\n#\n# CHANGE: filenames updated from effb0_* to shufflenetv2_*\n# ================================================================\n\nimport os\n\nSAVE_DIR = '/kaggle/working'\n\n# ── 1. PyTorch state-dict (already written by training loop) ────\npth_path = os.path.join(SAVE_DIR, 'shufflenetv2_best.pth')\ntorch.save(model.state_dict(), pth_path)\nprint(f'✅ State-dict saved      → {pth_path}')\n\n# ── 2. TorchScript traced model (framework-independent deploy) ──\nmodel.eval()\ndummy_input  = torch.randn(1, 3, IMG_SIZE, IMG_SIZE).to(DEVICE)\ntraced_model = torch.jit.trace(model, dummy_input)\nts_path      = os.path.join(SAVE_DIR, 'shufflenetv2_traced.pt')\ntraced_model.save(ts_path)\nprint(f'✅ TorchScript saved     → {ts_path}')\n\n# ── 3. SVM classifier ───────────────────────────────────────────\nsvm_path = os.path.join(SAVE_DIR, 'svm_model_shufflenetv2.pkl')\njoblib.dump(best_svm, svm_path)\nprint(f'✅ SVM model saved       → {svm_path}')\n\n# ── 4. Feature scaler ───────────────────────────────────────────\nscaler_path = os.path.join(SAVE_DIR, 'scaler_shufflenetv2.pkl')\njoblib.dump(scaler, scaler_path)\nprint(f'✅ Feature scaler saved  → {scaler_path}')\n\n# ── Summary ─────────────────────────────────────────────────────\nprint('\\n' + '─'*55)\nprint('  Deployment artefacts')\nprint('─'*55)\nartefacts = [\n    (pth_path,     'PyTorch state-dict  (load with model.load_state_dict)'),\n    (ts_path,      'TorchScript model   (torch.jit.load — no source needed)'),\n    (svm_path,     'SVM classifier      (joblib.load)'),\n    (scaler_path,  'Feature scaler      (joblib.load)'),\n]\nfor path, desc in artefacts:\n    size_kb = os.path.getsize(path) / 1024\n    print(f'  {os.path.basename(path):45s}  {size_kb:7.1f} KB  — {desc}')\nprint('─'*55)\nprint('\\nInference usage:')\nprint('  model = torch.jit.load(\"shufflenetv2_traced.pt\")')\nprint('  svm   = joblib.load(\"svm_model_shufflenetv2.pkl\")')\nprint('  sc    = joblib.load(\"scaler_shufflenetv2.pkl\")')\nprint('  feats = model(img_tensor)  # after apply_clahe_fast + val_tfm')\nprint('  pred  = svm.predict(sc.transform(feats.numpy()))')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:46:11.378788Z","iopub.execute_input":"2026-06-06T04:46:11.379218Z","iopub.status.idle":"2026-06-06T04:46:12.946774Z","shell.execute_reply.started":"2026-06-06T04:46:11.379177Z","shell.execute_reply":"2026-06-06T04:46:12.946084Z"}},"outputs":[],"execution_count":null},{"id":"feb7a7f8-5105-45cd-bd97-56986991d971","cell_type":"code","source":"# ================================================================\n# Save Paper Figures — all outputs already written during run\n# ================================================================\n\nfigures = [\n    '/kaggle/working/shufflenetv2_training_curves.png',\n    '/kaggle/working/shufflenetv2_confusion_matrix.png',\n    '/kaggle/working/shufflenetv2_gradcam.png',\n]\n\nprint('Paper figures saved:')\nfor f in figures:\n    exists = '✅' if os.path.exists(f) else '❌ MISSING'\n    print(f'  {exists}  {f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T04:46:12.947739Z","iopub.execute_input":"2026-06-06T04:46:12.948089Z","iopub.status.idle":"2026-06-06T04:46:12.953521Z","shell.execute_reply.started":"2026-06-06T04:46:12.948066Z","shell.execute_reply":"2026-06-06T04:46:12.952653Z"}},"outputs":[],"execution_count":null}]}