{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"datasetVersion","sourceId":2822650,"datasetId":1715304,"databundleVersionId":2869088}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-26T16:57:54.113030Z","iopub.execute_input":"2026-05-26T16:57:54.113541Z","iopub.status.idle":"2026-05-26T16:57:55.955357Z","shell.execute_reply.started":"2026-05-26T16:57:54.113500Z","shell.execute_reply":"2026-05-26T16:57:55.954769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, warnings, time\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as T\nimport torchvision.models as models\n\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score,\n    f1_score, confusion_matrix, classification_report,\n    roc_auc_score\n)\nfrom sklearn.utils.class_weight import compute_class_weight\n\nDEVICE      = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nSEED        = 42\nIMG_SIZE    = 224\nBATCH_SIZE  = 32\nMEAN        = [0.485, 0.456, 0.406]\nSTD         = [0.229, 0.224, 0.225]\nBASE        = '/kaggle/input/datasets/mariaherrerot/aptos2019'\nCLASS_NAMES = ['No DR','Mild','Moderate',\n               'Severe','Proliferative']\n\ntorch.manual_seed(SEED)\nnp.random.seed(SEED)\nprint(f'Device : {DEVICE}')\nprint('All imports done.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T16:57:55.956659Z","iopub.execute_input":"2026-05-26T16:57:55.956967Z","iopub.status.idle":"2026-05-26T16:57:55.965848Z","shell.execute_reply.started":"2026-05-26T16:57:55.956946Z","shell.execute_reply":"2026-05-26T16:57:55.964991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_clahe(image_path, img_size=224):\n    try:\n        img = cv2.imread(image_path)\n        if img is None:\n            raise ValueError\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, (img_size, img_size))\n        lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n        l, a, b = cv2.split(lab)\n        clahe   = cv2.createCLAHE(clipLimit=2.0,\n                                   tileGridSize=(8,8))\n        l_clahe   = clahe.apply(l)\n        lab_clahe = cv2.merge([l_clahe, a, b])\n        img_clahe = cv2.cvtColor(lab_clahe,\n                                  cv2.COLOR_LAB2RGB)\n        return Image.fromarray(img_clahe)\n    except:\n        return Image.open(image_path).convert('RGB')\\\n                    .resize((img_size, img_size))\n\ndef find_image(id_code, folders):\n    for folder in folders:\n        for ext in ['.png', '.jpeg', '.jpg']:\n            p = os.path.join(BASE, folder,\n                             str(id_code) + ext)\n            if os.path.exists(p):\n                return p\n    return None\n\nval_tfm = T.Compose([\n    T.ToTensor(),\n    T.Normalize(MEAN, STD),\n])\n\nclass FundusDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df        = df.reset_index(drop=True)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = apply_clahe(row['filepath'], IMG_SIZE)\n        if self.transform:\n            img = self.transform(img)\n        return img, int(row['label'])\n\nprint('Functions ready.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T16:57:55.966778Z","iopub.execute_input":"2026-05-26T16:57:55.967104Z","iopub.status.idle":"2026-05-26T16:57:55.983446Z","shell.execute_reply.started":"2026-05-26T16:57:55.967085Z","shell.execute_reply":"2026-05-26T16:57:55.982844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, warnings\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\nwarnings.filterwarnings('ignore')\n\nBASE = '/kaggle/input/datasets/mariaherrerot/aptos2019'\n\nALL_FOLDERS = [\n    'train_images/train_images',\n    'val_images/val_images',\n    'test_images/test_images'\n]\n\nCLASS_NAMES = ['No DR','Mild','Moderate','Severe','Proliferative']\nSEED = 42\n\ndef find_image(id_code, folders):\n    for folder in folders:\n        for ext in ['.png', '.jpeg', '.jpg']:\n            p = os.path.join(BASE, folder, str(id_code) + ext)\n            if os.path.exists(p):\n                return p\n    return None\n\n# Load all 3 CSVs\ndf1 = pd.read_csv(f'{BASE}/train_1.csv')\ndf2 = pd.read_csv(f'{BASE}/valid.csv')\ndf3 = pd.read_csv(f'{BASE}/test.csv')\n\n# Combine all\ndf_all = pd.concat([df1, df2, df3], ignore_index=True)\ndf_all['label'] = df_all['diagnosis']\ndf_all['filepath'] = df_all['id_code'].apply(\n    lambda x: find_image(x, ALL_FOLDERS)\n)\ndf_all = df_all[df_all['filepath'].notna()].reset_index(drop=True)\n\n# 80/20 stratified split\ndf_tr, df_te = train_test_split(\n    df_all,\n    test_size=0.20,\n    stratify=df_all['label'],\n    random_state=SEED\n)\n\n# Further split train into train/val (90/10)\ndf_tr, df_va = train_test_split(\n    df_tr,\n    test_size=0.10,\n    stratify=df_tr['label'],\n    random_state=SEED\n)\n\nprint(f'Total images : {len(df_all)}')\nprint(f'Train        : {len(df_tr)}')\nprint(f'Val          : {len(df_va)}')\nprint(f'Test         : {len(df_te)}')\nprint('\\nClass distribution in TEST:')\nfor i, name in enumerate(CLASS_NAMES):\n    c = len(df_te[df_te['label']==i])\n    bar = '█' * (c // 5)\n    print(f'  {name:15s}: {c:4d} {bar}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T16:57:55.984237Z","iopub.execute_input":"2026-05-26T16:57:55.984523Z","iopub.status.idle":"2026-05-26T16:57:56.110506Z","shell.execute_reply.started":"2026-05-26T16:57:55.984504Z","shell.execute_reply":"2026-05-26T16:57:56.109815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EfficientNetB0_Extractor(nn.Module):\n    \"\"\"\n    EfficientNetB0 feature extractor.\n    Parameters : 5.3M (lightweight)\n    Output     : 1280-D feature vector\n    Same output dimension as EfficientNetV2-Small\n    in the paper — compatible with SVM directly.\n    \"\"\"\n    def __init__(self):\n        super(EfficientNetB0_Extractor, self).__init__()\n        base = models.efficientnet_b0(\n            weights=models.EfficientNet_B0_Weights.IMAGENET1K_V1\n        )\n        # Remove classifier — keep feature layers only\n        self.features   = base.features\n        self.avgpool    = base.avgpool\n        self.dropout    = nn.Dropout(p=0.2)\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.avgpool(x)\n        x = x.flatten(1)\n        x = self.dropout(x)\n        return x\n\nextractor = EfficientNetB0_Extractor()\n\n# Freeze all layers first\nfor p in extractor.parameters():\n    p.requires_grad = False\n\n# Unfreeze last 10 layers for fine tuning\nfor layer in list(extractor.features.children())[-10:]:\n    for p in layer.parameters():\n        p.requires_grad = True\n\nextractor = extractor.to(DEVICE)\n\ntotal     = sum(p.numel() for p in extractor.parameters())\ntrainable = sum(p.numel() for p in extractor.parameters()\n                if p.requires_grad)\nprint(f'Total params     : {total:,}')\nprint(f'Trainable params : {trainable:,}')\nprint(f'Output features  : 1280-D')\nprint('EfficientNetB0 ready.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T16:57:56.112014Z","iopub.execute_input":"2026-05-26T16:57:56.112324Z","iopub.status.idle":"2026-05-26T16:57:56.258344Z","shell.execute_reply.started":"2026-05-26T16:57:56.112303Z","shell.execute_reply":"2026-05-26T16:57:56.257635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef extract_features(model, loader):\n    model.eval()\n    feats, labels = [], []\n    for imgs, lbls in loader:\n        feats.append(\n            model(imgs.to(DEVICE)).cpu().numpy()\n        )\n        labels.append(lbls.numpy())\n    return np.vstack(feats), np.concatenate(labels)\n\ntrain_dl = DataLoader(\n    FundusDataset(df_tr, val_tfm),\n    batch_size=BATCH_SIZE, shuffle=False,\n    num_workers=2\n)\nval_dl = DataLoader(\n    FundusDataset(df_va, val_tfm),\n    batch_size=BATCH_SIZE, shuffle=False,\n    num_workers=2\n)\ntest_dl = DataLoader(\n    FundusDataset(df_te, val_tfm),\n    batch_size=BATCH_SIZE, shuffle=False,\n    num_workers=2\n)\n\nprint('Extracting train features ...')\nX_train, y_train = extract_features(extractor, train_dl)\nprint('Extracting val features ...')\nX_val,   y_val   = extract_features(extractor, val_dl)\nprint('Extracting test features ...')\nX_test,  y_test  = extract_features(extractor, test_dl)\n\nprint(f'\\nFeature shape : {X_train.shape}')\nprint(f'Train samples : {len(y_train)}')\nprint(f'Val samples   : {len(y_val)}')\nprint(f'Test samples  : {len(y_test)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T17:10:41.897351Z","iopub.execute_input":"2026-05-26T17:10:41.898206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Scaling features ...')\nscaler     = StandardScaler()\nX_train_sc = scaler.fit_transform(X_train)\nX_val_sc   = scaler.transform(X_val)\nX_test_sc  = scaler.transform(X_test)\n\n# Class weights for SVM\nclass_weights = compute_class_weight(\n    'balanced',\n    classes=np.array([0,1,2,3,4]),\n    y=y_train\n)\nclass_weight_dict = {i: w for i, w\n                     in enumerate(class_weights)}\n\nprint('\\nClass weights:')\nfor i, (n, w) in enumerate(zip(CLASS_NAMES,\n                                class_weights)):\n    print(f'  {n:15s}: {w:.4f}')\n\nprint('\\nGrid searching SVM hyperparameters ...')\nparam_grid = {\n    'C'    : [0.1, 1, 10, 100],\n    'gamma': ['scale', 'auto', 0.001, 0.01],\n}\ngs = GridSearchCV(\n    SVC(kernel='rbf', probability=True,\n        class_weight=class_weight_dict,\n        random_state=SEED),\n    param_grid, cv=3,\n    scoring='accuracy',\n    n_jobs=-1, verbose=1\n)\ngs.fit(X_train_sc, y_train)\n\nsvm = gs.best_estimator_\nprint(f'\\nBest params  : {gs.best_params_}')\nprint(f'Val accuracy : '\n      f'{accuracy_score(y_val, svm.predict(X_val_sc))*100:.2f}%')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T17:00:50.821736Z","iopub.execute_input":"2026-05-26T17:00:50.822027Z","iopub.status.idle":"2026-05-26T17:06:36.027688Z","shell.execute_reply.started":"2026-05-26T17:00:50.821999Z","shell.execute_reply":"2026-05-26T17:06:36.026829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = svm.predict(X_test_sc)\n\nacc  = accuracy_score(y_test, y_pred)\nprec = precision_score(y_test, y_pred,\n                       average='weighted',\n                       zero_division=0)\nrec  = recall_score(y_test, y_pred,\n                    average='weighted',\n                    zero_division=0)\nf1   = f1_score(y_test, y_pred,\n                average='weighted',\n                zero_division=0)\n\nprint('='*55)\nprint('   EfficientNetB0 + SVM — 5-CLASS RESULTS')\nprint('='*55)\nprint(f'  Accuracy  : {acc*100:.2f}%')\nprint(f'  Precision : {prec:.4f}')\nprint(f'  Recall    : {rec:.4f}')\nprint(f'  F1-Score  : {f1:.4f}')\nprint('='*55)\nprint(classification_report(\n    y_test, y_pred,\n    target_names=CLASS_NAMES,\n    zero_division=0\n))\nprint(f'Wiratama et al.  : 92.60%')\nprint(f'Our Model        : {acc*100:.2f}%')\nprint(f'Improvement      : +{acc*100-92.60:.2f}%')\nif acc*100 > 93:\n    print('TARGET ACHIEVED — Above 93%! ✅')\nelif acc*100 > 92.60:\n    print('Beats Wiratama et al.! ✅')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T17:06:36.029038Z","iopub.execute_input":"2026-05-26T17:06:36.029487Z","iopub.status.idle":"2026-05-26T17:06:37.274843Z","shell.execute_reply.started":"2026-05-26T17:06:36.029443Z","shell.execute_reply":"2026-05-26T17:06:37.274207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(16, 5))\nfig.suptitle(\n    'EfficientNetB0 + SVM — 5-Class DR Results',\n    fontsize=13, fontweight='bold'\n)\n\n# Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            ax=axes[0],\n            xticklabels=CLASS_NAMES,\n            yticklabels=CLASS_NAMES)\naxes[0].set_title('Confusion Matrix')\naxes[0].set_ylabel('True')\naxes[0].set_xlabel('Predicted')\n\n# Per class accuracy\nper_class = [cm[i][i]/max(cm[i].sum(),1)*100\n             for i in range(5)]\ncolors = ['#2ecc71','#3498db','#f39c12',\n          '#e74c3c','#9b59b6']\nbars = axes[1].bar(CLASS_NAMES, per_class,\n                   color=colors, width=0.6)\naxes[1].set_ylim(0, 115)\naxes[1].axhline(y=92.60, color='red',\n                linestyle='--', lw=2,\n                label='Wiratama 92.60%')\naxes[1].axhline(y=93.00, color='green',\n                linestyle='--', lw=2,\n                label='Target 93%')\naxes[1].set_title('Per-Class Accuracy (%)')\naxes[1].legend(fontsize=8)\nfor b, v in zip(bars, per_class):\n    axes[1].text(\n        b.get_x()+b.get_width()/2,\n        b.get_height()+1,\n        f'{v:.1f}%', ha='center',\n        fontsize=9, fontweight='bold'\n    )\n\n# Overall metrics\nnames = ['Accuracy','Precision','Recall','F1']\nvals  = [acc, prec, rec, f1]\nbars2 = axes[2].bar(names, vals,\n                    color=['#4C72B0','#DD8452',\n                           '#55A868','#C44E52'],\n                    width=0.5)\naxes[2].set_ylim(0, 1.15)\naxes[2].set_title('Overall Metrics')\nfor b, v in zip(bars2, vals):\n    axes[2].text(\n        b.get_x()+b.get_width()/2,\n        b.get_height()+0.02,\n        f'{v:.4f}', ha='center', fontsize=10\n    )\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T17:06:37.275858Z","iopub.execute_input":"2026-05-26T17:06:37.276173Z","iopub.status.idle":"2026-05-26T17:06:37.762638Z","shell.execute_reply.started":"2026-05-26T17:06:37.276152Z","shell.execute_reply":"2026-05-26T17:06:37.761796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib, torch\n\n# Save the SVM model\njoblib.dump(svm, 'svm_model.pkl')\n\n# Save the scaler\njoblib.dump(scaler, 'scaler.pkl')\n\n# Save the EfficientNetB0 feature extractor weights\ntorch.save(extractor.state_dict(), 'efficientnet_extractor.pth')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-26T17:06:37.763710Z","iopub.execute_input":"2026-05-26T17:06:37.764036Z","iopub.status.idle":"2026-05-26T17:06:37.857157Z","shell.execute_reply.started":"2026-05-26T17:06:37.764013Z","shell.execute_reply":"2026-05-26T17:06:37.856225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}