{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU"},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Evaluate the deployed model — no retraining\n\nPulls the trained model from the HuggingFace Hub (`suhanii23/retinopathy-model`)\nand evaluates it on the APTOS 2019 validation split.\n\n**Why this is valid without retraining:** the split is fully deterministic —\n`test_size=0.15`, `random_state=2006`, `stratify=y`. This notebook reconstructs\nthe identical 550-image validation set the model was originally selected on, so\nthe metrics it produces are the true metrics for this model.\n\n### Before you run\n1. **Add Data** -> search `aptos2019-blindness-detection` -> Add\n2. **Settings -> Accelerator -> GPU P100** (optional; ~3x faster)\n3. **Run All** (about 10-15 minutes, nearly all of it preprocessing)\n\n### When it finishes\nSend back `metrics.json` and the classification report.","metadata":{}},{"cell_type":"code","source":"!pip install -q huggingface_hub\n\nimport os, json\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tqdm.auto import tqdm\n\n%matplotlib inline\nprint(\"TensorFlow:\", tf.__version__)\n\nDATA_DIR = '/kaggle/input/competitions/aptos2019-blindness-detection'\nIMAGE_SIZE   = 299\nBATCH_SIZE   = 16\nVAL_FRACTION = 0.15\nSEED         = 2006          # must match the original training run\nCLASS_NAMES  = ['No DR', 'Mild', 'Moderate', 'Severe', 'Proliferative DR']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:15:22.602980Z","iopub.execute_input":"2026-07-27T14:15:22.603923Z","iopub.status.idle":"2026-07-27T14:15:26.820697Z","shell.execute_reply.started":"2026-07-27T14:15:22.603823Z","shell.execute_reply":"2026-07-27T14:15:26.819597Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Download the model from the Hub\n\nThis is the exact file your live Space loads, so whatever we measure here is\nwhat your deployed app is actually doing.","metadata":{}},{"cell_type":"code","source":"import os, shutil, numpy as np\nfrom huggingface_hub import hf_hub_download\nfrom tensorflow.keras.models import load_model\n\nREPO_ID = \"suhanii23/retinopathy-model\"\n\npath = hf_hub_download(repo_id=REPO_ID, filename=\"diabetic_retinopathy_model.keras\")\nprint(f\"Size: {os.path.getsize(path)/1024/1024:.1f} MB\")\n\n# The file is named .keras but is actually legacy HDF5. Keras 3 validates that a\n# .keras file is a zip archive and refuses to open it. Copying to a .h5 extension\n# routes it through the legacy HDF5 loader instead.\nh5_path = \"/kaggle/working/_model_legacy.h5\"\nshutil.copy(path, h5_path)\n\ntry:\n    model = load_model(h5_path, compile=False)\n    print(\"Loaded via Keras 3 legacy h5\")\nexcept Exception as e:\n    print(f\"Keras 3 failed: {type(e).__name__}: {str(e)[:200]}\")\n    os.system(\"pip install -q tf-keras\")\n    import tf_keras\n    model = tf_keras.models.load_model(h5_path, compile=False)\n    print(\"Loaded via tf_keras\")\n\nMODEL_PATH = h5_path\nprint(\"\\nOutput shape:\", model.output_shape)\nassert model.output_shape[-1] == 5\n\nprobe = model.predict(np.zeros((1, 299, 299, 3), dtype=np.float32), verbose=0)\nprint(f\"Probe sums to {probe.sum():.4f} -> {'softmax confirmed' if abs(probe.sum()-1) < 0.01 else 'NOT softmax'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:15:26.822913Z","iopub.execute_input":"2026-07-27T14:15:26.823419Z","iopub.status.idle":"2026-07-27T14:15:35.027616Z","shell.execute_reply.started":"2026-07-27T14:15:26.823383Z","shell.execute_reply":"2026-07-27T14:15:35.026044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Preprocessing\n\nIdentical to training: crop black border -> resize 299 -> Ben Graham high-pass\n(`4*I - 4*blur(I) + 128`) -> scale to **[-1, 1]** via Xception's `preprocess_input`.\n\nThe [-1, 1] range matters. Xception was pretrained on inputs in that range;\nusing [0, 1] shifts every input off the distribution its filters expect. (Your\ndeployed `app.py` was doing exactly that — see the action plan.)","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.xception import preprocess_input\n\ndef crop_image_from_gray(img, tol=7):\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    mask = gray > tol\n    if img[:, :, 0][np.ix_(mask.any(1), mask.any(0))].shape[0] == 0:\n        return img\n    return np.stack([img[:, :, i][np.ix_(mask.any(1), mask.any(0))] for i in range(3)], axis=-1)\n\ndef preprocess_image(path, sigma_x=10):\n    image = cv2.imread(path)\n    if image is None:\n        raise FileNotFoundError(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMAGE_SIZE, IMAGE_SIZE))\n    image = cv2.addWeighted(image, 4, cv2.GaussianBlur(image, (0, 0), sigma_x), -4, 128)\n    return preprocess_input(image.astype(np.float32))\n\ndef for_display(img):\n    return np.clip((img + 1) / 2, 0, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:15:35.029035Z","iopub.execute_input":"2026-07-27T14:15:35.029461Z","iopub.status.idle":"2026-07-27T14:15:35.040189Z","shell.execute_reply.started":"2026-07-27T14:15:35.029430Z","shell.execute_reply":"2026-07-27T14:15:35.039072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Rebuild the exact validation split","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_df = pd.read_csv(f'{DATA_DIR}/train.csv')\ntrain_df['diagnosis'] = train_df['diagnosis'].astype(int)\nprint(f\"Total images: {len(train_df)}\")\n\nX = np.empty((len(train_df), IMAGE_SIZE, IMAGE_SIZE, 3), dtype=np.float32)\nfor i, image_id in enumerate(tqdm(train_df['id_code'], desc='Preprocessing')):\n    X[i] = preprocess_image(f'{DATA_DIR}/train_images/{image_id}.png')\n\ny = train_df['diagnosis'].values\n\n_, x_val, _, y_val = train_test_split(\n    X, y, test_size=VAL_FRACTION, random_state=SEED, stratify=y)\n\ndel X   # free ~3.9 GB\n\nprint(f\"\\nValidation set: {len(x_val)} images\")\nfor i, name in enumerate(CLASS_NAMES):\n    print(f\"  {name:18s} {int((y_val == i).sum()):4d}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:15:35.042628Z","iopub.execute_input":"2026-07-27T14:15:35.043479Z","iopub.status.idle":"2026-07-27T14:27:47.404115Z","shell.execute_reply.started":"2026-07-27T14:15:35.043443Z","shell.execute_reply":"2026-07-27T14:27:47.402317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor root, dirs, files in os.walk('/kaggle/input'):\n    depth = root.count('/') - 2\n    if depth > 2: continue\n    print('  ' * depth + os.path.basename(root) + '/')\n    for f in files[:5]:\n        print('  ' * (depth+1) + f)\n    if len(files) > 5:\n        print('  ' * (depth+1) + f'... +{len(files)-5} more')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:27:47.406896Z","iopub.execute_input":"2026-07-27T14:27:47.407412Z","iopub.status.idle":"2026-07-27T14:27:49.118226Z","shell.execute_reply.started":"2026-07-27T14:27:47.407361Z","shell.execute_reply":"2026-07-27T14:27:49.116925Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. The numbers\n\n**QWK is the primary metric, not accuracy.** The labels are ordinal, so confusing\nNo DR with Proliferative is far worse than confusing Mild with Moderate — QWK\nweights errors by `(i-j)^2`. And with 49% of images being No DR, a model that\npredicts No DR for everything gets ~49% accuracy; kappa is chance-corrected so\nthat baseline scores ~0.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import (accuracy_score, classification_report,\n                             cohen_kappa_score, confusion_matrix)\n\nprobs  = model.predict(x_val, batch_size=BATCH_SIZE, verbose=1)\ny_pred = np.argmax(probs, axis=1)      # softmax head -> argmax\n\nacc = accuracy_score(y_val, y_pred)\nqwk = cohen_kappa_score(y_val, y_pred, weights='quadratic')\n\nprint(\"\\n\" + \"=\" * 62)\nprint(f\"  VALIDATION ACCURACY        : {acc:.4f}   ({acc*100:.2f}%)\")\nprint(f\"  QUADRATIC WEIGHTED KAPPA   : {qwk:.4f}\")\nprint(f\"  Validation set size        : {len(y_val)}\")\nprint(\"=\" * 62)\nprint(\"\\n>>> These two numbers go on the README and the resume. <<<\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:27:49.120044Z","iopub.execute_input":"2026-07-27T14:27:49.120480Z","iopub.status.idle":"2026-07-27T14:29:36.004552Z","shell.execute_reply.started":"2026-07-27T14:27:49.120437Z","shell.execute_reply":"2026-07-27T14:29:36.003127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(classification_report(y_val, y_pred, target_names=CLASS_NAMES, digits=3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:29:36.006128Z","iopub.execute_input":"2026-07-27T14:29:36.006565Z","iopub.status.idle":"2026-07-27T14:29:36.028484Z","shell.execute_reply.started":"2026-07-27T14:29:36.006522Z","shell.execute_reply":"2026-07-27T14:29:36.027361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What the old app.py was doing\n\nDemonstrating the deployed bug on your real validation predictions: the Space used\n`int((probs > 0.5).sum())` — ordinal thresholding — against a softmax head.","metadata":{}},{"cell_type":"code","source":"buggy = np.minimum((probs > 0.5).sum(axis=1), 4)\n\nprint(\"Class distribution of predictions:\\n\")\nprint(f\"{'':18s} {'correct (argmax)':>18s} {'deployed (>0.5 sum)':>22s}\")\nfor i, name in enumerate(CLASS_NAMES):\n    print(f\"{name:18s} {int((y_pred == i).sum()):>18d} {int((buggy == i).sum()):>22d}\")\n\nprint(f\"\\nAccuracy with correct argmax logic : {accuracy_score(y_val, y_pred):.4f}\")\nprint(f\"Accuracy with the deployed logic   : {accuracy_score(y_val, buggy):.4f}\")\nprint(f\"\\nClasses reachable by deployed app  : {sorted(set(buggy.tolist()))}\")\nprint(\"(only two of five -- Moderate, Severe and Proliferative were impossible)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:29:36.029941Z","iopub.execute_input":"2026-07-27T14:29:36.030241Z","iopub.status.idle":"2026-07-27T14:29:36.041955Z","shell.execute_reply.started":"2026-07-27T14:29:36.030214Z","shell.execute_reply":"2026-07-27T14:29:36.040927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Confusion matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_val, y_pred)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(17, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax1,\n            xticklabels=CLASS_NAMES, yticklabels=CLASS_NAMES)\nax1.set_title('Confusion Matrix (counts)')\nax1.set_ylabel('True'); ax1.set_xlabel('Predicted')\n\ncm_norm = cm.astype(float) / cm.sum(axis=1, keepdims=True)\nsns.heatmap(cm_norm, annot=True, fmt='.2f', cmap='Blues', ax=ax2, vmin=0, vmax=1,\n            xticklabels=CLASS_NAMES, yticklabels=CLASS_NAMES)\nax2.set_title('Row-normalised (= per-class recall)')\nax2.set_ylabel('True'); ax2.set_xlabel('Predicted')\n\nplt.tight_layout()\nplt.savefig('confusion_matrix.png', dpi=120)\nplt.show()\n\n# How far off are the errors? This is what QWK rewards and accuracy hides.\nerr = np.abs(y_val - y_pred); err = err[err > 0]\nif len(err):\n    print(f\"Of {len(err)} misclassifications:\")\n    for d in range(1, 5):\n        n = int((err == d).sum())\n        if n:\n            print(f\"  {d} class away: {n:4d}  ({100*n/len(err):5.1f}%)\")\n    print(f\"Mean error distance: {err.mean():.2f} classes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:29:36.043323Z","iopub.execute_input":"2026-07-27T14:29:36.043779Z","iopub.status.idle":"2026-07-27T14:29:37.190299Z","shell.execute_reply.started":"2026-07-27T14:29:36.043749Z","shell.execute_reply":"2026-07-27T14:29:37.189103Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Referable DR\n\nIn screening the operational question is not \"what exact grade?\" but \"does this\npatient need an ophthalmologist?\" — standard cutoff is Moderate or worse.\nThis is usually much stronger than the 5-class numbers suggest, because most\nerrors are between adjacent classes on the same side of the threshold.","metadata":{}},{"cell_type":"code","source":"yt = (y_val >= 2).astype(int)\nyp = (y_pred >= 2).astype(int)\n\ntp = int(((yt==1)&(yp==1)).sum()); tn = int(((yt==0)&(yp==0)).sum())\nfp = int(((yt==0)&(yp==1)).sum()); fn = int(((yt==1)&(yp==0)).sum())\n\nsens = tp/(tp+fn) if tp+fn else 0.0\nspec = tn/(tn+fp) if tn+fp else 0.0\n\nprint(\"Referable DR (Moderate or worse)\")\nprint(f\"  Sensitivity : {sens:.3f}   <- missing these is the costly error\")\nprint(f\"  Specificity : {spec:.3f}\")\nprint(f\"  TP={tp}  TN={tn}  FP={fp}  FN={fn}\")\n\nwith open('metrics.json', 'w') as f:\n    json.dump({\n        'val_accuracy': float(acc),\n        'qwk': float(qwk),\n        'n_val': int(len(y_val)),\n        'referable_sensitivity': float(sens),\n        'referable_specificity': float(spec),\n        'per_class_recall': {CLASS_NAMES[i]: float((y_pred[y_val==i]==i).mean())\n                             for i in range(5)},\n        'confusion_matrix': cm.tolist(),\n    }, f, indent=2)\n\nprint(\"\\nSaved metrics.json  <- send me this file\")\nprint(json.dumps(json.load(open('metrics.json')), indent=2)[:900])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:29:37.193415Z","iopub.execute_input":"2026-07-27T14:29:37.194445Z","iopub.status.idle":"2026-07-27T14:29:37.207932Z","shell.execute_reply.started":"2026-07-27T14:29:37.194410Z","shell.execute_reply":"2026-07-27T14:29:37.206785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Sample predictions","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(20, 10))\nfor i in range(8):\n    ax = plt.subplot(2, 8, 2*i+1)\n    ax.imshow(for_display(x_val[i])); ax.set_xticks([]); ax.set_yticks([])\n    ok = y_pred[i] == y_val[i]\n    ax.set_xlabel(f\"{CLASS_NAMES[y_pred[i]]} {100*probs[i].max():.0f}%\\n(true: {CLASS_NAMES[y_val[i]]})\",\n                  color='green' if ok else 'red', fontsize=8)\n    ax = plt.subplot(2, 8, 2*i+2)\n    bars = ax.bar(range(5), probs[i], color='#bbbbbb')\n    bars[y_pred[i]].set_color('red'); bars[y_val[i]].set_color('green')\n    ax.set_ylim([0,1]); ax.set_yticks([])\n    ax.set_xticks(range(5)); ax.set_xticklabels(CLASS_NAMES, rotation=90, fontsize=6)\nplt.tight_layout()\nplt.savefig('validation_predictions.png', dpi=120)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-27T14:29:37.209371Z","iopub.execute_input":"2026-07-27T14:29:37.209819Z","iopub.status.idle":"2026-07-27T14:29:39.085782Z","shell.execute_reply.started":"2026-07-27T14:29:37.209778Z","shell.execute_reply":"2026-07-27T14:29:39.084492Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Done\n\n`/kaggle/working/` now contains `metrics.json`, `confusion_matrix.png`,\n`validation_predictions.png`.\n\n**Send back `metrics.json` and the classification report output.**\n\nThen: **Save Version -> Save & Run All (Commit)**, and download from the completed\nversion in the Versions list so the outputs are preserved.","metadata":{}}]}