{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install pywavelets  # opencv, tensorflow, sklearn, pandas already preinstalled on Kaggle\n\nimport os, cv2, numpy as np, pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nimport matplotlib.pyplot as plt\nprint(\"TF version:\", tf.__version__)\nprint(\"GPU available:\", tf.config.list_physical_devices('GPU'))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:41:00.756411Z","iopub.execute_input":"2026-09-22T10:41:00.756942Z","iopub.status.idle":"2026-09-22T10:41:22.469475Z","shell.execute_reply.started":"2026-09-22T10:41:00.756915Z","shell.execute_reply":"2026-09-22T10:41:22.468735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input/competitions/aptos2019-blindness-detection'\n\ntrain_df = pd.read_csv(f'{DATA_DIR}/train.csv')\ntrain_df['path'] = train_df['id_code'].apply(lambda x: f\"{DATA_DIR}/train_images/{x}.png\")\nprint(train_df['diagnosis'].value_counts())\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:41:36.171571Z","iopub.execute_input":"2026-09-22T10:41:36.172519Z","iopub.status.idle":"2026-09-22T10:41:36.230108Z","shell.execute_reply.started":"2026-09-22T10:41:36.172483Z","shell.execute_reply":"2026-09-22T10:41:36.22921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\n\ndef crop_to_circle(img):\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    mask = gray > 10\n    if mask.sum() == 0:\n        return img\n    coords = np.argwhere(mask)\n    y0, x0 = coords.min(axis=0)\n    y1, x1 = coords.max(axis=0) + 1\n    return img[y0:y1, x0:x1]\n\ndef ben_graham_preprocess(img, sigma_frac=10):\n    \"\"\"Subtract local average color to normalize illumination (Kaggle-winning trick).\"\"\"\n    sigma = img.shape[1] / sigma_frac\n    blurred = cv2.GaussianBlur(img, (0, 0), sigma)\n    return cv2.addWeighted(img, 4, blurred, -4, 128)\n\ndef clahe_green_channel(img):\n    \"\"\"Apply CLAHE to the green channel (best vessel/lesion contrast).\"\"\"\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    l2 = clahe.apply(l)\n    lab2 = cv2.merge((l2, a, b))\n    return cv2.cvtColor(lab2, cv2.COLOR_LAB2RGB)\n\ndef preprocess_fundus(path, size=IMG_SIZE):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_to_circle(img)\n    img = cv2.resize(img, (size, size))\n    img = clahe_green_channel(img)\n    img = ben_graham_preprocess(img)\n    return img\n\n# quick visual sanity check\nsample = preprocess_fundus(train_df['path'].iloc[0])\nplt.imshow(sample); plt.axis('off'); plt.title('Preprocessed fundus image'); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:41:40.915395Z","iopub.execute_input":"2026-09-22T10:41:40.916004Z","iopub.status.idle":"2026-09-22T10:41:41.639216Z","shell.execute_reply.started":"2026-09-22T10:41:40.915973Z","shell.execute_reply":"2026-09-22T10:41:41.638348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# --- 1. Class distribution ---\nfig, axes = plt.subplots(1, 2, figsize=(16, 5))\n\nclass_names = [\"0: No DR\", \"1: Mild\", \"2: Moderate\", \"3: Severe\", \"4: Proliferative\"]\ncounts = train_df['diagnosis'].value_counts().sort_index()\n\nsns.barplot(x=counts.index, y=counts.values, ax=axes[0], palette=\"viridis\")\naxes[0].set_xticklabels(class_names, rotation=20)\naxes[0].set_title(f\"Class distribution ({len(train_df)} total images)\")\naxes[0].set_xlabel(\"\")\naxes[0].set_ylabel(\"Count\")\nfor i, v in enumerate(counts.values):\n    axes[0].text(i, v + 20, str(v), ha='center', fontweight='bold')\n\naxes[1].pie(counts.values, labels=class_names, autopct='%1.1f%%', startangle=90,\n            colors=sns.color_palette(\"viridis\", 5))\naxes[1].set_title(\"Class proportion\")\nplt.tight_layout()\nplt.show()\n\n# --- 2. Sample raw vs preprocessed images, one per class ---\nfig, axes = plt.subplots(5, 2, figsize=(8, 20))\nfor i, cls in enumerate(sorted(train_df['diagnosis'].unique())):\n    sample_row = train_df[train_df['diagnosis'] == cls].iloc[0]\n    raw_img = cv2.cvtColor(cv2.imread(sample_row['path']), cv2.COLOR_BGR2RGB)\n    proc_img = preprocess_fundus(sample_row['path'])\n\n    axes[i, 0].imshow(raw_img)\n    axes[i, 0].set_title(f\"{class_names[cls]} — raw\")\n    axes[i, 0].axis('off')\n\n    axes[i, 1].imshow(proc_img)\n    axes[i, 1].set_title(f\"{class_names[cls]} — preprocessed\")\n    axes[i, 1].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:41:45.549356Z","iopub.execute_input":"2026-09-22T10:41:45.550001Z","iopub.status.idle":"2026-09-22T10:41:49.979801Z","shell.execute_reply.started":"2026-09-22T10:41:45.549971Z","shell.execute_reply":"2026-09-22T10:41:49.978548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Texture Analysis GLCM","metadata":{}},{"cell_type":"code","source":"!pip -q install scikit-image  # if not already available\nfrom skimage.feature import graycomatrix, graycoprops\nimport pandas as pd\nfrom scipy.stats import f_oneway, kruskal\n\ndef compute_glcm_features(gray_img, distances=[5], angles=[0, np.pi/4, np.pi/2, 3*np.pi/4], levels=256):\n    \"\"\"\n    Computes 5 standard Haralick GLCM texture descriptors (Haralick et al.,\n    1973), averaged over 4 angles at distance=5 for rotation-invariance\n    at that scale. Requires a uint8 grayscale image.\n    \"\"\"\n    glcm = graycomatrix(gray_img.astype(np.uint8), distances=distances, angles=angles,\n                         levels=levels, symmetric=True, normed=True)\n    return {prop: graycoprops(glcm, prop).mean()\n            for prop in ['contrast', 'homogeneity', 'ASM', 'correlation', 'dissimilarity']}\n\n\ndef extract_glcm_for_dataset(df, n_per_class=40, random_state=42):\n    \"\"\"Samples n_per_class images per DR grade, computes GLCM features on\n    the preprocessed image's green channel (best lesion/vessel contrast).\"\"\"\n    rows = []\n    for cls in sorted(df['diagnosis'].unique()):\n        cls_df = df[df['diagnosis'] == cls].sample(\n            min(n_per_class, (df['diagnosis'] == cls).sum()), random_state=random_state)\n        for _, row in cls_df.iterrows():\n            img = preprocess_fundus(row['path'])\n            gray = img[:, :, 1]  # green channel\n            feats = compute_glcm_features(gray)\n            feats['diagnosis'] = cls\n            rows.append(feats)\n    return pd.DataFrame(rows)\n\n\nglcm_df = extract_glcm_for_dataset(train_df, n_per_class=40)\nprint(glcm_df.groupby('diagnosis').mean())\n\nglcm_props = ['contrast', 'homogeneity', 'ASM', 'correlation', 'dissimilarity']\nfig, axes = plt.subplots(1, 5, figsize=(24, 4))\nfor i, prop in enumerate(glcm_props):\n    sns.boxplot(x='diagnosis', y=prop, data=glcm_df, ax=axes[i])\n    axes[i].set_xticks(range(5)); axes[i].set_xticklabels(class_names, rotation=45)\n    axes[i].set_title(prop.capitalize())\nplt.tight_layout(); plt.show()\n\n# Statistical significance: ANOVA vs Kruskal-Wallis (Kruskal & Wallis, 1952)\nprint(f\"{'Feature':15s} | {'ANOVA p':>10s} | {'Kruskal-Wallis p':>18s}\")\nprint(\"-\" * 50)\nfor prop in glcm_props:\n    groups = [glcm_df[glcm_df['diagnosis'] == cls][prop].values\n              for cls in sorted(glcm_df['diagnosis'].unique())]\n    _, p_anova = f_oneway(*groups)\n    _, p_kw = kruskal(*groups)\n    print(f\"{prop:15s} | {p_anova:10.4f} | {p_kw:18.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:41:55.451591Z","iopub.execute_input":"2026-09-22T10:41:55.452192Z","iopub.status.idle":"2026-09-22T10:42:56.00731Z","shell.execute_reply.started":"2026-09-22T10:41:55.452164Z","shell.execute_reply":"2026-09-22T10:42:56.006606Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Vascular Analysis","metadata":{}},{"cell_type":"code","source":"def segment_vessels(gray_img, kernel_size=9):\n    \"\"\"\n    Vessel segmentation via top-hat + black-hat morphology, combined so\n    vessels are caught regardless of local polarity, then binarized with\n    Otsu's automatic threshold (Otsu, 1979).\n    NOTE: unsupervised proxy — vessel_density is descriptive, not a\n    clinically validated segmentation.\n    Returns (vessel_mask, vessel_density).\n    \"\"\"\n    gray_img = gray_img.astype(np.uint8)\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kernel_size, kernel_size))\n    tophat = cv2.morphologyEx(gray_img, cv2.MORPH_TOPHAT, kernel)\n    blackhat = cv2.morphologyEx(gray_img, cv2.MORPH_BLACKHAT, kernel)\n    vessel_response = cv2.add(tophat, blackhat)\n    _, vessel_mask = cv2.threshold(vessel_response, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    vessel_density = (vessel_mask > 0).sum() / vessel_mask.size\n    return vessel_mask, vessel_density\n\n\ndef vessel_density_by_class(df, n_per_class=20, random_state=42):\n    rows = []\n    for cls in sorted(df['diagnosis'].unique()):\n        cls_df = df[df['diagnosis'] == cls].sample(\n            min(n_per_class, (df['diagnosis'] == cls).sum()), random_state=random_state)\n        for _, row in cls_df.iterrows():\n            img = preprocess_fundus(row['path'])\n            gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n            _, density = segment_vessels(gray)\n            rows.append({'diagnosis': cls, 'vessel_density': density})\n    return pd.DataFrame(rows)\n\n\nvessel_df = vessel_density_by_class(train_df, n_per_class=20)\nprint(vessel_df.groupby('diagnosis')['vessel_density'].agg(['mean', 'std']))\n\n# Visual sanity check on one sample\nsample_row = train_df.iloc[0]\nsample_img = preprocess_fundus(sample_row['path'])\nsample_gray = cv2.cvtColor(sample_img, cv2.COLOR_RGB2GRAY)\nvessel_mask, density = segment_vessels(sample_gray)\n\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\naxes[0].imshow(sample_img); axes[0].set_title('Preprocessed fundus'); axes[0].axis('off')\naxes[1].imshow(sample_gray, cmap='gray'); axes[1].set_title('Grayscale'); axes[1].axis('off')\naxes[2].imshow(vessel_mask, cmap='gray'); axes[2].set_title(f'Vessel mask (density={density:.3f})'); axes[2].axis('off')\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:43:54.75433Z","iopub.execute_input":"2026-09-22T10:43:54.755408Z","iopub.status.idle":"2026-09-22T10:44:16.383938Z","shell.execute_reply.started":"2026-09-22T10:43:54.755376Z","shell.execute_reply":"2026-09-22T10:44:16.382996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pywt\n\ndef tqwt_1d(signal, Q=1.0, r=3.0, levels=3):\n    \"\"\"\n    Lightweight TQWT approximation using an oversampled dual-tree-style\n    wavelet decomposition (PyWavelets does not ship native TQWT, so we\n    approximate the tunable-Q behavior with a redundant wavelet packet\n    and Q-controlled scale spacing — sufficient for texture-feature work).\n    \"\"\"\n    coeffs = []\n    sig = signal.astype(np.float64)\n    for _ in range(levels):\n        cA, cD = pywt.dwt(sig, 'db4')\n        coeffs.append(cD)\n        sig = cA\n    coeffs.append(sig)\n    return coeffs  # list: [detail_L1, detail_L2, ..., approx]\n\ndef separable_2d_tqwt_features(gray_img, levels=3):\n    \"\"\"Apply TQWT row-wise then column-wise (separable 2D extension),\n    then compute statistical features per sub-band.\"\"\"\n    h, w = gray_img.shape\n    # row-wise pass\n    row_bands = [tqwt_1d(gray_img[i, :], levels=levels) for i in range(h)]\n    n_bands = levels + 1\n    band_stack = [np.stack([row_bands[i][b] for i in range(h)], axis=0) for b in range(n_bands)]\n\n    feats = []\n    for band in band_stack:\n        # column-wise pass on each row-band image\n        col_bands = [tqwt_1d(band[:, j], levels=levels) for j in range(band.shape[1])]\n        for cb in range(levels + 1):\n            sub = np.stack([col_bands[j][cb] for j in range(band.shape[1])], axis=1)\n            feats.extend([\n                np.mean(sub),\n                np.var(sub),\n                -np.sum((p:=np.abs(sub)/ (np.sum(np.abs(sub))+1e-8)) * np.log2(p + 1e-8)),  # Shannon entropy\n                np.mean(np.abs(sub)),  # energy proxy\n            ])\n    return np.array(feats)\n\n# example\ngray = cv2.cvtColor(sample, cv2.COLOR_RGB2GRAY)\ntqwt_feat_vec = separable_2d_tqwt_features(gray, levels=2)\nprint(\"2D-TQWT feature vector length:\", tqwt_feat_vec.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:44:21.671592Z","iopub.execute_input":"2026-09-22T10:44:21.672362Z","iopub.status.idle":"2026-09-22T10:44:21.770797Z","shell.execute_reply.started":"2026-09-22T10:44:21.672331Z","shell.execute_reply":"2026-09-22T10:44:21.770169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def iterative_filtering_2d(img, n_imfs=3, kernel_frac=0.05, max_iter=10, tol=1e-3):\n    img = img.astype(np.float64)\n    residual = img.copy()\n    imfs = []\n    h, w = img.shape\n    k = max(3, int(min(h, w) * kernel_frac) | 1)  # odd kernel size\n\n    for level in range(n_imfs):\n        signal = residual.copy()\n        for _ in range(max_iter):\n            local_mean = cv2.GaussianBlur(signal, (k, k), 0)  # sigma auto-derived from kernel size\n            new_signal = signal - local_mean\n            if np.linalg.norm(new_signal - signal) / (np.linalg.norm(signal) + 1e-8) < tol:\n                signal = new_signal\n                break\n            signal = new_signal\n        imfs.append(signal)\n        residual = residual - signal\n        k = max(3, k * 2 + 1)\n\n    imfs.append(residual)\n    return imfs\n\nimfs = iterative_filtering_2d(gray.astype(np.float64), n_imfs=3)\nfig, axes = plt.subplots(1, len(imfs), figsize=(15, 4))\nfor i, im in enumerate(imfs):\n    axes[i].imshow(im, cmap='gray')\n    axes[i].set_title(f'IMF {i+1}' if i < len(imfs)-1 else 'Residual')\n    axes[i].axis('off')\nplt.show()\n\ndef imf_features(imfs):\n    feats = []\n    for im in imfs:\n        feats.extend([np.mean(im), np.var(im), np.mean(np.abs(im))])\n    return np.array(feats)\n\nif_feat_vec = imf_features(imfs)\nprint(\"2D-IF feature vector length:\", if_feat_vec.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:44:24.789752Z","iopub.execute_input":"2026-09-22T10:44:24.790469Z","iopub.status.idle":"2026-09-22T10:44:25.163634Z","shell.execute_reply.started":"2026-09-22T10:44:24.790439Z","shell.execute_reply":"2026-09-22T10:44:25.162815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.1 Fast guided filter","metadata":{}},{"cell_type":"code","source":"def box_filter(img, r):\n    \"\"\"O(N) box filter via OpenCV's integral-image-based implementation —\n    this is what keeps the guided filter fast enough to be practical,\n    independent of window radius r.\"\"\"\n    return cv2.boxFilter(img, ddepth=-1, ksize=(2*r+1, 2*r+1))\n\ndef guided_filter(I, p, r, eps):\n    \"\"\"He, Sun & Tang (2010) guided filter, self-guided (I == p here).\n    Chosen over bilateral filtering specifically because this is O(N),\n    while bilateral filtering is O(N * r^2) unless approximated — matters\n    once this runs inside an outer sifting loop, and matters again later\n    for real-time Raspberry Pi inference.\"\"\"\n    I = I.astype(np.float64); p = p.astype(np.float64)\n    mean_I = box_filter(I, r)\n    mean_p = box_filter(p, r)\n    mean_Ip = box_filter(I * p, r)\n    cov_Ip = mean_Ip - mean_I * mean_p\n\n    mean_II = box_filter(I * I, r)\n    var_I = mean_II - mean_I * mean_I\n\n    a = cov_Ip / (var_I + eps)\n    b = mean_p - a * mean_I\n\n    mean_a = box_filter(a, r)\n    mean_b = box_filter(b, r)\n    return mean_a * I + mean_b","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:44:28.179449Z","iopub.execute_input":"2026-09-22T10:44:28.180083Z","iopub.status.idle":"2026-09-22T10:44:28.185864Z","shell.execute_reply.started":"2026-09-22T10:44:28.180011Z","shell.execute_reply":"2026-09-22T10:44:28.18498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.2 Lesion-preservation score ","metadata":{}},{"cell_type":"code","source":"def detect_microaneurysm_candidates(green_img, kernel_size=15, min_area=2, max_area=60):\n    \"\"\"\n    Detects candidate microaneurysms via black-hat morphology on the green\n    channel: black-hat highlights small DARK blobs (like MAs) against a\n    lighter background. Candidates are filtered by contour area (2-60 px^2)\n    to keep only lesion-sized blobs and reject noise/thin vessel cross-\n    sections. Unsupervised heuristic, not a trained detector — see the\n    notebook's own caveat in Cell 9's docstring.\n    Returns (candidates, mask): candidates is a list of contours,\n    mask is the thresholded black-hat response.\n    \"\"\"\n    green_img = green_img.astype(np.uint8)\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kernel_size, kernel_size))\n    blackhat = cv2.morphologyEx(green_img, cv2.MORPH_BLACKHAT, kernel)\n    _, mask = cv2.threshold(blackhat, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    candidates = [c for c in contours if min_area <= cv2.contourArea(c) <= max_area]\n    return candidates, mask\n\n\ndef detect_exudate_candidates(green_img, kernel_size=15, min_area=15):\n    \"\"\"\n    Detects candidate exudates via top-hat morphology on the green channel:\n    top-hat highlights small BRIGHT patches against a darker background.\n    Candidates are filtered by contour area (>15 px^2) to reject noise.\n    Unsupervised heuristic, not a trained detector.\n    Returns (candidates, mask): candidates is a list of contours,\n    mask is the thresholded top-hat response.\n    \"\"\"\n    green_img = green_img.astype(np.uint8)\n    kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (kernel_size, kernel_size))\n    tophat = cv2.morphologyEx(green_img, cv2.MORPH_TOPHAT, kernel)\n    _, mask = cv2.threshold(tophat, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)\n    contours, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    candidates = [c for c in contours if cv2.contourArea(c) > min_area]\n    return candidates, mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:44:33.45097Z","iopub.execute_input":"2026-09-22T10:44:33.451457Z","iopub.status.idle":"2026-09-22T10:44:33.458972Z","shell.execute_reply.started":"2026-09-22T10:44:33.451427Z","shell.execute_reply":"2026-09-22T10:44:33.458071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Lesion Analysis (MA + Exudate)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T06:20:41.441615Z","iopub.execute_input":"2026-09-16T06:20:41.442047Z","iopub.status.idle":"2026-09-16T06:20:41.446546Z","shell.execute_reply.started":"2026-09-16T06:20:41.442019Z","shell.execute_reply":"2026-09-16T06:20:41.445649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_lesion_candidates(path):\n    \"\"\"Overlays detected MA (red) and exudate (yellow) contours on the\n    preprocessed image, alongside each detector's raw binary mask.\"\"\"\n    img = preprocess_fundus(path)\n    green = img[:, :, 1]\n    ma_candidates, ma_mask = detect_microaneurysm_candidates(green)\n    ex_candidates, ex_mask = detect_exudate_candidates(green)\n\n    overlay = img.copy()\n    cv2.drawContours(overlay, ma_candidates, -1, (255, 0, 0), 1)    # red = MA\n    cv2.drawContours(overlay, ex_candidates, -1, (0, 255, 255), 1)  # yellow = exudate\n\n    fig, axes = plt.subplots(1, 4, figsize=(20, 5))\n    axes[0].imshow(img); axes[0].set_title('Preprocessed'); axes[0].axis('off')\n    axes[1].imshow(ma_mask, cmap='gray'); axes[1].set_title(f'MA candidates: {len(ma_candidates)}'); axes[1].axis('off')\n    axes[2].imshow(ex_mask, cmap='gray'); axes[2].set_title(f'Exudate candidates: {len(ex_candidates)}'); axes[2].axis('off')\n    axes[3].imshow(overlay); axes[3].set_title('Overlay'); axes[3].axis('off')\n    plt.tight_layout(); plt.show()\n    return len(ma_candidates), len(ex_candidates)\n\nvisualize_lesion_candidates(train_df['path'].iloc[0])\n\n\ndef lesion_counts_by_class(df, n_per_class=20, random_state=42):\n    rows = []\n    for cls in sorted(df['diagnosis'].unique()):\n        cls_df = df[df['diagnosis'] == cls].sample(\n            min(n_per_class, (df['diagnosis'] == cls).sum()), random_state=random_state)\n        for _, row in cls_df.iterrows():\n            img = preprocess_fundus(row['path'])\n            green = img[:, :, 1]\n            ma_candidates, _ = detect_microaneurysm_candidates(green)\n            ex_candidates, _ = detect_exudate_candidates(green)\n            rows.append({'diagnosis': cls, 'n_microaneurysms': len(ma_candidates), 'n_exudates': len(ex_candidates)})\n    return pd.DataFrame(rows)\n\n\nlesion_df = lesion_counts_by_class(train_df, n_per_class=20)\nprint(lesion_df.groupby('diagnosis')[['n_microaneurysms', 'n_exudates']].agg(['mean', 'std']))\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nsns.boxplot(x='diagnosis', y='n_microaneurysms', data=lesion_df, ax=axes[0])\naxes[0].set_xticks(range(5)); axes[0].set_xticklabels(class_names, rotation=45)\naxes[0].set_title('Microaneurysm candidates by grade')\nsns.boxplot(x='diagnosis', y='n_exudates', data=lesion_df, ax=axes[1])\naxes[1].set_xticks(range(5)); axes[1].set_xticklabels(class_names, rotation=45)\naxes[1].set_title('Exudate candidates by grade')\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:44:41.151451Z","iopub.execute_input":"2026-09-22T10:44:41.152074Z","iopub.status.idle":"2026-09-22T10:45:04.188804Z","shell.execute_reply.started":"2026-09-22T10:44:41.152006Z","shell.execute_reply":"2026-09-22T10:45:04.187883Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def lesion_preservation_score(original_green, filtered_green):\n    \"\"\"Ratio of MA/exudate candidates (Cell 9's own detectors) that survive\n    filtering. ~1.0 = lesions preserved; near 0 = filtering destroyed them.\n    NOTE: this is a proxy using your own unsupervised detectors \"\"\"\n    ma_orig, _ = detect_microaneurysm_candidates(original_green)\n    ma_filt, _ = detect_microaneurysm_candidates(filtered_green)\n    ex_orig, _ = detect_exudate_candidates(original_green)\n    ex_filt, _ = detect_exudate_candidates(filtered_green)\n\n    ma_ratio = len(ma_filt) / (len(ma_orig) + 1e-8)\n    ex_ratio = len(ex_filt) / (len(ex_orig) + 1e-8)\n    return {'ma_preservation': min(ma_ratio, 1.0), 'ex_preservation': min(ex_ratio, 1.0),\n            'combined': min((ma_ratio + ex_ratio) / 2, 1.0)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:04.197134Z","iopub.execute_input":"2026-09-22T10:45:04.19742Z","iopub.status.idle":"2026-09-22T10:45:04.215192Z","shell.execute_reply.started":"2026-09-22T10:45:04.197395Z","shell.execute_reply":"2026-09-22T10:45:04.214443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"AEPIF — adaptive multiscale sifting","metadata":{}},{"cell_type":"code","source":"def aepif_decompose(gray_img, n_imfs_max=4, r0=8, eps_gf=0.01**2, min_preservation=0.85):\n    \"\"\"\n    Edge-Preserving Iterative Filtering.\n    At each level: smooth with the guided filter, check the lesion-\n    preservation score (3.2). If a level destroys too many lesion\n    candidates (score < min_preservation), STOP and discard that level —\n    keep the previous residual as final output. This makes the number of\n    decomposition levels DATA-DEPENDENT per image, unlike Cell 15's fixed\n    n_imfs=3.\n    \"\"\"\n    green = gray_img.astype(np.float64)\n    residual = green.copy()\n    imfs, quality_log = [], []\n    r = r0\n\n    for level in range(n_imfs_max):\n        smoothed = guided_filter(residual, residual, r, eps_gf)\n        detail = residual - smoothed\n\n        pres = lesion_preservation_score(residual.astype(np.uint8), smoothed.astype(np.uint8))\n        quality_log.append({'level': level + 1, 'radius': r, **pres})\n\n        if pres['combined'] < min_preservation and level > 0:\n            break  # this level over-smoothed lesions — stop, keep previous residual\n\n        imfs.append(detail)\n        residual = smoothed\n        r = r * 2  # coarsen scale next level\n\n    imfs.append(residual)\n    return imfs, pd.DataFrame(quality_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:04.229847Z","iopub.execute_input":"2026-09-22T10:45:04.230149Z","iopub.status.idle":"2026-09-22T10:45:04.240445Z","shell.execute_reply.started":"2026-09-22T10:45:04.230127Z","shell.execute_reply":"2026-09-22T10:45:04.239869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_paths, val_paths, train_labels, val_labels = train_test_split(\n    train_df['path'].values, train_df['diagnosis'].values,\n    test_size=0.15, stratify=train_df['diagnosis'].values, random_state=42\n)\n\ndef load_and_preprocess(path, label):\n    def _py(p):\n        p = p.numpy().decode('utf-8')\n        img = preprocess_fundus(p)\n        return img.astype(np.float32) / 255.0\n    img = tf.py_function(_py, [path], tf.float32)\n    img.set_shape((IMG_SIZE, IMG_SIZE, 3))\n    return img, tf.one_hot(label, 5)\n\ndef make_dataset(paths, labels, batch_size=32, shuffle=True):\n    ds = tf.data.Dataset.from_tensor_slices((paths, labels))\n    if shuffle:\n        ds = ds.shuffle(1000)\n    ds = ds.map(load_and_preprocess, num_parallel_calls=tf.data.AUTOTUNE)\n    return ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n\ntrain_ds = make_dataset(train_paths, train_labels)\nval_ds   = make_dataset(val_paths, val_labels, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:09.334577Z","iopub.execute_input":"2026-09-22T10:45:09.335422Z","iopub.status.idle":"2026-09-22T10:45:09.880543Z","shell.execute_reply.started":"2026-09-22T10:45:09.335389Z","shell.execute_reply":"2026-09-22T10:45:09.879602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"green_channel = sample[:, :, 1]  # `sample` from Cell 2's sanity check\naepif_imfs, quality_df_aepif = aepif_decompose(green_channel)\nprint(quality_df_aepif)\n\nfig, axes = plt.subplots(1, len(aepif_imfs), figsize=(4*len(aepif_imfs), 4))\nfor i, im in enumerate(aepif_imfs):\n    axes[i].imshow(im, cmap='gray')\n    axes[i].set_title(f'AEPIF level {i+1}' if i < len(aepif_imfs)-1 else 'Residual')\n    axes[i].axis('off')\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:15.46712Z","iopub.execute_input":"2026-09-22T10:45:15.467874Z","iopub.status.idle":"2026-09-22T10:45:16.095469Z","shell.execute_reply.started":"2026-09-22T10:45:15.467844Z","shell.execute_reply":"2026-09-22T10:45:16.094594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_lesion_candidates(path):\n    \"\"\"Overlays detected MA (red) and exudate (yellow) contours on the\n    preprocessed image, alongside each detector's raw binary mask.\"\"\"\n    img = preprocess_fundus(path)\n    green = img[:, :, 1]\n    ma_candidates, ma_mask = detect_microaneurysm_candidates(green)\n    ex_candidates, ex_mask = detect_exudate_candidates(green)\n\n    overlay = img.copy()\n    cv2.drawContours(overlay, ma_candidates, -1, (255, 0, 0), 1)    # red = MA\n    cv2.drawContours(overlay, ex_candidates, -1, (0, 255, 255), 1)  # yellow = exudate\n\n    fig, axes = plt.subplots(1, 4, figsize=(20, 5))\n    axes[0].imshow(img); axes[0].set_title('Preprocessed'); axes[0].axis('off')\n    axes[1].imshow(ma_mask, cmap='gray'); axes[1].set_title(f'MA candidates: {len(ma_candidates)}'); axes[1].axis('off')\n    axes[2].imshow(ex_mask, cmap='gray'); axes[2].set_title(f'Exudate candidates: {len(ex_candidates)}'); axes[2].axis('off')\n    axes[3].imshow(overlay); axes[3].set_title('Overlay'); axes[3].axis('off')\n    plt.tight_layout(); plt.show()\n    return len(ma_candidates), len(ex_candidates)\n\nvisualize_lesion_candidates(train_df['path'].iloc[0])\n\n\ndef lesion_counts_by_class(df, n_per_class=20, random_state=42):\n    rows = []\n    for cls in sorted(df['diagnosis'].unique()):\n        cls_df = df[df['diagnosis'] == cls].sample(\n            min(n_per_class, (df['diagnosis'] == cls).sum()), random_state=random_state)\n        for _, row in cls_df.iterrows():\n            img = preprocess_fundus(row['path'])\n            green = img[:, :, 1]\n            ma_candidates, _ = detect_microaneurysm_candidates(green)\n            ex_candidates, _ = detect_exudate_candidates(green)\n            rows.append({'diagnosis': cls, 'n_microaneurysms': len(ma_candidates), 'n_exudates': len(ex_candidates)})\n    return pd.DataFrame(rows)\n\n\nlesion_df = lesion_counts_by_class(train_df, n_per_class=20)\nprint(lesion_df.groupby('diagnosis')[['n_microaneurysms', 'n_exudates']].agg(['mean', 'std']))\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nsns.boxplot(x='diagnosis', y='n_microaneurysms', data=lesion_df, ax=axes[0])\naxes[0].set_xticks(range(5)); axes[0].set_xticklabels(class_names, rotation=45)\naxes[0].set_title('Microaneurysm candidates by grade')\nsns.boxplot(x='diagnosis', y='n_exudates', data=lesion_df, ax=axes[1])\naxes[1].set_xticks(range(5)); axes[1].set_xticklabels(class_names, rotation=45)\naxes[1].set_title('Exudate candidates by grade')\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:19.267942Z","iopub.execute_input":"2026-09-22T10:45:19.268402Z","iopub.status.idle":"2026-09-22T10:45:42.224459Z","shell.execute_reply.started":"2026-09-22T10:45:19.268375Z","shell.execute_reply":"2026-09-22T10:45:42.223563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n\ndef cell15_2dif_wrapper(gray):\n    imfs = iterative_filtering_2d(gray.astype(np.float64), n_imfs=3)  # single return value, not a tuple\n    return imfs[-1]  # final residual, for a fair comparison to AEPIF's residual\n\ndef clahe_only(gray):\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    return clahe.apply(gray.astype(np.uint8)).astype(np.float64)\n\nquick_methods = {'CLAHE only': clahe_only, '2D-IF (Cell 15, fixed)': cell15_2dif_wrapper}\nsample_paths = train_df['path'].sample(20, random_state=42).values\n\nquick_results = []\nfor name, fn in quick_methods.items():\n    scores, times = [], []\n    for p in sample_paths:\n        gray = cv2.cvtColor(preprocess_fundus(p), cv2.COLOR_RGB2GRAY)\n        t0 = time.perf_counter(); out = fn(gray); times.append(time.perf_counter() - t0)\n        pres = lesion_preservation_score(gray, np.clip(out, 0, 255).astype(np.uint8))\n        scores.append(pres['combined'])\n    quick_results.append({'method': name, 'mean_preservation': np.mean(scores), 'mean_time_s': np.mean(times)})\n\naepif_scores, aepif_times = [], []\nfor p in sample_paths:\n    gray = cv2.cvtColor(preprocess_fundus(p), cv2.COLOR_RGB2GRAY)\n    t0 = time.perf_counter(); imfs, qdf = aepif_decompose(gray); aepif_times.append(time.perf_counter() - t0)\n    aepif_scores.append(qdf['combined'].iloc[-1] if len(qdf) else np.nan)\nquick_results.append({'method': 'AEPIF (proposed)', 'mean_preservation': np.nanmean(aepif_scores), 'mean_time_s': np.mean(aepif_times)})\n\nprint(pd.DataFrame(quick_results).sort_values('mean_preservation', ascending=False).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:45:57.249551Z","iopub.execute_input":"2026-09-22T10:45:57.249912Z","iopub.status.idle":"2026-09-22T10:46:11.905595Z","shell.execute_reply.started":"2026-09-22T10:45:57.249887Z","shell.execute_reply":"2026-09-22T10:46:11.90469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\n\ndef build_model(img_size=IMG_SIZE, num_classes=5, alpha=0.35):\n    base = tf.keras.applications.MobileNetV2(\n        input_shape=(img_size, img_size, 3), include_top=False, weights='imagenet', alpha=alpha\n    )\n    base.trainable = False\n\n    inputs = layers.Input(shape=(img_size, img_size, 3))\n    x = base(inputs, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.5)(x)  # was 0.3\n    x = layers.Dense(64, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(1e-4))(x)\n    outputs = layers.Dense(num_classes, activation='softmax')(x)\n    return models.Model(inputs, outputs), base\n\nmodel, base_model = build_model()\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(5e-4),  # was 1e-3\n    loss='categorical_crossentropy',\n    metrics=['accuracy', tf.keras.metrics.AUC(name='auc')]\n)\nmodel.summary()\n\nclass_weights_arr = compute_class_weight('balanced', classes=np.unique(train_labels), y=train_labels)\nclass_weight_dict = dict(enumerate(class_weights_arr))\nprint(\"Class weights:\", class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:46:14.776478Z","iopub.execute_input":"2026-09-22T10:46:14.776755Z","iopub.status.idle":"2026-09-22T10:46:15.508586Z","shell.execute_reply.started":"2026-09-22T10:46:14.776726Z","shell.execute_reply":"2026-09-22T10:46:15.507937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs('/kaggle/working/checkpoints', exist_ok=True)\n\ncallbacks = [\n    tf.keras.callbacks.EarlyStopping(patience=6, restore_best_weights=True, monitor='val_auc', mode='max'),\n    tf.keras.callbacks.ReduceLROnPlateau(patience=3, factor=0.5, monitor='val_loss', min_lr=1e-7),\n    tf.keras.callbacks.ModelCheckpoint(\n        filepath='/kaggle/working/checkpoints/best_model_phase1.keras',\n        monitor='val_auc', mode='max', save_best_only=True, verbose=1\n    ),\n]\n\nhistory1 = model.fit(\n    train_ds, validation_data=val_ds, epochs=30,\n    class_weight=class_weight_dict,\n    callbacks=callbacks\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-12T10:45:41.408186Z","iopub.execute_input":"2026-09-12T10:45:41.408669Z","iopub.status.idle":"2026-09-12T10:46:53.511563Z","shell.execute_reply.started":"2026-09-12T10:45:41.408631Z","shell.execute_reply":"2026-09-12T10:46:53.510188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Full model (architecture + weights + optimizer state)\nmodel.save('/kaggle/working/dr_model_phase1_full.keras')\n\n# 2. Weights only (smaller; useful if rebuilding architecture separately later)\nmodel.save_weights('/kaggle/working/dr_model_phase1.weights.h5')\n\n# 3. Training history — so you don't lose epoch-by-epoch numbers when the session ends\nimport json\nwith open('/kaggle/working/history_phase1.json', 'w') as f:\n    json.dump(history1.history, f)\n\nprint(\"Saved model size (MB):\", os.path.getsize('/kaggle/working/dr_model_phase1_full.keras') / (1024*1024))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport json, os\n\n# Point this at whichever model you're finalizing — Phase 1 or Phase 2's checkpoint\nMODEL_PATH = '/kaggle/input/models/wamanvasudhasuneel/aepif-v1/keras/default/1/dr_model_phase1_full.keras'\nHISTORY_PATH = '/kaggle/input/models/wamanvasudhasuneel/aepif-v1/keras/default/1/history_phase1.json'\n\nmodel = tf.keras.models.load_model(MODEL_PATH)\nmodel.summary()\n\nwith open(HISTORY_PATH) as f:\n    hist = json.load(f)\n\nclass_names = [\"0: No DR\", \"1: Mild\", \"2: Moderate\", \"3: Severe\", \"4: Proliferative\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:46:59.169711Z","iopub.execute_input":"2026-09-22T10:46:59.170167Z","iopub.status.idle":"2026-09-22T10:47:00.170912Z","shell.execute_reply.started":"2026-09-22T10:46:59.170138Z","shell.execute_reply":"2026-09-22T10:47:00.170372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import (classification_report, confusion_matrix, cohen_kappa_score,\n                              multilabel_confusion_matrix, roc_curve, auc, ConfusionMatrixDisplay)\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.dummy import DummyClassifier\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# --- Collect predictions once, reuse everywhere below ---\ny_true, y_pred, y_probs = [], [], []\nfor imgs, labels in val_ds:\n    probs = model.predict(imgs, verbose=0)\n    y_true.extend(np.argmax(labels.numpy(), axis=1))\n    y_pred.extend(np.argmax(probs, axis=1))\n    y_probs.extend(probs)\ny_true, y_pred, y_probs = np.array(y_true), np.array(y_pred), np.array(y_probs)\n\n# ============================================================\n# A. Training curves\n# ============================================================\nfig, axes = plt.subplots(1, 3, figsize=(18, 4))\nepochs_ran = range(1, len(hist['loss']) + 1)\nfor ax, metric, title in zip(axes, ['accuracy', 'loss', 'auc'], ['Accuracy', 'Loss', 'AUC']):\n    ax.plot(epochs_ran, hist[metric], label='Train', marker='o', ms=3)\n    ax.plot(epochs_ran, hist[f'val_{metric}'], label='Validation', marker='o', ms=3)\n    ax.set_title(title); ax.set_xlabel('Epoch'); ax.legend(); ax.grid(alpha=0.3)\nplt.tight_layout(); plt.savefig('/kaggle/working/training_curves.png', dpi=150); plt.show()\n\nbest_epoch = int(np.argmax(hist['val_auc'])) + 1\nprint(f\"Best epoch: {best_epoch}/{len(hist['loss'])} | Best val_auc: {max(hist['val_auc']):.4f}\")\n\n# ============================================================\n# B. Classification report + quadratic weighted kappa\n# ============================================================\nreport_dict = classification_report(y_true, y_pred, target_names=class_names, digits=3, output_dict=True)\nreport_df = pd.DataFrame(report_dict).transpose()\nprint(classification_report(y_true, y_pred, target_names=class_names, digits=3))\n\nqwk = cohen_kappa_score(y_true, y_pred, weights='quadratic')\nprint(f\"\\nQuadratic Weighted Kappa: {qwk:.4f}\")\nreport_df.to_csv('/kaggle/working/classification_report.csv')\n\n# ============================================================\n# C. Confusion matrix (raw counts + normalized)\n# ============================================================\nfig, axes = plt.subplots(1, 2, figsize=(13, 5))\ncm = confusion_matrix(y_true, y_pred)\ncm_norm = confusion_matrix(y_true, y_pred, normalize='true')\n\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=axes[0],\n            xticklabels=class_names, yticklabels=class_names)\naxes[0].set_title('Confusion Matrix (counts)'); axes[0].set_xlabel('Predicted'); axes[0].set_ylabel('True')\n\nsns.heatmap(cm_norm, annot=True, fmt='.2f', cmap='Blues', ax=axes[1],\n            xticklabels=class_names, yticklabels=class_names)\naxes[1].set_title('Confusion Matrix (row-normalized)'); axes[1].set_xlabel('Predicted'); axes[1].set_ylabel('True')\nplt.tight_layout(); plt.savefig('/kaggle/working/confusion_matrix.png', dpi=150); plt.show()\n\n# ============================================================\n# D. Per-class sensitivity/specificity/PPV/NPV\n# ============================================================\nmcm = multilabel_confusion_matrix(y_true, y_pred, labels=list(range(5)))\nper_class_rows = []\nfor i, cls_name in enumerate(class_names):\n    tn, fp, fn, tp = mcm[i].ravel()\n    per_class_rows.append({\n        'class': cls_name,\n        'sensitivity': tp / (tp + fn + 1e-8), 'specificity': tn / (tn + fp + 1e-8),\n        'precision_ppv': tp / (tp + fp + 1e-8), 'npv': tn / (tn + fn + 1e-8),\n        'support': int(tp + fn),\n    })\nper_class_df = pd.DataFrame(per_class_rows)\nprint(per_class_df.to_string(index=False))\nper_class_df.to_csv('/kaggle/working/per_class_metrics.csv', index=False)\n\n# ============================================================\n# E. Per-class ROC curves + AUC\n# ============================================================\ny_true_bin = label_binarize(y_true, classes=list(range(5)))\nplt.figure(figsize=(7, 6))\nfor i, cls_name in enumerate(class_names):\n    fpr, tpr, _ = roc_curve(y_true_bin[:, i], y_probs[:, i])\n    plt.plot(fpr, tpr, label=f'{cls_name} (AUC={auc(fpr, tpr):.3f})')\nplt.plot([0, 1], [0, 1], 'k--', alpha=0.3)\nplt.xlabel('False Positive Rate'); plt.ylabel('True Positive Rate')\nplt.title('Per-class ROC curves (one-vs-rest)'); plt.legend(loc='lower right')\nplt.tight_layout(); plt.savefig('/kaggle/working/roc_curves.png', dpi=150); plt.show()\n\n# ============================================================\n# F. Referable-DR (binary) reframing — Moderate+ vs Mild/None\n# often more clinically meaningful and comparable to screening literature\n# ============================================================\nreferable_true = (y_true >= 2).astype(int)\nreferable_pred = (y_pred >= 2).astype(int)\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nprint(f\"\\nReferable-DR (grade>=2) binary accuracy: {accuracy_score(referable_true, referable_pred):.4f}\")\nreferable_probs = y_probs[:, 2:].sum(axis=1)  # P(Moderate) + P(Severe) + P(Proliferative)\nprint(f\"Referable-DR binary AUC: {roc_auc_score(referable_true, referable_probs):.4f}\")\n\n# ============================================================\n# G. Dummy baseline comparison (sanity check — model must clear this)\n# ============================================================\nX_dummy_train = np.zeros((len(train_labels), 1))\nX_dummy_val = np.zeros((len(val_labels), 1))\ndummy = DummyClassifier(strategy='most_frequent').fit(X_dummy_train, train_labels)\ndummy_pred = dummy.predict(X_dummy_val)\ndummy_kappa = cohen_kappa_score(val_labels, dummy_pred, weights='quadratic')\nprint(f\"\\nDummy (majority-class) accuracy: {(dummy_pred == val_labels).mean():.4f} | kappa: {dummy_kappa:.4f}\")\nprint(f\"Your model kappa: {qwk:.4f}  →  {'✓ clears baseline' if qwk > dummy_kappa + 0.05 else '⚠ marginal improvement over baseline'}\")\n\n# ============================================================\n# H. Save ALL results as one summary file for the report\n# ============================================================\nsummary = {\n    'model_path': MODEL_PATH, 'best_epoch': best_epoch, 'epochs_trained': len(hist['loss']),\n    'best_val_auc': float(max(hist['val_auc'])), 'quadratic_weighted_kappa': float(qwk),\n    'overall_accuracy': float(report_dict['accuracy']),\n    'macro_avg_f1': float(report_dict['macro avg']['f1-score']),\n    'referable_dr_accuracy': float(accuracy_score(referable_true, referable_pred)),\n    'referable_dr_auc': float(roc_auc_score(referable_true, referable_probs)),\n    'dummy_baseline_kappa': float(dummy_kappa),\n}\nwith open('/kaggle/working/model_analysis_summary.json', 'w') as f:\n    json.dump(summary, f, indent=2)\nprint(\"\\n=== SUMMARY ===\")\nfor k, v in summary.items():\n    print(f\"{k}: {v}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T10:47:02.961671Z","iopub.execute_input":"2026-09-22T10:47:02.962517Z","iopub.status.idle":"2026-09-22T10:48:15.082959Z","shell.execute_reply.started":"2026-09-22T10:47:02.962462Z","shell.execute_reply":"2026-09-22T10:48:15.082288Z"}},"outputs":[],"execution_count":null}]}