{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport kagglehub\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow.keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:56:29.604341Z","iopub.execute_input":"2026-09-29T12:56:29.605095Z","iopub.status.idle":"2026-09-29T12:56:55.777656Z","shell.execute_reply.started":"2026-09-29T12:56:29.605064Z","shell.execute_reply":"2026-09-29T12:56:55.776678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to the APTOS 2019 dataset\nBASE = '/kaggle/input/competitions/aptos2019-blindness-detection'\n\n# Load the training CSV\ndf = pd.read_csv(f'{BASE}/train.csv')\n\n# Diagnosis class names\nnames = {\n    0: 'No DR',\n    1: 'Mild',\n    2: 'Moderate',\n    3: 'Severe',\n    4: 'Proliferative'\n}\n\n# Split the labeled dataset into:\n# 85% Train + Validation\n# 15% Holdout Test\n#\n# Stratification ensures that the class proportions\n# are approximately preserved in both splits.\ntrainval_df, test_df = train_test_split(\n    df,\n    test_size=0.15,\n    stratify=df['diagnosis'],\n    random_state=42\n)\n\n# Display split sizes\nprint('Train + Validation:', len(trainval_df))\nprint('Holdout Test:', len(test_df))\nprint('Total:', len(trainval_df) + len(test_df))\n\n# Display class distribution in both splits\ndistribution = pd.concat(\n    [\n        trainval_df['diagnosis']\n        .value_counts()\n        .sort_index()\n        .rename('Train + Validation'),\n\n        test_df['diagnosis']\n        .value_counts()\n        .sort_index()\n        .rename('Holdout Test')\n    ],\n    axis=1\n)\n\n# Replace numeric labels with class names\ndistribution.index = [\n    names[i] for i in distribution.index\n]\n\nprint('\\nClass distribution:')\nprint(distribution)\n\n# Create output directory\nos.makedirs('/kaggle/working', exist_ok=True)\n\n# Save the split datasets\ntrainval_df.to_csv(\n    '/kaggle/working/trainval.csv',\n    index=False\n)\n\ntest_df.to_csv(\n    '/kaggle/working/test_holdout.csv',\n    index=False\n)\n\nprint('\\nFiles saved successfully:')\nprint('/kaggle/working/trainval.csv')\nprint('/kaggle/working/test_holdout.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:57:13.662061Z","iopub.execute_input":"2026-09-29T12:57:13.662943Z","iopub.status.idle":"2026-09-29T12:57:13.777183Z","shell.execute_reply.started":"2026-09-29T12:57:13.662909Z","shell.execute_reply":"2026-09-29T12:57:13.776325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Image size\nIMG_SIZE = 224\n\n\n# Crop, make square, and resize image\ndef crop_and_square(img, thresh=10):\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    mask = gray > thresh\n\n    if mask.any():\n        ys, xs = np.where(mask)\n\n        img = img[\n            ys.min():ys.max() + 1,\n            xs.min():xs.max() + 1\n        ]\n\n    h, w = img.shape[:2]\n    s = max(h, w)\n\n    canvas = np.zeros(\n        (s, s, 3),\n        dtype=img.dtype\n    )\n\n    canvas[\n        (s - h) // 2:(s - h) // 2 + h,\n        (s - w) // 2:(s - w) // 2 + w\n    ] = img\n\n    return cv2.resize(\n        canvas,\n        (IMG_SIZE, IMG_SIZE),\n        interpolation=cv2.INTER_AREA\n    )\n\n\n# Apply CLAHE enhancement\ndef clahe_enhance(img):\n    lab = cv2.cvtColor(\n        img,\n        cv2.COLOR_RGB2LAB\n    )\n\n    l, a, b = cv2.split(lab)\n\n    l = cv2.createCLAHE(\n        clipLimit=2.0,\n        tileGridSize=(8, 8)\n    ).apply(l)\n\n    return cv2.cvtColor(\n        cv2.merge((l, a, b)),\n        cv2.COLOR_LAB2RGB\n    )\n\n\n# Apply Ben Graham enhancement\ndef ben_graham(img):\n    blur = cv2.GaussianBlur(\n        img,\n        (0, 0),\n        IMG_SIZE / 30\n    )\n\n    return cv2.addWeighted(\n        img,\n        4,\n        blur,\n        -4,\n        128\n    )\n\n\n# Apply masked Ben Graham enhancement\ndef ben_graham_masked(img):\n    out = ben_graham(img)\n\n    gray = cv2.cvtColor(\n        img,\n        cv2.COLOR_RGB2GRAY\n    )\n\n    mask = (\n        gray > 8\n    ).astype(np.uint8)\n\n    mask = cv2.morphologyEx(\n        mask,\n        cv2.MORPH_CLOSE,\n        np.ones((15, 15), np.uint8)\n    )\n\n    mask = cv2.erode(\n        mask,\n        np.ones((5, 5), np.uint8),\n        iterations=3\n    )\n\n    return np.where(\n        mask[..., None] > 0,\n        out,\n        128\n    ).astype(np.uint8)\n\n\n# Load image in RGB format\ndef load_rgb(image_id):\n    image_path = f'{BASE}/train_images/{image_id}.png'\n\n    img = cv2.imread(image_path)\n\n    if img is None:\n        raise FileNotFoundError(\n            f'Could not read image: {image_path}'\n        )\n\n    return cv2.cvtColor(\n        img,\n        cv2.COLOR_BGR2RGB\n    )\n\n\nprint(\"Preprocessing functions defined successfully.\")\nprint(f\"Image size: {IMG_SIZE} × {IMG_SIZE}\")\nprint(f\"Dataset path: {BASE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:59:01.446196Z","iopub.execute_input":"2026-09-29T12:59:01.44654Z","iopub.status.idle":"2026-09-29T12:59:01.457324Z","shell.execute_reply.started":"2026-09-29T12:59:01.446512Z","shell.execute_reply":"2026-09-29T12:59:01.456738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select one random image from each class\nsample_ids = [\n    df[df['diagnosis'] == c]\n    .sample(1, random_state=3)['id_code']\n    .values[0]\n    for c in range(5)\n]\n\n# Create comparison figure\nfig, axes = plt.subplots(\n    5,\n    5,\n    figsize=(15, 15)\n)\n\n# Process and display images\nfor r, (cls, image_id) in enumerate(\n    zip(range(5), sample_ids)\n):\n    orig = load_rgb(image_id)\n\n    sq = crop_and_square(orig)\n\n    panels = [\n        ('Original', orig),\n        ('Cropped + resized', sq),\n        ('CLAHE', clahe_enhance(sq)),\n        ('Ben Graham', ben_graham(sq)),\n        ('Ben Graham (masked)', ben_graham_masked(sq))\n    ]\n\n    for c, (title, img) in enumerate(panels):\n        axes[r][c].imshow(img)\n        axes[r][c].axis('off')\n        axes[r][c].set_title(\n            f'{names[cls]}: {title}',\n            fontsize=8\n        )\n\n# Adjust layout\nplt.tight_layout()\n\n# Save the comparison figure\noutput_path = '/kaggle/working/preprocessing_comparison.png'\n\nplt.savefig(\n    output_path,\n    dpi=150,\n    bbox_inches='tight'\n)\n\n# Display the figure\nplt.show()\n\nprint(f\"Figure saved successfully at: {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:59:39.850069Z","iopub.execute_input":"2026-09-29T12:59:39.850592Z","iopub.status.idle":"2026-09-29T12:59:50.472442Z","shell.execute_reply.started":"2026-09-29T12:59:39.850563Z","shell.execute_reply":"2026-09-29T12:59:50.471471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define preprocessing methods\nMETHODS = {\n    'clahe': clahe_enhance,\n    'bengraham': ben_graham_masked\n}\n\n\n# Build preprocessed image arrays\ndef build_arrays(frame):\n    out = {\n        method: np.zeros(\n            (len(frame), IMG_SIZE, IMG_SIZE, 3),\n            dtype=np.uint8\n        )\n        for method in METHODS\n    }\n\n    for i, image_id in enumerate(\n        tqdm(\n            frame['id_code'].values,\n            desc='Processing images'\n        )\n    ):\n        sq = crop_and_square(\n            load_rgb(image_id)\n        )\n\n        for method, fn in METHODS.items():\n            out[method][i] = fn(sq)\n\n    labels = frame['diagnosis'].values.astype(\n        np.int64\n    )\n\n    return out, labels\n\n\n# Build arrays for train + validation data\narrs_tv, y_tv = build_arrays(trainval_df)\n\n\n# Build arrays for holdout test data\narrs_te, y_te = build_arrays(test_df)\n\n\n# Save preprocessed image arrays\nfor method in METHODS:\n    np.save(\n        f'/kaggle/working/X_trainval_{method}.npy',\n        arrs_tv[method]\n    )\n\n    np.save(\n        f'/kaggle/working/X_test_{method}.npy',\n        arrs_te[method]\n    )\n\n\n# Save labels\nnp.save(\n    '/kaggle/working/y_trainval.npy',\n    y_tv\n)\n\nnp.save(\n    '/kaggle/working/y_test.npy',\n    y_te\n)\n\n\n# Verify saved files\nprint(\"\\nPreprocessed arrays saved successfully.\")\n\nfor method in METHODS:\n    print(\n        f\"\\n{method}:\"\n        f\"\\n  Train + Validation: \"\n        f\"{arrs_tv[method].shape}\"\n        f\"\\n  Holdout Test: \"\n        f\"{arrs_te[method].shape}\"\n    )\n\nprint(f\"\\ny_trainval shape: {y_tv.shape}\")\nprint(f\"y_test shape: {y_te.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T13:00:32.872918Z","iopub.execute_input":"2026-09-29T13:00:32.873704Z","iopub.status.idle":"2026-09-29T13:18:33.864045Z","shell.execute_reply.started":"2026-09-29T13:00:32.873676Z","shell.execute_reply":"2026-09-29T13:18:33.863103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define data augmentation\naugment = tf.keras.Sequential([\n    layers.RandomFlip(\n        'horizontal_and_vertical'\n    ),\n    layers.RandomRotation(\n        0.05\n    ),  # about ±18 degrees\n    layers.RandomZoom(\n        0.1\n    ),\n    layers.RandomBrightness(\n        0.1,\n        value_range=(0, 255)\n    )\n])\n\n\n# Visualize augmentation for each preprocessing method\nfor method in ['clahe', 'bengraham']:\n\n    base_img = arrs_tv[method][0]\n\n    img = base_img[None].astype(\n        'float32'\n    )\n\n    fig, axes = plt.subplots(\n        2,\n        4,\n        figsize=(12, 6)\n    )\n\n    # Display original preprocessed image\n    axes[0][0].imshow(base_img)\n    axes[0][0].set_title(\n        f'Preprocessed ({method})'\n    )\n    axes[0][0].axis('off')\n\n    # Generate and display augmented images\n    for k, ax in enumerate(\n        axes.flatten()[1:]\n    ):\n        aug = np.clip(\n            augment(\n                img,\n                training=True\n            )[0].numpy(),\n            0,\n            255\n        ).astype('uint8')\n\n        ax.imshow(aug)\n        ax.set_title(\n            f'Augmented {k + 1}'\n        )\n        ax.axis('off')\n\n    # Adjust layout\n    plt.tight_layout()\n\n    # Save augmentation visualization\n    output_path = (\n        f'/kaggle/working/'\n        f'augmentation_{method}.png'\n    )\n\n    plt.savefig(\n        output_path,\n        dpi=150,\n        bbox_inches='tight'\n    )\n\n    # Display the figure\n    plt.show()\n\n    print(\n        f\"Augmentation figure saved at: \"\n        f\"{output_path}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T13:20:04.434587Z","iopub.execute_input":"2026-09-29T13:20:04.434973Z","iopub.status.idle":"2026-09-29T13:20:08.973006Z","shell.execute_reply.started":"2026-09-29T13:20:04.434945Z","shell.execute_reply":"2026-09-29T13:20:08.971895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List files saved in the working directory\nprint(\"Files in /kaggle/working:\\n\")\n\nfor file_name in sorted(\n    os.listdir('/kaggle/working')\n):\n    file_path = f'/kaggle/working/{file_name}'\n\n    # Calculate file size in MB\n    size_mb = os.path.getsize(file_path) / 1e6\n\n    print(\n        f'{file_name:40s} '\n        f'{size_mb:8.1f} MB'\n    )\n\n\n# Display array shapes\nprint()\n\nfor method in ['clahe', 'bengraham']:\n    print(\n        f'{method}: '\n        f'trainval: {arrs_tv[method].shape} '\n        f' test: {arrs_te[method].shape}'\n    )\n\n\n# Display label shapes\nprint(\n    f'labels: {y_tv.shape} {y_te.shape}'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T13:20:49.232587Z","iopub.execute_input":"2026-09-29T13:20:49.233053Z","iopub.status.idle":"2026-09-29T13:20:49.241683Z","shell.execute_reply.started":"2026-09-29T13:20:49.233024Z","shell.execute_reply":"2026-09-29T13:20:49.240816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing & Augmentation Conclusions\n\n- The APTOS 2019 images vary in resolution and aspect ratio, so all images are cropped, padded to a square and resized to 224 × 224.\n- The 3,662 labelled images were split (stratified, random_state=42) into 3,112 train+validation and 550 held-out test images (15%). The test set is not used until final evaluation.\n- Two contrast variants were generated for comparison: CLAHE, and Ben Graham enhancement with a retina mask to remove edge artifacts. The final choice will be made from validation results (macro F1, per-class recall, quadratic weighted kappa) using an identical CNN.\n- Augmentation (flips, rotation about ±18°, 10% zoom, 10% brightness) is applied on the fly to training data only.\n- Outputs: X_trainval and X_test arrays for each method, y_trainval and y_test, plus the two split CSVs.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}