{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom concurrent.futures import ProcessPoolExecutor, as_completed\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\n\n# Input path\nAPTOS_ROOT  = Path(\"/kaggle/input/competitions/aptos2019-blindness-detection\")\nINPUT_DIR   = APTOS_ROOT / \"train_images\"\nLABELS_CSV  = APTOS_ROOT / \"train.csv\"\n\n# Output path\nOUTPUT_DIR  = Path(\"/kaggle/working/aptos_processed\")\nIMG_DIR     = OUTPUT_DIR / \"images_512\"\nIMG_DIR.mkdir(parents=True, exist_ok=True)\n\n# Config\nSIZE    = 512\nWORKERS = 4\nSEED    = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:26:09.005917Z","iopub.execute_input":"2026-09-30T09:26:09.006262Z","iopub.status.idle":"2026-09-30T09:26:09.053287Z","shell.execute_reply.started":"2026-09-30T09:26:09.006224Z","shell.execute_reply":"2026-09-30T09:26:09.052353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ben Graham's crop - Removes black border from images\ndef crop_image_from_gray(img, tol=7):\n    \"\"\"Ben Graham's crop — remove black border by thresholding.\"\"\"\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    mask = gray > tol\n    if mask.sum() == 0:\n        return img\n    return img[np.ix_(mask.any(1), mask.any(0))]\n\n# CLAHE - Enhance contrast\ndef apply_clahe(rgb_img, clip_limit=2.0, tile_grid=(8, 8)):\n    \"\"\"CLAHE on the L channel of LAB — preserves color, boosts local contrast.\"\"\"\n    lab = cv2.cvtColor(rgb_img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=clip_limit, tileGridSize=tile_grid)\n    l = clahe.apply(l)\n    lab = cv2.merge((l, a, b))\n    return cv2.cvtColor(lab, cv2.COLOR_LAB2RGB)\n\n# Remove black circle border and resize to 512x512\ndef crop_circle_and_resize(rgb_img, size=512):\n    \"\"\"Circular mask + resize to (size, size).\"\"\"\n    h, w = rgb_img.shape[:2]\n    r = min(h, w) // 2\n    mask = np.zeros((h, w), dtype=np.uint8)\n    cv2.circle(mask, (w // 2, h // 2), r, 255, -1)\n    mask = cv2.GaussianBlur(mask, (51, 51), 0)\n\n    rgb_float = rgb_img.astype(np.float32)\n    mask_3ch = mask[..., None].astype(np.float32) / 255.0\n    masked = (rgb_float * mask_3ch + 255.0 * (1 - mask_3ch)).astype(np.uint8)\n\n    return cv2.resize(masked, (size, size), interpolation=cv2.INTER_AREA)\n\n# Save as .png file\ndef preprocess_one(args):\n    src_path, dst_path, size = args\n    try:\n        img = cv2.imread(src_path)\n        if img is None:\n            return (src_path, False, \"unreadable\")\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = crop_image_from_gray(img)\n        img = apply_clahe(img, 2.0, (8, 8))\n        img = crop_circle_and_resize(img, size=size)\n        out_bgr = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        cv2.imwrite(dst_path, out_bgr, [cv2.IMWRITE_PNG_COMPRESSION, 3])\n        return (src_path, True, None)\n    except Exception as e:\n        return (src_path, False, str(e))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:26:09.910293Z","iopub.execute_input":"2026-09-30T09:26:09.910642Z","iopub.status.idle":"2026-09-30T09:26:09.923335Z","shell.execute_reply.started":"2026-09-30T09:26:09.910614Z","shell.execute_reply":"2026-09-30T09:26:09.922317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load labels\ndf = pd.read_csv(LABELS_CSV)\ndf.columns = [c.strip().lower() for c in df.columns]\ndf = df[[\"id_code\", \"diagnosis\"]]\nprint(f\"Loaded {len(df)} rows\")\nprint(df[\"diagnosis\"].value_counts().sort_index())\n\n# Build task list\ntasks, kept_rows = [], []\nfor _, row in df.iterrows():\n    src = INPUT_DIR / f\"{row['id_code']}.png\"\n    if not src.exists():\n        print(f\"[WARN] Missing: {row['id_code']}\")\n        continue\n    dst = IMG_DIR / f\"{row['id_code']}.png\"\n    tasks.append((str(src), str(dst), SIZE))\n    kept_rows.append({\n        \"id_code\":   row[\"id_code\"],\n        \"diagnosis\": int(row[\"diagnosis\"]),\n        \"filepath\":  str(dst),\n    })\n\nprint(f\"\\nPreprocessing {len(tasks)} images with {WORKERS} workers...\")\n\nfailures = []\nwith ProcessPoolExecutor(max_workers=WORKERS) as ex:\n    futures = [ex.submit(preprocess_one, t) for t in tasks]\n    for fut in tqdm(as_completed(futures), total=len(futures), desc=\"Preprocessing\"):\n        src, ok, err = fut.result()\n        if not ok:\n            failures.append((src, err))\n\nif failures:\n    print(f\"[WARN] {len(failures)} failures (showing first 5):\")\n    for f in failures[:5]:\n        print(\"   \", f)\n\nprint(\"Done.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:26:09.925135Z","iopub.execute_input":"2026-09-30T09:26:09.925621Z","iopub.status.idle":"2026-09-30T09:37:59.089675Z","shell.execute_reply.started":"2026-09-30T09:26:09.92559Z","shell.execute_reply":"2026-09-30T09:37:59.08794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clean_df = pd.DataFrame(kept_rows)\nclean_df = clean_df[clean_df[\"filepath\"].apply(os.path.exists)].reset_index(drop=True)\nprint(f\"Successfully processed: {len(clean_df)} / {len(df)} images\")\n\n# 80/10/10 split (still keeps the propotion)\ntrain_df, temp_df = train_test_split(\n    clean_df, test_size=0.20, stratify=clean_df[\"diagnosis\"], random_state=SEED\n)\nval_df, test_df = train_test_split(\n    temp_df, test_size=0.50, stratify=temp_df[\"diagnosis\"], random_state=SEED\n)\n\ntrain_df = train_df.reset_index(drop=True)\nval_df   = val_df.reset_index(drop=True)\ntest_df  = test_df.reset_index(drop=True)\n\ntrain_df.to_csv(OUTPUT_DIR / \"train.csv\", index=False)\nval_df.to_csv(OUTPUT_DIR / \"val.csv\",     index=False)\ntest_df.to_csv(OUTPUT_DIR / \"test.csv\",   index=False)\n\nprint(f\"\\nSplit sizes: train={len(train_df)}  val={len(val_df)}  test={len(test_df)}\")\nfor name, split in [(\"train\", train_df), (\"val\", val_df), (\"test\", test_df)]:\n    counts = split[\"diagnosis\"].value_counts().sort_index()\n    pct = (counts / len(split) * 100).round(2)\n    print(f\"  {name:5s} -> \" +\n          \", \".join(f\"{k}:{v}({p}%)\" for k, v, p in\n                    zip(counts.index, counts.values, pct.values)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:37:59.09226Z","iopub.execute_input":"2026-09-30T09:37:59.092658Z","iopub.status.idle":"2026-09-30T09:37:59.1854Z","shell.execute_reply.started":"2026-09-30T09:37:59.092619Z","shell.execute_reply":"2026-09-30T09:37:59.184175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class-weight computation\nclasses = np.sort(train_df[\"diagnosis\"].unique())\nweights = compute_class_weight(\n    class_weight=\"balanced\",\n    classes=classes,\n    y=train_df[\"diagnosis\"].values,\n)\nclass_weight_dict = {int(c): float(w) for c, w in zip(classes, weights)}\n\npd.Series(class_weight_dict).to_csv(\n    OUTPUT_DIR / \"class_weights.csv\", header=[\"weight\"]\n)\nprint(\"Class weights:\", class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:37:59.186595Z","iopub.execute_input":"2026-09-30T09:37:59.187057Z","iopub.status.idle":"2026-09-30T09:37:59.200632Z","shell.execute_reply.started":"2026-09-30T09:37:59.187016Z","shell.execute_reply":"2026-09-30T09:37:59.199766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig, axes = plt.subplots(2, 5, figsize=(15, 6))\nfor ax, (_, row) in zip(axes.flat, train_df.sample(10, random_state=0).iterrows()):\n    img = cv2.cvtColor(cv2.imread(row[\"filepath\"]), cv2.COLOR_BGR2RGB)\n    ax.imshow(img); ax.axis(\"off\")\n    ax.set_title(f\"Grade {row['diagnosis']}\")\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-30T09:37:59.20273Z","iopub.execute_input":"2026-09-30T09:37:59.203057Z"}},"outputs":[],"execution_count":null}]}