{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T06:03:01.270222Z","iopub.execute_input":"2026-09-16T06:03:01.270932Z","iopub.status.idle":"2026-09-16T06:03:10.215691Z","shell.execute_reply.started":"2026-09-16T06:03:01.270901Z","shell.execute_reply":"2026-09-16T06:03:10.215141Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')\n# CELL 1: IMPORTS\n# ==========================================\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nprint(\"✅ All imports successful\")\n\n# CELL 2: CONFIGURATION AND FOLDER CREATION\n# ==========================================\nBASE_PATH = '/kaggle/input/competitions/aptos2019-blindness-detection/'\nIMAGE_PATH = os.path.join(BASE_PATH, 'train_images/')\nCSV_PATH = os.path.join(BASE_PATH, 'train.csv')\n\n# Define multiple target sizes\nTARGET_SIZES = [(224, 224), (384, 384), (512, 512)]\nRANDOM_SEED = 42\n\n# Output folder structure for all sizes\nOUTPUT_BASE = '/kaggle/working/APTOS_Preprocessed/'\n\n# Create base output directory\nos.makedirs(OUTPUT_BASE, exist_ok=True)\n\n# Create folders for each size\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    os.makedirs(os.path.join(size_dir, 'train'), exist_ok=True)\n    os.makedirs(os.path.join(size_dir, 'val'), exist_ok=True)\n    os.makedirs(os.path.join(size_dir, 'test'), exist_ok=True)\n\nprint(f\"✅ Folder structure created for sizes: {TARGET_SIZES}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"   {size_str}:\")\n    print(f\"   ├── train/\")\n    print(f\"   ├── val/\")\n    print(f\"   └── test/\")\n\n# CELL 3: LOAD DATASET\n# ==========================================\ndf = pd.read_csv(CSV_PATH)\nprint(f\"Total images in dataset: {len(df)}\")\nprint(f\"\\nClass distribution:\\n{df['diagnosis'].value_counts().sort_index()}\")\n\n# CELL 4: TRAIN / VALIDATION / TEST SPLIT (70-15-15)\n# ==========================================\n# Split 1: 70% Train, 30% Temp (which will be split into Val + Test)\nfrom sklearn.model_selection import train_test_split\ntrain_df, temp_df = train_test_split(\n    df, \n    test_size=0.30, \n    stratify=df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Split 2: 15% Val, 15% Test (50% of 30% = 15%)\nval_df, test_df = train_test_split(\n    temp_df, \n    test_size=0.50,  # 50% of 30% = 15% total\n    stratify=temp_df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Reset indices for cleanliness\ntrain_df = train_df.reset_index(drop=True)\nval_df = val_df.reset_index(drop=True)\ntest_df = test_df.reset_index(drop=True)\n\nprint(f\"\\n✅ Split Complete:\")\nprint(f\"Train: {len(train_df)} images\")\nprint(f\"Val:   {len(val_df)} images\")\nprint(f\"Test:  {len(test_df)} images\")\n\nprint(f\"\\nTrain class distribution:\\n{train_df['diagnosis'].value_counts().sort_index()}\")\n\n# CELL 5: CREATE OUTPUT FOLDER STRUCTURE\n# ==========================================\n# (Already created in Cell 2, this is for documentation)\nprint(\"✅ Output folder structure is ready:\")\nprint(f\"   {OUTPUT_BASE}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"   ├── {size_str}/\")\n    print(f\"   │   ├── train/\")\n    print(f\"   │   ├── val/\")\n    print(f\"   │   └── test/\")\n\n# CELL 6: SAVE SPLIT CSV FILES\n# ==========================================\ntrain_csv_path = os.path.join(OUTPUT_BASE, 'train_split.csv')\nval_csv_path = os.path.join(OUTPUT_BASE, 'val_split.csv')\ntest_csv_path = os.path.join(OUTPUT_BASE, 'test_split.csv')\n\ntrain_df.to_csv(train_csv_path, index=False)\nval_df.to_csv(val_csv_path, index=False)\ntest_df.to_csv(test_csv_path, index=False)\n\nprint(\"✅ Split CSVs saved:\")\nprint(f\"   {train_csv_path}\")\nprint(f\"   {val_csv_path}\")\nprint(f\"   {test_csv_path}\")\nprint(\"\\n⚠️  DO NOT open test_split.csv until final evaluation!\")\n\n# CELL 7: DEFINE PREPROCESSING FUNCTIONS\n# ==========================================\n\ndef crop_image_from_gray(img, tol=7):\n    \"\"\"Remove black borders by cropping to the fundus circle.\"\"\"\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        mask = gray_img > tol\n        if mask.sum() == 0: \n            return img\n        img1 = img[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n        img2 = img[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n        img3 = img[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n        return np.stack([img1, img2, img3], axis=-1)\n\ndef apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8)):\n    \"\"\"Apply CLAHE on the L-channel of LAB to preserve color.\"\"\"\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=clipLimit, tileGridSize=tileGridSize)\n    l_enhanced = clahe.apply(l)\n    enhanced_img = cv2.merge((l_enhanced, a, b))\n    return cv2.cvtColor(enhanced_img, cv2.COLOR_LAB2RGB)\n\ndef preprocess_image(image_path, target_size):\n    \"\"\"\n    Apply the FULL preprocessing pipeline with specified target size.\n    ORDER: Crop -> Ben Graham -> RESIZE -> CLAHE -> float conversion [0, 1]\n    \n    Returns: float32 image in range [0, 1], ready for normalization.\n    \"\"\"\n    # 1. Read image\n    img = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n    \n    # 2. Circle Crop\n    img = crop_image_from_gray(img, tol=7)\n    \n    # 3. Ben Graham Illumination Correction\n    \n    # 4. RESIZE (CRITICAL: Do this BEFORE CLAHE)\n    img = cv2.resize(img, target_size, interpolation=cv2.INTER_AREA)\n    \n    # 5. CLAHE (Now applied on the exact target grid)\n    img = apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8))\n    \n    # 6. Convert to float [0, 1]\n    img = img.astype(np.float32) / 255.0\n    \n    return img\n\nprint(\"✅ All preprocessing functions defined\")\n\n# CELL 8-10: PREPROCESS AND SAVE IMAGES FOR EACH SIZE\n# ==========================================\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n{'='*60}\")\n    print(f\"Processing size: {size_str}\")\n    print(f\"{'='*60}\")\n    \n    # Set directories for this size\n    size_base = os.path.join(OUTPUT_BASE, size_str)\n    train_dir = os.path.join(size_base, 'train')\n    val_dir = os.path.join(size_base, 'val')\n    test_dir = os.path.join(size_base, 'test')\n    \n    # Process TRAINING images\n    print(f\"\\n⏳ Preprocessing and saving TRAINING images for {size_str}...\")\n    print(f\"Total training images to process: {len(train_df)}\")\n    \n    for idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(train_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(train_df)} training images saved to {train_dir}\")\n    \n    # Process VALIDATION images\n    print(f\"\\n⏳ Preprocessing and saving VALIDATION images for {size_str}...\")\n    print(f\"Total validation images to process: {len(val_df)}\")\n    \n    for idx, row in tqdm(val_df.iterrows(), total=len(val_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(val_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(val_df)} validation images saved to {val_dir}\")\n    \n    # Process TEST images\n    print(f\"\\n⏳ Preprocessing and saving TEST images for {size_str}...\")\n    print(f\"Total test images to process: {len(test_df)}\")\n    \n    for idx, row in tqdm(test_df.iterrows(), total=len(test_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(test_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(test_df)} test images saved to {test_dir}\")\n\n# CELL 11: COMPUTE MEAN & STD FOR EACH SIZE\n# ==========================================\nall_means = {}\nall_stds = {}\n\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    train_dir = os.path.join(size_dir, 'train')\n    \n    print(f\"\\n⏳ Computing Mean and Std for {size_str} from saved TRAINING images...\")\n    print(f\"Total training images to analyze: {len(train_df)}\")\n    \n    all_pixels = []\n    \n    for idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n        # Load the SAVED preprocessed image\n        npy_path = os.path.join(train_dir, row['id_code'] + '.npy')\n        processed_img = np.load(npy_path)\n        \n        # Flatten\n        all_pixels.append(processed_img.reshape(-1, 3))\n    \n    # Stack all pixels into a single array\n    all_pixels = np.vstack(all_pixels)\n    \n    # Calculate global mean and std per channel\n    dataset_mean = np.mean(all_pixels, axis=0)\n    dataset_std = np.std(all_pixels, axis=0)\n    \n    all_means[size_str] = dataset_mean\n    all_stds[size_str] = dataset_std\n    \n    print(f\"\\n✅ Dataset Statistics for {size_str}:\")\n    print(f\"   Mean (R, G, B): [{dataset_mean[0]:.4f}, {dataset_mean[1]:.4f}, {dataset_mean[2]:.4f}]\")\n    print(f\"   Std  (R, G, B): [{dataset_std[0]:.4f}, {dataset_std[1]:.4f}, {dataset_std[2]:.4f}]\")\n    \n    # Save normalization statistics for this size\n    mean_path = os.path.join(size_dir, 'retina_mean.npy')\n    std_path = os.path.join(size_dir, 'retina_std.npy')\n    \n    np.save(mean_path, dataset_mean)\n    np.save(std_path, dataset_std)\n    \n    print(f\"✅ Normalization statistics saved:\")\n    print(f\"   {mean_path}\")\n    print(f\"   {std_path}\")\n\n# CELL 12: VISUALIZATION - COMPARE ALL SIZES\n# ==========================================\nprint(\"\\n⏳ Creating comparison visualization for all sizes...\")\n\n# Pick one image from train split\nsample_id = train_df.iloc[0]['id_code']\noriginal_path = os.path.join(IMAGE_PATH, sample_id + '.png')\noriginal = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n\nfig, axes = plt.subplots(3, 4, figsize=(24, 16))\ntitles = ['Original', 'Cropped', 'CLAHE', 'Final']\n\n# Show original and processing steps for each size\nfor row, target_size in enumerate(TARGET_SIZES):\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    npy_path = os.path.join(size_dir, 'train', sample_id + '.npy')\n    \n    # Load the preprocessed image\n    processed_img = np.load(npy_path)\n    \n    # Column 0: Original (same for all rows)\n    if row == 0:\n        axes[row, 0].imshow(original)\n        axes[row, 0].set_title(\"Original (Raw)\")\n    else:\n        axes[row, 0].imshow(original)\n        axes[row, 0].set_title(\"Original (Raw)\")\n    \n    # Column 1: Cropped\n    cropped = crop_image_from_gray(original)\n    axes[row, 1].imshow(cropped)\n    axes[row, 1].set_title(f\"Cropped\")\n    \n    # Column 2: CLAHE (on resized)\n    cropped = crop_image_from_gray(original)\n    resized = cv2.resize(cropped, target_size, interpolation=cv2.INTER_AREA)\n    clahe_img = apply_clahe(resized)\n    axes[row, 2].imshow(clahe_img)\n    axes[row, 2].set_title(f\"CLAHE on {size_str}\")\n    \n    # Column 3: Final processed image (from saved .npy)\n    axes[row, 3].imshow(processed_img)\n    axes[row, 3].set_title(f\"Final {size_str}\")\n    \n    # Add row label\n    axes[row, 0].set_ylabel(size_str, fontsize=12, rotation=90, labelpad=20)\n\n# Turn off axes for cleanliness\nfor ax in axes.flat:\n    ax.axis('off')\n\nviz_compare_path = os.path.join(OUTPUT_BASE, 'size_comparison.png')\nplt.tight_layout()\nplt.savefig(viz_compare_path, dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(f\"✅ Size comparison visualization saved: {viz_compare_path}\")\n\n# CELL 13: 20-IMAGE VISUALIZATION BY CLASS FOR EACH SIZE\n# ==========================================\nclass_names = {\n    0: 'Class 0 - No DR',\n    1: 'Class 1 - Mild',\n    2: 'Class 2 - Moderate',\n    3: 'Class 3 - Severe',\n    4: 'Class 4 - Proliferative'\n}\n\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n⏳ Creating 20-image visualization for {size_str}...\")\n    \n    num_rows = 20\n    num_cols = 4\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(24, 4 * num_rows))\n    \n    sample_seed = 42\n    row_idx = 0\n    \n    for class_idx in range(5):\n        # Filter the training dataframe for this class\n        class_df = train_df[train_df['diagnosis'] == class_idx]\n        \n        # Sample 4 images (if less than 4, take all available)\n        n_samples = min(4, len(class_df))\n        samples = class_df.sample(n=n_samples, random_state=sample_seed + class_idx)\n        \n        for _, sample_row in samples.iterrows():\n            img_id = sample_row['id_code']\n            \n            # Load preprocessed image for this size\n            size_dir = os.path.join(OUTPUT_BASE, size_str)\n            npy_path = os.path.join(size_dir, 'train', img_id + '.npy')\n            processed_img = np.load(npy_path)\n            \n            # --- Column 0: Original ---\n            original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n            original = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n            axes[row_idx, 0].imshow(original)\n            \n            # --- Column 1: Cropped ---\n            cropped = crop_image_from_gray(original)\n            axes[row_idx, 1].imshow(cropped)\n            \n            # --- Column 2: Resized (with CLAHE) ---\n            cropped = crop_image_from_gray(original)\n            resized = cv2.resize(cropped, target_size, interpolation=cv2.INTER_AREA)\n            clahe_img = apply_clahe(resized)\n            axes[row_idx, 2].imshow(clahe_img)\n            \n            # --- Column 3: Final preprocessed ---\n            axes[row_idx, 3].imshow(processed_img)\n            \n            # Add class label on the far left\n            if row_idx % 4 == 0:\n                axes[row_idx, 0].set_ylabel(class_names[class_idx], fontsize=11, rotation=90, labelpad=15)\n            \n            row_idx += 1\n    \n    # Set column titles on the first row\n    col_titles = ['Original', '1. Cropped', f'2. CLAHE ({size_str})', '3. Final Preprocessed']\n    for col, title in enumerate(col_titles):\n        axes[0, col].set_title(title, fontsize=12, fontweight='bold')\n    \n    # Turn off all axes for a clean look\n    for ax in axes.flat:\n        ax.axis('off')\n    \n    viz_grid_path = os.path.join(OUTPUT_BASE, f'preprocessing_grid_{size_str}.png')\n    plt.tight_layout()\n    plt.savefig(viz_grid_path, dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    print(f\"✅ 20-image preprocessing grid saved: {viz_grid_path}\")\n\n# ==========================================\n# FINAL SUMMARY\n# ==========================================\nprint(\"\\n\" + \"=\"*60)\nprint(\"✅ PREPROCESSING PIPELINE COMPLETE FOR ALL SIZES\")\nprint(\"=\"*60)\nprint(f\"\\nOutput directory structure:\")\nprint(f\"  {OUTPUT_BASE}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"  ├── {size_str}/\")\n    print(f\"  │   ├── train/              ({len(train_df)} images)\")\n    print(f\"  │   ├── val/                ({len(val_df)} images)\")\n    print(f\"  │   ├── test/               ({len(test_df)} images)\")\n    print(f\"  │   ├── retina_mean.npy     {all_means[size_str]}\")\n    print(f\"  │   └── retina_std.npy      {all_stds[size_str]}\")\nprint(f\"  ├── train_split.csv\")\nprint(f\"  ├── val_split.csv\")\nprint(f\"  ├── test_split.csv\")\nprint(f\"  ├── size_comparison.png\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"  └── preprocessing_grid_{size_str}.png\")\nprint(\"\\n✅ All sizes processed successfully!\")\nprint(\"✅ Ready for PyTorch DataLoader with multiple input sizes!\")\nprint(\"=\"*60)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T06:03:10.216813Z","iopub.execute_input":"2026-09-16T06:03:10.217209Z","iopub.status.idle":"2026-09-16T06:37:17.330611Z","shell.execute_reply.started":"2026-09-16T06:03:10.217184Z","shell.execute_reply":"2026-09-16T06:37:17.329414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}