{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:43:35.739944Z","iopub.execute_input":"2026-07-10T13:43:35.740206Z","iopub.status.idle":"2026-07-10T13:43:48.623505Z","shell.execute_reply.started":"2026-07-10T13:43:35.740173Z","shell.execute_reply":"2026-07-10T13:43:48.622856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 1: IMPORTS\n# ==========================================\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nprint(\"✅ All imports successful\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:44:18.230558Z","iopub.execute_input":"2026-07-10T13:44:18.231270Z","iopub.status.idle":"2026-07-10T13:44:19.649563Z","shell.execute_reply.started":"2026-07-10T13:44:18.231239Z","shell.execute_reply":"2026-07-10T13:44:19.648739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 2: CONFIGURATION AND FOLDER CREATION\n# ==========================================\nBASE_PATH = '/kaggle/input/competitions/aptos2019-blindness-detection/'\nIMAGE_PATH = os.path.join(BASE_PATH, 'train_images/')\nCSV_PATH = os.path.join(BASE_PATH, 'train.csv')\n\nTARGET_SIZE = (384, 384)  # Goldilocks zone for Ensemble\nRANDOM_SEED = 42\n\n# Output folder structure\nOUTPUT_BASE = '/kaggle/working/APTOS_Preprocessed/'\nTRAIN_DIR = os.path.join(OUTPUT_BASE, 'train/')\nVAL_DIR = os.path.join(OUTPUT_BASE, 'val/')\nTEST_DIR = os.path.join(OUTPUT_BASE, 'test/')\n\n# Create folder structure\nos.makedirs(TRAIN_DIR, exist_ok=True)\nos.makedirs(VAL_DIR, exist_ok=True)\nos.makedirs(TEST_DIR, exist_ok=True)\n\nprint(f\"✅ Folder structure created:\")\nprint(f\"   Train: {TRAIN_DIR}\")\nprint(f\"   Val:   {VAL_DIR}\")\nprint(f\"   Test:  {TEST_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:44:41.716562Z","iopub.execute_input":"2026-07-10T13:44:41.717527Z","iopub.status.idle":"2026-07-10T13:44:41.725022Z","shell.execute_reply.started":"2026-07-10T13:44:41.717498Z","shell.execute_reply":"2026-07-10T13:44:41.724248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3: LOAD DATASET\n# ==========================================\ndf = pd.read_csv(CSV_PATH)\nprint(f\"Total images in dataset: {len(df)}\")\nprint(f\"\\nClass distribution:\\n{df['diagnosis'].value_counts().sort_index()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:50:57.210511Z","iopub.execute_input":"2026-07-10T13:50:57.211351Z","iopub.status.idle":"2026-07-10T13:50:57.329003Z","shell.execute_reply.started":"2026-07-10T13:50:57.211313Z","shell.execute_reply":"2026-07-10T13:50:57.328369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 4: TRAIN / VALIDATION / TEST SPLIT (70-15-15)\n# ==========================================\n# Split 1: 70% Train, 30% Temp (which will be split into Val + Test)\ntrain_df, temp_df = train_test_split(\n    df, \n    test_size=0.30, \n    stratify=df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Split 2: 15% Val, 15% Test (50% of 30% = 15%)\nval_df, test_df = train_test_split(\n    temp_df, \n    test_size=0.50,  # 50% of 30% = 15% total\n    stratify=temp_df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Reset indices for cleanliness\ntrain_df = train_df.reset_index(drop=True)\nval_df = val_df.reset_index(drop=True)\ntest_df = test_df.reset_index(drop=True)\n\nprint(f\"\\n✅ Split Complete:\")\nprint(f\"Train: {len(train_df)} images\")\nprint(f\"Val:   {len(val_df)} images\")\nprint(f\"Test:  {len(test_df)} images\")\n\nprint(f\"\\nTrain class distribution:\\n{train_df['diagnosis'].value_counts().sort_index()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:51:03.730392Z","iopub.execute_input":"2026-07-10T13:51:03.731059Z","iopub.status.idle":"2026-07-10T13:51:03.748952Z","shell.execute_reply.started":"2026-07-10T13:51:03.731029Z","shell.execute_reply":"2026-07-10T13:51:03.748123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 5: CREATE OUTPUT FOLDER STRUCTURE\n# ==========================================\n# (Already created in Cell 2, this is for documentation)\nprint(\"✅ Output folder structure is ready:\")\nprint(f\"   {OUTPUT_BASE}\")\nprint(f\"   ├── train/\")\nprint(f\"   ├── val/\")\nprint(f\"   ├── test/\")\nprint(f\"   └── (split CSVs and stats will be saved here)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:51:07.217685Z","iopub.execute_input":"2026-07-10T13:51:07.217937Z","iopub.status.idle":"2026-07-10T13:51:07.223010Z","shell.execute_reply.started":"2026-07-10T13:51:07.217915Z","shell.execute_reply":"2026-07-10T13:51:07.222197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 6: SAVE SPLIT CSV FILES\n# ==========================================\ntrain_csv_path = os.path.join(OUTPUT_BASE, 'train_split.csv')\nval_csv_path = os.path.join(OUTPUT_BASE, 'val_split.csv')\ntest_csv_path = os.path.join(OUTPUT_BASE, 'test_split.csv')\n\ntrain_df.to_csv(train_csv_path, index=False)\nval_df.to_csv(val_csv_path, index=False)\ntest_df.to_csv(test_csv_path, index=False)\n\nprint(\"✅ Split CSVs saved:\")\nprint(f\"   {train_csv_path}\")\nprint(f\"   {val_csv_path}\")\nprint(f\"   {test_csv_path}\")\nprint(\"\\n⚠️  DO NOT open test_split.csv until final evaluation!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:51:10.180112Z","iopub.execute_input":"2026-07-10T13:51:10.180422Z","iopub.status.idle":"2026-07-10T13:51:10.201066Z","shell.execute_reply.started":"2026-07-10T13:51:10.180399Z","shell.execute_reply":"2026-07-10T13:51:10.200438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 7: DEFINE PREPROCESSING FUNCTIONS\n# ==========================================\n\ndef crop_image_from_gray(img, tol=7):\n    \"\"\"Remove black borders by cropping to the fundus circle.\"\"\"\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        mask = gray_img > tol\n        if mask.sum() == 0: \n            return img\n        img1 = img[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n        img2 = img[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n        img3 = img[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n        return np.stack([img1, img2, img3], axis=-1)\n\n\ndef ben_graham_process(img, sigmaX=30):\n    \"\"\"Standardize uneven illumination (Kaggle Grandmaster trick).\"\"\"\n    return cv2.addWeighted(img, 4, cv2.GaussianBlur(img, (0, 0), sigmaX), -4, 128)\n\n\ndef apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8)):\n    \"\"\"Apply CLAHE on the L-channel of LAB to preserve color.\"\"\"\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=clipLimit, tileGridSize=tileGridSize)\n    l_enhanced = clahe.apply(l)\n    enhanced_img = cv2.merge((l_enhanced, a, b))\n    return cv2.cvtColor(enhanced_img, cv2.COLOR_LAB2RGB)\n\n\ndef preprocess_image(image_path):\n    \"\"\"\n    Apply the FULL preprocessing pipeline.\n    ORDER: Crop -> Ben Graham -> RESIZE -> CLAHE -> float conversion [0, 1]\n    \n    Returns: float32 image in range [0, 1], ready for normalization.\n    \"\"\"\n    # 1. Read image\n    img = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n    \n    # 2. Circle Crop\n    img = crop_image_from_gray(img, tol=7)\n    \n    # 3. Ben Graham Illumination Correction\n    img = ben_graham_process(img, sigmaX=30)\n    \n    # 4. RESIZE (CRITICAL: Do this BEFORE CLAHE)\n    img = cv2.resize(img, TARGET_SIZE, interpolation=cv2.INTER_AREA)\n    \n    # 5. CLAHE (Now applied on the exact 384x384 grid)\n    img = apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8))\n    \n    # 6. Convert to float [0, 1]\n    img = img.astype(np.float32) / 255.0\n    \n    return img\n\n\nprint(\"✅ All preprocessing functions defined\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:51:16.176322Z","iopub.execute_input":"2026-07-10T13:51:16.176576Z","iopub.status.idle":"2026-07-10T13:51:16.186434Z","shell.execute_reply.started":"2026-07-10T13:51:16.176556Z","shell.execute_reply":"2026-07-10T13:51:16.185581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 8: PREPROCESS AND SAVE TRAINING IMAGES\n# ==========================================\nprint(\"\\n⏳ Preprocessing and saving TRAINING images...\")\nprint(f\"Total training images to process: {len(train_df)}\")\n\nfor idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n    img_name = row['id_code'] + '.png'\n    img_path = os.path.join(IMAGE_PATH, img_name)\n    \n    # Preprocess\n    processed_img = preprocess_image(img_path)\n    \n    # Save as .npy (preserves float32 precision)\n    output_path = os.path.join(TRAIN_DIR, row['id_code'] + '.npy')\n    np.save(output_path, processed_img)\n\nprint(f\"✅ {len(train_df)} training images saved to {TRAIN_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T13:51:20.770202Z","iopub.execute_input":"2026-07-10T13:51:20.770791Z","iopub.status.idle":"2026-07-10T14:15:10.376652Z","shell.execute_reply.started":"2026-07-10T13:51:20.770764Z","shell.execute_reply":"2026-07-10T14:15:10.375583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 9: PREPROCESS AND SAVE VALIDATION IMAGES\n# ==========================================\nprint(\"\\n⏳ Preprocessing and saving VALIDATION images...\")\nprint(f\"Total validation images to process: {len(val_df)}\")\n\nfor idx, row in tqdm(val_df.iterrows(), total=len(val_df)):\n    img_name = row['id_code'] + '.png'\n    img_path = os.path.join(IMAGE_PATH, img_name)\n    \n    # Preprocess\n    processed_img = preprocess_image(img_path)\n    \n    # Save as .npy\n    output_path = os.path.join(VAL_DIR, row['id_code'] + '.npy')\n    np.save(output_path, processed_img)\n\nprint(f\"✅ {len(val_df)} validation images saved to {VAL_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:15:26.753075Z","iopub.execute_input":"2026-07-10T14:15:26.753792Z","iopub.status.idle":"2026-07-10T14:20:40.557777Z","shell.execute_reply.started":"2026-07-10T14:15:26.753761Z","shell.execute_reply":"2026-07-10T14:20:40.556877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 10: PREPROCESS AND SAVE TEST IMAGES\n# ==========================================\nprint(\"\\n⏳ Preprocessing and saving TEST images...\")\nprint(f\"Total test images to process: {len(test_df)}\")\n\nfor idx, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    img_name = row['id_code'] + '.png'\n    img_path = os.path.join(IMAGE_PATH, img_name)\n    \n    # Preprocess\n    processed_img = preprocess_image(img_path)\n    \n    # Save as .npy\n    output_path = os.path.join(TEST_DIR, row['id_code'] + '.npy')\n    np.save(output_path, processed_img)\n\nprint(f\"✅ {len(test_df)} test images saved to {TEST_DIR}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:22:07.373003Z","iopub.execute_input":"2026-07-10T14:22:07.373876Z","iopub.status.idle":"2026-07-10T14:27:06.930738Z","shell.execute_reply.started":"2026-07-10T14:22:07.373842Z","shell.execute_reply":"2026-07-10T14:27:06.929686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 11: COMPUTE MEAN & STD FROM SAVED TRAINING IMAGES\n# ==========================================\nprint(\"\\n⏳ Computing Mean and Std from saved TRAINING images...\")\nprint(f\"Total training images to analyze: {len(train_df)}\")\n\nall_pixels = []\n\nfor idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n    # Load the SAVED preprocessed image (not the original)\n    npy_path = os.path.join(TRAIN_DIR, row['id_code'] + '.npy')\n    processed_img = np.load(npy_path)  # Already float32, range [0, 1]\n    \n    # Flatten: (384, 384, 3) -> (147456, 3)\n    all_pixels.append(processed_img.reshape(-1, 3))\n\n# Stack all pixels into a single array\nall_pixels = np.vstack(all_pixels)\n\n# Calculate global mean and std per channel (R, G, B)\ndataset_mean = np.mean(all_pixels, axis=0)\ndataset_std = np.std(all_pixels, axis=0)\n\nprint(f\"\\n✅ Dataset Statistics Calculated (from saved TRAIN images):\")\nprint(f\"   Mean (R, G, B): [{dataset_mean[0]:.4f}, {dataset_mean[1]:.4f}, {dataset_mean[2]:.4f}]\")\nprint(f\"   Std  (R, G, B): [{dataset_std[0]:.4f}, {dataset_std[1]:.4f}, {dataset_std[2]:.4f}]\")\n\n# ==========================================\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:45:46.103230Z","iopub.execute_input":"2026-07-10T14:45:46.104023Z","iopub.status.idle":"2026-07-10T14:46:18.493480Z","shell.execute_reply.started":"2026-07-10T14:45:46.103994Z","shell.execute_reply":"2026-07-10T14:46:18.492731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 12: SAVE NORMALIZATION STATISTICS\n# ==========================================\nmean_path = os.path.join(OUTPUT_BASE, 'retina_mean.npy')\nstd_path = os.path.join(OUTPUT_BASE, 'retina_std.npy')\n\nnp.save(mean_path, dataset_mean)\nnp.save(std_path, dataset_std)\n\nprint(f\"✅ Normalization statistics saved:\")\nprint(f\"   {mean_path}\")\nprint(f\"   {std_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:46:18.494855Z","iopub.execute_input":"2026-07-10T14:46:18.495211Z","iopub.status.idle":"2026-07-10T14:46:18.501442Z","shell.execute_reply.started":"2026-07-10T14:46:18.495180Z","shell.execute_reply":"2026-07-10T14:46:18.500778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 13: VISUALIZATION - PREPROCESSING PIPELINE CHECK (3x6)\n# ==========================================\nprint(\"\\n⏳ Creating 3x6 preprocessing pipeline visualization...\")\n\n# Pick one image from each split from the SAVED preprocessed images\nsample_ids = {\n    'Train': train_df.iloc[0]['id_code'],\n    'Val': val_df.iloc[0]['id_code'],\n    'Test': test_df.iloc[0]['id_code']\n}\n\nfig, axes = plt.subplots(3, 5, figsize=(24, 12))\ntitles = ['Original', 'Cropped', 'Ben Graham', 'Resized (384)', 'CLAHE']\n\nfor row, (split_name, img_id) in enumerate(sample_ids.items()):\n    # Determine which directory to load from\n    if split_name == 'Train':\n        npy_path = os.path.join(TRAIN_DIR, img_id + '.npy')\n        original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n    elif split_name == 'Val':\n        npy_path = os.path.join(VAL_DIR, img_id + '.npy')\n        original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n    else:  # Test\n        npy_path = os.path.join(TEST_DIR, img_id + '.npy')\n        original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n    \n    # --- Column 0: Original (from original source) ---\n    original = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n    axes[row, 0].imshow(original)\n    axes[row, 0].set_title(f\"{split_name} - Original\")\n    \n    # --- Column 1: Cropped ---\n    cropped = crop_image_from_gray(original)\n    axes[row, 1].imshow(cropped)\n    axes[row, 1].set_title(\"1. Cropped\")\n    \n    # --- Column 2: Ben Graham ---\n    bg = ben_graham_process(cropped)\n    axes[row, 2].imshow(bg)\n    axes[row, 2].set_title(\"2. Ben Graham\")\n    \n    # --- Column 3: Resized ---\n    resized = cv2.resize(bg, TARGET_SIZE, interpolation=cv2.INTER_AREA)\n    axes[row, 3].imshow(resized)\n    axes[row, 3].set_title(f\"3. Resized {TARGET_SIZE}\")\n    \n    # --- Column 4: CLAHE ---\n    clahe_img = apply_clahe(resized)\n    axes[row, 4].imshow(clahe_img)\n    axes[row, 4].set_title(\"4. CLAHE\")\n    \n    \n\n# Turn off axes for cleanliness\nfor ax in axes.flat:\n    ax.axis('off')\n\nviz1_path = os.path.join(OUTPUT_BASE, 'preprocessing_pipeline_check.png')\nplt.tight_layout()\nplt.savefig(viz1_path, dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(f\"✅ Preprocessing pipeline visualization saved: {viz1_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:54:55.369899Z","iopub.execute_input":"2026-07-10T14:54:55.370743Z","iopub.status.idle":"2026-07-10T14:55:06.722888Z","shell.execute_reply.started":"2026-07-10T14:54:55.370712Z","shell.execute_reply":"2026-07-10T14:55:06.721759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 14: 20-IMAGE VISUALIZATION BY CLASS\n# ==========================================\nprint(\"\\n⏳ Creating 20-image (5 classes × 4 images) preprocessing grid...\")\n\nclass_names = {\n    0: 'Class 0 - No DR',\n    1: 'Class 1 - Mild',\n    2: 'Class 2 - Moderate',\n    3: 'Class 3 - Severe',\n    4: 'Class 4 - Proliferative'\n}\n\nnum_rows = 20\nnum_cols = 5\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(24, 4 * num_rows))\n\nsample_seed = 42\nrow_idx = 0\n\nfor class_idx in range(5):\n    # Filter the training dataframe for this class\n    class_df = train_df[train_df['diagnosis'] == class_idx]\n    \n    # Sample 4 images (if less than 4, take all available)\n    n_samples = min(4, len(class_df))\n    samples = class_df.sample(n=n_samples, random_state=sample_seed + class_idx)\n    \n    for _, sample_row in samples.iterrows():\n        img_id = sample_row['id_code']\n        \n        # --- Load original for stages 1-4 ---\n        original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n        original = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n        \n        # --- Column 0: Original ---\n        axes[row_idx, 0].imshow(original)\n        \n        # --- Column 1: Cropped ---\n        cropped = crop_image_from_gray(original)\n        axes[row_idx, 1].imshow(cropped)\n        \n        # --- Column 2: Ben Graham ---\n        bg = ben_graham_process(cropped)\n        axes[row_idx, 2].imshow(bg)\n        \n        # --- Column 3: Resized ---\n        resized = cv2.resize(bg, TARGET_SIZE, interpolation=cv2.INTER_AREA)\n        axes[row_idx, 3].imshow(resized)\n        \n        # --- Column 4: CLAHE ---\n        clahe_img = apply_clahe(resized)\n        axes[row_idx, 4].imshow(clahe_img)\n        \n        \n        # Add class label on the far left for the first image of each class\n        if row_idx % 4 == 0:\n            axes[row_idx, 0].set_ylabel(class_names[class_idx], fontsize=11, rotation=90, labelpad=15)\n        \n        row_idx += 1\n\n# Set column titles on the first row\ncol_titles = ['Original', '1. Cropped', '2. Ben Graham', '3. Resized (384)', '4. CLAHE']\nfor col, title in enumerate(col_titles):\n    axes[0, col].set_title(title, fontsize=12, fontweight='bold')\n\n# Turn off all axes for a clean look\nfor ax in axes.flat:\n    ax.axis('off')\n\nviz2_path = os.path.join(OUTPUT_BASE, 'full_preprocessing_grid_20images.png')\nplt.tight_layout()\nplt.savefig(viz2_path, dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(f\"✅ 20-image preprocessing grid saved: {viz2_path}\")\n\n# ==========================================\n# FINAL SUMMARY\n# ==========================================\nprint(\"\\n\" + \"=\"*60)\nprint(\"✅ PREPROCESSING PIPELINE COMPLETE\")\nprint(\"=\"*60)\nprint(f\"\\nOutput directory structure:\")\nprint(f\"  {OUTPUT_BASE}\")\nprint(f\"  ├── train/              ({len(train_df)} images)\")\nprint(f\"  ├── val/                ({len(val_df)} images)\")\nprint(f\"  ├── test/               ({len(test_df)} images)\")\nprint(f\"  ├── train_split.csv\")\nprint(f\"  ├── val_split.csv\")\nprint(f\"  ├── test_split.csv\")\nprint(f\"  ├── retina_mean.npy     {dataset_mean}\")\nprint(f\"  ├── retina_std.npy      {dataset_std}\")\nprint(f\"  ├── preprocessing_pipeline_check.png\")\nprint(f\"  └── full_preprocessing_grid_20images.png\")\nprint(\"\\n✅ Ready for PyTorch DataLoader!\")\nprint(\"=\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T14:55:23.535523Z","iopub.execute_input":"2026-07-10T14:55:23.536348Z","iopub.status.idle":"2026-07-10T14:56:40.111045Z","shell.execute_reply.started":"2026-07-10T14:55:23.536316Z","shell.execute_reply":"2026-07-10T14:56:40.109788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = np.load(os.path.join(TRAIN_DIR, train_df.iloc[0][\"id_code\"] + \".npy\"))\nprint(sample.shape)\nprint(sample.dtype)\nprint(sample.min(), sample.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T15:01:32.168551Z","iopub.execute_input":"2026-07-10T15:01:32.169659Z","iopub.status.idle":"2026-07-10T15:01:32.183356Z","shell.execute_reply.started":"2026-07-10T15:01:32.169614Z","shell.execute_reply":"2026-07-10T15:01:32.182392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(dataset_mean)\nprint(dataset_std)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T15:01:45.516020Z","iopub.execute_input":"2026-07-10T15:01:45.516795Z","iopub.status.idle":"2026-07-10T15:01:45.522844Z","shell.execute_reply.started":"2026-07-10T15:01:45.516763Z","shell.execute_reply":"2026-07-10T15:01:45.521904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Train files:\", len(os.listdir(TRAIN_DIR)))\nprint(\"Val files:\", len(os.listdir(VAL_DIR)))\nprint(\"Test files:\", len(os.listdir(TEST_DIR)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T15:03:52.108889Z","iopub.execute_input":"2026-07-10T15:03:52.109699Z","iopub.status.idle":"2026-07-10T15:03:52.116977Z","shell.execute_reply.started":"2026-07-10T15:03:52.109669Z","shell.execute_reply":"2026-07-10T15:03:52.116210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = np.load(os.path.join(TRAIN_DIR,\n                              train_df.iloc[0][\"id_code\"] + \".npy\"))\n\nprint(sample.shape)\nprint(sample.dtype)\nprint(sample.min(), sample.max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-10T15:04:03.626547Z","iopub.execute_input":"2026-07-10T15:04:03.627257Z","iopub.status.idle":"2026-07-10T15:04:03.634940Z","shell.execute_reply.started":"2026-07-10T15:04:03.627227Z","shell.execute_reply":"2026-07-10T15:04:03.633976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}