{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-28T12:17:32.755553Z","iopub.execute_input":"2026-07-28T12:17:32.755857Z","iopub.status.idle":"2026-07-28T12:17:39.848054Z","shell.execute_reply.started":"2026-07-28T12:17:32.755822Z","shell.execute_reply":"2026-07-28T12:17:39.847159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')\n# CELL 1: IMPORTS\n# ==========================================\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nprint(\"✅ All imports successful\")\n\n# CELL 2: CONFIGURATION AND FOLDER CREATION\n# ==========================================\nBASE_PATH = '/kaggle/input/competitions/aptos2019-blindness-detection/'\nIMAGE_PATH = os.path.join(BASE_PATH, 'train_images/')\nCSV_PATH = os.path.join(BASE_PATH, 'train.csv')\n\n# Define multiple target sizes\nTARGET_SIZES = [(224, 224), (384, 384), (512, 512)]\nRANDOM_SEED = 42\n\n# Output folder structure for all sizes\nOUTPUT_BASE = '/kaggle/working/APTOS_Preprocessed/'\n\n# Create base output directory\nos.makedirs(OUTPUT_BASE, exist_ok=True)\n\n# Create folders for each size\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    os.makedirs(os.path.join(size_dir, 'train'), exist_ok=True)\n    os.makedirs(os.path.join(size_dir, 'val'), exist_ok=True)\n    os.makedirs(os.path.join(size_dir, 'test'), exist_ok=True)\n\nprint(f\"✅ Folder structure created for sizes: {TARGET_SIZES}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"   {size_str}:\")\n    print(f\"   ├── train/\")\n    print(f\"   ├── val/\")\n    print(f\"   └── test/\")\n\n# CELL 3: LOAD DATASET\n# ==========================================\ndf = pd.read_csv(CSV_PATH)\nprint(f\"Total images in dataset: {len(df)}\")\nprint(f\"\\nClass distribution:\\n{df['diagnosis'].value_counts().sort_index()}\")\n\n# CELL 4: TRAIN / VALIDATION / TEST SPLIT (70-15-15)\n# ==========================================\n# Split 1: 70% Train, 30% Temp (which will be split into Val + Test)\nfrom sklearn.model_selection import train_test_split\ntrain_df, temp_df = train_test_split(\n    df, \n    test_size=0.30, \n    stratify=df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Split 2: 15% Val, 15% Test (50% of 30% = 15%)\nval_df, test_df = train_test_split(\n    temp_df, \n    test_size=0.50,  # 50% of 30% = 15% total\n    stratify=temp_df['diagnosis'], \n    random_state=RANDOM_SEED\n)\n\n# Reset indices for cleanliness\ntrain_df = train_df.reset_index(drop=True)\nval_df = val_df.reset_index(drop=True)\ntest_df = test_df.reset_index(drop=True)\n\nprint(f\"\\n✅ Split Complete:\")\nprint(f\"Train: {len(train_df)} images\")\nprint(f\"Val:   {len(val_df)} images\")\nprint(f\"Test:  {len(test_df)} images\")\n\nprint(f\"\\nTrain class distribution:\\n{train_df['diagnosis'].value_counts().sort_index()}\")\n\n# CELL 5: CREATE OUTPUT FOLDER STRUCTURE\n# ==========================================\n# (Already created in Cell 2, this is for documentation)\nprint(\"✅ Output folder structure is ready:\")\nprint(f\"   {OUTPUT_BASE}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"   ├── {size_str}/\")\n    print(f\"   │   ├── train/\")\n    print(f\"   │   ├── val/\")\n    print(f\"   │   └── test/\")\n\n# CELL 6: SAVE SPLIT CSV FILES\n# ==========================================\ntrain_csv_path = os.path.join(OUTPUT_BASE, 'train_split.csv')\nval_csv_path = os.path.join(OUTPUT_BASE, 'val_split.csv')\ntest_csv_path = os.path.join(OUTPUT_BASE, 'test_split.csv')\n\ntrain_df.to_csv(train_csv_path, index=False)\nval_df.to_csv(val_csv_path, index=False)\ntest_df.to_csv(test_csv_path, index=False)\n\nprint(\"✅ Split CSVs saved:\")\nprint(f\"   {train_csv_path}\")\nprint(f\"   {val_csv_path}\")\nprint(f\"   {test_csv_path}\")\nprint(\"\\n⚠️  DO NOT open test_split.csv until final evaluation!\")\n\n# CELL 7: DEFINE PREPROCESSING FUNCTIONS\n# ==========================================\n\ndef crop_image_from_gray(img, tol=7):\n    \"\"\"Remove black borders by cropping to the fundus circle.\"\"\"\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n    elif img.ndim == 3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        mask = gray_img > tol\n        if mask.sum() == 0: \n            return img\n        img1 = img[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n        img2 = img[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n        img3 = img[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n        return np.stack([img1, img2, img3], axis=-1)\n\ndef apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8)):\n    \"\"\"Apply CLAHE on the L-channel of LAB to preserve color.\"\"\"\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=clipLimit, tileGridSize=tileGridSize)\n    l_enhanced = clahe.apply(l)\n    enhanced_img = cv2.merge((l_enhanced, a, b))\n    return cv2.cvtColor(enhanced_img, cv2.COLOR_LAB2RGB)\n\ndef preprocess_image(image_path, target_size):\n    \"\"\"\n    Apply the FULL preprocessing pipeline with specified target size.\n    ORDER: Crop -> Ben Graham -> RESIZE -> CLAHE -> float conversion [0, 1]\n    \n    Returns: float32 image in range [0, 1], ready for normalization.\n    \"\"\"\n    # 1. Read image\n    img = cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\n    \n    # 2. Circle Crop\n    img = crop_image_from_gray(img, tol=7)\n    \n    # 3. Ben Graham Illumination Correction\n    \n    # 4. RESIZE (CRITICAL: Do this BEFORE CLAHE)\n    img = cv2.resize(img, target_size, interpolation=cv2.INTER_AREA)\n    \n    # 5. CLAHE (Now applied on the exact target grid)\n    img = apply_clahe(img, clipLimit=2.0, tileGridSize=(8, 8))\n    \n    # 6. Convert to float [0, 1]\n    img = img.astype(np.float32) / 255.0\n    \n    return img\n\nprint(\"✅ All preprocessing functions defined\")\n\n# CELL 8-10: PREPROCESS AND SAVE IMAGES FOR EACH SIZE\n# ==========================================\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n{'='*60}\")\n    print(f\"Processing size: {size_str}\")\n    print(f\"{'='*60}\")\n    \n    # Set directories for this size\n    size_base = os.path.join(OUTPUT_BASE, size_str)\n    train_dir = os.path.join(size_base, 'train')\n    val_dir = os.path.join(size_base, 'val')\n    test_dir = os.path.join(size_base, 'test')\n    \n    # Process TRAINING images\n    print(f\"\\n⏳ Preprocessing and saving TRAINING images for {size_str}...\")\n    print(f\"Total training images to process: {len(train_df)}\")\n    \n    for idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(train_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(train_df)} training images saved to {train_dir}\")\n    \n    # Process VALIDATION images\n    print(f\"\\n⏳ Preprocessing and saving VALIDATION images for {size_str}...\")\n    print(f\"Total validation images to process: {len(val_df)}\")\n    \n    for idx, row in tqdm(val_df.iterrows(), total=len(val_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(val_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(val_df)} validation images saved to {val_dir}\")\n    \n    # Process TEST images\n    print(f\"\\n⏳ Preprocessing and saving TEST images for {size_str}...\")\n    print(f\"Total test images to process: {len(test_df)}\")\n    \n    for idx, row in tqdm(test_df.iterrows(), total=len(test_df)):\n        img_name = row['id_code'] + '.png'\n        img_path = os.path.join(IMAGE_PATH, img_name)\n        \n        # Preprocess with target size\n        processed_img = preprocess_image(img_path, target_size)\n        \n        # Save as .npy\n        output_path = os.path.join(test_dir, row['id_code'] + '.npy')\n        np.save(output_path, processed_img)\n    \n    print(f\"✅ {len(test_df)} test images saved to {test_dir}\")\n\n# CELL 11: COMPUTE MEAN & STD FOR EACH SIZE\n# ==========================================\nall_means = {}\nall_stds = {}\n\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    train_dir = os.path.join(size_dir, 'train')\n    \n    print(f\"\\n⏳ Computing Mean and Std for {size_str} from saved TRAINING images...\")\n    print(f\"Total training images to analyze: {len(train_df)}\")\n    \n    all_pixels = []\n    \n    for idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n        # Load the SAVED preprocessed image\n        npy_path = os.path.join(train_dir, row['id_code'] + '.npy')\n        processed_img = np.load(npy_path)\n        \n        # Flatten\n        all_pixels.append(processed_img.reshape(-1, 3))\n    \n    # Stack all pixels into a single array\n    all_pixels = np.vstack(all_pixels)\n    \n    # Calculate global mean and std per channel\n    dataset_mean = np.mean(all_pixels, axis=0)\n    dataset_std = np.std(all_pixels, axis=0)\n    \n    all_means[size_str] = dataset_mean\n    all_stds[size_str] = dataset_std\n    \n    print(f\"\\n✅ Dataset Statistics for {size_str}:\")\n    print(f\"   Mean (R, G, B): [{dataset_mean[0]:.4f}, {dataset_mean[1]:.4f}, {dataset_mean[2]:.4f}]\")\n    print(f\"   Std  (R, G, B): [{dataset_std[0]:.4f}, {dataset_std[1]:.4f}, {dataset_std[2]:.4f}]\")\n    \n    # Save normalization statistics for this size\n    mean_path = os.path.join(size_dir, 'retina_mean.npy')\n    std_path = os.path.join(size_dir, 'retina_std.npy')\n    \n    np.save(mean_path, dataset_mean)\n    np.save(std_path, dataset_std)\n    \n    print(f\"✅ Normalization statistics saved:\")\n    print(f\"   {mean_path}\")\n    print(f\"   {std_path}\")\n\n# CELL 12: VISUALIZATION - COMPARE ALL SIZES\n# ==========================================\nprint(\"\\n⏳ Creating comparison visualization for all sizes...\")\n\n# Pick one image from train split\nsample_id = train_df.iloc[0]['id_code']\noriginal_path = os.path.join(IMAGE_PATH, sample_id + '.png')\noriginal = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n\nfig, axes = plt.subplots(3, 4, figsize=(24, 16))\ntitles = ['Original', 'Cropped', 'CLAHE', 'Final']\n\n# Show original and processing steps for each size\nfor row, target_size in enumerate(TARGET_SIZES):\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(OUTPUT_BASE, size_str)\n    npy_path = os.path.join(size_dir, 'train', sample_id + '.npy')\n    \n    # Load the preprocessed image\n    processed_img = np.load(npy_path)\n    \n    # Column 0: Original (same for all rows)\n    if row == 0:\n        axes[row, 0].imshow(original)\n        axes[row, 0].set_title(\"Original (Raw)\")\n    else:\n        axes[row, 0].imshow(original)\n        axes[row, 0].set_title(\"Original (Raw)\")\n    \n    # Column 1: Cropped\n    cropped = crop_image_from_gray(original)\n    axes[row, 1].imshow(cropped)\n    axes[row, 1].set_title(f\"Cropped\")\n    \n    # Column 2: CLAHE (on resized)\n    cropped = crop_image_from_gray(original)\n    resized = cv2.resize(cropped, target_size, interpolation=cv2.INTER_AREA)\n    clahe_img = apply_clahe(resized)\n    axes[row, 2].imshow(clahe_img)\n    axes[row, 2].set_title(f\"CLAHE on {size_str}\")\n    \n    # Column 3: Final processed image (from saved .npy)\n    axes[row, 3].imshow(processed_img)\n    axes[row, 3].set_title(f\"Final {size_str}\")\n    \n    # Add row label\n    axes[row, 0].set_ylabel(size_str, fontsize=12, rotation=90, labelpad=20)\n\n# Turn off axes for cleanliness\nfor ax in axes.flat:\n    ax.axis('off')\n\nviz_compare_path = os.path.join(OUTPUT_BASE, 'size_comparison.png')\nplt.tight_layout()\nplt.savefig(viz_compare_path, dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(f\"✅ Size comparison visualization saved: {viz_compare_path}\")\n\n# CELL 13: 20-IMAGE VISUALIZATION BY CLASS FOR EACH SIZE\n# ==========================================\nclass_names = {\n    0: 'Class 0 - No DR',\n    1: 'Class 1 - Mild',\n    2: 'Class 2 - Moderate',\n    3: 'Class 3 - Severe',\n    4: 'Class 4 - Proliferative'\n}\n\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n⏳ Creating 20-image visualization for {size_str}...\")\n    \n    num_rows = 20\n    num_cols = 4\n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(24, 4 * num_rows))\n    \n    sample_seed = 42\n    row_idx = 0\n    \n    for class_idx in range(5):\n        # Filter the training dataframe for this class\n        class_df = train_df[train_df['diagnosis'] == class_idx]\n        \n        # Sample 4 images (if less than 4, take all available)\n        n_samples = min(4, len(class_df))\n        samples = class_df.sample(n=n_samples, random_state=sample_seed + class_idx)\n        \n        for _, sample_row in samples.iterrows():\n            img_id = sample_row['id_code']\n            \n            # Load preprocessed image for this size\n            size_dir = os.path.join(OUTPUT_BASE, size_str)\n            npy_path = os.path.join(size_dir, 'train', img_id + '.npy')\n            processed_img = np.load(npy_path)\n            \n            # --- Column 0: Original ---\n            original_path = os.path.join(IMAGE_PATH, img_id + '.png')\n            original = cv2.cvtColor(cv2.imread(original_path), cv2.COLOR_BGR2RGB)\n            axes[row_idx, 0].imshow(original)\n            \n            # --- Column 1: Cropped ---\n            cropped = crop_image_from_gray(original)\n            axes[row_idx, 1].imshow(cropped)\n            \n            # --- Column 2: Resized (with CLAHE) ---\n            cropped = crop_image_from_gray(original)\n            resized = cv2.resize(cropped, target_size, interpolation=cv2.INTER_AREA)\n            clahe_img = apply_clahe(resized)\n            axes[row_idx, 2].imshow(clahe_img)\n            \n            # --- Column 3: Final preprocessed ---\n            axes[row_idx, 3].imshow(processed_img)\n            \n            # Add class label on the far left\n            if row_idx % 4 == 0:\n                axes[row_idx, 0].set_ylabel(class_names[class_idx], fontsize=11, rotation=90, labelpad=15)\n            \n            row_idx += 1\n    \n    # Set column titles on the first row\n    col_titles = ['Original', '1. Cropped', f'2. CLAHE ({size_str})', '3. Final Preprocessed']\n    for col, title in enumerate(col_titles):\n        axes[0, col].set_title(title, fontsize=12, fontweight='bold')\n    \n    # Turn off all axes for a clean look\n    for ax in axes.flat:\n        ax.axis('off')\n    \n    viz_grid_path = os.path.join(OUTPUT_BASE, f'preprocessing_grid_{size_str}.png')\n    plt.tight_layout()\n    plt.savefig(viz_grid_path, dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    print(f\"✅ 20-image preprocessing grid saved: {viz_grid_path}\")\n\n# ==========================================\n# FINAL SUMMARY\n# ==========================================\nprint(\"\\n\" + \"=\"*60)\nprint(\"✅ PREPROCESSING PIPELINE COMPLETE FOR ALL SIZES\")\nprint(\"=\"*60)\nprint(f\"\\nOutput directory structure:\")\nprint(f\"  {OUTPUT_BASE}\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"  ├── {size_str}/\")\n    print(f\"  │   ├── train/              ({len(train_df)} images)\")\n    print(f\"  │   ├── val/                ({len(val_df)} images)\")\n    print(f\"  │   ├── test/               ({len(test_df)} images)\")\n    print(f\"  │   ├── retina_mean.npy     {all_means[size_str]}\")\n    print(f\"  │   └── retina_std.npy      {all_stds[size_str]}\")\nprint(f\"  ├── train_split.csv\")\nprint(f\"  ├── val_split.csv\")\nprint(f\"  ├── test_split.csv\")\nprint(f\"  ├── size_comparison.png\")\nfor size in TARGET_SIZES:\n    size_str = f\"{size[0]}x{size[1]}\"\n    print(f\"  └── preprocessing_grid_{size_str}.png\")\nprint(\"\\n✅ All sizes processed successfully!\")\nprint(\"✅ Ready for PyTorch DataLoader with multiple input sizes!\")\nprint(\"=\"*60)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-28T12:17:39.849836Z","iopub.execute_input":"2026-07-28T12:17:39.850194Z","iopub.status.idle":"2026-07-28T12:47:38.577334Z","shell.execute_reply.started":"2026-07-28T12:17:39.850168Z","shell.execute_reply":"2026-07-28T12:47:38.574012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# CELL 15: PYTORCH DATASET & DATALOADERS (MULTI-SIZE VERSION)\n# ==========================================\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport albumentations as A\nfrom tqdm import tqdm\nimport glob\n\nprint(\"🔍 Looking for preprocessed data...\")\n\n# --- 1. FIND THE CORRECT PATH ---\n# Try multiple possible locations\npossible_paths = [\n    '/kaggle/input/notebooks/sahithishiny/fyp-review1/APTOS_Preprocessed',\n    '/kaggle/input/notebooks/sreecharithakadiri/fyp-review1/APTOS_Preprocessed',\n    '/kaggle/input/aptos-preprocessed/APTOS_Preprocessed',\n    '/kaggle/working/APTOS_Preprocessed',\n    '/kaggle/input/notebooks/fyp-review1/APTOS_Preprocessed'\n]\n\ndata_base = None\nfor path in possible_paths:\n    if os.path.exists(path):\n        data_base = path\n        print(f\"✅ Found data at: {data_base}\")\n        break\n\n# If not found, search for APTOS_Preprocessed folder\nif data_base is None:\n    print(\"⚠️ Searching for APTOS_Preprocessed folder...\")\n    for root, dirs, files in os.walk('/kaggle/'):\n        if 'APTOS_Preprocessed' in dirs:\n            data_base = os.path.join(root, 'APTOS_Preprocessed')\n            print(f\"✅ Found data at: {data_base}\")\n            break\n\nif data_base is None:\n    # Check if data was preprocessed and saved in working directory\n    if os.path.exists('/kaggle/working/APTOS_Preprocessed'):\n        data_base = '/kaggle/working/APTOS_Preprocessed'\n        print(f\"✅ Found data at: {data_base}\")\n    else:\n        raise FileNotFoundError(f\"\"\"\n❌ APTOS_Preprocessed folder not found!\n\nPlease check:\n1. Did you run the preprocessing code first?\n2. Is the data in one of these locations:\n   {possible_paths}\n3. Or run the preprocessing code from the previous cells\n\nCurrent working directory contents:\n{os.listdir('/kaggle/working/') if os.path.exists('/kaggle/working/') else 'Cannot access /kaggle/working/'}\n\nInput directory contents:\n{os.listdir('/kaggle/input/') if os.path.exists('/kaggle/input/') else 'Cannot access /kaggle/input/'}\n\"\"\")\n\nprint(f\"✅ Data path found: {data_base}\")\n\n# --- 2. DEFINE SIZES AND PATHS ---\nTARGET_SIZES = [(224, 224), (384, 384), (512, 512)]\n\n# Output paths (Writeable - for saving models and new results)\nOUTPUT_BASE = '/kaggle/working/'\nos.makedirs(OUTPUT_BASE, exist_ok=True)\n\n# Check which sizes are available\navailable_sizes = []\nfor target_size in TARGET_SIZES:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(data_base, size_str)\n    if os.path.exists(size_dir):\n        available_sizes.append(target_size)\n        print(f\"✅ Found {size_str} data\")\n    else:\n        print(f\"⚠️ {size_str} data not found at {size_dir}\")\n\nif not available_sizes:\n    raise FileNotFoundError(f\"No size folders found in {data_base}. Contents: {os.listdir(data_base)}\")\n\nprint(f\"📊 Available sizes: {available_sizes}\")\n\n# --- 3. LOAD OR COMPUTE MEAN/STD FOR EACH SIZE ---\nmean_dict = {}\nstd_dict = {}\n\nfor target_size in available_sizes:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    size_dir = os.path.join(data_base, size_str)\n    \n    mean_path = os.path.join(size_dir, 'retina_mean.npy')\n    std_path = os.path.join(size_dir, 'retina_std.npy')\n    \n    if os.path.exists(mean_path) and os.path.exists(std_path):\n        mean_dict[size_str] = np.load(mean_path)\n        std_dict[size_str] = np.load(std_path)\n        print(f\"✅ Loaded Mean/Std for {size_str}\")\n    else:\n        print(f\"⚠️ Mean/Std files not found for {size_str}. Computing...\")\n        train_df = pd.read_csv(os.path.join(data_base, 'train_split.csv'))\n        train_dir = os.path.join(size_dir, 'train')\n        all_pixels = []\n        for idx, row in tqdm(train_df.iterrows(), total=len(train_df)):\n            npy_path = os.path.join(train_dir, row['id_code'] + '.npy')\n            if os.path.exists(npy_path):\n                img = np.load(npy_path)\n                all_pixels.append(img.reshape(-1, 3))\n        if all_pixels:\n            all_pixels = np.vstack(all_pixels)\n            dataset_mean = np.mean(all_pixels, axis=0)\n            dataset_std = np.std(all_pixels, axis=0)\n            mean_dict[size_str] = dataset_mean\n            std_dict[size_str] = dataset_std\n            np.save(os.path.join(OUTPUT_BASE, f'retina_mean_{size_str}.npy'), dataset_mean)\n            np.save(os.path.join(OUTPUT_BASE, f'retina_std_{size_str}.npy'), dataset_std)\n            print(f\"✅ Computed Mean/Std for {size_str}\")\n        else:\n            print(f\"❌ No .npy files found in {train_dir}\")\n\n# --- 4. PYTORCH DATASET CLASS ---\nclass APTOSDataset(Dataset):\n    def __init__(self, csv_path, data_dir, phase='train', mean_tensor=None, std_tensor=None):\n        self.df = pd.read_csv(csv_path)\n        self.data_dir = data_dir\n        self.phase = phase\n        self.mean = mean_tensor\n        self.std = std_tensor\n        \n        # LIGHT AUGMENTATIONS (Only for training)\n        if phase == 'train':\n            self.transform = A.Compose([\n                A.HorizontalFlip(p=0.5),\n                A.Rotate(limit=15, border_mode=cv2.BORDER_CONSTANT, fill_value=0, p=0.8),\n                A.RandomBrightnessContrast(brightness_limit=0.05, contrast_limit=0.05, p=0.2),\n            ])\n        else:\n            self.transform = None\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_id = row['id_code']\n        label = int(row['diagnosis'])\n        \n        npy_path = os.path.join(self.data_dir, img_id + '.npy')\n        \n        # Check if file exists\n        if not os.path.exists(npy_path):\n            print(f\"⚠️ File not found: {npy_path}\")\n            # Return a dummy image if file is missing (shouldn't happen)\n            dummy_img = np.zeros((224, 224, 3), dtype=np.float32)\n            img = dummy_img\n        else:\n            img = np.load(npy_path)\n        \n        # --- TARGETED AUGMENTATION FOR MINORITY CLASSES ---\n        if self.phase == 'train' and label in [1, 3]:\n            strong_aug = A.Compose([\n                A.HorizontalFlip(p=0.5),\n                A.Rotate(limit=30, border_mode=cv2.BORDER_CONSTANT, fill_value=0, p=0.9),\n                A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=20, \n                                   border_mode=cv2.BORDER_CONSTANT, value=0, p=0.5),\n                A.RandomBrightnessContrast(brightness_limit=0.15, contrast_limit=0.15, p=0.5),\n                A.GaussianBlur(blur_limit=(3, 5), p=0.3),\n                A.CLAHE(clip_limit=2.0, tile_grid_size=(8, 8), p=0.3)\n            ])\n            augmented = strong_aug(image=img)\n            img = augmented['image']\n        elif self.phase == 'train':\n            light_aug = A.Compose([\n                A.HorizontalFlip(p=0.5),\n                A.Rotate(limit=15, border_mode=cv2.BORDER_CONSTANT, fill_value=0, p=0.8),\n                A.RandomBrightnessContrast(brightness_limit=0.05, contrast_limit=0.05, p=0.2),\n            ])\n            augmented = light_aug(image=img)\n            img = augmented['image']\n        \n        # Convert to tensor: (H, W, C) -> (C, H, W)\n        img = torch.from_numpy(img).permute(2, 0, 1).float()\n        \n        # Final normalization\n        img = (img - self.mean.squeeze(0)) / self.std.squeeze(0)\n        \n        return img, label\n\n# --- 5. CREATE DATALOADERS FOR EACH SIZE ---\nBATCH_SIZE = 16\ndataloaders = {}\n\n# Load split CSVs\ntrain_csv = os.path.join(data_base, 'train_split.csv')\nval_csv = os.path.join(data_base, 'val_split.csv')\ntest_csv = os.path.join(data_base, 'test_split.csv')\n\nif not all([os.path.exists(f) for f in [train_csv, val_csv, test_csv]]):\n    print(\"⚠️ Split CSV files not found. Creating from scratch...\")\n    df = pd.read_csv(os.path.join(os.path.dirname(data_base), 'train.csv') if os.path.exists(os.path.join(os.path.dirname(data_base), 'train.csv')) \n                     else 'train.csv')\n    # Use your original split logic here if needed\n\nfor target_size in available_sizes:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n📦 Creating DataLoaders for {size_str}...\")\n    \n    size_dir = os.path.join(data_base, size_str)\n    train_dir = os.path.join(size_dir, 'train')\n    val_dir = os.path.join(size_dir, 'val')\n    test_dir = os.path.join(size_dir, 'test')\n    \n    # Convert mean/std to tensors\n    mean_tensor = torch.tensor(mean_dict[size_str], dtype=torch.float32).view(1, 3, 1, 1)\n    std_tensor = torch.tensor(std_dict[size_str], dtype=torch.float32).view(1, 3, 1, 1)\n    \n    # Create datasets\n    train_dataset = APTOSDataset(\n        csv_path=os.path.join(data_base, 'train_split.csv'),\n        data_dir=train_dir,\n        phase='train',\n        mean_tensor=mean_tensor,\n        std_tensor=std_tensor\n    )\n    val_dataset = APTOSDataset(\n        csv_path=os.path.join(data_base, 'val_split.csv'),\n        data_dir=val_dir,\n        phase='val',\n        mean_tensor=mean_tensor,\n        std_tensor=std_tensor\n    )\n    test_dataset = APTOSDataset(\n        csv_path=os.path.join(data_base, 'test_split.csv'),\n        data_dir=test_dir,\n        phase='test',\n        mean_tensor=mean_tensor,\n        std_tensor=std_tensor\n    )\n    \n    # Create dataloaders\n    train_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, \n                              num_workers=2, pin_memory=True)\n    val_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, \n                            num_workers=2, pin_memory=True)\n    test_loader = DataLoader(test_dataset, batch_size=BATCH_SIZE, shuffle=False, \n                             num_workers=2, pin_memory=True)\n    \n    dataloaders[size_str] = {\n        'train': train_loader,\n        'val': val_loader,\n        'test': test_loader,\n        'train_dataset': train_dataset,\n        'val_dataset': val_dataset,\n        'test_dataset': test_dataset\n    }\n    \n    print(f\"✅ DataLoaders for {size_str} created\")\n    print(f\"   Train batches: {len(train_loader)}\")\n    print(f\"   Val batches:   {len(val_loader)}\")\n    print(f\"   Test batches:  {len(test_loader)}\")\n\nprint(\"\\n✅ All DataLoaders Created Successfully!\")\n\n# Sanity check\nfor size_str, loaders in dataloaders.items():\n    train_loader = loaders['train']\n    for images, labels in train_loader:\n        print(f\"\\n✅ Sanity Check for {size_str}:\")\n        print(f\"   Images shape: {images.shape}\")\n        print(f\"   Labels: {labels[:5]}\")\n        print(f\"   Image min: {images.min():.3f}, max: {images.max():.3f}\")\n        break\n\n# ==========================================\n# CELL 16: CLASS WEIGHTS & FOCAL LOSS\n# ==========================================\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.utils.class_weight import compute_class_weight\n\n# Load training labels\nif os.path.exists(os.path.join(data_base, 'train_split.csv')):\n    train_df = pd.read_csv(os.path.join(data_base, 'train_split.csv'))\n    train_labels = train_df['diagnosis'].values\nelse:\n    # Fallback: use original CSV\n    original_csv = os.path.join(os.path.dirname(data_base), 'train.csv')\n    if os.path.exists(original_csv):\n        df = pd.read_csv(original_csv)\n        train_labels = df['diagnosis'].values\n    else:\n        raise FileNotFoundError(\"Could not find training labels\")\n\nclass_weights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.array([0, 1, 2, 3, 4]),\n    y=train_labels\n)\nclass_weights = torch.tensor(class_weights, dtype=torch.float32)\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"📊 CLASS WEIGHTS (Inverse Frequency)\")\nprint(\"=\"*50)\nfor i, w in enumerate(class_weights):\n    print(f\"   Class {i} (Count: {np.sum(train_labels == i)}): {w:.4f}\")\nprint(\"=\"*50)\n\nclass FocalLoss(nn.Module):\n    def __init__(self, alpha=None, gamma=2.0):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n\n    def forward(self, inputs, targets):\n        ce_loss = F.cross_entropy(inputs, targets, reduction='none')\n        pt = torch.exp(-ce_loss)\n        if self.alpha is not None:\n            alpha_t = self.alpha[targets]\n            focal_loss = alpha_t * (1 - pt) ** self.gamma * ce_loss\n        else:\n            focal_loss = (1 - pt) ** self.gamma * ce_loss\n        return focal_loss.mean()\n\nprint(\"\\n✅ Focal Loss initialized successfully!\")\n\n# ==========================================\n# CELL 17: TRAINING LOOP FOR ALL SIZES\n# ==========================================\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torchvision.models as models\nfrom tqdm import tqdm\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport warnings\nwarnings.filterwarnings('ignore')\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"🚀 Using device: {device}\")\n\nOUTPUT_BASE = '/kaggle/working/'\nos.makedirs(OUTPUT_BASE, exist_ok=True)\n\n# Create subdirectories for each size\nfor target_size in available_sizes:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    os.makedirs(os.path.join(OUTPUT_BASE, size_str), exist_ok=True)\n\nEPOCHS = 15\nresults = {}\n\nfor target_size in available_sizes:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    print(f\"\\n{'='*60}\")\n    print(f\"🏋️  TRAINING FOR {size_str}\")\n    print(f\"{'='*60}\")\n    \n    train_loader = dataloaders[size_str]['train']\n    val_loader = dataloaders[size_str]['val']\n    train_dataset = dataloaders[size_str]['train_dataset']\n    val_dataset = dataloaders[size_str]['val_dataset']\n    \n    model = models.densenet121(weights=models.DenseNet121_Weights.IMAGENET1K_V1)\n    model.classifier = nn.Linear(model.classifier.in_features, 5)\n    model = model.to(device)\n    \n    print(f\"   Model: DenseNet-121\")\n    print(f\"   Total parameters: {sum(p.numel() for p in model.parameters()) / 1e6:.2f}M\")\n    \n    criterion = FocalLoss(alpha=class_weights.to(device), gamma=2.0)\n    optimizer = optim.AdamW(model.parameters(), lr=1e-4, weight_decay=1e-4)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer, mode='min', factor=0.5, patience=2\n    )\n    \n    best_val_kappa = 0.0\n    best_epoch = 0\n    \n    history = {\n        'train_loss': [],\n        'val_loss': [],\n        'val_acc': [],\n        'val_kappa': [],\n        'lr': []\n    }\n    \n    for epoch in range(1, EPOCHS + 1):\n        # Training Phase\n        model.train()\n        train_loss = 0.0\n        train_correct = 0\n        train_total = 0\n        progress = tqdm(train_loader, desc=f\"Epoch {epoch}/{EPOCHS} [Train]\")\n        for images, labels in progress:\n            images, labels = images.to(device), labels.to(device)\n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item() * images.size(0)\n            _, preds = torch.max(outputs, 1)\n            train_correct += (preds == labels).sum().item()\n            train_total += labels.size(0)\n            progress.set_postfix({\n                'loss': f\"{loss.item():.4f}\",\n                'acc': f\"{(train_correct/train_total)*100:.1f}%\"\n            })\n        avg_train_loss = train_loss / len(train_dataset)\n        train_acc = train_correct / train_total\n\n        # Validation Phase\n        model.eval()\n        val_loss = 0.0\n        all_preds, all_labels = [], []\n        with torch.no_grad():\n            for images, labels in tqdm(val_loader, desc=f\"Epoch {epoch}/{EPOCHS} [Val]\", leave=False):\n                images, labels = images.to(device), labels.to(device)\n                outputs = model(images)\n                loss = criterion(outputs, labels)\n                val_loss += loss.item() * images.size(0)\n                _, preds = torch.max(outputs, 1)\n                all_preds.extend(preds.cpu().numpy())\n                all_labels.extend(labels.cpu().numpy())\n        avg_val_loss = val_loss / len(val_dataset)\n        val_acc = accuracy_score(all_labels, all_preds)\n        val_kappa = cohen_kappa_score(all_labels, all_preds, weights='quadratic')\n\n        history['train_loss'].append(avg_train_loss)\n        history['val_loss'].append(avg_val_loss)\n        history['val_acc'].append(val_acc)\n        history['val_kappa'].append(val_kappa)\n        history['lr'].append(optimizer.param_groups[0]['lr'])\n\n        print(f\"\\n📊 Epoch {epoch}/{EPOCHS} Summary:\")\n        print(f\"   Train Loss: {avg_train_loss:.4f} | Train Acc: {train_acc*100:.2f}%\")\n        print(f\"   Val Loss:   {avg_val_loss:.4f} | Val Acc:   {val_acc*100:.2f}% | QWK: {val_kappa:.4f}\")\n        print(f\"   LR: {optimizer.param_groups[0]['lr']:.2e}\")\n\n        scheduler.step(avg_val_loss)\n\n        if val_kappa > best_val_kappa:\n            best_val_kappa = val_kappa\n            best_epoch = epoch\n            size_dir = os.path.join(OUTPUT_BASE, size_str)\n            torch.save(model.state_dict(), os.path.join(size_dir, f'densenet121_{size_str}_best.pth'))\n            print(f\"   ✅ New best model saved! (QWK = {val_kappa:.4f})\")\n        print(\"-\"*60)\n\n    print(\"\\n\" + \"=\"*60)\n    print(f\"🏁 TRAINING FOR {size_str} COMPLETE!\")\n    print(f\"🏆 Best Validation QWK: {best_val_kappa:.4f} (Epoch {best_epoch})\")\n    print(\"=\"*60)\n    \n    results[size_str] = {\n        'best_kappa': best_val_kappa,\n        'best_epoch': best_epoch,\n        'history': history\n    }\n\n# ==========================================\n# CELL 18: FINAL SUMMARY\n# ==========================================\nprint(\"\\n\" + \"=\"*60)\nprint(\"✅ MULTI-SIZE TRAINING COMPLETE!\")\nprint(\"=\"*60)\n\nprint(\"\\n📊 Best Results Summary:\")\nprint(\"-\"*50)\nfor size_str, result in results.items():\n    print(f\"{size_str}: QWK = {result['best_kappa']:.4f} (Epoch {result['best_epoch']})\")\nprint(\"-\"*50)\n\nif results:\n    best_size = max(results.items(), key=lambda x: x[1]['best_kappa'])\n    print(f\"\\n🏆 Best performing model: {best_size[0]} with QWK = {best_size[1]['best_kappa']:.4f}\")\n\nprint(f\"\\n📁 Models saved to: {OUTPUT_BASE}\")\nfor target_size in available_sizes:\n    size_str = f\"{target_size[0]}x{target_size[1]}\"\n    if os.path.exists(os.path.join(OUTPUT_BASE, size_str, f'densenet121_{size_str}_best.pth')):\n        print(f\"   ✅ {size_str}/densenet121_{size_str}_best.pth\")\n    else:\n        print(f\"   ❌ {size_str}/densenet121_{size_str}_best.pth (not saved)\")\n\nprint(\"\\n\" + \"=\"*60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-28T16:14:59.361936Z","iopub.execute_input":"2026-07-28T16:14:59.362636Z","execution_failed":"2026-07-28T18:25:33.905Z"}},"outputs":[],"execution_count":null}]}