{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# System and File Handling\nimport os\n\n# Data Manipulation, Numerical Processing & Statistical Testing\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\n\n# Computer Vision & Image Processing\nimport cv2\nfrom PIL import Image\n\n# Data Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Progress Monitoring\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-21T15:43:24.613019Z","iopub.execute_input":"2026-09-21T15:43:24.613376Z","iopub.status.idle":"2026-09-21T15:43:26.39529Z","shell.execute_reply.started":"2026-09-21T15:43:24.613346Z","shell.execute_reply":"2026-09-21T15:43:26.394322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dataset\ndf = pd.read_csv('/kaggle/input/competitions/aptos2019-blindness-detection/train.csv')\n\nimage_dir = '/kaggle/input/competitions/aptos2019-blindness-detection/train_images'\n\n# Add a column with the direct file path for each image\ndf['file_path'] = df['id_code'].apply(lambda x: os.path.join(image_dir, f\"{x}.png\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-21T15:43:29.992634Z","iopub.execute_input":"2026-09-21T15:43:29.993117Z","iopub.status.idle":"2026-09-21T15:43:30.023728Z","shell.execute_reply.started":"2026-09-21T15:43:29.993071Z","shell.execute_reply":"2026-09-21T15:43:30.022897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to scan all image paths for file or image-level corruption\ndef scan_for_corrupted_images(df, min_brightness_threshold=5):\n    corrupted_files = []\n    \n    print(f\"Scanning {len(df)} images for corruption and errors...\\n\")\n    \n    for idx, row in tqdm(df.iterrows(), total=len(df)):\n        path = row['file_path']\n        \n        # 1. Check if the file actually exists on disk\n        if not os.path.exists(path):\n            corrupted_files.append({\n                'index': idx, 'file_path': path, 'reason': 'File Not Found'\n            })\n            continue\n            \n        # 2. Check for zero-byte / empty files\n        if os.path.getsize(path) == 0:\n            corrupted_files.append({\n                'index': idx, 'file_path': path, 'reason': 'Zero-Byte File (Empty)'\n            })\n            continue\n            \n        # 3. Attempt to read the image via OpenCV\n        img = cv2.imread(path)\n        if img is None:\n            corrupted_files.append({\n                'index': idx, 'file_path': path, 'reason': 'Unreadable / Corrupted Image'\n            })\n            continue\n            \n        # 4. Check for extreme low-intensity \"blackout\" images\n        mean_val = img.mean()\n        if mean_val < min_brightness_threshold:\n            corrupted_files.append({\n                'index': idx, 'file_path': path, 'reason': f'Severe Blackout Image (Mean Pixel = {mean_val:.2f})'\n            })\n\n    # Summary Output\n    corrupted_df = pd.DataFrame(corrupted_files)\n    \n    if len(corrupted_df) == 0:\n        print(\"\\nSuccess! All images are valid, readable, and non-corrupted.\")\n    else:\n        print(f\"\\nWarning: Found {len(corrupted_df)} problematic files!\")\n        \n    return corrupted_df\n\n# Execute scan and output findings report\ncorrupted_report = scan_for_corrupted_images(df)\n\nif not corrupted_report.empty:\n    display(corrupted_report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-21T15:43:33.291941Z","iopub.execute_input":"2026-09-21T15:43:33.292269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Class Distribution (Brighter for highest counts, darker for lower counts)\nplt.figure(figsize=(8, 4))\nax = sns.countplot(x='diagnosis', data=df)\n\nplt.title('Class Imbalance: DR Severity Grades (0–4)', fontsize=12)\nplt.xlabel('Diagnosis Grade (0: No DR, 1: Mild, 2: Moderate, 3: Severe, 4: Proliferative)', fontsize=10)\nplt.ylabel('Image Count', fontsize=10)\n\ntotal = len(df)\nfor p in ax.patches:\n    height = p.get_height()\n    percentage = (height / total) * 100\n    ax.annotate(\n        f'{int(height)}\\n({percentage:.1f}%)', \n        (p.get_x() + p.get_width() / 2., height),\n        ha='center', va='bottom', \n        xytext=(0, 3), textcoords='offset points',\n        fontsize=9\n    )\n\nplt.ylim(0, max([p.get_height() for p in ax.patches]) * 1.15)\nplt.tight_layout()\nplt.show()\n\n# Print summary metrics and compute loss weights\nprint(f\"Total Dataset Size: {len(df)} images\\n\")\nprint(\"Class Counts & Percentages:\")\nbreakdown = pd.DataFrame({\n    'Count': df['diagnosis'].value_counts(),\n    'Percentage (%)': np.round(df['diagnosis'].value_counts(normalize=True) * 100, 2)\n}).sort_index()\ndisplay(breakdown)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:23:54.512731Z","iopub.execute_input":"2026-09-20T14:23:54.513588Z","iopub.status.idle":"2026-09-20T14:23:54.871472Z","shell.execute_reply.started":"2026-09-20T14:23:54.513555Z","shell.execute_reply":"2026-09-20T14:23:54.870799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"widths, heights, aspect_ratios = [], [], []\n\n# Extract image dimensions from headers\nfor path in df['file_path']:\n    with Image.open(path) as img:\n        w, h = img.size\n        widths.append(w)\n        heights.append(h)\n        aspect_ratios.append(w / h)\n\n# Plot Spatial Geometry Distributions\nfig, ax = plt.subplots(2, 2, figsize=(14, 8))\n\n# 1. Width Distribution\nsns.histplot(widths, ax=ax[0, 0], color='teal', kde=True, bins=20)\nax[0, 0].set_title('Raw Image Widths Distribution', fontsize=11)\nax[0, 0].set_xlabel('Width (pixels)')\nax[0, 0].set_ylabel('Count')\n\n# 2. Height Distribution\nsns.histplot(heights, ax=ax[0, 1], color='teal', kde=True, bins=20)\nax[0, 1].set_title('Raw Image Heights Distribution', fontsize=11)\nax[0, 1].set_xlabel('Height (pixels)')\nax[0, 1].set_ylabel('Count')\n\n# 3. Width vs Height Scatter\nsns.scatterplot(x=widths, y=heights, ax=ax[1, 0], alpha=0.5, color='purple')\nax[1, 0].set_title('Width vs. Height Scatter Plot', fontsize=11)\nax[1, 0].set_xlabel('Width (pixels)')\nax[1, 0].set_ylabel('Height (pixels)')\n\n# 4. Aspect Ratio Distribution\nsns.histplot(aspect_ratios, ax=ax[1, 1], color='coral', kde=True, bins=20)\nax[1, 1].set_title('Aspect Ratio (W / H) Distribution', fontsize=11)\nax[1, 1].set_xlabel('Aspect Ratio')\nax[1, 1].set_ylabel('Count')\n\nplt.tight_layout()\nplt.show()\n\n# Print Aspect Ratio summary\nunique_ratios = sorted(set([round(r, 2) for r in aspect_ratios]))\nprint(f\"Unique Aspect Ratios found across dataset: {unique_ratios}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:24:10.954521Z","iopub.execute_input":"2026-09-20T14:24:10.954919Z","iopub.status.idle":"2026-09-20T14:24:14.110797Z","shell.execute_reply.started":"2026-09-20T14:24:10.954886Z","shell.execute_reply":"2026-09-20T14:24:14.109583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to calculate mean brightness across RGB channels for a single image\ndef get_mean_intensity(path):\n    img = cv2.imread(path)\n    if img is None:\n        return np.nan\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    return np.mean(img)\n\n# 1. Compute mean intensity with progress tracking\ntqdm.pandas(desc=\"Calculating Image Intensities\")\ndf['mean_intensity'] = df['file_path'].progress_apply(get_mean_intensity)\n\n# 2. Plot overall brightness distribution using Seaborn with KDE overlay\nplt.figure(figsize=(8, 4))\nsns.histplot(\n    data=df, \n    x='mean_intensity', \n    bins=30, \n    color='darkorange', \n    kde=True, \n    edgecolor='black'\n)\n\nplt.title('Raw Image Overall Brightness Distribution', fontsize=12)\nplt.xlabel('Mean Pixel Intensity (0 = Pitch Black, 255 = Pure White)', fontsize=10)\nplt.ylabel('Image Count', fontsize=10)\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:24:32.272486Z","iopub.execute_input":"2026-09-20T14:24:32.27282Z","iopub.status.idle":"2026-09-20T14:31:10.841609Z","shell.execute_reply.started":"2026-09-20T14:24:32.272779Z","shell.execute_reply":"2026-09-20T14:31:10.840626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Visualize Brightness Distribution Across Diagnosis Grades\nplt.figure(figsize=(10, 5))\n\n# Boxplot shows the median, quartiles, and outliers per grade\nsns.boxplot(\n    x='diagnosis', \n    y='mean_intensity', \n    data=df, \n    showmeans=True,\n    meanprops={\"marker\": \"o\", \"markerfacecolor\": \"red\", \"markeredgecolor\": \"red\"}\n)\n\n# Overlay individual data points (jittered) to see density\nsns.stripplot(\n    x='diagnosis', \n    y='mean_intensity', \n    data=df, \n    color='black', \n    alpha=0.15, \n    jitter=0.2\n)\n\nplt.title('Image Brightness (Mean Intensity) Across Retinopathy Severity Grades', fontsize=12)\nplt.xlabel('Diagnosis Grade (0: No DR, 1: Mild, 2: Moderate, 3: Severe, 4: Proliferative)', fontsize=10)\nplt.ylabel('Mean Pixel Intensity (0–255)', fontsize=10)\nplt.grid(axis='y', linestyle='--', alpha=0.5)\n\nplt.tight_layout()\nplt.show()\n\n# 2. Statistical Correlation Check (One-Way ANOVA Test)\n# Group intensities by diagnosis grade\ngrade_groups = [group['mean_intensity'].dropna().values for _, group in df.groupby('diagnosis')]\n\n# Perform One-Way ANOVA test\nf_stat, p_val = stats.f_oneway(*grade_groups)\n\nprint(\"--- Statistical Analysis (One-Way ANOVA) ---\")\nprint(f\"F-Statistic: {f_stat:.4f}\")\nprint(f\"p-value:     {p_val:.4e}\")\n\nif p_val > 0.05:\n    print(\"\\nTakeaway: No statistically significant brightness difference across grades (p > 0.05).\")\n    print(\"Lighting variations are independent of disease severity—safe to apply uniform CLAHE/normalization.\")\nelse:\n    print(\"\\nTakeaway: Statistically significant brightness difference detected across grades (p <= 0.05).\")\n    print(\"Careful: Normalize lighting across all grades so the model doesn't learn brightness as a shortcut feature.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:31:46.408713Z","iopub.execute_input":"2026-09-20T14:31:46.409567Z","iopub.status.idle":"2026-09-20T14:31:46.727122Z","shell.execute_reply.started":"2026-09-20T14:31:46.409534Z","shell.execute_reply":"2026-09-20T14:31:46.726282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Register tqdm with pandas for progress tracking\ntqdm.pandas(desc=\"Extracting RGB Channel Means\")\n\ndef get_rgb_channel_means(path):\n    img = cv2.imread(path)\n    if img is None:\n        return np.nan, np.nan, np.nan\n    \n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    # Calculate mean across height and width for each channel (0: R, 1: G, 2: B)\n    return img[:, :, 0].mean(), img[:, :, 1].mean(), img[:, :, 2].mean()\n\n# 1. Apply function across dataset and expand into separate DataFrame columns\nrgb_means = df['file_path'].progress_apply(get_rgb_channel_means)\ndf[['red_mean', 'green_mean', 'blue_mean']] = pd.DataFrame(rgb_means.tolist(), index=df.index)\n\n# 2. Plot Channel Density Distributions\nplt.figure(figsize=(10, 5))\nsns.kdeplot(df['red_mean'], color='red', label='Red Channel', fill=True, alpha=0.3)\nsns.kdeplot(df['green_mean'], color='green', label='Green Channel', fill=True, alpha=0.3)\nsns.kdeplot(df['blue_mean'], color='blue', label='Blue Channel', fill=True, alpha=0.3)\n\nplt.title('Color Channel Mean Distribution Across Retinal Fundus Images', fontsize=12)\nplt.xlabel('Mean Pixel Intensity (0 - 255)', fontsize=10)\nplt.ylabel('Density', fontsize=10)\nplt.legend(title='Channels')\nplt.grid(axis='x', linestyle='--', alpha=0.5)\n\nplt.tight_layout()\nplt.show()\n\n# 3. Print channel metric summary\nprint(\"--- RGB Channel Averages ---\")\nprint(f\"Red Channel Mean:   {df['red_mean'].mean():.2f}\")\nprint(f\"Green Channel Mean: {df['green_mean'].mean():.2f}\")\nprint(f\"Blue Channel Mean:  {df['blue_mean'].mean():.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:32:07.076288Z","iopub.execute_input":"2026-09-20T14:32:07.077222Z","iopub.status.idle":"2026-09-20T14:39:02.53059Z","shell.execute_reply.started":"2026-09-20T14:32:07.077171Z","shell.execute_reply":"2026-09-20T14:39:02.529678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Register tqdm with pandas for progress bar tracking\ntqdm.pandas(desc=\"Calculating Eye Tissue Ratio\")\n\ndef get_eye_tissue_ratio(path):\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        return np.nan\n        \n    # Threshold: Consider pixels > 10 as foreground \"eye content\"\n    eye_pixels = np.sum(img > 10)\n    total_pixels = img.size\n    \n    # Return percentage of active eye tissue\n    return (eye_pixels / total_pixels) * 100\n\n# 1. Apply calculation across dataset with progress tracking\ndf['eye_tissue_ratio'] = df['file_path'].progress_apply(get_eye_tissue_ratio)\n\n# 2. Plot Tissue Ratio Distribution\nplt.figure(figsize=(8, 4))\nsns.histplot(\n    data=df, \n    x='eye_tissue_ratio', \n    color='teal', \n    kde=True, \n    bins=25,\n    edgecolor='black'\n)\n\nplt.title('Ocular Globe Area vs. Frame Size (% Eye Content - Full Dataset)', fontsize=12)\nplt.xlabel('Percentage of Image Covered by Eye Tissue (%)', fontsize=10)\nplt.ylabel('Image Count', fontsize=10)\nplt.grid(axis='y', linestyle='--', alpha=0.5)\n\nplt.tight_layout()\nplt.show()\n\n# 3. Print summary metrics\nprint(\"--- Eye Tissue Coverage Metrics ---\")\nprint(f\"Average Eye Frame Coverage: {df['eye_tissue_ratio'].mean():.2f}%\")\nprint(f\"Min Eye Coverage Found:     {df['eye_tissue_ratio'].min():.2f}%\")\nprint(f\"Max Eye Coverage Found:     {df['eye_tissue_ratio'].max():.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T14:40:20.430574Z","iopub.execute_input":"2026-09-20T14:40:20.430999Z","iopub.status.idle":"2026-09-20T14:46:44.881967Z","shell.execute_reply.started":"2026-09-20T14:40:20.430969Z","shell.execute_reply":"2026-09-20T14:46:44.88065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:44:10.118708Z","iopub.execute_input":"2026-09-20T15:44:10.119089Z","iopub.status.idle":"2026-09-20T15:44:10.144445Z","shell.execute_reply.started":"2026-09-20T15:44:10.11904Z","shell.execute_reply":"2026-09-20T15:44:10.143457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df['id_code'].duplicated().sum())\nprint(df['id_code'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:45:01.786403Z","iopub.execute_input":"2026-09-20T15:45:01.786812Z","iopub.status.idle":"2026-09-20T15:45:01.796245Z","shell.execute_reply.started":"2026-09-20T15:45:01.786781Z","shell.execute_reply":"2026-09-20T15:45:01.795278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df['id_code'].head(20).tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:45:43.419439Z","iopub.execute_input":"2026-09-20T15:45:43.419795Z","iopub.status.idle":"2026-09-20T15:45:43.425221Z","shell.execute_reply.started":"2026-09-20T15:45:43.419743Z","shell.execute_reply":"2026-09-20T15:45:43.42417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df['diagnosis'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:46:23.663057Z","iopub.execute_input":"2026-09-20T15:46:23.66341Z","iopub.status.idle":"2026-09-20T15:46:23.683673Z","shell.execute_reply.started":"2026-09-20T15:46:23.663379Z","shell.execute_reply":"2026-09-20T15:46:23.682556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\n\nsample_paths = df['file_path'].sample(6, random_state=42)\n\nplt.figure(figsize=(15, 8))\n\nfor i, path in enumerate(sample_paths):\n    img = Image.open(path)\n\n    plt.subplot(2, 3, i + 1)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(f\"{img.size}\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:51:53.737232Z","iopub.execute_input":"2026-09-20T15:51:53.737561Z","iopub.status.idle":"2026-09-20T15:51:57.91112Z","shell.execute_reply.started":"2026-09-20T15:51:53.737525Z","shell.execute_reply":"2026-09-20T15:51:57.909581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\nsample_paths = df['file_path'].sample(6, random_state=42)\n\nplt.figure(figsize=(15, 8))\n\nfor i, path in enumerate(sample_paths):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    plt.subplot(2, 3, i + 1)\n    plt.imshow(img)\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:53:32.914007Z","iopub.execute_input":"2026-09-20T15:53:32.914391Z","iopub.status.idle":"2026-09-20T15:53:36.154142Z","shell.execute_reply.started":"2026-09-20T15:53:32.914361Z","shell.execute_reply":"2026-09-20T15:53:36.151877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef crop_black_borders(img, threshold=10):\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    mask = gray > threshold\n\n    coords = cv2.findNonZero(mask.astype('uint8'))\n\n    if coords is None:\n        return img\n\n    x, y, w, h = cv2.boundingRect(coords)\n\n    return img[y:y+h, x:x+w]\n\n\nsample_paths = df['file_path'].sample(6, random_state=42)\n\nplt.figure(figsize=(14, 12))\n\nfor i, path in enumerate(sample_paths):\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    # الصورة الأصلية\n    plt.subplot(6, 2, 2*i + 1)\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(\"قبل\")\n\n    # بعد القص\n    plt.subplot(6, 2, 2*i + 2)\n    plt.imshow(cropped)\n    plt.axis('off')\n    plt.title(\"بعد\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:54:41.376389Z","iopub.execute_input":"2026-09-20T15:54:41.376723Z","iopub.status.idle":"2026-09-20T15:54:45.823202Z","shell.execute_reply.started":"2026-09-20T15:54:41.376697Z","shell.execute_reply":"2026-09-20T15:54:45.822137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef resize_with_padding(img, target_size=(224, 224)):\n    target_w, target_h = target_size\n\n    h, w = img.shape[:2]\n\n    # نحسب معامل التصغير مع الحفاظ على نسبة الأبعاد\n    scale = min(target_w / w, target_h / h)\n\n    new_w = int(w * scale)\n    new_h = int(h * scale)\n\n    # تصغير الصورة\n    resized = cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_AREA)\n\n    # إنشاء خلفية سوداء بالحجم المطلوب\n    padded = np.zeros((target_h, target_w, 3), dtype=np.uint8)\n\n    # نحسب مكان وضع الصورة في المنتصف\n    x_offset = (target_w - new_w) // 2\n    y_offset = (target_h - new_h) // 2\n\n    padded[y_offset:y_offset + new_h,\n           x_offset:x_offset + new_w] = resized\n\n    return padded\n\n\nsample_paths = df['file_path'].sample(6, random_state=42)\n\nplt.figure(figsize=(12, 12))\n\nfor i, path in enumerate(sample_paths):\n\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # أولًا: إزالة الحواف السوداء\n    cropped = crop_black_borders(img)\n\n    # ثانيًا: Resize + Padding\n    processed = resize_with_padding(cropped, (224, 224))\n\n    plt.subplot(3, 2, i + 1)\n    plt.imshow(processed)\n    plt.axis('off')\n    plt.title(f\"Original: {img.shape[1]}×{img.shape[0]} → 224×224\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T15:57:32.069394Z","iopub.execute_input":"2026-09-20T15:57:32.069899Z","iopub.status.idle":"2026-09-20T15:57:34.050271Z","shell.execute_reply.started":"2026-09-20T15:57:32.069849Z","shell.execute_reply":"2026-09-20T15:57:34.049316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sizes = [(224, 224), (512, 512)]\n\nplt.figure(figsize=(12, 12))\n\nfor i, path in enumerate(sample_paths):\n\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    for j, size in enumerate(sizes):\n        processed = resize_with_padding(cropped, size)\n\n        plt.subplot(6, 2, i * 2 + j + 1)\n        plt.imshow(processed)\n        plt.axis('off')\n        plt.title(f\"{size[0]}×{size[1]}\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:00:27.721509Z","iopub.execute_input":"2026-09-20T16:00:27.721936Z","iopub.status.idle":"2026-09-20T16:00:29.79733Z","shell.execute_reply.started":"2026-09-20T16:00:27.721902Z","shell.execute_reply":"2026-09-20T16:00:29.795662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(img, target_size=(224, 224)):\n    cropped = crop_black_borders(img)\n    processed = resize_with_padding(cropped, target_size)\n    return processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:02:19.713947Z","iopub.execute_input":"2026-09-20T16:02:19.714328Z","iopub.status.idle":"2026-09-20T16:02:19.720514Z","shell.execute_reply.started":"2026-09-20T16:02:19.714295Z","shell.execute_reply":"2026-09-20T16:02:19.719368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprocessed_dir = \"/kaggle/working/processed_images\"\n\nos.makedirs(processed_dir, exist_ok=True)\n\nprint(processed_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:03:21.514717Z","iopub.execute_input":"2026-09-20T16:03:21.515104Z","iopub.status.idle":"2026-09-20T16:03:21.522256Z","shell.execute_reply.started":"2026-09-20T16:03:21.515073Z","shell.execute_reply":"2026-09-20T16:03:21.521094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torchvision import transforms\nfrom PIL import Image\n\npath = df['file_path'].iloc[0]\n\nimg = Image.open(path).convert(\"RGB\")\n\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(\n        mean=[0.485, 0.456, 0.406],\n        std=[0.229, 0.224, 0.225]\n    )\n])\n\ntensor_img = transform(img)\n\nprint(\"Original size:\", img.size)\nprint(\"Tensor shape:\", tensor_img.shape)\nprint(\"Minimum:\", tensor_img.min().item())\nprint(\"Maximum:\", tensor_img.max().item())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:04:42.384503Z","iopub.execute_input":"2026-09-20T16:04:42.384861Z","iopub.status.idle":"2026-09-20T16:04:50.547021Z","shell.execute_reply.started":"2026-09-20T16:04:42.384832Z","shell.execute_reply":"2026-09-20T16:04:50.545638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom torchvision import transforms\nfrom PIL import Image\n\npath = df['file_path'].iloc[0]\nimg = Image.open(path).convert(\"RGB\")\n\naugmentations = {\n    \"Original\": transforms.Compose([]),\n\n    \"Horizontal Flip\": transforms.Compose([\n        transforms.RandomHorizontalFlip(p=1.0)\n    ]),\n\n    \"Small Rotation\": transforms.Compose([\n        transforms.RandomRotation(degrees=15)\n    ]),\n\n    \"Small Zoom\": transforms.Compose([\n        transforms.RandomResizedCrop(\n            size=img.size[::-1],\n            scale=(0.9, 1.0),\n            ratio=(0.95, 1.05)\n        )\n    ])\n}\n\nplt.figure(figsize=(14, 10))\n\nfor i, (name, transform) in enumerate(augmentations.items()):\n\n    augmented = transform(img)\n\n    plt.subplot(2, 2, i + 1)\n    plt.imshow(augmented)\n    plt.axis(\"off\")\n    plt.title(name)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:07:46.502071Z","iopub.execute_input":"2026-09-20T16:07:46.50283Z","iopub.status.idle":"2026-09-20T16:07:49.781134Z","shell.execute_reply.started":"2026-09-20T16:07:46.502792Z","shell.execute_reply":"2026-09-20T16:07:49.778502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vertical_flip = transforms.RandomVerticalFlip(p=1.0)\n\nflipped = vertical_flip(img)\n\nplt.figure(figsize=(10, 5))\n\nplt.subplot(1, 2, 1)\nplt.imshow(img)\nplt.axis(\"off\")\nplt.title(\"Original\")\n\nplt.subplot(1, 2, 2)\nplt.imshow(flipped)\nplt.axis(\"off\")\nplt.title(\"Vertical Flip\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:09:49.252846Z","iopub.execute_input":"2026-09-20T16:09:49.253355Z","iopub.status.idle":"2026-09-20T16:09:50.65928Z","shell.execute_reply.started":"2026-09-20T16:09:49.253317Z","shell.execute_reply":"2026-09-20T16:09:50.658258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"brightness_contrast = transforms.ColorJitter(\n    brightness=0.1,\n    contrast=0.1\n)\n\naugmented = brightness_contrast(img)\n\nplt.figure(figsize=(10, 5))\n\nplt.subplot(1, 2, 1)\nplt.imshow(img)\nplt.axis(\"off\")\nplt.title(\"Original\")\n\nplt.subplot(1, 2, 2)\nplt.imshow(augmented)\nplt.axis(\"off\")\nplt.title(\"Brightness + Contrast\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:37:13.579639Z","iopub.execute_input":"2026-09-20T16:37:13.580606Z","iopub.status.idle":"2026-09-20T16:37:14.852867Z","shell.execute_reply.started":"2026-09-20T16:37:13.58056Z","shell.execute_reply":"2026-09-20T16:37:14.851854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df['diagnosis'].value_counts().sort_index()\n\nprint(\"عدد الصور في كل فئة:\")\nprint(class_counts)\n\nprint(\"\\nنسبة كل فئة:\")\nprint((class_counts / len(df) * 100).round(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:38:57.292421Z","iopub.execute_input":"2026-09-20T16:38:57.292817Z","iopub.status.idle":"2026-09-20T16:38:57.306501Z","shell.execute_reply.started":"2026-09-20T16:38:57.292785Z","shell.execute_reply":"2026-09-20T16:38:57.30529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom torchvision import transforms\n\n# التحويلات اللي اخترناها مبدئيًا\nhorizontal_flip = transforms.RandomHorizontalFlip(p=1.0)\n\nrotation = transforms.RandomRotation(degrees=15)\n\nbrightness_contrast = transforms.ColorJitter(\n    brightness=0.1,\n    contrast=0.1\n)\n\n# نختار صورة واحدة عشوائية من كل فئة\nsamples = []\n\nfor grade in sorted(df['diagnosis'].unique()):\n    row = df[df['diagnosis'] == grade].sample(n=1).iloc[0]\n    img = Image.open(row['file_path']).convert(\"RGB\")\n    samples.append((grade, img))\n\n# عرض النتائج\nfig, axes = plt.subplots(5, 4, figsize=(12, 18))\n\nfor i, (grade, img) in enumerate(samples):\n\n    # الصورة الأصلية\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(f\"Grade {grade} - Original\")\n    axes[i, 0].axis(\"off\")\n\n    # انعكاس أفقي\n    flipped = horizontal_flip(img)\n    axes[i, 1].imshow(flipped)\n    axes[i, 1].set_title(\"Horizontal Flip\")\n    axes[i, 1].axis(\"off\")\n\n    # دوران\n    rotated = rotation(img)\n    axes[i, 2].imshow(rotated)\n    axes[i, 2].set_title(\"Rotation ±15°\")\n    axes[i, 2].axis(\"off\")\n\n    # سطوع + تباين\n    adjusted = brightness_contrast(img)\n    axes[i, 3].imshow(adjusted)\n    axes[i, 3].set_title(\"Brightness + Contrast\")\n    axes[i, 3].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:41:05.845139Z","iopub.execute_input":"2026-09-20T16:41:05.845535Z","iopub.status.idle":"2026-09-20T16:41:14.620305Z","shell.execute_reply.started":"2026-09-20T16:41:05.845505Z","shell.execute_reply":"2026-09-20T16:41:14.61879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df['diagnosis'].value_counts().sort_index()\n\ntarget_count = int(class_counts.median())\n\nprint(\"العدد المستهدف المبدئي:\", target_count)\nprint(\"\\nعدد الصور الحالية والإضافية المقترحة:\")\n\nfor grade, count in class_counts.items():\n    additional = max(0, target_count - count)\n\n    print(\n        f\"الدرجة {grade}: \"\n        f\"الحالي = {count}, \"\n        f\"الإضافي = {additional}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:48:46.415029Z","iopub.execute_input":"2026-09-20T16:48:46.415371Z","iopub.status.idle":"2026-09-20T16:48:46.424936Z","shell.execute_reply.started":"2026-09-20T16:48:46.415338Z","shell.execute_reply":"2026-09-20T16:48:46.423817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df['diagnosis'].value_counts().sort_index()\n\ntarget_count = int(class_counts.median())\n\nprint(\"الزيادة المطلوبة كنسبة مئوية:\\n\")\n\nfor grade, count in class_counts.items():\n    additional = max(0, target_count - count)\n    percentage_increase = (additional / count) * 100\n\n    print(\n        f\"الدرجة {grade}: \"\n        f\"زيادة {additional} صورة \"\n        f\"({percentage_increase:.1f}%)\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:49:40.390808Z","iopub.execute_input":"2026-09-20T16:49:40.391172Z","iopub.status.idle":"2026-09-20T16:49:40.400072Z","shell.execute_reply.started":"2026-09-20T16:49:40.391135Z","shell.execute_reply":"2026-09-20T16:49:40.398562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df['diagnosis'].value_counts().sort_index()\n\ntotal = len(df)\nnum_classes = len(class_counts)\n\nclass_weights = total / (num_classes * class_counts)\n\nprint(\"أوزان الفئات:\")\nfor grade, weight in class_weights.items():\n    print(f\"الدرجة {grade}: {weight:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:50:58.32302Z","iopub.execute_input":"2026-09-20T16:50:58.323409Z","iopub.status.idle":"2026-09-20T16:50:58.331941Z","shell.execute_reply.started":"2026-09-20T16:50:58.323376Z","shell.execute_reply":"2026-09-20T16:50:58.330835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df['diagnosis'].value_counts().sort_index()\n\naugmentation_factor = {\n    0: 0,\n    1: 1,\n    2: 0,\n    3: 2,\n    4: 1\n}\n\nprint(\"الصور الإضافية المقترحة مبدئيًا:\\n\")\n\nfor grade, count in class_counts.items():\n    extra = count * augmentation_factor[grade]\n    print(\n        f\"الدرجة {grade}: \"\n        f\"{int(extra)} صورة إضافية \"\n        f\"({augmentation_factor[grade]} نسخة إضافية لكل صورة)\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:51:53.85369Z","iopub.execute_input":"2026-09-20T16:51:53.854079Z","iopub.status.idle":"2026-09-20T16:51:53.861785Z","shell.execute_reply.started":"2026-09-20T16:51:53.854046Z","shell.execute_reply":"2026-09-20T16:51:53.86102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef crop_black_borders(img, threshold=10):\n    \"\"\"\n    إزالة المناطق السوداء غير المفيدة حول صورة العين.\n    \"\"\"\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    mask = gray > threshold\n\n    coords = cv2.findNonZero(mask.astype(np.uint8))\n\n    if coords is None:\n        return img\n\n    x, y, w, h = cv2.boundingRect(coords)\n\n    return img[y:y+h, x:x+w]\n\n\ndef resize_with_padding(img, target_size=(224, 224)):\n    \"\"\"\n    تغيير الحجم مع الحفاظ على نسبة الأبعاد،\n    ثم إضافة Padding للوصول إلى الحجم المطلوب.\n    \"\"\"\n    target_w, target_h = target_size\n\n    h, w = img.shape[:2]\n\n    scale = min(target_w / w, target_h / h)\n\n    new_w = int(w * scale)\n    new_h = int(h * scale)\n\n    resized = cv2.resize(\n        img,\n        (new_w, new_h),\n        interpolation=cv2.INTER_AREA\n    )\n\n    padded = np.zeros(\n        (target_h, target_w, 3),\n        dtype=np.uint8\n    )\n\n    x_offset = (target_w - new_w) // 2\n    y_offset = (target_h - new_h) // 2\n\n    padded[\n        y_offset:y_offset + new_h,\n        x_offset:x_offset + new_w\n    ] = resized\n\n    return padded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:54:27.180389Z","iopub.execute_input":"2026-09-20T16:54:27.180787Z","iopub.status.idle":"2026-09-20T16:54:27.190947Z","shell.execute_reply.started":"2026-09-20T16:54:27.180741Z","shell.execute_reply":"2026-09-20T16:54:27.189657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# نختار أول صورة من الداتا\npath = df['file_path'].iloc[0]\n\n# قراءة الصورة وتحويلها من BGR إلى RGB\nimg = cv2.imread(path)\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n# تطبيق الـ preprocessing\ncropped = crop_black_borders(img)\nprocessed = resize_with_padding(cropped, target_size=(224, 224))\n\n# عرض النتيجة\nplt.figure(figsize=(10, 5))\n\nplt.subplot(1, 2, 1)\nplt.imshow(img)\nplt.title(\"Original\")\nplt.axis(\"off\")\n\nplt.subplot(1, 2, 2)\nplt.imshow(processed)\nplt.title(\"Preprocessed 224×224\")\nplt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n\nprint(\"Original shape:\", img.shape)\nprint(\"After cropping:\", cropped.shape)\nprint(\"Final shape:\", processed.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:55:33.203464Z","iopub.execute_input":"2026-09-20T16:55:33.20384Z","iopub.status.idle":"2026-09-20T16:55:34.225782Z","shell.execute_reply.started":"2026-09-20T16:55:33.203809Z","shell.execute_reply":"2026-09-20T16:55:34.224738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# نختار 3 صور من فئات مختلفة\nselected_grades = [0, 2, 3]\n\nfig, axes = plt.subplots(3, 3, figsize=(12, 12))\n\nfor i, grade in enumerate(selected_grades):\n\n    # اختيار صورة عشوائية من الفئة\n    row = df[df['diagnosis'] == grade].sample(n=1).iloc[0]\n\n    # قراءة الصورة\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # تطبيق Crop\n    cropped = crop_black_borders(img)\n\n    # تطبيق Crop + Resize + Padding\n    processed = resize_with_padding(\n        cropped,\n        target_size=(224, 224)\n    )\n\n    # الأصل\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(f\"Grade {grade} - Original\")\n    axes[i, 0].axis(\"off\")\n\n    # بعد Crop\n    axes[i, 1].imshow(cropped)\n    axes[i, 1].set_title(\"After Crop\")\n    axes[i, 1].axis(\"off\")\n\n    # النتيجة النهائية\n    axes[i, 2].imshow(processed)\n    axes[i, 2].set_title(\"Crop + Resize + Padding\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T16:58:07.939016Z","iopub.execute_input":"2026-09-20T16:58:07.939371Z","iopub.status.idle":"2026-09-20T16:58:10.18809Z","shell.execute_reply.started":"2026-09-20T16:58:07.939344Z","shell.execute_reply":"2026-09-20T16:58:10.186479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# اختيار 12 صورة من مختلف درجات المرض\nsample_df = df.groupby('diagnosis', group_keys=False).apply(\n    lambda x: x.sample(min(len(x), 3), random_state=42)\n).reset_index(drop=True)\n\nfig, axes = plt.subplots(len(sample_df), 3, figsize=(12, 4 * len(sample_df)))\n\nfor i, (_, row) in enumerate(sample_df.iterrows()):\n\n    # قراءة الصورة\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # Crop\n    cropped = crop_black_borders(img)\n\n    # Crop + Resize + Padding\n    processed = resize_with_padding(\n        cropped,\n        target_size=(224, 224)\n    )\n\n    # Original\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(f\"Grade {row['diagnosis']} - Original\")\n    axes[i, 0].axis(\"off\")\n\n    # Crop\n    axes[i, 1].imshow(cropped)\n    axes[i, 1].set_title(\"After Crop\")\n    axes[i, 1].axis(\"off\")\n\n    # Final\n    axes[i, 2].imshow(processed)\n    axes[i, 2].set_title(\"Crop + Resize + Padding\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:02:45.751617Z","iopub.execute_input":"2026-09-20T17:02:45.752028Z","iopub.status.idle":"2026-09-20T17:03:00.666349Z","shell.execute_reply.started":"2026-09-20T17:02:45.751997Z","shell.execute_reply":"2026-09-20T17:03:00.664724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:09:47.935292Z","iopub.execute_input":"2026-09-20T17:09:47.935656Z","iopub.status.idle":"2026-09-20T17:09:47.940274Z","shell.execute_reply.started":"2026-09-20T17:09:47.935628Z","shell.execute_reply":"2026-09-20T17:09:47.93932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade4 = df[df['diagnosis'] == 4].sample(n=6, random_state=42)\n\nfig, axes = plt.subplots(6, 3, figsize=(12, 20))\n\nfor i, (_, row) in enumerate(grade4.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    processed = resize_with_padding(\n        cropped,\n        target_size=(224, 224)\n    )\n\n    # متوسط السطوع\n    original_brightness = img.mean()\n    cropped_brightness = cropped.mean()\n    processed_brightness = processed.mean()\n\n    # Original\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(\n        f\"Original\\nBrightness: {original_brightness:.1f}\"\n    )\n    axes[i, 0].axis(\"off\")\n\n    # Crop\n    axes[i, 1].imshow(cropped)\n    axes[i, 1].set_title(\n        f\"Crop\\nBrightness: {cropped_brightness:.1f}\"\n    )\n    axes[i, 1].axis(\"off\")\n\n    # Final\n    axes[i, 2].imshow(processed)\n    axes[i, 2].set_title(\n        f\"Final 224×224\\nBrightness: {processed_brightness:.1f}\"\n    )\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:11:13.512021Z","iopub.execute_input":"2026-09-20T17:11:13.512399Z","iopub.status.idle":"2026-09-20T17:11:20.79062Z","shell.execute_reply.started":"2026-09-20T17:11:13.512367Z","shell.execute_reply":"2026-09-20T17:11:20.788272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef apply_clahe(img, clip_limit=2.0, tile_grid_size=(8, 8)):\n    # تحويل RGB إلى LAB\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n\n    # فصل قناة الإضاءة\n    l_channel, a_channel, b_channel = cv2.split(lab)\n\n    # إنشاء CLAHE\n    clahe = cv2.createCLAHE(\n        clipLimit=clip_limit,\n        tileGridSize=tile_grid_size\n    )\n\n    # تحسين الإضاءة\n    l_enhanced = clahe.apply(l_channel)\n\n    # إعادة دمج القنوات\n    enhanced_lab = cv2.merge(\n        [l_enhanced, a_channel, b_channel]\n    )\n\n    # الرجوع إلى RGB\n    enhanced = cv2.cvtColor(\n        enhanced_lab,\n        cv2.COLOR_LAB2RGB\n    )\n\n    return enhanced","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:18:57.707701Z","iopub.execute_input":"2026-09-20T17:18:57.70808Z","iopub.status.idle":"2026-09-20T17:18:57.715374Z","shell.execute_reply.started":"2026-09-20T17:18:57.70805Z","shell.execute_reply":"2026-09-20T17:18:57.714142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade4 = df[df['diagnosis'] == 4].sample(\n    n=3,\n    random_state=42\n)\n\nfig, axes = plt.subplots(3, 3, figsize=(12, 12))\n\nfor i, (_, row) in enumerate(grade4.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    enhanced = apply_clahe(cropped)\n\n    # Original\n    axes[i, 0].imshow(cropped)\n    axes[i, 0].set_title(\"After Crop\")\n    axes[i, 0].axis(\"off\")\n\n    # CLAHE\n    axes[i, 1].imshow(enhanced)\n    axes[i, 1].set_title(\"After CLAHE\")\n    axes[i, 1].axis(\"off\")\n\n    # الفرق في السطوع\n    axes[i, 2].imshow(\n        np.abs(\n            enhanced.astype(float) -\n            cropped.astype(float)\n        ).mean(axis=2),\n        cmap=\"gray\"\n    )\n    axes[i, 2].set_title(\"Change Map\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:19:11.361714Z","iopub.execute_input":"2026-09-20T17:19:11.362125Z","iopub.status.idle":"2026-09-20T17:19:18.644503Z","shell.execute_reply.started":"2026-09-20T17:19:11.362086Z","shell.execute_reply":"2026-09-20T17:19:18.641903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# اختيار 3 صور من كل Grade\nsample_df = (\n    df.groupby('diagnosis', group_keys=False)\n      .sample(n=3, random_state=42)\n      .reset_index(drop=True)\n)\n\nfig, axes = plt.subplots(\n    len(sample_df), 2,\n    figsize=(10, 4 * len(sample_df))\n)\n\nfor i, (_, row) in enumerate(sample_df.iterrows()):\n\n    # قراءة الصورة\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # إزالة الحدود السوداء\n    cropped = crop_black_borders(img)\n\n    # تطبيق CLAHE\n    enhanced = apply_clahe(cropped)\n\n    # قبل CLAHE\n    axes[i, 0].imshow(cropped)\n    axes[i, 0].set_title(\n        f\"Grade {row['diagnosis']} - Before CLAHE\"\n    )\n    axes[i, 0].axis(\"off\")\n\n    # بعد CLAHE\n    axes[i, 1].imshow(enhanced)\n    axes[i, 1].set_title(\n        f\"Grade {row['diagnosis']} - After CLAHE\"\n    )\n    axes[i, 1].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T17:24:31.104346Z","iopub.execute_input":"2026-09-20T17:24:31.10484Z","iopub.status.idle":"2026-09-20T17:24:47.864882Z","shell.execute_reply.started":"2026-09-20T17:24:31.104794Z","shell.execute_reply":"2026-09-20T17:24:47.86387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:11:34.890172Z","iopub.execute_input":"2026-09-20T18:11:34.890506Z","iopub.status.idle":"2026-09-20T18:11:35.351495Z","shell.execute_reply.started":"2026-09-20T18:11:34.890466Z","shell.execute_reply":"2026-09-20T18:11:35.350463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndf = pd.read_csv(\n    '/kaggle/input/competitions/aptos2019-blindness-detection/train.csv'\n)\n\nimage_dir = '/kaggle/input/competitions/aptos2019-blindness-detection/train_images'\n\ndf['file_path'] = df['id_code'].apply(\n    lambda x: os.path.join(image_dir, f\"{x}.png\")\n)\n\nprint(df.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:12:57.347783Z","iopub.execute_input":"2026-09-20T18:12:57.348161Z","iopub.status.idle":"2026-09-20T18:12:57.725674Z","shell.execute_reply.started":"2026-09-20T18:12:57.348113Z","shell.execute_reply":"2026-09-20T18:12:57.724593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_black_borders(img, threshold=10):\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    mask = gray > threshold\n\n    coords = cv2.findNonZero(\n        mask.astype('uint8')\n    )\n\n    if coords is None:\n        return img\n\n    x, y, w, h = cv2.boundingRect(coords)\n\n    return img[y:y+h, x:x+w]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:13:52.713093Z","iopub.execute_input":"2026-09-20T18:13:52.713459Z","iopub.status.idle":"2026-09-20T18:13:52.721203Z","shell.execute_reply.started":"2026-09-20T18:13:52.713431Z","shell.execute_reply":"2026-09-20T18:13:52.720245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_clahe(img, clip_limit=2.0, tile_grid_size=(8, 8)):\n    lab = cv2.cvtColor(img, cv2.COLOR_RGB2LAB)\n\n    l_channel, a_channel, b_channel = cv2.split(lab)\n\n    clahe = cv2.createCLAHE(\n        clipLimit=clip_limit,\n        tileGridSize=tile_grid_size\n    )\n\n    l_enhanced = clahe.apply(l_channel)\n\n    enhanced_lab = cv2.merge(\n        [l_enhanced, a_channel, b_channel]\n    )\n\n    enhanced = cv2.cvtColor(\n        enhanced_lab,\n        cv2.COLOR_LAB2RGB\n    )\n\n    return enhanced","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:14:27.513641Z","iopub.execute_input":"2026-09-20T18:14:27.514132Z","iopub.status.idle":"2026-09-20T18:14:27.520531Z","shell.execute_reply.started":"2026-09-20T18:14:27.514097Z","shell.execute_reply":"2026-09-20T18:14:27.519519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:16:11.844776Z","iopub.execute_input":"2026-09-20T18:16:11.845133Z","iopub.status.idle":"2026-09-20T18:16:11.849946Z","shell.execute_reply.started":"2026-09-20T18:16:11.845105Z","shell.execute_reply":"2026-09-20T18:16:11.84911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade4 = df[df['diagnosis'] == 4].sample(\n    n=3,\n    random_state=42\n)\n\nfig, axes = plt.subplots(\n    3, 3,\n    figsize=(12, 12)\n)\n\nfor i, (_, row) in enumerate(grade4.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    clahe_2 = apply_clahe(\n        cropped,\n        clip_limit=2.0\n    )\n\n    clahe_15 = apply_clahe(\n        cropped,\n        clip_limit=1.5\n    )\n\n    axes[i, 0].imshow(cropped)\n    axes[i, 0].set_title(\"Before CLAHE\")\n    axes[i, 0].axis(\"off\")\n\n    axes[i, 1].imshow(clahe_2)\n    axes[i, 1].set_title(\"CLAHE 2.0\")\n    axes[i, 1].axis(\"off\")\n\n    axes[i, 2].imshow(clahe_15)\n    axes[i, 2].set_title(\"CLAHE 1.5\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:16:18.719933Z","iopub.execute_input":"2026-09-20T18:16:18.720373Z","iopub.status.idle":"2026-09-20T18:16:23.509326Z","shell.execute_reply.started":"2026-09-20T18:16:18.720342Z","shell.execute_reply":"2026-09-20T18:16:23.506825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade4 = df[df['diagnosis'] == 4].sample(\n    n=3,\n    random_state=42\n)\n\nfig, axes = plt.subplots(\n    3, 3,\n    figsize=(12, 12)\n)\n\nfor i, (_, row) in enumerate(grade4.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    clahe_2 = apply_clahe(\n        cropped,\n        clip_limit=2.0\n    )\n\n    clahe_15 = apply_clahe(\n        cropped,\n        clip_limit=1.5\n    )\n\n    axes[i, 0].imshow(cropped)\n    axes[i, 0].set_title(\"Before CLAHE\")\n    axes[i, 0].axis(\"off\")\n\n    axes[i, 1].imshow(clahe_2)\n    axes[i, 1].set_title(\"CLAHE 2.0\")\n    axes[i, 1].axis(\"off\")\n\n    axes[i, 2].imshow(clahe_15)\n    axes[i, 2].set_title(\"CLAHE 1.5\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:13:02.593434Z","iopub.execute_input":"2026-09-20T18:13:02.593767Z","iopub.status.idle":"2026-09-20T18:13:02.610855Z","shell.execute_reply.started":"2026-09-20T18:13:02.593738Z","shell.execute_reply":"2026-09-20T18:13:02.60949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grade4 = df[df['diagnosis'] == 4].sample(\n    n=3,\n    random_state=42\n)\n\nfig, axes = plt.subplots(\n    3, 3,\n    figsize=(12, 12)\n)\n\nfor i, (_, row) in enumerate(grade4.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n\n    clahe_2 = apply_clahe(\n        cropped,\n        clip_limit=2.0\n    )\n\n    clahe_15 = apply_clahe(\n        cropped,\n        clip_limit=1.5\n    )\n\n    axes[i, 0].imshow(cropped)\n    axes[i, 0].set_title(\"Before CLAHE\")\n    axes[i, 0].axis(\"off\")\n\n    axes[i, 1].imshow(clahe_2)\n    axes[i, 1].set_title(\"CLAHE 2.0\")\n    axes[i, 1].axis(\"off\")\n\n    axes[i, 2].imshow(clahe_15)\n    axes[i, 2].set_title(\"CLAHE 1.5\")\n    axes[i, 2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2026-09-20T18:11:03.624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nhorizontal_flip = transforms.RandomHorizontalFlip(p=1.0)\n\nrotation = transforms.RandomRotation(degrees=15)\n\nbrightness_contrast = transforms.ColorJitter(\n    brightness=0.1,\n    contrast=0.1\n)\n\nsample = df.sample(n=1, random_state=42).iloc[0]\n\nimg = Image.open(sample['file_path']).convert(\"RGB\")\n\naugmented_images = [\n    (\"Original\", img),\n    (\"Horizontal Flip\", horizontal_flip(img)),\n    (\"Rotation ±15°\", rotation(img)),\n    (\"Brightness + Contrast\", brightness_contrast(img))\n]\n\nfig, axes = plt.subplots(\n    1, 4,\n    figsize=(16, 4)\n)\n\nfor ax, (title, image) in zip(axes, augmented_images):\n    ax.imshow(image)\n    ax.set_title(title)\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:23:03.268702Z","iopub.execute_input":"2026-09-20T18:23:03.269742Z","iopub.status.idle":"2026-09-20T18:23:12.640595Z","shell.execute_reply.started":"2026-09-20T18:23:03.269678Z","shell.execute_reply":"2026-09-20T18:23:12.639537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nhorizontal_flip = transforms.RandomHorizontalFlip(p=1.0)\n\nrotation = transforms.RandomRotation(degrees=15)\n\nbrightness_contrast = transforms.ColorJitter(\n    brightness=0.1,\n    contrast=0.1\n)\n\nsample_df = (\n    df.groupby('diagnosis', group_keys=False)\n      .sample(n=1, random_state=42)\n      .reset_index(drop=True)\n)\n\nfig, axes = plt.subplots(\n    5, 4,\n    figsize=(12, 18)\n)\n\nfor i, (_, row) in enumerate(sample_df.iterrows()):\n\n    img = Image.open(row['file_path']).convert(\"RGB\")\n\n    augmented_images = [\n        (\"Original\", img),\n        (\"Horizontal Flip\", horizontal_flip(img)),\n        (\"Rotation ±15°\", rotation(img)),\n        (\"Brightness + Contrast\", brightness_contrast(img))\n    ]\n\n    for j, (title, image) in enumerate(augmented_images):\n        axes[i, j].imshow(image)\n        axes[i, j].set_title(\n            f\"Grade {row['diagnosis']} - {title}\"\n        )\n        axes[i, j].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:29:34.772488Z","iopub.execute_input":"2026-09-20T18:29:34.773136Z","iopub.status.idle":"2026-09-20T18:29:43.07543Z","shell.execute_reply.started":"2026-09-20T18:29:34.773085Z","shell.execute_reply":"2026-09-20T18:29:43.074323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:34:52.147074Z","iopub.execute_input":"2026-09-20T18:34:52.147424Z","iopub.status.idle":"2026-09-20T18:34:52.152073Z","shell.execute_reply.started":"2026-09-20T18:34:52.147397Z","shell.execute_reply":"2026-09-20T18:34:52.151094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resize_with_padding(img, target_size=(224, 224)):\n    target_w, target_h = target_size\n\n    h, w = img.shape[:2]\n\n    scale = min(target_w / w, target_h / h)\n\n    new_w = int(w * scale)\n    new_h = int(h * scale)\n\n    resized = cv2.resize(\n        img,\n        (new_w, new_h),\n        interpolation=cv2.INTER_AREA\n    )\n\n    padded = np.zeros(\n        (target_h, target_w, 3),\n        dtype=np.uint8\n    )\n\n    x_offset = (target_w - new_w) // 2\n    y_offset = (target_h - new_h) // 2\n\n    padded[\n        y_offset:y_offset + new_h,\n        x_offset:x_offset + new_w\n    ] = resized\n\n    return padded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:34:58.130606Z","iopub.execute_input":"2026-09-20T18:34:58.131041Z","iopub.status.idle":"2026-09-20T18:34:58.137947Z","shell.execute_reply.started":"2026-09-20T18:34:58.131012Z","shell.execute_reply":"2026-09-20T18:34:58.13706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\npath = df['file_path'].iloc[0]\n\n# قراءة الصورة\nimg = cv2.imread(path)\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n# 1. Crop black borders\ncropped = crop_black_borders(img)\n\n# 2. Resize + Padding\nresized = resize_with_padding(\n    cropped,\n    target_size=(224, 224)\n)\n\n# 3. CLAHE\nenhanced = apply_clahe(\n    resized,\n    clip_limit=2.0,\n    tile_grid_size=(8, 8)\n)\n\n# عرض المراحل\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\n\naxes[0].imshow(img)\naxes[0].set_title(\"Original\")\naxes[0].axis(\"off\")\n\naxes[1].imshow(cropped)\naxes[1].set_title(\"After Cropping\")\naxes[1].axis(\"off\")\n\naxes[2].imshow(enhanced)\naxes[2].set_title(\"Final Preprocessing\")\naxes[2].axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n\nprint(\"Original:\", img.shape)\nprint(\"Cropped:\", cropped.shape)\nprint(\"Final:\", enhanced.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:35:04.968496Z","iopub.execute_input":"2026-09-20T18:35:04.969059Z","iopub.status.idle":"2026-09-20T18:35:06.530382Z","shell.execute_reply.started":"2026-09-20T18:35:04.969019Z","shell.execute_reply":"2026-09-20T18:35:06.529315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = (\n    df.groupby('diagnosis', group_keys=False)\n      .sample(n=1, random_state=42)\n      .reset_index(drop=True)\n)\n\nfig, axes = plt.subplots(5, 2, figsize=(8, 18))\n\nfor i, (_, row) in enumerate(sample_df.iterrows()):\n\n    img = cv2.imread(row['file_path'])\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    cropped = crop_black_borders(img)\n    processed = resize_with_padding(cropped, (224, 224))\n    processed = apply_clahe(\n        processed,\n        clip_limit=2.0,\n        tile_grid_size=(8, 8)\n    )\n\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(f\"Grade {row['diagnosis']} - Original\")\n    axes[i, 0].axis(\"off\")\n\n    axes[i, 1].imshow(processed)\n    axes[i, 1].set_title(f\"Grade {row['diagnosis']} - Processed\")\n    axes[i, 1].axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:36:17.03994Z","iopub.execute_input":"2026-09-20T18:36:17.040504Z","iopub.status.idle":"2026-09-20T18:36:20.694261Z","shell.execute_reply.started":"2026-09-20T18:36:17.040453Z","shell.execute_reply":"2026-09-20T18:36:20.693031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\n\nprocessed_dir = \"/kaggle/working/processed_images\"\nos.makedirs(processed_dir, exist_ok=True)\n\nprint(\"Output folder:\", processed_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:41:16.517676Z","iopub.execute_input":"2026-09-20T18:41:16.518125Z","iopub.status.idle":"2026-09-20T18:41:16.52453Z","shell.execute_reply.started":"2026-09-20T18:41:16.518092Z","shell.execute_reply":"2026-09-20T18:41:16.523629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(path):\n    # Read image\n    img = cv2.imread(path)\n\n    if img is None:\n        return None\n\n    # Convert BGR → RGB\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # 1. Crop black borders\n    cropped = crop_black_borders(img)\n\n    # 2. Resize + Padding\n    resized = resize_with_padding(\n        cropped,\n        target_size=(224, 224)\n    )\n\n    # 3. CLAHE\n    processed = apply_clahe(\n        resized,\n        clip_limit=2.0,\n        tile_grid_size=(8, 8)\n    )\n\n    return processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:42:00.456653Z","iopub.execute_input":"2026-09-20T18:42:00.457218Z","iopub.status.idle":"2026-09-20T18:42:00.464099Z","shell.execute_reply.started":"2026-09-20T18:42:00.457181Z","shell.execute_reply":"2026-09-20T18:42:00.462383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image = preprocess_image(df['file_path'].iloc[0])\n\nprint(\"Processed shape:\", test_image.shape)\nprint(\"Data type:\", test_image.dtype)\n\nplt.figure(figsize=(5, 5))\nplt.imshow(test_image)\nplt.axis(\"off\")\nplt.title(\"Preprocessed Image\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-20T18:42:32.498043Z","iopub.execute_input":"2026-09-20T18:42:32.498457Z","iopub.status.idle":"2026-09-20T18:42:32.90092Z","shell.execute_reply.started":"2026-09-20T18:42:32.498424Z","shell.execute_reply":"2026-09-20T18:42:32.900108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"failed_images = []\n\nfor _, row in tqdm(\n    df.iterrows(),\n    total=len(df),\n    desc=\"Preprocessing images\"\n):\n    image_id = row[\"id_code\"]\n    output_path = os.path.join(\n        processed_dir,\n        f\"{image_id}.png\"\n    )\n\n    try:\n        processed = preprocess_image(row[\"file_path\"])\n\n        if processed is None:\n            failed_images.append(image_id)\n            continue\n\n        cv2.imwrite(\n            output_path,\n            cv2.cvtColor(processed, cv2.COLOR_RGB2BGR)\n        )\n\n    except Exception as e:\n        failed_images.append((image_id, str(e)))\n\nprint(\"\\nProcessing finished.\")\nprint(\"Total images:\", len(df))\nprint(\"Failed:\", len(failed_images))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-21T14:42:19.862056Z","iopub.execute_input":"2026-09-21T14:42:19.862627Z","iopub.status.idle":"2026-09-21T14:42:19.88655Z","shell.execute_reply.started":"2026-09-21T14:42:19.86256Z","shell.execute_reply":"2026-09-21T14:42:19.885242Z"}},"outputs":[],"execution_count":null}]}