{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport kagglehub\nimport matplotlib.pyplot as plt\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:34:06.737759Z","iopub.execute_input":"2026-09-29T12:34:06.738122Z","iopub.status.idle":"2026-09-29T12:34:08.634699Z","shell.execute_reply.started":"2026-09-29T12:34:06.738088Z","shell.execute_reply":"2026-09-29T12:34:08.633975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Base Kaggle input directory\nINPUT_PATH = '/kaggle/input'\n\n# Display available directories and number of files\nprint(\"Available Kaggle inputs:\\n\")\n\nfor root, dirs, files in os.walk(INPUT_PATH):\n    print(f\"{root} -> {len(files)} files\")\n\n# APTOS 2019 dataset path\nDATASET_PATH = '/kaggle/input/competitions/aptos2019-blindness-detection'\n\n# Check that the dataset exists\nif os.path.exists(DATASET_PATH):\n    print(\"\\nAPTOS 2019 dataset found!\")\n    print(\"\\nContents:\")\n    print(os.listdir(DATASET_PATH))\nelse:\n    print(\"\\nAPTOS 2019 dataset not found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:34:12.238642Z","iopub.execute_input":"2026-09-29T12:34:12.238915Z","iopub.status.idle":"2026-09-29T12:34:19.885576Z","shell.execute_reply.started":"2026-09-29T12:34:12.238891Z","shell.execute_reply":"2026-09-29T12:34:19.884915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Path to the APTOS 2019 dataset\nBASE = '/kaggle/input/competitions/aptos2019-blindness-detection'\n\n# Load the training CSV file\ndf = pd.read_csv(f'{BASE}/train.csv')\n\n# Display dataset shape\nprint(\"Dataset shape:\", df.shape)\n\n# Display first 5 rows\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:34:37.801421Z","iopub.execute_input":"2026-09-29T12:34:37.801854Z","iopub.status.idle":"2026-09-29T12:34:37.835391Z","shell.execute_reply.started":"2026-09-29T12:34:37.801827Z","shell.execute_reply":"2026-09-29T12:34:37.83485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map diagnosis labels to their corresponding class names\nnames = {\n    0: 'No DR',\n    1: 'Mild',\n    2: 'Moderate',\n    3: 'Severe',\n    4: 'Proliferative'\n}\n\n# Count images in each class\ncounts = df['diagnosis'].value_counts().sort_index()\n\n# Create bar chart\nplt.figure(figsize=(9, 5))\n\nplt.bar(\n    [names[i] for i in counts.index],\n    counts.values\n)\n\nplt.title('APTOS 2019 Class Distribution')\nplt.xlabel('Diabetic Retinopathy Class')\nplt.ylabel('Number of Images')\n\n# Add count values above each bar\nfor i, v in enumerate(counts.values):\n    plt.text(\n        i,\n        v + 10,\n        str(v),\n        ha='center'\n    )\n\nplt.tight_layout()\nplt.show()\n\n# Print class counts\nprint(\"Class Distribution:\")\nprint(counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:34:54.142701Z","iopub.execute_input":"2026-09-29T12:34:54.143234Z","iopub.status.idle":"2026-09-29T12:34:54.35611Z","shell.execute_reply.started":"2026-09-29T12:34:54.143204Z","shell.execute_reply":"2026-09-29T12:34:54.355478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create figure with 5 rows × 3 columns\nfig, axes = plt.subplots(5, 3, figsize=(9, 15))\n\n# Display 3 random images from each class\nfor cls in range(5):\n\n    # Select 3 images from the current class\n    ids = (\n        df[df['diagnosis'] == cls]\n        .sample(3, random_state=1)['id_code']\n        .values\n    )\n\n    for j, image_id in enumerate(ids):\n\n        # Construct image path\n        image_path = f'{BASE}/train_images/{image_id}.png'\n\n        # Read image\n        img = cv2.imread(image_path)\n\n        # Check whether image was loaded successfully\n        if img is None:\n            axes[cls][j].axis('off')\n            axes[cls][j].set_title('Image not found')\n            continue\n\n        # Convert BGR → RGB\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        # Display image\n        axes[cls][j].imshow(img)\n        axes[cls][j].axis('off')\n\n        # Display class name and image dimensions\n        axes[cls][j].set_title(\n            f'{names[cls]} ({img.shape[1]}×{img.shape[0]})',\n            fontsize=8\n        )\n\n# Add spacing between rows and columns\nplt.tight_layout()\n\n# Display the figure\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:35:14.723501Z","iopub.execute_input":"2026-09-29T12:35:14.723764Z","iopub.status.idle":"2026-09-29T12:35:21.518911Z","shell.execute_reply.started":"2026-09-29T12:35:14.723742Z","shell.execute_reply":"2026-09-29T12:35:21.517939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lists to store image dimensions and unreadable images\nsizes = []\nbad = []\n\n# Check every training image\nfor image_id in df['id_code']:\n\n    # Construct image path\n    image_path = f'{BASE}/train_images/{image_id}.png'\n\n    # Read image\n    img = cv2.imread(image_path)\n\n    # Check whether image was loaded successfully\n    if img is None:\n        bad.append(image_id)\n    else:\n        # Store image height and width\n        sizes.append(img.shape[:2])\n\n# Print corruption results\nprint('Total training images checked:', len(df))\nprint('Corrupt / unreadable images:', len(bad))\n\nif bad:\n    print('\\nCorrupt image IDs:')\n    print(bad)\nelse:\n    print('No corrupt or unreadable images found.')\n\n# Print image size distribution\nprint('\\nMost common image sizes:')\nprint(pd.Series(sizes).value_counts().head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-29T12:35:33.021875Z","iopub.execute_input":"2026-09-29T12:35:33.022783Z","iopub.status.idle":"2026-09-29T12:42:39.140167Z","shell.execute_reply.started":"2026-09-29T12:35:33.022752Z","shell.execute_reply":"2026-09-29T12:42:39.139413Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA Conclusions\n\n- The APTOS 2019 training dataset contains 3,662 images with a significant class imbalance. The No DR class is the largest, while the Severe class is the smallest.\n- The training images have varying resolutions and aspect ratios, so image resizing will be required before training the CNN model.\n- All 3,662 training images were checked and no corrupt or unreadable images were found.\n- Because of the class imbalance, model evaluation will not rely on accuracy alone. Per-class recall and F1-score will also be reported. Class weighting or another suitable imbalance-handling technique will be evaluated during model training.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}