{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4104,"databundleVersionId":46661}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:41:53.604985Z","iopub.execute_input":"2026-03-22T10:41:53.605241Z","iopub.status.idle":"2026-03-22T10:41:54.791270Z","shell.execute_reply.started":"2026-03-22T10:41:53.605220Z","shell.execute_reply":"2026-03-22T10:41:54.790638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Perform EDA on the Dataset","metadata":{}},{"cell_type":"markdown","source":"The first task is to inspect the content of the mounted dataset. There is both image and textual dataset given here.\n* The **trainLabels.csv** contains labelled output of each test image for training purposes\n* Each **test.zip** contains a set of left and right images corresponsing to the csv file","metadata":{}},{"cell_type":"markdown","source":"## EDA on Labelled Dataset","metadata":{}},{"cell_type":"code","source":"# List the items in the provided dataset list\nINPUT_DIR = Path(\"/kaggle/input/competitions/diabetic-retinopathy-detection/\")\nfor item in os.listdir(INPUT_DIR):\n    print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:41:58.534020Z","iopub.execute_input":"2026-03-22T10:41:58.534263Z","iopub.status.idle":"2026-03-22T10:41:58.538668Z","shell.execute_reply.started":"2026-03-22T10:41:58.534244Z","shell.execute_reply":"2026-03-22T10:41:58.538060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\n\n# create a function to extract data from a zip file. Easier in next iterations\ndef extract_file(zip_path,extract_path):\n    os.makedirs(extract_path, exist_ok=True)\n    \n    with zipfile.ZipFile(zip_path, 'r') as zip_ref:\n        zip_ref.extractall(extract_path)\n    print(f\"Extracted {zip_path.name} successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:01.232242Z","iopub.execute_input":"2026-03-22T10:42:01.232522Z","iopub.status.idle":"2026-03-22T10:42:01.236381Z","shell.execute_reply.started":"2026-03-22T10:42:01.232503Z","shell.execute_reply":"2026-03-22T10:42:01.235835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Extracted labelled dataset inside the working directory\ncsv_zip_path = Path(\"/kaggle/input/competitions/diabetic-retinopathy-detection/trainLabels.csv.zip\")\ncsv_extract_path = Path(\"/kaggle/working/eyedata\")\nextract_file(csv_zip_path,csv_extract_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:02.362665Z","iopub.execute_input":"2026-03-22T10:42:02.362935Z","iopub.status.idle":"2026-03-22T10:42:02.378626Z","shell.execute_reply.started":"2026-03-22T10:42:02.362915Z","shell.execute_reply":"2026-03-22T10:42:02.378007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/working/eyedata/trainLabels.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:03.643572Z","iopub.execute_input":"2026-03-22T10:42:03.643868Z","iopub.status.idle":"2026-03-22T10:42:03.689260Z","shell.execute_reply.started":"2026-03-22T10:42:03.643847Z","shell.execute_reply":"2026-03-22T10:42:03.688648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape # 35126 records and 2 columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:04.933335Z","iopub.execute_input":"2026-03-22T10:42:04.933609Z","iopub.status.idle":"2026-03-22T10:42:04.938742Z","shell.execute_reply.started":"2026-03-22T10:42:04.933589Z","shell.execute_reply":"2026-03-22T10:42:04.937992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:05.763575Z","iopub.execute_input":"2026-03-22T10:42:05.764337Z","iopub.status.idle":"2026-03-22T10:42:05.787529Z","shell.execute_reply.started":"2026-03-22T10:42:05.764309Z","shell.execute_reply":"2026-03-22T10:42:05.786679Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum() # no null values in the dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:06.913051Z","iopub.execute_input":"2026-03-22T10:42:06.913285Z","iopub.status.idle":"2026-03-22T10:42:06.921189Z","shell.execute_reply.started":"2026-03-22T10:42:06.913267Z","shell.execute_reply":"2026-03-22T10:42:06.920538Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Upto now the datatype of the fields seem to be correct and there is no indication of null values present","metadata":{}},{"cell_type":"code","source":"# Break the images into a subject id and eye_side for easier processing\ndf['subject_id'] = df['image'].str.extract(r'^(\\d+)_')\ndf['eye_side']   = df['image'].str.extract(r'_(left|right)')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:08.437482Z","iopub.execute_input":"2026-03-22T10:42:08.437749Z","iopub.status.idle":"2026-03-22T10:42:08.577362Z","shell.execute_reply.started":"2026-03-22T10:42:08.437731Z","shell.execute_reply":"2026-03-22T10:42:08.576711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check if same image is labelled more than once or if some image labels are missing\nsubject_counts = df['subject_id'].value_counts()\nover_twice = subject_counts[subject_counts > 2]\nunder_twice = subject_counts[subject_counts < 2]\n\n# This seems the same id is not labelled more than twice\nprint(f\"Number of subjects appearing more than twice: {len(over_twice)}\") \nprint(f\"Number of subjects appearing less than twice: {len(under_twice)}\")\n\nprint(f\"Eye side distribution: {df['eye_side'].value_counts()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:10.504345Z","iopub.execute_input":"2026-03-22T10:42:10.504665Z","iopub.status.idle":"2026-03-22T10:42:10.522585Z","shell.execute_reply.started":"2026-03-22T10:42:10.504642Z","shell.execute_reply":"2026-03-22T10:42:10.521723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Images Length: {len(df.image)}\")\nprint(\"Class Count : {} \\n\".format(len(df['level'].value_counts())))\n\nprint(\"Count the number of images in each class\")\nprint(df['level'].value_counts())\n\nprint(\"Class Distribution (%):\")\nprint((df['level'].value_counts() / len(df) * 100).round(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:12.225954Z","iopub.execute_input":"2026-03-22T10:42:12.226223Z","iopub.status.idle":"2026-03-22T10:42:12.235868Z","shell.execute_reply.started":"2026-03-22T10:42:12.226200Z","shell.execute_reply":"2026-03-22T10:42:12.234921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=df, x='level')\nplt.title('Distribution of Diabetic Retinopathy Severity Levels')\nplt.xlabel('\\n Image Level')\nplt.ylabel('Count Image')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:14.559757Z","iopub.execute_input":"2026-03-22T10:42:14.559987Z","iopub.status.idle":"2026-03-22T10:42:14.803963Z","shell.execute_reply.started":"2026-03-22T10:42:14.559970Z","shell.execute_reply":"2026-03-22T10:42:14.803282Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As seen here there is a class imbalance among the 5 classes with training data being heavy on level 0","metadata":{}},{"cell_type":"code","source":"# Perform a bivariate analysis between the eye_side(left or right) and the level\n# This is to figure out if there is any correlation between the level and the eye level\npd.crosstab(df['eye_side'], df['level'],normalize='index') * 100","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T11:00:29.788831Z","iopub.execute_input":"2026-03-22T11:00:29.789111Z","iopub.status.idle":"2026-03-22T11:00:29.806999Z","shell.execute_reply.started":"2026-03-22T11:00:29.789092Z","shell.execute_reply":"2026-03-22T11:00:29.806267Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The distributions are almost identical and there doesnt seem to be much of a correlation among the two","metadata":{}},{"cell_type":"markdown","source":"## EDA on Training Images","metadata":{}},{"cell_type":"code","source":"# Collect all training data into one set\nparts = sorted(INPUT_DIR.glob(\"train.zip.00*\"))\nprint(f\"Found {len(parts)} parts:\")\ntotal_size = 0\nfor p in parts:\n    size = p.stat().st_size / 1e9\n    print(f\"  {p.name}  {size:.2f} GB\")\n    total_size += size\nprint(f\"Total training dataset size: {round(total_size, 2)} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:17.424995Z","iopub.execute_input":"2026-03-22T10:42:17.425249Z","iopub.status.idle":"2026-03-22T10:42:17.435918Z","shell.execute_reply.started":"2026-03-22T10:42:17.425227Z","shell.execute_reply":"2026-03-22T10:42:17.435189Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As seen here the compressed training dataset itself is around **35GB** which exceeds kaggles session limit of **20GB**.\n\nTo tackle this one option is to resize and perform pre-processing on the images to reduce the size and remove the original in batch.\n\nBefore moving on it is necessary to inspect how the dataset looks using the **sample.zip**","metadata":{}},{"cell_type":"code","source":"SAMPLE_ZIP  = Path(\"/kaggle/input/competitions/diabetic-retinopathy-detection/sample.zip\")\nSAMPLE_DIR  = Path(\"/kaggle/working/eyedata/dataset\")\n\n#Extract sample.zip\nextract_file(SAMPLE_ZIP,SAMPLE_DIR)\n\nSAMPLE_IMAGE_DIR = Path(f\"{SAMPLE_DIR}/sample\")\nimages = list(SAMPLE_IMAGE_DIR.glob(\"*.jpeg\"))\n\nprint(f\"Sample images found: {len(images)}\")\nimg_list = []\nfor img in images:\n    print(f\"{img.name}\")\n    img_list.append(img.stem)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:25.380368Z","iopub.execute_input":"2026-03-22T10:42:25.380659Z","iopub.status.idle":"2026-03-22T10:42:25.606210Z","shell.execute_reply.started":"2026-03-22T10:42:25.380639Z","shell.execute_reply":"2026-03-22T10:42:25.605580Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The sample zip only contains 10 sample images and 5 pairs which is insufficient for a proper EDA","metadata":{}},{"cell_type":"code","source":"# As seen here majority of images on the sample.zip is from level 0 while level 3 is completly missing\ndf[df['image'].isin(img_list)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:28.839526Z","iopub.execute_input":"2026-03-22T10:42:28.839797Z","iopub.status.idle":"2026-03-22T10:42:28.851574Z","shell.execute_reply.started":"2026-03-22T10:42:28.839774Z","shell.execute_reply":"2026-03-22T10:42:28.850918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Following challanges are encounterd:\n* The sample.zip doesn't contain enough images of class balance to perform a proper EDA\n* Loading the full dataset in impractical in one kaggle session\n\nFor this purpose basic EDA is done on the sample.zip to understand basic information such as resoultion, size and other factors to design the pre-processing pipeline","metadata":{}},{"cell_type":"code","source":"records = []\nfor img_path in sorted(images):\n\n    #Open each image\n    with Image.open(img_path) as img:\n        w, h     = img.size\n        mode     = img.mode\n        arr      = np.array(img).astype(np.float32)\n        file_kb  = img_path.stat().st_size / 1024\n\n        records.append({\n            \"name\"    : img_path.name,\n            \"width\"   : w,\n            \"height\"  : h,\n            \"mode\"    : mode,\n            \"file_kb\" : round(file_kb, 1),\n        })\n\n\ndf_sample = pd.DataFrame(records)\ndf_sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:32.192819Z","iopub.execute_input":"2026-03-22T10:42:32.193095Z","iopub.status.idle":"2026-03-22T10:42:33.281517Z","shell.execute_reply.started":"2026-03-22T10:42:32.193075Z","shell.execute_reply":"2026-03-22T10:42:33.280689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Resolution summary ────────────────────────────────────────────────────────\nprint(\"Unique resolutions found:\")\nprint(df_sample.groupby([\"width\", \"height\"]).size().reset_index(name=\"count\"))\n\nprint(f\"\\nMin width : {df_sample['width'].min()}px\")\nprint(f\"Max width : {df_sample['width'].max()}px\")\nprint(f\"Min height: {df_sample['height'].min()}px\")\nprint(f\"Max height: {df_sample['height'].max()}px\")\nprint(f\"All RGB   : {(df_sample['mode'] == 'RGB').all()}\")\nprint(f\"Average Size: {df_sample['file_kb'].mean()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:36.022425Z","iopub.execute_input":"2026-03-22T10:42:36.022703Z","iopub.status.idle":"2026-03-22T10:42:36.035510Z","shell.execute_reply.started":"2026-03-22T10:42:36.022680Z","shell.execute_reply":"2026-03-22T10:42:36.034426Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As seen from the sample there is a varying sizes among each images and most of the images are quite large. A preprocessing pipeline must be built to reduce the size.","metadata":{}},{"cell_type":"markdown","source":"As seen from above results most images in the sample.zip is to the higher end which needs to be dealt with during preprocessing part","metadata":{}},{"cell_type":"code","source":"# Complete the 3-panel plot\nfig, axes = plt.subplots(1, 3, figsize=(14, 4))\n\n# Panel 1 — Resolution scatter\naxes[0].scatter(df_sample[\"width\"], df_sample[\"height\"], color=\"#378ADD\", s=80)\nfor _, row in df_sample.iterrows():\n    axes[0].annotate(row[\"name\"].replace(\".jpeg\",\"\"),\n                     (row[\"width\"], row[\"height\"]), fontsize=7)\naxes[0].set_xlabel(\"Width (px)\")\naxes[0].set_ylabel(\"Height (px)\")\naxes[0].set_title(\"Resolution per Image\")\n\n# Panel 2 — File size bar chart\naxes[1].barh(df_sample[\"name\"], df_sample[\"file_kb\"], color=\"#378ADD\")\naxes[1].set_xlabel(\"File Size (KB)\")\naxes[1].set_title(\"File Size per Image\")\n\n# Panel 3 — Aspect ratio distribution\ndf_sample[\"aspect_ratio\"] = df_sample[\"width\"] / df_sample[\"height\"]\naxes[2].hist(df_sample[\"aspect_ratio\"], bins=5, color=\"#378ADD\", edgecolor=\"white\")\naxes[2].set_xlabel(\"Aspect Ratio (W/H)\")\naxes[2].set_title(\"Aspect Ratio Distribution\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:42:46.393824Z","iopub.execute_input":"2026-03-22T10:42:46.394080Z","iopub.status.idle":"2026-03-22T10:42:46.768858Z","shell.execute_reply.started":"2026-03-22T10:42:46.394059Z","shell.execute_reply":"2026-03-22T10:42:46.767907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualise sample images in a grid with their labels\nfig, axes = plt.subplots(2, 5, figsize=(18, 7))\naxes = axes.flatten()\n\nfor i, img_path in enumerate(sorted(images)[:10]):\n    img_name = img_path.stem   # e.g. \"10_left\"\n    label_row = df[df['image'] == img_name]\n    level = label_row['level'].values[0] if len(label_row) > 0 else \"N/A\"\n\n    with Image.open(img_path) as img:\n        axes[i].imshow(img)\n        axes[i].set_title(f\"{img_path.name}\\nLevel: {level}\", fontsize=8)\n        axes[i].axis(\"off\")\n\nplt.suptitle(\"Sample Retinal Images with DR Severity Level\", fontsize=13)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:01:57.516995Z","iopub.execute_input":"2026-03-22T10:01:57.517433Z","iopub.status.idle":"2026-03-22T10:02:05.874728Z","shell.execute_reply.started":"2026-03-22T10:01:57.517406Z","shell.execute_reply":"2026-03-22T10:02:05.873406Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"According to the dataset description - The images in the dataset come from different models and types of cameras, which can affect the visual appearance of left vs. right and Some images are shown as one would see the retina anatomically (macula on the left, optic nerve on the right for the right eye) and others are shown inverted.\n\n* It is inverted if the macula (the small dark central area) is slightly higher than the midline through the optic nerve. If the macula is lower than the midline of the optic nerve, it's not inverted.\n* If there is a **notch on the side of the image** (square, triangle, or circle) then it's not inverted. If there is no notch, it's inverted.","metadata":{}},{"cell_type":"code","source":"brightness = []\nfor img_path in sorted(images):\n    label_row = df[df['image'] == img_path.stem]\n    level = label_row['level'].values[0] if len(label_row) > 0 else -1\n    \n    with Image.open(img_path) as img:\n        arr = np.array(img).astype(np.float32)\n        brightness.append({\n            'image'   : img_path.stem,\n            'level'   : level,\n            'mean_R'  : arr[:,:,0].mean(),\n            'mean_G'  : arr[:,:,1].mean(),\n            'mean_B'  : arr[:,:,2].mean(),\n            'mean_all': arr.mean()\n        })\n\ndf_bright = pd.DataFrame(brightness)\ndf_bright","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:57:34.855023Z","iopub.execute_input":"2026-03-22T10:57:34.855273Z","iopub.status.idle":"2026-03-22T10:57:36.093634Z","shell.execute_reply.started":"2026-03-22T10:57:34.855252Z","shell.execute_reply":"2026-03-22T10:57:36.093101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot mean channel intensities per image\ndf_bright.set_index('image')[['mean_R','mean_G','mean_B']].plot(\n    kind='bar', figsize=(12, 4), color=['#e74c3c','#2ecc71','#3498db']\n)\nplt.title('Mean Pixel Intensity per Channel per Image')\nplt.ylabel('Mean Pixel Value (0-255)')\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:48:30.539077Z","iopub.execute_input":"2026-03-22T10:48:30.539347Z","iopub.status.idle":"2026-03-22T10:48:30.777517Z","shell.execute_reply.started":"2026-03-22T10:48:30.539325Z","shell.execute_reply":"2026-03-22T10:48:30.776670Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There is a serious variance among the brigtness and exposure of the images in the sample dataset with images like **10_right having low exposure**. This needs to be generalized when doing preprocessing.\n\nIt is reasonable to assume that due to the different camera models used other images in the training set would also be the same","metadata":{}},{"cell_type":"code","source":"df_bright['quality_flag'] = 'normal'\ndf_bright.loc[df_bright['mean_all'] < 30,  'quality_flag'] = 'underexposed'\ndf_bright.loc[df_bright['mean_all'] > 220, 'quality_flag'] = 'overexposed'\nprint(df_bright[['image', 'level', 'mean_all', 'quality_flag']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-22T10:48:46.457960Z","iopub.execute_input":"2026-03-22T10:48:46.458198Z","iopub.status.idle":"2026-03-22T10:48:46.466553Z","shell.execute_reply.started":"2026-03-22T10:48:46.458181Z","shell.execute_reply":"2026-03-22T10:48:46.465745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing and Transformation","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}