{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:25.202975Z","iopub.execute_input":"2025-04-10T19:24:25.20324Z","iopub.status.idle":"2025-04-10T19:24:30.645677Z","shell.execute_reply.started":"2025-04-10T19:24:25.203219Z","shell.execute_reply":"2025-04-10T19:24:30.644912Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploring the DATA SET ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:30.647123Z","iopub.execute_input":"2025-04-10T19:24:30.647876Z","iopub.status.idle":"2025-04-10T19:24:31.377461Z","shell.execute_reply.started":"2025-04-10T19:24:30.647849Z","shell.execute_reply":"2025-04-10T19:24:31.376941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Show first few rows\nprint(df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.37795Z","iopub.execute_input":"2025-04-10T19:24:31.378238Z","iopub.status.idle":"2025-04-10T19:24:31.403759Z","shell.execute_reply.started":"2025-04-10T19:24:31.37822Z","shell.execute_reply":"2025-04-10T19:24:31.403268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# shape of the dataset\nprint(\"Shape of dataset:\", df.shape)\n\n# columns\nprint(\"Columns:\", df.columns.tolist())\n\n# missing values\nprint(df.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.404495Z","iopub.execute_input":"2025-04-10T19:24:31.404725Z","iopub.status.idle":"2025-04-10T19:24:31.409921Z","shell.execute_reply.started":"2025-04-10T19:24:31.404697Z","shell.execute_reply":"2025-04-10T19:24:31.409312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#VISULAIZATION\nsns.countplot(x='diagnosis', data= df)\nplt.title('Class Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.411463Z","iopub.execute_input":"2025-04-10T19:24:31.411686Z","iopub.status.idle":"2025-04-10T19:24:31.693897Z","shell.execute_reply.started":"2025-04-10T19:24:31.411666Z","shell.execute_reply":"2025-04-10T19:24:31.693248Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"(0=No DR, 1=Mild, 2=Moderate, 3=Severe, 4=Proliferative DR)","metadata":{}},{"cell_type":"markdown","source":"# Image PRE PROCESSING\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport cv2\nimport os\nfrom tqdm import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.694621Z","iopub.execute_input":"2025-04-10T19:24:31.694867Z","iopub.status.idle":"2025-04-10T19:24:31.987537Z","shell.execute_reply.started":"2025-04-10T19:24:31.694845Z","shell.execute_reply":"2025-04-10T19:24:31.986819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess_image(image_path, img_size=224):\n    # Step 1: Read image\n    image = cv2.imread(image_path)\n    \n    # Step 2: Convert from BGR (OpenCV default) to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Step 3: Resize to smaller size (224x224)\n    image = cv2.resize(image, (img_size, img_size))\n    \n    # Step 4: Normalize pixel values (0-255 -> 0-1)\n    image = image / 255.0\n    \n    return image\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.988627Z","iopub.execute_input":"2025-04-10T19:24:31.988849Z","iopub.status.idle":"2025-04-10T19:24:31.992627Z","shell.execute_reply.started":"2025-04-10T19:24:31.988828Z","shell.execute_reply":"2025-04-10T19:24:31.992117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Directory where images are stored\nimg_folder = '/kaggle/input/aptos2019-blindness-detection/train_images/'\n\n# Create empty lists to save data\nimages = []\nlabels = []\n\n# Loop through first 100 images\nfor i in tqdm(range(100)):\n    img_name = df.loc[i, 'id_code']  # get image name\n    label = df.loc[i, 'diagnosis']   # get disease label\n    \n    img_path = os.path.join(img_folder, img_name + '.png')  # full path to image\n    \n    img = preprocess_image(img_path)  # preprocess the image\n    images.append(img)                # save the image\n    labels.append(label)              # save the label\n\n# Convert to arrays\nimages = np.array(images)\nlabels = np.array(labels)\n\nprint(\"Images shape:\", images.shape)  # (100, 224, 224, 3)\nprint(\"Labels shape:\", labels.shape)  # (100,)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:31.993309Z","iopub.execute_input":"2025-04-10T19:24:31.994088Z","iopub.status.idle":"2025-04-10T19:24:43.802674Z","shell.execute_reply.started":"2025-04-10T19:24:31.994018Z","shell.execute_reply":"2025-04-10T19:24:43.8019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot first 5 images\nplt.figure(figsize=(15,5))\n\nfor i in range(5):\n    plt.subplot(1, 5, i+1)\n    plt.imshow(images[i])\n    plt.title(f\"Label: {labels[i]}\")\n    plt.axis('off')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:43.803346Z","iopub.execute_input":"2025-04-10T19:24:43.803602Z","iopub.status.idle":"2025-04-10T19:24:44.168518Z","shell.execute_reply.started":"2025-04-10T19:24:43.803583Z","shell.execute_reply":"2025-04-10T19:24:44.167814Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA AGUMENTATION","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport albumentations as A\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm  # Progress bar\n\n# 1. Set paths\nINPUT_DIR = '/kaggle/input/aptos2019-blindness-detection/train_images'\nOUTPUT_DIR = '/kaggle/working/augmented_images'\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# 2. Define a lightweight augmentation pipeline\ntransform = A.Compose([\n    A.RandomBrightnessContrast(p=0.5),\n    A.HorizontalFlip(p=0.5),\n    A.Rotate(limit=15, p=0.5),\n    A.Resize(224, 224)  # Resize to 224x224 for faster processing\n])\n\n# 3. List images\nimage_list = os.listdir(INPUT_DIR)\nprint(f\"Found {len(image_list)} images.\")\n\n# 4. Loop with tqdm progress bar\nfor img_name in tqdm(image_list, desc=\"Augmenting Images\"):\n    img_path = os.path.join(INPUT_DIR, img_name)\n    image = cv2.imread(img_path)\n    if image is None:\n        continue  # Skip broken images\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # 5. Apply augmentations 2 times per image\n    for i in range(2):\n        augmented = transform(image=image)['image']\n        save_name = f\"{img_name.split('.')[0]}_aug{i}.png\"\n        save_path = os.path.join(OUTPUT_DIR, save_name)\n        cv2.imwrite(save_path, cv2.cvtColor(augmented, cv2.COLOR_RGB2BGR))\n\nprint(\" Augmentation done!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:24:44.169387Z","iopub.execute_input":"2025-04-10T19:24:44.169603Z","iopub.status.idle":"2025-04-10T19:32:55.492667Z","shell.execute_reply.started":"2025-04-10T19:24:44.169585Z","shell.execute_reply":"2025-04-10T19:32:55.491866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\n#  few augmented images\naugmented_images = os.listdir(OUTPUT_DIR)\n\n# Randomly pick 9 images to plot\nsample_images = random.sample(augmented_images, 9)\n\nplt.figure(figsize=(12, 12))\n\nfor idx, img_name in enumerate(sample_images):\n    img_path = os.path.join(OUTPUT_DIR, img_name)\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    plt.subplot(3, 3, idx+1)\n    plt.imshow(img)\n    plt.title(img_name, fontsize=8)\n    plt.axis('off')\n\nplt.suptitle('Sample Augmented Images', fontsize=16)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:32:55.493531Z","iopub.execute_input":"2025-04-10T19:32:55.494307Z","iopub.status.idle":"2025-04-10T19:32:56.781003Z","shell.execute_reply.started":"2025-04-10T19:32:55.494285Z","shell.execute_reply":"2025-04-10T19:32:56.780119Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"** now we can combine agumented and original data set **","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os\n\n# Folders\nORIGINAL_DIR = '/kaggle/input/aptos2019-blindness-detection/train_images'\nAUGMENTED_DIR = '/kaggle/working/augmented_images'\n\n# List all original images\noriginal_images = os.listdir(ORIGINAL_DIR)\n\n# Copy each original image to augmented folder\nfor img in original_images:\n    src = os.path.join(ORIGINAL_DIR, img)\n    dst = os.path.join(AUGMENTED_DIR, img)\n    shutil.copy(src, dst)\n\nprint(f\"Copied {len(original_images)} original images to augmented folder.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:32:56.782196Z","iopub.execute_input":"2025-04-10T19:32:56.782477Z","iopub.status.idle":"2025-04-10T19:33:21.145349Z","shell.execute_reply.started":"2025-04-10T19:32:56.782449Z","shell.execute_reply":"2025-04-10T19:33:21.144647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\naugmented_images = os.listdir('/kaggle/working/augmented_images')\nprint(f\"Total images after augmentation + copying: {len(augmented_images)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-10T19:33:21.146088Z","iopub.execute_input":"2025-04-10T19:33:21.14628Z","iopub.status.idle":"2025-04-10T19:33:21.155926Z","shell.execute_reply.started":"2025-04-10T19:33:21.146264Z","shell.execute_reply":"2025-04-10T19:33:21.155418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training the Data ","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nfrom PIL import Image\nimport cv2\nfrom sklearn.model_selection import train_test_split\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:39:02.042741Z","iopub.execute_input":"2025-04-11T02:39:02.043537Z","iopub.status.idle":"2025-04-11T02:39:02.341616Z","shell.execute_reply.started":"2025-04-11T02:39:02.043508Z","shell.execute_reply":"2025-04-11T02:39:02.340962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the CSV file\ndf = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/train.csv')\n\n# Image paths\nimage_dir = '/kaggle/input/aptos2019-blindness-detection/train_images'\n\n# Split into train and validation\ntrain_df, val_df = train_test_split(df, test_size=0.2, stratify=df['diagnosis'], random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:39:27.257194Z","iopub.execute_input":"2025-04-11T02:39:27.257759Z","iopub.status.idle":"2025-04-11T02:39:27.268941Z","shell.execute_reply.started":"2025-04-11T02:39:27.257739Z","shell.execute_reply":"2025-04-11T02:39:27.268284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DRDataset(Dataset):\n    def __init__(self, dataframe, image_dir, transform=None):\n        self.labels = dataframe\n        self.image_dir = image_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels)\n\n    def __getitem__(self, idx):\n        img_id = self.labels.iloc[idx, 0]\n        label = self.labels.iloc[idx, 1]\n        img_path = os.path.join(self.image_dir, img_name)\n        \n        # Read image\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        # Convert to PIL image\n        image = Image.fromarray(image)\n\n        # Apply transformations\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:44:27.337955Z","iopub.execute_input":"2025-04-11T02:44:27.338615Z","iopub.status.idle":"2025-04-11T02:44:27.343747Z","shell.execute_reply.started":"2025-04-11T02:44:27.338594Z","shell.execute_reply":"2025-04-11T02:44:27.343057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Image augmentations for training\ntrain_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(20),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n    transforms.ToTensor(),\n])\n\n# Validation (no augmentation)\nval_transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:40:25.622739Z","iopub.execute_input":"2025-04-11T02:40:25.623038Z","iopub.status.idle":"2025-04-11T02:40:25.628361Z","shell.execute_reply.started":"2025-04-11T02:40:25.622989Z","shell.execute_reply":"2025-04-11T02:40:25.627571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 32\n\ntrain_dataset = DRDataset(train_df, image_dir, transform=train_transform)\nval_dataset = DRDataset(val_df, image_dir, transform=val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:40:46.112601Z","iopub.execute_input":"2025-04-11T02:40:46.112868Z","iopub.status.idle":"2025-04-11T02:40:46.117265Z","shell.execute_reply.started":"2025-04-11T02:40:46.112848Z","shell.execute_reply":"2025-04-11T02:40:46.116517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import models\nfrom torchvision.models import EfficientNet_B0_Weights\n\n# Load EfficientNet-B0 with pretrained weights\nmodel = models.efficientnet_b0(weights=EfficientNet_B0_Weights.IMAGENET1K_V1)\nmodel.classifier[1] = nn.Linear(model.classifier[1].in_features, 5)  # For 5 classes\nmodel = model.to(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:47:33.725783Z","iopub.execute_input":"2025-04-11T02:47:33.726381Z","iopub.status.idle":"2025-04-11T02:47:34.018188Z","shell.execute_reply.started":"2025-04-11T02:47:33.726359Z","shell.execute_reply":"2025-04-11T02:47:34.017403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:51:50.969588Z","iopub.execute_input":"2025-04-11T02:51:50.970249Z","iopub.status.idle":"2025-04-11T02:51:50.975806Z","shell.execute_reply.started":"2025-04-11T02:51:50.970227Z","shell.execute_reply":"2025-04-11T02:51:50.975087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training and Validation","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm  \n\nnum_epochs = 5  # You can increase if you want\n\nfor epoch in range(num_epochs):\n    model.train()\n    train_loss = 0\n\n    # Wrap train_loader with tqdm\n    loop = tqdm(train_loader, leave=True)  # leave=True keeps the last progress bar printed\n\n    for images, labels in loop:\n        images, labels = images.to(device), labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        train_loss += loss.item()\n\n        # Update tqdm description\n        loop.set_description(f\"Epoch [{epoch+1}/{num_epochs}]\")\n        loop.set_postfix(loss=loss.item())\n\n    print(f\"Epoch {epoch+1}/{num_epochs}, Training Loss: {train_loss/len(train_loader):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T02:52:22.942467Z","iopub.execute_input":"2025-04-11T02:52:22.943172Z","iopub.status.idle":"2025-04-11T03:22:35.465369Z","shell.execute_reply.started":"2025-04-11T02:52:22.943147Z","shell.execute_reply":"2025-04-11T03:22:35.464703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()  # Set model to evaluation mode\ncorrect = 0\ntotal = 0\nval_loss = 0\n\nwith torch.no_grad():  # Disable gradient calculation\n    for images, labels in val_loader:\n        images, labels = images.to(device), labels.to(device)\n\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        val_loss += loss.item()\n\n        _, predicted = torch.max(outputs.data, 1)  # Pick highest probability\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n\navg_val_loss = val_loss / len(val_loader)\naccuracy = 100 * correct / total\n\nprint(f\"Validation Loss: {avg_val_loss:.4f}, Validation Accuracy: {accuracy:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T03:31:00.138687Z","iopub.execute_input":"2025-04-11T03:31:00.138988Z","iopub.status.idle":"2025-04-11T03:32:34.457556Z","shell.execute_reply.started":"2025-04-11T03:31:00.138968Z","shell.execute_reply":"2025-04-11T03:32:34.456895Z"}},"outputs":[],"execution_count":null}]}