{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This project focuses on detecting retinal health disorders using deep learning and transfer learning techniques. The APTOS 2019 retina dataset was used, where retina fundus images were classified into Healthy and Diseased categories. The images were preprocessed by resizing them to 224×224 pixels and normalizing pixel values before training. A MobileNetV2 pre-trained convolutional neural network was used for binary image classification, and the model was trained for 3 epochs. The performance of the model was evaluated using accuracy, precision, recall, and confusion matrix metrics. Grad-CAM visualization was implemented to highlight the retinal regions responsible for the model’s predictions, improving explainability in medical image analysis. The results demonstrate that transfer learning can effectively assist in automated retinal disease detection, while future improvements can include larger datasets, advanced preprocessing, and fine-tuning deeper network layers for better performance.","metadata":{}},{"cell_type":"code","source":"# ============================================\n# 1. DOWNLOAD DATASET\n# ============================================\n\nimport kagglehub\n\n# Download dataset\npath = kagglehub.competition_download(\n    'aptos2019-blindness-detection'\n)\n\nprint(\"Dataset Path:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T11:11:43.313615Z","iopub.execute_input":"2026-05-06T11:11:43.314349Z","iopub.status.idle":"2026-05-06T11:11:43.850614Z","shell.execute_reply.started":"2026-05-06T11:11:43.314319Z","shell.execute_reply":"2026-05-06T11:11:43.850035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 2. IMPORT LIBRARIES\n# ============================================\n\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\n\nimport tensorflow as tf\n\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.optimizers import Adam\n\nprint(\"TensorFlow Version:\", tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:42:10.602838Z","iopub.execute_input":"2026-05-06T10:42:10.603445Z","iopub.status.idle":"2026-05-06T10:42:10.609223Z","shell.execute_reply.started":"2026-05-06T10:42:10.603416Z","shell.execute_reply":"2026-05-06T10:42:10.608463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 3. CHECK DATASET FILES\n# ============================================\n\nprint(os.listdir(path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:42:24.355139Z","iopub.execute_input":"2026-05-06T10:42:24.355906Z","iopub.status.idle":"2026-05-06T10:42:24.364365Z","shell.execute_reply.started":"2026-05-06T10:42:24.355875Z","shell.execute_reply":"2026-05-06T10:42:24.363529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 4. DEFINE PATHS\n# ============================================\n\ncsv_path = os.path.join(path, \"train.csv\")\n\nimage_folder = os.path.join(path, \"train_images\")\n\nprint(\"CSV Path:\", csv_path)\nprint(\"Image Folder:\", image_folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:42:35.825942Z","iopub.execute_input":"2026-05-06T10:42:35.826677Z","iopub.status.idle":"2026-05-06T10:42:35.831327Z","shell.execute_reply.started":"2026-05-06T10:42:35.826645Z","shell.execute_reply":"2026-05-06T10:42:35.830448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 5. LOAD CSV FILE\n# ============================================\n\ndf = pd.read_csv(csv_path)\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:42:51.917307Z","iopub.execute_input":"2026-05-06T10:42:51.917917Z","iopub.status.idle":"2026-05-06T10:42:51.952665Z","shell.execute_reply.started":"2026-05-06T10:42:51.917886Z","shell.execute_reply":"2026-05-06T10:42:51.952092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 6. CONVERT LABELS\n# ============================================\n\n# 0 -> Healthy\n# 1,2,3,4 -> Diseased\n\ndf['binary_label'] = df['diagnosis'].apply(\n    lambda x: 0 if x == 0 else 1\n)\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:43:06.002425Z","iopub.execute_input":"2026-05-06T10:43:06.003255Z","iopub.status.idle":"2026-05-06T10:43:06.013714Z","shell.execute_reply.started":"2026-05-06T10:43:06.003224Z","shell.execute_reply":"2026-05-06T10:43:06.013155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 7. TAKE SMALL SUBSET\n# ============================================\n\nhealthy_df = df[df['binary_label'] == 0].sample(\n    150,\n    random_state=42\n)\n\ndiseased_df = df[df['binary_label'] == 1].sample(\n    150,\n    random_state=42\n)\n\ndf = pd.concat([healthy_df, diseased_df])\n\ndf = df.sample(frac=1).reset_index(drop=True)\n\nprint(\"Total Images:\", len(df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:43:20.310421Z","iopub.execute_input":"2026-05-06T10:43:20.310679Z","iopub.status.idle":"2026-05-06T10:43:20.325548Z","shell.execute_reply.started":"2026-05-06T10:43:20.310657Z","shell.execute_reply":"2026-05-06T10:43:20.324792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 8. LOAD AND PREPROCESS IMAGES\n# ============================================\n\nIMG_SIZE = 224\n\nimages = []\nlabels = []\n\nfor index, row in df.iterrows():\n\n    image_id = row['id_code']\n\n    label = row['binary_label']\n\n    image_path = os.path.join(\n        image_folder,\n        image_id + \".png\"\n    )\n\n    try:\n\n        img = cv2.imread(image_path)\n\n        img = cv2.cvtColor(\n            img,\n            cv2.COLOR_BGR2RGB\n        )\n\n        img = cv2.resize(\n            img,\n            (IMG_SIZE, IMG_SIZE)\n        )\n\n        # Optional enhancement\n        img = cv2.GaussianBlur(img, (5,5), 0)\n\n        # Normalize\n        img = img / 255.0\n\n        images.append(img)\n\n        labels.append(label)\n\n    except:\n        pass\n\nimages = np.array(images)\n\nlabels = np.array(labels)\n\nprint(\"Images Shape:\", images.shape)\n\nprint(\"Labels Shape:\", labels.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:43:37.155701Z","iopub.execute_input":"2026-05-06T10:43:37.156241Z","iopub.status.idle":"2026-05-06T10:44:06.802517Z","shell.execute_reply.started":"2026-05-06T10:43:37.156213Z","shell.execute_reply":"2026-05-06T10:44:06.801608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 9. VISUALIZE SAMPLE IMAGES\n# ============================================\n\nplt.figure(figsize=(12,6))\n\nfor i in range(6):\n\n    plt.subplot(2,3,i+1)\n\n    plt.imshow(images[i])\n\n    if labels[i] == 0:\n        plt.title(\"Healthy\")\n    else:\n        plt.title(\"Diseased\")\n\n    plt.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T11:06:11.583956Z","iopub.execute_input":"2026-05-06T11:06:11.584492Z","iopub.status.idle":"2026-05-06T11:06:11.927019Z","shell.execute_reply.started":"2026-05-06T11:06:11.584465Z","shell.execute_reply":"2026-05-06T11:06:11.926167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 10. TRAIN TEST SPLIT\n# ============================================\n\nX_train, X_test, y_train, y_test = train_test_split(\n    images,\n    labels,\n    test_size=0.2,\n    random_state=42,\n    stratify=labels\n)\n\nprint(\"Training Images:\", X_train.shape)\n\nprint(\"Testing Images:\", X_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T11:07:49.732851Z","iopub.execute_input":"2026-05-06T11:07:49.733439Z","iopub.status.idle":"2026-05-06T11:07:49.844245Z","shell.execute_reply.started":"2026-05-06T11:07:49.733411Z","shell.execute_reply":"2026-05-06T11:07:49.843516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 11. BUILD TRANSFER LEARNING MODEL\n# ============================================\n\nbase_model = MobileNetV2(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224,224,3)\n)\n\nbase_model.trainable = False\n\nx = base_model.output\n\nx = GlobalAveragePooling2D()(x)\n\nx = Dense(128, activation='relu')(x)\n\nx = Dropout(0.3)(x)\n\noutput = Dense(1, activation='sigmoid')(x)\n\nmodel = Model(\n    inputs=base_model.input,\n    outputs=output\n)\n\nmodel.compile(\n    optimizer=Adam(learning_rate=0.0001),\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:44:39.657013Z","iopub.execute_input":"2026-05-06T10:44:39.657277Z","iopub.status.idle":"2026-05-06T10:44:43.763505Z","shell.execute_reply.started":"2026-05-06T10:44:39.657253Z","shell.execute_reply":"2026-05-06T10:44:43.762923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 12. TRAIN MODEL\n# ============================================\n\nhistory = model.fit(\n    X_train,\n    y_train,\n    validation_data=(X_test, y_test),\n    epochs=3,\n    batch_size=16\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:45:11.505465Z","iopub.execute_input":"2026-05-06T10:45:11.505905Z","iopub.status.idle":"2026-05-06T10:45:44.544413Z","shell.execute_reply.started":"2026-05-06T10:45:11.505876Z","shell.execute_reply":"2026-05-06T10:45:44.543761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 13. PLOT ACCURACY AND LOSS\n# ============================================\n\nplt.figure(figsize=(12,5))\n\n# Accuracy Plot\nplt.subplot(1,2,1)\n\nplt.plot(\n    history.history['accuracy'],\n    label='Train Accuracy'\n)\n\nplt.plot(\n    history.history['val_accuracy'],\n    label='Validation Accuracy'\n)\n\nplt.title(\"Accuracy\")\n\nplt.xlabel(\"Epoch\")\n\nplt.ylabel(\"Accuracy\")\n\nplt.legend()\n\n# Loss Plot\nplt.subplot(1,2,2)\n\nplt.plot(\n    history.history['loss'],\n    label='Train Loss'\n)\n\nplt.plot(\n    history.history['val_loss'],\n    label='Validation Loss'\n)\n\nplt.title(\"Loss\")\n\nplt.xlabel(\"Epoch\")\n\nplt.ylabel(\"Loss\")\n\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:45:44.545604Z","iopub.execute_input":"2026-05-06T10:45:44.545956Z","iopub.status.idle":"2026-05-06T10:45:44.848734Z","shell.execute_reply.started":"2026-05-06T10:45:44.545931Z","shell.execute_reply":"2026-05-06T10:45:44.847924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 14. MODEL EVALUATION\n# ============================================\n\npredictions = model.predict(X_test)\n\npredictions = (predictions > 0.5).astype(int)\n\nprint(\n    classification_report(\n        y_test,\n        predictions\n    )\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:46:16.204584Z","iopub.execute_input":"2026-05-06T10:46:16.205072Z","iopub.status.idle":"2026-05-06T10:46:39.141104Z","shell.execute_reply.started":"2026-05-06T10:46:16.205033Z","shell.execute_reply":"2026-05-06T10:46:39.140239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 15. CONFUSION MATRIX\n# ============================================\n\ncm = confusion_matrix(\n    y_test,\n    predictions\n)\n\nplt.figure(figsize=(6,5))\n\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt='d',\n    cmap='Blues',\n    xticklabels=['Healthy', 'Diseased'],\n    yticklabels=['Healthy', 'Diseased']\n)\n\nplt.xlabel(\"Predicted\")\n\nplt.ylabel(\"Actual\")\n\nplt.title(\"Confusion Matrix\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:46:40.351314Z","iopub.execute_input":"2026-05-06T10:46:40.351928Z","iopub.status.idle":"2026-05-06T10:46:40.482452Z","shell.execute_reply.started":"2026-05-06T10:46:40.351897Z","shell.execute_reply":"2026-05-06T10:46:40.481798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 16. SAVE MODEL\n# ============================================\n\nmodel.save(\"retina_model.keras\")\n\nprint(\"Model Saved Successfully\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T10:47:23.467160Z","iopub.execute_input":"2026-05-06T10:47:23.467582Z","iopub.status.idle":"2026-05-06T10:47:23.908418Z","shell.execute_reply.started":"2026-05-06T10:47:23.467551Z","shell.execute_reply":"2026-05-06T10:47:23.907796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 17. GRAD-CAM SETUP\n# ============================================\n\nlast_conv_layer_name = \"Conv_1\"\n\ngrad_model = tf.keras.models.Model(\n    inputs=model.inputs,\n    outputs=[\n        model.get_layer(last_conv_layer_name).output,\n        model.output\n    ]\n)\n\ndef make_gradcam_heatmap(img_array):\n\n    img_array = tf.convert_to_tensor(\n        img_array,\n        dtype=tf.float32\n    )\n\n    with tf.GradientTape() as tape:\n\n        conv_outputs, predictions = grad_model(\n            img_array\n        )\n\n        loss = predictions[:, 0]\n\n    grads = tape.gradient(\n        loss,\n        conv_outputs\n    )\n\n    pooled_grads = tf.reduce_mean(\n        grads,\n        axis=(0,1,2)\n    )\n\n    conv_outputs = conv_outputs[0]\n\n    heatmap = tf.reduce_sum(\n        tf.multiply(\n            pooled_grads,\n            conv_outputs\n        ),\n        axis=-1\n    )\n\n    heatmap = tf.maximum(\n        heatmap,\n        0\n    )\n\n    heatmap = heatmap / (\n        tf.reduce_max(heatmap) + 1e-8\n    )\n\n    return heatmap.numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T11:02:37.296682Z","iopub.execute_input":"2026-05-06T11:02:37.297153Z","iopub.status.idle":"2026-05-06T11:02:37.312293Z","shell.execute_reply.started":"2026-05-06T11:02:37.297123Z","shell.execute_reply":"2026-05-06T11:02:37.311632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# 18. GENERATE HEALTHY + DISEASED GRAD-CAM\n# ============================================\n\nplt.figure(figsize=(15,18))\n\ncount = 0\n\nhealthy_found = 0\ndiseased_found = 0\n\nfor i in range(len(X_test)):\n\n    sample_image = X_test[i]\n\n    img_array = np.expand_dims(\n        sample_image,\n        axis=0\n    )\n\n    # Prediction\n    prediction = model.predict(\n        img_array,\n        verbose=0\n    )[0][0]\n\n    if prediction > 0.5:\n        predicted_label = \"Diseased\"\n    else:\n        predicted_label = \"Healthy\"\n\n    # Keep:\n    # 2 Healthy samples\n    # 1 Diseased sample\n\n    if predicted_label == \"Healthy\" and healthy_found < 2:\n\n        healthy_found += 1\n\n    elif predicted_label == \"Diseased\" and diseased_found < 1:\n\n        diseased_found += 1\n\n    else:\n        continue\n\n    # Generate heatmap\n    heatmap = make_gradcam_heatmap(\n        img_array\n    )\n\n    # Resize heatmap\n    heatmap = cv2.resize(\n        heatmap,\n        (224,224)\n    )\n\n    # Convert heatmap\n    heatmap = np.uint8(\n        255 * heatmap\n    )\n\n    heatmap = cv2.applyColorMap(\n        heatmap,\n        cv2.COLORMAP_JET\n    )\n\n    # Overlay heatmap\n    superimposed_img = (\n        heatmap * 0.7 +\n        (sample_image * 255)\n    )\n\n    superimposed_img = np.clip(\n        superimposed_img,\n        0,\n        255\n    ).astype(\"uint8\")\n\n    # ORIGINAL IMAGE\n    plt.subplot(\n        3,\n        2,\n        count*2 + 1\n    )\n\n    plt.imshow(sample_image)\n\n    plt.title(\n        f\"Original Retina\\nPrediction: {predicted_label}\"\n    )\n\n    plt.axis(\"off\")\n\n    # GRAD-CAM IMAGE\n    plt.subplot(\n        3,\n        2,\n        count*2 + 2\n    )\n\n    plt.imshow(superimposed_img)\n\n    plt.title(\"Grad-CAM Heatmap\")\n\n    plt.axis(\"off\")\n\n    count += 1\n\n    # Stop after 3 samples\n    if count == 3:\n        break\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T11:16:42.795045Z","iopub.execute_input":"2026-05-06T11:16:42.795325Z","iopub.status.idle":"2026-05-06T11:16:44.972518Z","shell.execute_reply.started":"2026-05-06T11:16:42.795302Z","shell.execute_reply":"2026-05-06T11:16:44.971791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}