{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 📦 STEP 1: Import Libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:54:48.733546Z","iopub.execute_input":"2025-04-02T19:54:48.733736Z","iopub.status.idle":"2025-04-02T19:55:02.61093Z","shell.execute_reply.started":"2025-04-02T19:54:48.733718Z","shell.execute_reply":"2025-04-02T19:55:02.610038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 32\nNUM_CLASSES = 5\nEPOCHS = 25\nSEED = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:17.75133Z","iopub.execute_input":"2025-04-02T19:55:17.751653Z","iopub.status.idle":"2025-04-02T19:55:17.755139Z","shell.execute_reply.started":"2025-04-02T19:55:17.751626Z","shell.execute_reply":"2025-04-02T19:55:17.754288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/aptos2019-blindness-detection/'\ndf = pd.read_csv(BASE_PATH + 'train.csv')\ndf['image_path'] = BASE_PATH + 'train_images/' + df['id_code'] + '.png'\ndf['diagnosis'] = df['diagnosis'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:24.756223Z","iopub.execute_input":"2025-04-02T19:55:24.756535Z","iopub.status.idle":"2025-04-02T19:55:24.781222Z","shell.execute_reply.started":"2025-04-02T19:55:24.756509Z","shell.execute_reply":"2025-04-02T19:55:24.780308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------- EDA --------------------\n# Class distribution\nplt.figure(figsize=(6,4))\nsns.countplot(x='diagnosis', data=df)\nplt.title('Class Distribution')\nplt.xlabel('Diagnosis Class')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:26.523932Z","iopub.execute_input":"2025-04-02T19:55:26.52422Z","iopub.status.idle":"2025-04-02T19:55:26.789907Z","shell.execute_reply.started":"2025-04-02T19:55:26.524198Z","shell.execute_reply":"2025-04-02T19:55:26.788883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show sample images per class\nfig, axes = plt.subplots(1, 5, figsize=(20, 5))\nfor i in range(5):\n    sample_path = df[df['diagnosis'] == str(i)].iloc[0]['image_path']\n    img = cv2.imread(sample_path)\n    img = cv2.cvtColor(cv2.resize(img, (IMG_SIZE, IMG_SIZE)), cv2.COLOR_BGR2RGB)\n    axes[i].imshow(img)\n    axes[i].axis('off')\n    axes[i].set_title(f\"Class {i}\")\nplt.suptitle(\"Sample Images from Each Class\", fontsize=16)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:30.61102Z","iopub.execute_input":"2025-04-02T19:55:30.611351Z","iopub.status.idle":"2025-04-02T19:55:31.907277Z","shell.execute_reply.started":"2025-04-02T19:55:30.611324Z","shell.execute_reply":"2025-04-02T19:55:31.906368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Data Splitting\ntrain_df, val_df = train_test_split(df, test_size=0.2, stratify=df['diagnosis'], random_state=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:43.657589Z","iopub.execute_input":"2025-04-02T19:55:43.657956Z","iopub.status.idle":"2025-04-02T19:55:43.667742Z","shell.execute_reply.started":"2025-04-02T19:55:43.657926Z","shell.execute_reply":"2025-04-02T19:55:43.666882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=10,\n    zoom_range=0.1,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='image_path',\n    y_col='diagnosis',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=SEED\n)\n\nval_gen = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    x_col='image_path',\n    y_col='diagnosis',\n    target_size=(IMG_SIZE, IMG_SIZE),\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T19:55:47.20258Z","iopub.execute_input":"2025-04-02T19:55:47.202939Z","iopub.status.idle":"2025-04-02T19:55:51.477236Z","shell.execute_reply.started":"2025-04-02T19:55:47.202911Z","shell.execute_reply":"2025-04-02T19:55:51.476547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = EfficientNetB0(include_top=False, weights='imagenet', input_shape=(IMG_SIZE, IMG_SIZE, 3))\n\nx = GlobalAveragePooling2D()(base_model.output)\nx = tf.keras.layers.BatchNormalization()(x)  # 🔪 Added normalization to stabilize training\nx = Dropout(0.5)(x)\nx = Dense(256, activation='relu')(x)  # 🔪 Increased capacity of classification head\nx = Dropout(0.3)(x)\noutput = Dense(NUM_CLASSES, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=output)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T20:00:15.456066Z","iopub.execute_input":"2025-04-02T20:00:15.456416Z","iopub.status.idle":"2025-04-02T20:00:16.41349Z","shell.execute_reply.started":"2025-04-02T20:00:15.45639Z","shell.execute_reply":"2025-04-02T20:00:16.412855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------- Training\nfrom sklearn.utils import class_weight\n\nclass_weights = class_weight.compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(train_df['diagnosis'].astype(int)),\n    y=train_df['diagnosis'].astype(int)\n)\nclass_weights = dict(enumerate(class_weights))\n\n# 🆕 Live Accuracy Callback\nclass LiveAccuracyCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        acc = logs.get('accuracy')\n        val_acc = logs.get('val_accuracy')\n        print(f\"\\n📊 Epoch {epoch+1}: Accuracy = {acc:.4f}, Val Accuracy = {val_acc:.4f}\")\n\ncallbacks = [\n    EarlyStopping(patience=5, restore_best_weights=True),\n    ReduceLROnPlateau(factor=0.3, patience=1, min_lr=1e-6) , # 🔧 More aggressive LR drop\n    ModelCheckpoint('best_model.keras', save_best_only=True),\n    LiveAccuracyCallback()\n]\n# 🔁 Train full model directly (no two-phase)\nbase_model.trainable = True\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=EPOCHS,\n    callbacks=callbacks,\n    class_weight=class_weights\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T21:08:46.994593Z","iopub.execute_input":"2025-04-02T21:08:46.994942Z","iopub.status.idle":"2025-04-02T21:08:48.472797Z","shell.execute_reply.started":"2025-04-02T21:08:46.994915Z","shell.execute_reply":"2025-04-02T21:08:48.471492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Evaluation\nplt.plot(history.history['accuracy'], label='Train Accuracy')\nplt.plot(history.history['val_accuracy'], label='Val Accuracy')\nplt.legend()\nplt.title('Model Accuracy Over Epochs')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T20:57:47.06883Z","iopub.execute_input":"2025-04-02T20:57:47.069194Z","iopub.status.idle":"2025-04-02T20:57:47.258709Z","shell.execute_reply.started":"2025-04-02T20:57:47.069168Z","shell.execute_reply":"2025-04-02T20:57:47.257846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion matrix\nval_preds = model.predict(val_gen)\ny_true = val_df['diagnosis'].astype(int).values\ny_pred = np.argmax(val_preds, axis=1)\n\nprint(\"\\nClassification Report:\\n\")\nprint(classification_report(y_true, y_pred))\n\nconf_matrix = confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(6,5))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted')\nplt.ylabel('Actual')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-02T20:59:21.319796Z","iopub.execute_input":"2025-04-02T20:59:21.32012Z","iopub.status.idle":"2025-04-02T21:00:41.276294Z","shell.execute_reply.started":"2025-04-02T20:59:21.320097Z","shell.execute_reply":"2025-04-02T21:00:41.275253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------- Grad-CAM Visualization --------------------\n# Select a sample image from validation set\ndef get_img_array(path):\n    img = tf.keras.utils.load_img(path, target_size=(IMG_SIZE, IMG_SIZE))\n    array = tf.keras.utils.img_to_array(img)\n    array = np.expand_dims(array, axis=0)\n    return array / 255.0\n\nsample_path = val_df.iloc[0]['image_path']\nimg_array = get_img_array(sample_path)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-02T19:54:23.337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Grad-CAM implementation\ngrad_model = Model([model.inputs], [model.get_layer('top_conv').output, model.output])\nwith tf.GradientTape() as tape:\n    conv_outputs, predictions = grad_model(img_array)\n    class_idx = tf.argmax(predictions[0])\n    loss = predictions[:, class_idx]\n\ngrads = tape.gradient(loss, conv_outputs)[0]\npooled_grads = tf.reduce_mean(grads, axis=(0, 1))\nconv_outputs = conv_outputs[0]\nheatmap = tf.reduce_sum(tf.multiply(pooled_grads, conv_outputs), axis=-1)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-02T19:54:23.337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalize heatmap\nheatmap = np.maximum(heatmap, 0)\nheatmap /= tf.math.reduce_max(heatmap)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-02T19:54:23.338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Superimpose on image\nimg = cv2.imread(sample_path)\nimg = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\nheatmap = cv2.resize(heatmap.numpy(), (IMG_SIZE, IMG_SIZE))\nheatmap = np.uint8(255 * heatmap)\nheatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n\nsuperimposed_img = cv2.addWeighted(img, 0.6, heatmap, 0.4, 0)\nplt.figure(figsize=(6,6))\nplt.imshow(cv2.cvtColor(superimposed_img, cv2.COLOR_BGR2RGB))\nplt.title(\"Grad-CAM Heatmap\")\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-02T19:54:23.338Z"}},"outputs":[],"execution_count":null}]}