{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\nimport seaborn as sns\nfrom sklearn.utils.class_weight import compute_class_weight  # 🗡️ Required for class_weights\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-14T08:59:50.528035Z","iopub.execute_input":"2025-04-14T08:59:50.528603Z","iopub.status.idle":"2025-04-14T09:00:04.46261Z","shell.execute_reply.started":"2025-04-14T08:59:50.528578Z","shell.execute_reply":"2025-04-14T09:00:04.461823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 300\nBATCH_SIZE = 32\nNUM_CLASSES = 5\nEPOCHS = 25\nSEED = 42","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:04.463449Z","iopub.execute_input":"2025-04-14T09:00:04.464435Z","iopub.status.idle":"2025-04-14T09:00:04.468098Z","shell.execute_reply.started":"2025-04-14T09:00:04.464409Z","shell.execute_reply":"2025-04-14T09:00:04.467387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/aptos2019-blindness-detection\"\n\ndf = pd.read_csv(f\"{BASE_PATH}/train.csv\")\ndf['diagnosis'] = df['diagnosis'].astype(str)\ndf['id_code'] = df['id_code'] + '.png'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:04.468817Z","iopub.execute_input":"2025-04-14T09:00:04.469012Z","iopub.status.idle":"2025-04-14T09:00:04.537403Z","shell.execute_reply.started":"2025-04-14T09:00:04.468989Z","shell.execute_reply":"2025-04-14T09:00:04.536687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Diagnosis Distribution\nsns.countplot(x='diagnosis', data=df)\nplt.title('Diagnosis Distribution')\nplt.xlabel('Diagnosis')\nplt.ylabel('Count')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:04.538233Z","iopub.execute_input":"2025-04-14T09:00:04.538511Z","iopub.status.idle":"2025-04-14T09:00:04.758813Z","shell.execute_reply.started":"2025-04-14T09:00:04.538463Z","shell.execute_reply":"2025-04-14T09:00:04.758142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🗡️ Data Augmentation Setup\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    zoom_range=0.15,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    validation_split=0.2\n)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=os.path.join(BASE_PATH, \"train_images\"),\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    subset=\"training\",\n    seed=42\n)\n\nval_generator = train_datagen.flow_from_dataframe(\n    dataframe=df,\n    directory=os.path.join(BASE_PATH, \"train_images\"),\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    subset=\"validation\",\n    seed=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:04.759585Z","iopub.execute_input":"2025-04-14T09:00:04.759849Z","iopub.status.idle":"2025-04-14T09:00:09.156357Z","shell.execute_reply.started":"2025-04-14T09:00:04.759826Z","shell.execute_reply":"2025-04-14T09:00:09.155846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize Raw Images\nfor label in df['diagnosis'].unique():\n    sample = df[df['diagnosis'] == label].iloc[0]\n    img_path = os.path.join(BASE_PATH, \"train_images\", sample['id_code'])\n    if os.path.exists(img_path):\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        plt.figure()\n        plt.imshow(img)\n        plt.title(f\"Raw Image - Diagnosis: {label}\")\n        plt.axis('off')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:09.158216Z","iopub.execute_input":"2025-04-14T09:00:09.158403Z","iopub.status.idle":"2025-04-14T09:00:13.05578Z","shell.execute_reply.started":"2025-04-14T09:00:09.158389Z","shell.execute_reply":"2025-04-14T09:00:13.055064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # ✂️ Train/Validation Split\n# X_train, X_val, y_train, y_val = train_test_split(X, y, stratify=labels, test_size=0.2, random_state=42)\n\n# print(\"X_train shape:\", X_train.shape)\n# print(\"X_val shape:\", X_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.062388Z","iopub.execute_input":"2025-04-14T09:00:13.063251Z","iopub.status.idle":"2025-04-14T09:00:13.215318Z","shell.execute_reply.started":"2025-04-14T09:00:13.063225Z","shell.execute_reply":"2025-04-14T09:00:13.21465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🗡️ Check Label Distribution in Train/Val Sets\ntrain_labels = train_generator.classes\nval_labels = val_generator.classes\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nsns.countplot(x=train_labels)\nplt.title('Training Set Label Distribution')\n\nplt.subplot(1, 2, 2)\nsns.countplot(x=val_labels)\nplt.title('Validation Set Label Distribution')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.216187Z","iopub.execute_input":"2025-04-14T09:00:13.216524Z","iopub.status.idle":"2025-04-14T09:00:13.528189Z","shell.execute_reply.started":"2025-04-14T09:00:13.216499Z","shell.execute_reply":"2025-04-14T09:00:13.527382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🔥 Focal Loss Function\ndef focal_loss(gamma=2., alpha=0.25):\n    def focal_loss_fixed(y_true, y_pred):\n        epsilon = tf.keras.backend.epsilon()\n        y_pred = tf.clip_by_value(y_pred, epsilon, 1. - epsilon)\n        cross_entropy = -y_true * tf.math.log(y_pred)\n        weight = alpha * tf.math.pow(1 - y_pred, gamma)\n        loss = weight * cross_entropy\n        return tf.reduce_mean(tf.reduce_sum(loss, axis=1))\n    return focal_loss_fixed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.528849Z","iopub.execute_input":"2025-04-14T09:00:13.529094Z","iopub.status.idle":"2025-04-14T09:00:13.53425Z","shell.execute_reply.started":"2025-04-14T09:00:13.529076Z","shell.execute_reply":"2025-04-14T09:00:13.533357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧠 Custom CNN Model (Your Proposed Model)\ndef build_custom_cnn(input_shape=(300, 300, 3), num_classes=5):\n    inputs = layers.Input(shape=input_shape)\n    x = layers.Conv2D(32, (3,3), activation='swish', padding='same')(inputs)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D()(x)\n\n    x = layers.Conv2D(64, (3,3), activation='swish', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D()(x)\n\n    x = layers.Conv2D(128, (3,3), activation='swish', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.MaxPooling2D()(x)\n\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.Dense(64, activation='swish')(x)\n    x = layers.Dropout(0.25)(x)\n    outputs = layers.Dense(num_classes, activation='softmax')(x)\n\n    model = models.Model(inputs, outputs)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.535075Z","iopub.execute_input":"2025-04-14T09:00:13.535372Z","iopub.status.idle":"2025-04-14T09:00:13.553171Z","shell.execute_reply.started":"2025-04-14T09:00:13.535351Z","shell.execute_reply":"2025-04-14T09:00:13.552437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🗡️ Compute Class Weights FROM train_generator\ny_train_labels = train_generator.classes  # 🗡️ Provided by ImageDataGenerator\nclass_weights = compute_class_weight(class_weight='balanced', classes=np.unique(y_train_labels), y=y_train_labels)\nclass_weights = dict(enumerate(class_weights))\nprint(\"Class Weights:\", class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.568745Z","iopub.execute_input":"2025-04-14T09:00:13.568938Z","iopub.status.idle":"2025-04-14T09:00:13.589901Z","shell.execute_reply.started":"2025-04-14T09:00:13.568914Z","shell.execute_reply":"2025-04-14T09:00:13.589036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🗡️ Visualize Augmented Training Images (Optional)\naugmented_imgs, labels = next(train_generator)\n\nplt.figure(figsize=(12,6))\nfor i in range(6):\n    plt.subplot(2, 3, i+1)\n    plt.imshow(augmented_imgs[i])\n    plt.title(f\"Label: {np.argmax(labels[i])}\")\n    plt.axis(\"off\")\nplt.suptitle(\"Augmented Samples from train_generator\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T09:00:13.590953Z","iopub.execute_input":"2025-04-14T09:00:13.591261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models, Input\n\n# ✅ Choose and Compile Model\n# model = build_pretrained_efficientnet()  # Optional baseline\nmodel_custom = build_custom_cnn()\nmodel_custom.compile(optimizer='adam', loss=focal_loss(), metrics=['accuracy'])\n\nhistory_custom = model_custom.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=20,\n    callbacks=[\n        tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True),\n        tf.keras.callbacks.ReduceLROnPlateau(patience=3, factor=0.3)\n    ],\n    class_weight=class_weights\n)\n","metadata":{"trusted":true,"execution":{"iopub.execute_input":"2025-04-14T09:00:19.041131Z","iopub.status.idle":"2025-04-14T11:14:34.282684Z","shell.execute_reply.started":"2025-04-14T09:00:19.04111Z","shell.execute_reply":"2025-04-14T11:14:34.28192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score\n\n# 🧾 Evaluate Custom CNN\ny_true = val_generator.classes\ny_pred_custom_probs = model_custom.predict(val_generator)\ny_pred_custom = np.argmax(y_pred_custom_probs, axis=1)\nkappa_custom = cohen_kappa_score(y_true, y_pred_custom, weights='quadratic')\nprint(\"Custom CNN Kappa:\", kappa_custom)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:14:34.283672Z","iopub.execute_input":"2025-04-14T11:14:34.283875Z","iopub.status.idle":"2025-04-14T11:16:06.774455Z","shell.execute_reply.started":"2025-04-14T11:14:34.283857Z","shell.execute_reply":"2025-04-14T11:16:06.773867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Save model\nmodel_custom.save(\"/kaggle/working/custom_cnn_model.h5\")\nprint(\"✅ Custom CNN model saved.\")\n# ✅ Save results (accuracy + kappa)\nresults_custom = {\n    \"model\": \"Custom CNN\",\n    \"accuracy\": history_custom.history['val_accuracy'][-1],\n    \"kappa\": cohen_kappa_score(val_generator.classes,\n                                np.argmax(model_custom.predict(val_generator), axis=1),\n                                weights='quadratic')\n}\nnp.save(\"/kaggle/working/custom_results.npy\", results_custom)\nprint(\"✅ Custom CNN results saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:16:06.77536Z","iopub.execute_input":"2025-04-14T11:16:06.775966Z","iopub.status.idle":"2025-04-14T11:17:38.958852Z","shell.execute_reply.started":"2025-04-14T11:16:06.77594Z","shell.execute_reply":"2025-04-14T11:17:38.958236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Visualize a Few Predictions on Validation Set\nval_generator.reset()  # Reset generator to start from beginning\nimages, labels = next(val_generator)  # Get one batch of validation data\n\n# Predict\npred_probs = model_custom.predict(images)\npred_classes = np.argmax(pred_probs, axis=1)\ntrue_classes = np.argmax(labels, axis=1)\n\n# Show first 5 predictions\nimport matplotlib.pyplot as plt\n\nfor i in range(5):\n    plt.imshow(images[i])\n    plt.title(f\"Predicted: {pred_classes[i]} | Actual: {true_classes[i]}\")\n    plt.axis(\"off\")\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:38.959562Z","iopub.execute_input":"2025-04-14T11:17:38.959785Z","iopub.status.idle":"2025-04-14T11:17:43.944623Z","shell.execute_reply.started":"2025-04-14T11:17:38.959768Z","shell.execute_reply":"2025-04-14T11:17:43.943871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 🔁 Evaluate EfficientNet\n# y_pred_eff = model_eff.predict(X_val)\n# y_pred_eff_class = np.argmax(y_pred_eff, axis=1)\n# kappa_eff = cohen_kappa_score(y_true, y_pred_eff_class, weights='quadratic')\n# print(\"EfficientNetB3 Kappa:\", kappa_eff)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:43.94531Z","iopub.execute_input":"2025-04-14T11:17:43.945527Z","iopub.status.idle":"2025-04-14T11:17:43.949028Z","shell.execute_reply.started":"2025-04-14T11:17:43.94551Z","shell.execute_reply":"2025-04-14T11:17:43.948259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 🔁 Compare Both Models\n# acc_custom = history_custom.history['val_accuracy'][-1]\n# acc_eff = history_eff.history['val_accuracy'][-1]\n\n# import pandas as pd\n\n# results = pd.DataFrame({\n#     \"Model\": [\"Custom CNN\", \"EfficientNetB3\"],\n#     \"Accuracy\": [acc_custom, acc_eff],\n#     \"Kappa\": [kappa_custom, kappa_eff]\n# })\n# display(results)\n\n# # 📊 Bar plot\n# results.set_index(\"Model\").plot(kind='bar', title=\"Model Comparison (Validation)\", ylim=(0, 1), figsize=(8,5))\n# plt.ylabel(\"Score\")\n# plt.grid(True)\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:43.949765Z","iopub.execute_input":"2025-04-14T11:17:43.949995Z","iopub.status.idle":"2025-04-14T11:17:43.964994Z","shell.execute_reply.started":"2025-04-14T11:17:43.949971Z","shell.execute_reply":"2025-04-14T11:17:43.964379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 🔁 (Optional) Fine-Tune EfficientNetB3 for Better Results\n# model_eff.layers[1].trainable = True\n# for layer in model_eff.layers[1].layers[:-20]:\n#     layer.trainable = False\n\n# model_eff.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss=focal_loss(), metrics=['accuracy'])\n\n# fine_tune_history = model_eff.fit(X_train, y_train,\n#                                   validation_data=(X_val, y_val),\n#                                   epochs=10,\n#                                   batch_size=32,\n#                                   callbacks=[\n#                                       tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True),\n#                                       tf.keras.callbacks.ReduceLROnPlateau(patience=3, factor=0.3)\n#                                   ],\n#                                   class_weight=class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:43.967917Z","iopub.execute_input":"2025-04-14T11:17:43.968158Z","iopub.status.idle":"2025-04-14T11:17:43.984064Z","shell.execute_reply.started":"2025-04-14T11:17:43.968142Z","shell.execute_reply":"2025-04-14T11:17:43.983346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🧾 Evaluate using Quadratic Weighted Kappa\n# y_pred_probs = model.predict(X_val)\n# y_true = np.argmax(y_val, axis=1)\n# y_pred = np.argmax(y_pred_probs, axis=1)\n\n# kappa = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n# print(\"Quadratic Weighted Kappa:\", kappa)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:43.98481Z","iopub.execute_input":"2025-04-14T11:17:43.985003Z","iopub.status.idle":"2025-04-14T11:17:43.998035Z","shell.execute_reply.started":"2025-04-14T11:17:43.984989Z","shell.execute_reply":"2025-04-14T11:17:43.997524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Predict on validation set\ny_pred_custom_probs = model_custom.predict(val_generator)\ny_pred_custom = np.argmax(y_pred_custom_probs, axis=1)\ny_true = val_generator.classes\n\n# ✅ See class prediction counts\nprint(\"Predicted class distribution:\", np.bincount(y_pred_custom))\nprint(\"True class distribution:\", np.bincount(y_true))\n\n# ✅ Kappa score (quality of predictions)\nfrom sklearn.metrics import cohen_kappa_score\nkappa_custom = cohen_kappa_score(y_true, y_pred_custom, weights='quadratic')\nprint(\"Custom CNN Kappa:\", kappa_custom)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T11:17:43.998724Z","iopub.execute_input":"2025-04-14T11:17:43.998969Z","iopub.status.idle":"2025-04-14T11:19:15.627361Z","shell.execute_reply.started":"2025-04-14T11:17:43.998948Z","shell.execute_reply":"2025-04-14T11:19:15.626775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}