{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":10792662,"sourceType":"datasetVersion","datasetId":6697698},{"sourceId":10799648,"sourceType":"datasetVersion","datasetId":6702828}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam\n\n# --- Paths ---\ntrain_dir = \"/kaggle/input/preprocessed-images/preprocessed_train\"\nval_dir = \"/kaggle/input/preprocessed-images/preprocessed_val\"\ncsv_path = \"/kaggle/input/aptos2019-blindness-detection/train.csv\"\n\n# --- Load and preprocess dataset ---\ndf = pd.read_csv(csv_path)\ndf[\"id_code\"] += \".npy\"\nvalid_ids = set(os.listdir(train_dir)).union(set(os.listdir(val_dir)))\ndf = df[df[\"id_code\"].isin(valid_ids)].copy()\ndf = df[df[\"diagnosis\"] != 0].copy()\ndf[\"diagnosis\"] = df[\"diagnosis\"] - 1\ndf[\"subset\"] = \"training\"\ndf.loc[df.groupby(\"diagnosis\", group_keys=False).apply(lambda x: x.sample(frac=0.2, random_state=42)).index, \"subset\"] = \"validation\"\n\n# --- Class weights ---\nclasses = np.unique(df[\"diagnosis\"])\nweights = compute_class_weight(class_weight=\"balanced\", classes=classes, y=df[\"diagnosis\"])\nclass_weights = {i: w for i, w in enumerate(weights)}\n\n# --- Generator ---\nclass NPYDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, directory, batch_size=32, target_size=(224, 224), shuffle=True):\n        self.df = df.reset_index(drop=True)\n        self.dir = directory\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.df))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return len(self.df) // self.batch_size\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __getitem__(self, index):\n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        batch_df = self.df.iloc[indexes]\n        X, y = [], []\n        for _, row in batch_df.iterrows():\n            path = os.path.join(self.dir, row[\"id_code\"])\n            if not os.path.exists(path): continue\n            try:\n                img = np.load(path)\n                img = tf.image.central_crop(img, 0.85)\n                img = tf.image.resize(img, self.target_size)\n                if img.shape[-1] == 1:\n                    img = tf.image.grayscale_to_rgb(img)\n                img = tf.image.convert_image_dtype(img, tf.float32)\n                img = tf.image.random_flip_left_right(img)\n                img = tf.image.random_brightness(img, 0.3)\n                img = tf.image.random_contrast(img, 0.6, 1.4)\n                X.append(img)\n                y.append(tf.keras.utils.to_categorical(row[\"diagnosis\"], num_classes=4))\n            except:\n                continue\n        if len(X) == 0 or len(y) == 0:\n            return self.__getitem__((index + 1) % self.__len__())\n        return np.array(X), np.array(y)\n\ntrain_gen = NPYDataGenerator(df[df[\"subset\"] == \"training\"], train_dir)\nval_gen = NPYDataGenerator(df[df[\"subset\"] == \"validation\"], val_dir)\n\n# --- Build ResNet50 Model ---\nbase_model = ResNet50(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))\nfor layer in base_model.layers[:100]:\n    layer.trainable = False\nfor layer in base_model.layers[100:]:\n    layer.trainable = True\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\nx = Dense(512, activation='relu')(x)\nx = BatchNormalization()(x)\nx = Dropout(0.5)(x)\noutputs = Dense(4, activation=\"softmax\")(x)\nmodel = Model(inputs=base_model.input, outputs=outputs)\n\n# --- Compile ---\nmodel.compile(\n    optimizer=Adam(learning_rate=1e-4),\n    loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.05),\n    metrics=[\"accuracy\"]\n)\n\n# --- Train ---\ncallbacks = [\n    EarlyStopping(monitor=\"val_accuracy\", patience=10, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.3, patience=3, min_lr=1e-7)\n]\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    class_weight=class_weights,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# --- Save Model ---\nmodel.save(\"resnet50_dr_stage_classifier_4class.h5\")\n\n# --- Max Accuracy ---\nbest_val_acc = max(history.history[\"val_accuracy\"])\nprint(f\"\\n Best Validation Accuracy: {best_val_acc:.4f}\")\n\n# --- Plot Accuracy and Loss ---\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history[\"accuracy\"], label=\"Train Accuracy\", marker='o')\nplt.plot(history.history[\"val_accuracy\"], label=\"Val Accuracy\", marker='o')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.grid(True)\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history[\"loss\"], label=\"Train Loss\", marker='o')\nplt.plot(history.history[\"val_loss\"], label=\"Val Loss\", marker='o')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.grid(True)\nplt.legend()\nplt.tight_layout()\nplt.show()\n\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-31T06:46:00.462531Z","iopub.execute_input":"2025-07-31T06:46:00.462856Z","iopub.status.idle":"2025-07-31T06:51:38.664885Z","shell.execute_reply.started":"2025-07-31T06:46:00.462832Z","shell.execute_reply":"2025-07-31T06:51:38.664079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Evaluation ---\n# Get predictions\ny_true, y_pred = [], []\nfor i in range(len(val_gen)):\n    X_batch, y_batch = val_gen[i]\n    preds = model.predict(X_batch, verbose=0)\n    y_true.extend(np.argmax(y_batch, axis=1))\n    y_pred.extend(np.argmax(preds, axis=1))\n\n# --- Classification Report ---\nprint(\"\\n Classification Report:\\n\")\nprint(classification_report(y_true, y_pred, target_names=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"], zero_division=0))\n\n# --- Confusion Matrix ---\ncm = confusion_matrix(y_true, y_pred, labels=[0, 1, 2, 3])\nprint(\"\\n Confusion Matrix:\\n\", cm)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"], yticklabels=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"])\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T06:53:55.315546Z","iopub.execute_input":"2025-07-31T06:53:55.315863Z","iopub.status.idle":"2025-07-31T06:54:05.605076Z","shell.execute_reply.started":"2025-07-31T06:53:55.315841Z","shell.execute_reply":"2025-07-31T06:54:05.604322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VGG19 Modal","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.applications import VGG19\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, BatchNormalization, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam\n\n# --- Paths ---\ntrain_dir = \"/kaggle/input/preprocessed-images/preprocessed_train\"\nval_dir = \"/kaggle/input/preprocessed-images/preprocessed_val\"\ncsv_path = \"/kaggle/input/aptos2019-blindness-detection/train.csv\"\n\n# --- Load and preprocess dataset ---\ndf = pd.read_csv(csv_path)\ndf[\"id_code\"] += \".npy\"\nvalid_ids = set(os.listdir(train_dir)).union(set(os.listdir(val_dir)))\ndf = df[df[\"id_code\"].isin(valid_ids)].copy()\n\n#  Remove No DR and shift labels to 0–3\ndf = df[df[\"diagnosis\"] != 0].copy()\ndf[\"diagnosis\"] = df[\"diagnosis\"] - 1\n\n# --- Stratified split ---\ndf[\"subset\"] = \"training\"\ndf.loc[df.groupby(\"diagnosis\", group_keys=False).apply(\n    lambda x: x.sample(frac=0.2, random_state=42)).index, \"subset\"] = \"validation\"\n\n# --- Class weights ---\nclasses = np.unique(df[\"diagnosis\"])\nweights = compute_class_weight(class_weight=\"balanced\", classes=classes, y=df[\"diagnosis\"])\nclass_weights = {i: w for i, w in enumerate(weights)}\n\n# --- Generator ---\nclass NPYDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, directory, batch_size=32, target_size=(224, 224), shuffle=True):\n        self.df = df.reset_index(drop=True)\n        self.dir = directory\n        self.batch_size = batch_size\n        self.target_size = target_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.df))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return len(self.df) // self.batch_size\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __getitem__(self, index):\n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        batch_df = self.df.iloc[indexes]\n        X, y = [], []\n        for _, row in batch_df.iterrows():\n            path = os.path.join(self.dir, row[\"id_code\"])\n            if not os.path.exists(path):\n                continue\n            try:\n                img = np.load(path)\n                img = tf.image.central_crop(img, 0.85)\n                img = tf.image.resize(img, self.target_size)\n                if img.shape[-1] == 1:\n                    img = tf.image.grayscale_to_rgb(img)\n                img = tf.image.convert_image_dtype(img, tf.float32)\n                img = tf.image.random_flip_left_right(img)\n                img = tf.image.random_brightness(img, 0.3)\n                img = tf.image.random_contrast(img, 0.6, 1.4)\n                X.append(img)\n                y.append(tf.keras.utils.to_categorical(row[\"diagnosis\"], num_classes=4))\n            except:\n                continue\n\n        if len(X) == 0 or len(y) == 0:\n            return self.__getitem__((index + 1) % self.__len__())\n        return np.array(X), np.array(y)\n\ntrain_gen = NPYDataGenerator(df[df[\"subset\"] == \"training\"], train_dir)\nval_gen = NPYDataGenerator(df[df[\"subset\"] == \"validation\"], val_dir)\n\n# --- Build VGG19 Model ---\nbase_model = VGG19(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))\nfor layer in base_model.layers[:15]:\n    layer.trainable = False\nfor layer in base_model.layers[15:]:\n    layer.trainable = True\n\nx = base_model.output\nx = Flatten()(x)  # VGG19 typically uses Flatten instead of GAP\nx = Dense(512, activation='relu')(x)\nx = BatchNormalization()(x)\nx = Dropout(0.5)(x)\noutputs = Dense(4, activation=\"softmax\")(x)\n\nmodel = Model(inputs=base_model.input, outputs=outputs)\n\n# --- Compile ---\nmodel.compile(\n    optimizer=Adam(learning_rate=1e-4),\n    loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.05),\n    metrics=[\"accuracy\"]\n)\n\n# --- Callbacks ---\ncallbacks = [\n    EarlyStopping(monitor=\"val_accuracy\", patience=10, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.3, patience=3, min_lr=1e-7)\n]\n\n# --- Train ---\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    class_weight=class_weights,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# --- Save Model ---\nmodel.save(\"vgg19_dr_stage_classifier_4class.h5\")\n\n# --- Best Validation Accuracy ---\nbest_val_acc = max(history.history[\"val_accuracy\"])\nprint(f\"\\n Best Validation Accuracy: {best_val_acc:.4f}\")\n\n# --- Plot Accuracy and Loss ---\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(history.history[\"accuracy\"], label=\"Train Accuracy\", marker='o')\nplt.plot(history.history[\"val_accuracy\"], label=\"Val Accuracy\", marker='o')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.grid(True)\nplt.legend()\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history[\"loss\"], label=\"Train Loss\", marker='o')\nplt.plot(history.history[\"val_loss\"], label=\"Val Loss\", marker='o')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.grid(True)\nplt.legend()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T06:57:24.870383Z","iopub.execute_input":"2025-07-31T06:57:24.871096Z","iopub.status.idle":"2025-07-31T07:01:23.078538Z","shell.execute_reply.started":"2025-07-31T06:57:24.871072Z","shell.execute_reply":"2025-07-31T07:01:23.07782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# --- Get predictions ---\ny_true, y_pred = [], []\nfor i in range(len(val_gen)):\n    X_batch, y_batch = val_gen[i]\n    preds = model.predict(X_batch, verbose=0)\n    y_true.extend(np.argmax(y_batch, axis=1))\n    y_pred.extend(np.argmax(preds, axis=1))\n\ny_true = np.array(y_true)\ny_pred = np.array(y_pred)\n\n# --- Classification Report (force all 4 classes) ---\nprint(\"\\n Classification Report:\\n\")\nprint(classification_report(\n    y_true, \n    y_pred, \n    labels=[0,1,2,3],   # Explicitly include all labels\n    target_names=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"],\n    digits=2,\n    zero_division=0\n))\n\n# --- Overall Accuracy ---\naccuracy = accuracy_score(y_true, y_pred)\nprint(f\"\\n Overall Validation Accuracy: {accuracy*100:.2f}%\")\n\n# --- Confusion Matrix ---\ncm = confusion_matrix(y_true, y_pred, labels=[0,1,2,3])\n\nplt.figure(figsize=(6,5))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\",\n            xticklabels=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"],\n            yticklabels=[\"Mild\", \"Moderate\", \"Severe\", \"Proliferative\"])\nplt.title(\"Confusion Matrix (Validation Set)\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T07:01:49.31503Z","iopub.execute_input":"2025-07-31T07:01:49.31556Z","iopub.status.idle":"2025-07-31T07:01:52.077243Z","shell.execute_reply.started":"2025-07-31T07:01:49.315532Z","shell.execute_reply":"2025-07-31T07:01:52.076551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Overall accuracy values\naccuracy_model1 = 0.55   # 55%\naccuracy_model2 = 0.57   # 57%\n\n# # Paths to your trained models\n# model_paths = [\n#     \"/kaggle/working/resnet50_dr_stage_classifier_4class.h5\",   # Replace with actual path\n#     \"/kaggle/working/vgg19_dr_stage_classifier_4class.h5\"\n# ]\n\nmodels = ['RestNet50', 'VGG19']\naccuracy_values = [accuracy_model1, accuracy_model2]\n\nplt.figure(figsize=(6,5))\nplt.bar(models, accuracy_values, color=['blue', 'green'])\nplt.ylim(0,1)\nplt.ylabel(\"Accuracy\")\nplt.title(\"Model Accuracy Comparison\")\n\n# Show percentage on top of bars\nfor i, v in enumerate(accuracy_values):\n    plt.text(i, v + 0.02, f\"{v*100:.2f}%\", ha='center', fontsize=12)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T07:09:08.315074Z","iopub.execute_input":"2025-07-31T07:09:08.315798Z","iopub.status.idle":"2025-07-31T07:09:08.428608Z","shell.execute_reply.started":"2025-07-31T07:09:08.315771Z","shell.execute_reply":"2025-07-31T07:09:08.427851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tensorflow.keras.models import load_model\n\n# Paths to your trained models\nmodel_paths = [\n    \"/kaggle/working/resnet50_dr_stage_classifier_4class.h5\",  \n    \"/kaggle/working/vgg19_dr_stage_classifier_4class.h5\"\n]\n\nmodels = [\"RestNet50\", \"VGG19\"]\naccuracy_values = []\n\n# Load each model and evaluate accuracy on validation/test generator\nfor path in model_paths:\n    model = load_model(path)\n    loss, acc = model.evaluate(val_gen, verbose=0)  # val_gen = your validation data generator\n    accuracy_values.append(acc)\n\n#  Plot\nplt.figure(figsize=(6,5))\nplt.bar(models, accuracy_values, color=['blue', 'green'])\nplt.ylim(0,1)\nplt.ylabel(\"Accuracy\")\nplt.title(\"Model Accuracy Comparison\")\n\n# Show percentage on top of bars\nfor i, v in enumerate(accuracy_values):\n    plt.text(i, v + 0.02, f\"{v*100:.2f}%\", ha='center', fontsize=12)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-31T07:07:23.050896Z","iopub.execute_input":"2025-07-31T07:07:23.05162Z","iopub.status.idle":"2025-07-31T07:07:34.197784Z","shell.execute_reply.started":"2025-07-31T07:07:23.051596Z","shell.execute_reply":"2025-07-31T07:07:34.197061Z"}},"outputs":[],"execution_count":null}]}