{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import DenseNet121\nfrom tensorflow.keras.applications.densenet import preprocess_input\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\nprint(f\"TensorFlow Version: {tf.__version__}\")\n\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    print(f\"GPU(s) available: {len(gpus)}\")\nelse:\n    print(\"No GPU detected.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:05.613733Z","iopub.execute_input":"2026-03-07T15:57:05.614456Z","iopub.status.idle":"2026-03-07T15:57:05.620504Z","shell.execute_reply.started":"2026-03-07T15:57:05.614426Z","shell.execute_reply":"2026-03-07T15:57:05.619696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_CSV = '/kaggle/input/competitions/aptos2019-blindness-detection/train.csv'\nTRAIN_IMG_DIR = '/kaggle/input/competitions/aptos2019-blindness-detection/train_images'\n\nIMG_SIZE = (256,256)\nBATCH_SIZE = 16\nSEED = 42\nNUM_CLASSES = 5\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:21.395969Z","iopub.execute_input":"2026-03-07T15:57:21.396316Z","iopub.status.idle":"2026-03-07T15:57:21.401055Z","shell.execute_reply.started":"2026-03-07T15:57:21.396288Z","shell.execute_reply":"2026-03-07T15:57:21.400274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\n\ndf['id_code'] = df['id_code'].astype(str) + '.png'\ndf['diagnosis'] = df['diagnosis'].astype(int)\n\nprint(\"Total Images:\",len(df))\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:24.754193Z","iopub.execute_input":"2026-03-07T15:57:24.754545Z","iopub.status.idle":"2026-03-07T15:57:24.772808Z","shell.execute_reply.started":"2026-03-07T15:57:24.754515Z","shell.execute_reply":"2026-03-07T15:57:24.772022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class mapping\nclass_mapping = {\n    0: 'No DR',\n    1: 'Mild',\n    2: 'Moderate',\n    3: 'Severe',\n    4: 'Proliferative DR'\n}\n\n# Count images per class\nclass_counts = df['diagnosis'].value_counts().sort_index()\n\n# Print counts\nfor cls, count in class_counts.items():\n    print(f\"{class_mapping[cls]} (Class {cls}) : {count}\")\n\n# Bar plot\nplt.figure(figsize=(7,5))\nsns.barplot(x=class_counts.index, y=class_counts.values, palette='viridis')\n\nplt.title(\"Class Distribution\")\nplt.xlabel(\"Diagnosis Class\")\nplt.ylabel(\"Number of Images\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:28.834832Z","iopub.execute_input":"2026-03-07T15:57:28.835552Z","iopub.status.idle":"2026-03-07T15:57:29.005446Z","shell.execute_reply.started":"2026-03-07T15:57:28.835506Z","shell.execute_reply":"2026-03-07T15:57:29.004650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(5,5))\n\nplt.pie(class_counts.values,\n        labels=class_counts.index,\n        autopct='%1.1f%%')\n\nplt.title(\"Class Distribution\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:33.314360Z","iopub.execute_input":"2026-03-07T15:57:33.314949Z","iopub.status.idle":"2026-03-07T15:57:33.381254Z","shell.execute_reply.started":"2026-03-07T15:57:33.314919Z","shell.execute_reply":"2026-03-07T15:57:33.380564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df['diagnosis'],\n    random_state=SEED\n)\n\nprint(\"Train Images:\",len(train_df))\nprint(\"Validation Images:\",len(val_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:37.235240Z","iopub.execute_input":"2026-03-07T15:57:37.236140Z","iopub.status.idle":"2026-03-07T15:57:37.245690Z","shell.execute_reply.started":"2026-03-07T15:57:37.236109Z","shell.execute_reply":"2026-03-07T15:57:37.244880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Ben Graham Preprocessing\n# def ben_graham_processing(img):\n\n#     img = img.astype(np.uint8)\n\n#     img = cv2.resize(img, IMG_SIZE)\n\n#     blur = cv2.GaussianBlur(img,(0,0),sigmaX=10)\n\n#     img = cv2.addWeighted(img,4,blur,-4,128)\n\n#     img = preprocess_input(img.astype(np.float32))\n\n#     return img","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Ben Graham Preprocessing+cropping\n\ndef crop_image_from_gray(img, tol=7):\n\n    if img.ndim == 2:\n        mask = img > tol\n        return img[np.ix_(mask.any(1), mask.any(0))]\n\n    elif img.ndim == 3:\n        gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray > tol\n\n        if img[:, :, 0][np.ix_(mask.any(1), mask.any(0))].size == 0:\n            return img\n\n        img1 = img[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n        img2 = img[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n        img3 = img[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n\n        img = np.stack([img1, img2, img3], axis=-1)\n\n    return img\n\n\ndef ben_graham_processing(img):\n\n    img = img.astype(np.uint8)\n\n    # remove dark borders\n    img = crop_image_from_gray(img)\n\n    # resize\n    img = cv2.resize(img, IMG_SIZE)\n\n    # gaussian blur\n    blur = cv2.GaussianBlur(img, (0,0), sigmaX=10)\n\n    # ben graham contrast enhancement\n    img = cv2.addWeighted(img, 4, blur, -4, 128)\n\n    # DenseNet preprocessing\n    img = preprocess_input(img.astype(np.float32))\n\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:46.725808Z","iopub.execute_input":"2026-03-07T15:57:46.726112Z","iopub.status.idle":"2026-03-07T15:57:46.735352Z","shell.execute_reply.started":"2026-03-07T15:57:46.726086Z","shell.execute_reply":"2026-03-07T15:57:46.734617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show one image per class (Before vs After preprocessing)\n\nclass_mapping = {\n    0: 'No DR',\n    1: 'Mild',\n    2: 'Moderate',\n    3: 'Severe',\n    4: 'Proliferative DR'\n}\n\nplt.figure(figsize=(15,6))\n\nfor i in range(5):\n\n    # get one image for each class\n    sample = df[df['diagnosis'] == i].iloc[0]['id_code']\n    \n    img_path = os.path.join(TRAIN_IMG_DIR, sample)\n\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    processed = ben_graham_processing(img)\n\n    # normalize for display\n    processed_display = (processed - processed.min())/(processed.max()-processed.min())\n\n    # original image\n    plt.subplot(2,5,i+1)\n    plt.imshow(img)\n    plt.title(class_mapping[i])\n    plt.axis(\"off\")\n\n    # processed image\n    plt.subplot(2,5,i+6)\n    plt.imshow(processed_display)\n    plt.title(\"Processed\")\n    plt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:57:57.029340Z","iopub.execute_input":"2026-03-07T15:57:57.030154Z","iopub.status.idle":"2026-03-07T15:58:00.157999Z","shell.execute_reply.started":"2026-03-07T15:57:57.030116Z","shell.execute_reply":"2026-03-07T15:58:00.157126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['diagnosis'] = train_df['diagnosis'].astype(str)\nval_df['diagnosis'] = val_df['diagnosis'].astype(str)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:58:06.890509Z","iopub.execute_input":"2026-03-07T15:58:06.890837Z","iopub.status.idle":"2026-03-07T15:58:06.896666Z","shell.execute_reply.started":"2026-03-07T15:58:06.890810Z","shell.execute_reply":"2026-03-07T15:58:06.895807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#data generators\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=ben_graham_processing,\n    rotation_range=25,\n    zoom_range=0.2,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    horizontal_flip=True\n)\n\nval_datagen = ImageDataGenerator(\n    preprocessing_function=ben_graham_processing\n)\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    train_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=True\n)\n\nval_gen = val_datagen.flow_from_dataframe(\n    val_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:58:09.677678Z","iopub.execute_input":"2026-03-07T15:58:09.677990Z","iopub.status.idle":"2026-03-07T15:58:11.263688Z","shell.execute_reply.started":"2026-03-07T15:58:09.677967Z","shell.execute_reply":"2026-03-07T15:58:11.262906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## : Calculate Class Weights\nclasses = np.unique(train_df['diagnosis'])\n\nweights = compute_class_weight(\n    class_weight='balanced',\n    classes=classes,\n    y=train_df['diagnosis']\n)\n\nclass_weight_dict = dict(zip(classes,weights))\n\nprint(class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:58:15.712085Z","iopub.execute_input":"2026-03-07T15:58:15.712777Z","iopub.status.idle":"2026-03-07T15:58:15.720483Z","shell.execute_reply.started":"2026-03-07T15:58:15.712745Z","shell.execute_reply":"2026-03-07T15:58:15.719675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# tf.keras.backend.clear_session()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base = DenseNet121(\n    include_top=False,\n    weights='imagenet',\n    input_shape=IMG_SIZE+(3,)\n)\n\nfor layer in base.layers[:150]:\n    layer.trainable=False\n\nfor layer in base.layers[150:]:\n    layer.trainable=True\n\n\ninputs = layers.Input(shape=IMG_SIZE+(3,))\n\nx = base(inputs)\n\nx = layers.GlobalAveragePooling2D()(x)\n\nx = layers.BatchNormalization()(x)\n\nx = layers.Dense(1024,activation='relu')(x)\n\nx = layers.Dropout(0.6)(x)\n\noutputs = layers.Dense(NUM_CLASSES,activation='softmax')(x)\n\nmodel = Model(inputs,outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:58:19.132206Z","iopub.execute_input":"2026-03-07T15:58:19.132573Z","iopub.status.idle":"2026-03-07T15:58:22.420675Z","shell.execute_reply.started":"2026-03-07T15:58:19.132544Z","shell.execute_reply":"2026-03-07T15:58:22.420082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    '/kaggle/working/dbestfile.keras',\n    monitor='val_accuracy',\n    save_best_only=True,\n    mode='max',\n    verbose=1\n)\n\nearly_stop = EarlyStopping(\n    monitor='val_accuracy',\n    patience=6,\n    mode='max',\n    restore_best_weights=True,\n    verbose=1\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.3,\n    patience=3,\n    min_lr=1e-6\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T15:58:29.952791Z","iopub.execute_input":"2026-03-07T15:58:29.953448Z","iopub.status.idle":"2026-03-07T15:58:29.957810Z","shell.execute_reply.started":"2026-03-07T15:58:29.953419Z","shell.execute_reply":"2026-03-07T15:58:29.957077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    class_weight=class_weight_dict,\n    callbacks=[checkpoint,early_stop,reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T16:03:01.640982Z","iopub.execute_input":"2026-03-07T16:03:01.641316Z","iopub.status.idle":"2026-03-07T19:25:41.328548Z","shell.execute_reply.started":"2026-03-07T16:03:01.641287Z","shell.execute_reply":"2026-03-07T19:25:41.327666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/dbestmodel.keras')\nmodel.save('/kaggle/working/dbestmodel.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T19:30:12.973625Z","iopub.execute_input":"2026-03-07T19:30:12.974315Z","iopub.status.idle":"2026-03-07T19:30:15.352715Z","shell.execute_reply.started":"2026-03-07T19:30:12.974285Z","shell.execute_reply":"2026-03-07T19:30:15.352056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/working'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T19:30:38.593515Z","iopub.execute_input":"2026-03-07T19:30:38.594052Z","iopub.status.idle":"2026-03-07T19:30:38.598374Z","shell.execute_reply.started":"2026-03-07T19:30:38.594021Z","shell.execute_reply":"2026-03-07T19:30:38.597715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Plot Training History\nplt.figure(figsize=(14,5))\n\n# Accuracy plot\nplt.subplot(1,2,1)\nplt.plot(epochs, history.history['accuracy'])\nplt.plot(epochs, history.history['val_accuracy'])\nplt.title(\"Accuracy\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.legend(['Training', 'Validation'])\n\n# Loss plot\nplt.subplot(1,2,2)\nplt.plot(epochs, history.history['loss'])\nplt.plot(epochs, history.history['val_loss'])\nplt.title(\"Loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.legend(['Training', 'Validation'])\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T19:40:28.122091Z","iopub.execute_input":"2026-03-07T19:40:28.122841Z","iopub.status.idle":"2026-03-07T19:40:28.382308Z","shell.execute_reply.started":"2026-03-07T19:40:28.122811Z","shell.execute_reply":"2026-03-07T19:40:28.381600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Evaluate the Model\nmodel.load_weights('/kaggle/working/dbestmodel.keras')\n\n## classification report\npreds=model.predict(val_gen)\npred_classes=np.argmax(preds,axis=1)\ntrue_classes=val_gen.classes\nprint(classification_report(true_classes,pred_classes))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T19:44:08.044644Z","iopub.execute_input":"2026-03-07T19:44:08.045332Z","iopub.status.idle":"2026-03-07T19:45:28.805671Z","shell.execute_reply.started":"2026-03-07T19:44:08.045303Z","shell.execute_reply":"2026-03-07T19:45:28.804940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display Confusion Matrix\ncm = confusion_matrix(true_classes, pred_classes)\nplt.figure(figsize=(7, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=classes, yticklabels=classes)\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-07T19:46:23.123145Z","iopub.execute_input":"2026-03-07T19:46:23.123959Z","iopub.status.idle":"2026-03-07T19:46:23.314132Z","shell.execute_reply.started":"2026-03-07T19:46:23.123930Z","shell.execute_reply":"2026-03-07T19:46:23.313331Z"}},"outputs":[],"execution_count":null}]}