{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ## Cell 1: Imports and Setup\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix, cohen_kappa_score\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\nprint(f\"TensorFlow Version: {tf.__version__}\")\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    print(f\"GPU(s) available: {len(gpus)}\")\nelse:\n    print(\"No GPU detected.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:20.513123Z","iopub.execute_input":"2026-02-23T09:45:20.514170Z","iopub.status.idle":"2026-02-23T09:45:20.520714Z","shell.execute_reply.started":"2026-02-23T09:45:20.514139Z","shell.execute_reply":"2026-02-23T09:45:20.519789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 2: Configuration\n\nTRAIN_CSV = '/kaggle/input/competitions/aptos2019-blindness-detection/train.csv'\nTRAIN_IMG_DIR = '/kaggle/input/competitions/aptos2019-blindness-detection/train_images'\n\nIMG_SIZE = (224, 224)\nBATCH_SIZE = 16\nSEED = 42\nNUM_CLASSES = 5\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:24.571422Z","iopub.execute_input":"2026-02-23T09:45:24.571771Z","iopub.status.idle":"2026-02-23T09:45:24.576643Z","shell.execute_reply.started":"2026-02-23T09:45:24.571734Z","shell.execute_reply":"2026-02-23T09:45:24.575783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 3: Load and Prepare Data\n\ndf = pd.read_csv(TRAIN_CSV)\ndf['id_code'] = df['id_code'].astype(str) + '.png'\ndf['diagnosis'] = df['diagnosis'].astype(str)\n\nprint(\"Data loaded successfully.\")\nprint(f\"Total samples: {len(df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:28.487042Z","iopub.execute_input":"2026-02-23T09:45:28.487772Z","iopub.status.idle":"2026-02-23T09:45:28.512984Z","shell.execute_reply.started":"2026-02-23T09:45:28.487731Z","shell.execute_reply":"2026-02-23T09:45:28.512284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 4: Class Distribution Analysis (With Class Names)\n\n# Mapping numeric labels to class names\nclass_mapping = {\n    '0': 'No DR',\n    '1': 'Mild',\n    '2': 'Moderate',\n    '3': 'Severe',\n    '4': 'Proliferative DR'\n}\n\nclass_counts = df['diagnosis'].value_counts().sort_index()\n\nprint(\"📊 Image Count Per Class:\\n\")\nfor cls, count in class_counts.items():\n    print(f\"{class_mapping[cls]} (Class {cls}): {count} images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:32.131518Z","iopub.execute_input":"2026-02-23T09:45:32.132236Z","iopub.status.idle":"2026-02-23T09:45:32.149672Z","shell.execute_reply.started":"2026-02-23T09:45:32.132205Z","shell.execute_reply":"2026-02-23T09:45:32.148804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ----- BAR PLOT -----\nplt.figure(figsize=(10, 5))\nsns.barplot(x=class_counts.index, y=class_counts.values, palette='viridis')\nplt.title(\"Class Distribution (Bar Chart)\")\nplt.xlabel(\"Diagnosis Level\")\nplt.ylabel(\"Number of Images\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:37.221990Z","iopub.execute_input":"2026-02-23T09:45:37.222349Z","iopub.status.idle":"2026-02-23T09:45:37.476757Z","shell.execute_reply.started":"2026-02-23T09:45:37.222309Z","shell.execute_reply":"2026-02-23T09:45:37.475951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ----- PIE CHART -----\nplt.figure(figsize=(5, 5))\nplt.pie(\n    class_counts.values,\n    labels=class_counts.index,\n    autopct='%1.1f%%',\n    startangle=140\n)\nplt.title(\"Class Imbalance (Pie Chart)\")\nplt.axis('equal')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:42.586694Z","iopub.execute_input":"2026-02-23T09:45:42.587030Z","iopub.status.idle":"2026-02-23T09:45:42.676666Z","shell.execute_reply.started":"2026-02-23T09:45:42.587007Z","shell.execute_reply":"2026-02-23T09:45:42.676019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 5: Train/Validation Split\n\ntrain_df, val_df = train_test_split(\n    df,\n    test_size=0.20,\n    random_state=SEED,\n    stratify=df['diagnosis']\n)\n\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Validation samples: {len(val_df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:45:59.977056Z","iopub.execute_input":"2026-02-23T09:45:59.977326Z","iopub.status.idle":"2026-02-23T09:45:59.989777Z","shell.execute_reply.started":"2026-02-23T09:45:59.977303Z","shell.execute_reply":"2026-02-23T09:45:59.989039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 6:  Gaussian Preprocessing\n\ndef apply_kaggle_gaussian(img):\n    img = cv2.resize(img, IMG_SIZE)\n    blurred = cv2.GaussianBlur(img, (0, 0), sigmaX=10)\n    img = cv2.addWeighted(img, 4, blurred, -4, 128)\n    return img\n\ndef kaggle_preprocessing(img):\n    img = img.astype(np.uint8)\n    img = apply_kaggle_gaussian(img)\n    img = preprocess_input(img.astype(np.float32))\n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:46:05.646198Z","iopub.execute_input":"2026-02-23T09:46:05.646971Z","iopub.status.idle":"2026-02-23T09:46:05.651948Z","shell.execute_reply.started":"2026-02-23T09:46:05.646939Z","shell.execute_reply":"2026-02-23T09:46:05.650905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 7: Data Generators\n\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=kaggle_preprocessing,\n    rotation_range=15,\n    horizontal_flip=True,\n    zoom_range=0.1,\n    width_shift_range=0.1,\n    height_shift_range=0.1\n)\n\nval_datagen = ImageDataGenerator(\n    preprocessing_function=kaggle_preprocessing\n)\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=True,\n    seed=SEED\n)\n\nval_gen = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:46:09.837501Z","iopub.execute_input":"2026-02-23T09:46:09.838231Z","iopub.status.idle":"2026-02-23T09:46:19.710337Z","shell.execute_reply.started":"2026-02-23T09:46:09.838205Z","shell.execute_reply":"2026-02-23T09:46:19.709675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 8: Visualize One Image Per Class After Preprocessing\n\nunique_classes = sorted(df['diagnosis'].unique())\n\nplt.figure(figsize=(15, 8))\n\nfor i, cls in enumerate(unique_classes):\n    sample_path = train_df[train_df['diagnosis'] == cls]['id_code'].iloc[0]\n    img_path = os.path.join(TRAIN_IMG_DIR, sample_path)\n\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = kaggle_preprocessing(img)\n\n    img_display = (img - img.min()) / (img.max() - img.min())\n\n    plt.subplot(1, 5, i+1)\n    plt.imshow(img_display)\n    plt.title(f\"Class {cls}\")\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:46:25.576330Z","iopub.execute_input":"2026-02-23T09:46:25.577072Z","iopub.status.idle":"2026-02-23T09:46:27.039076Z","shell.execute_reply.started":"2026-02-23T09:46:25.577045Z","shell.execute_reply":"2026-02-23T09:46:27.038306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 9: Calculate Class Weights\n# Correctly compute class weights on the integer labels to handle imbalance\nclasses = np.unique(train_df['diagnosis'])\nclass_weights = compute_class_weight(\n    'balanced',\n    classes=classes,\n    y=train_df['diagnosis'].values\n)\nclass_weight_dict = {c: w for c, w in zip(classes, class_weights)}\n\nprint(\"Class weights calculated:\")\nprint(class_weight_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:46:32.641250Z","iopub.execute_input":"2026-02-23T09:46:32.641540Z","iopub.status.idle":"2026-02-23T09:46:32.649825Z","shell.execute_reply.started":"2026-02-23T09:46:32.641516Z","shell.execute_reply":"2026-02-23T09:46:32.648960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 10: Build the Model\ndef build_simple_model(input_shape=IMG_SIZE + (3,), n_classes=NUM_CLASSES):\n    \"\"\"Builds a simple, robust EfficientNetB0 model for single-phase training.\"\"\"\n    # Base model\n    base = EfficientNetB0(\n        include_top=False,\n        weights='imagenet',\n        input_shape=input_shape\n    )\n    #base.trainable = True # Train the whole model\n    # Freeze early layers\n    for layer in base.layers[:100]:\n        layer.trainable = False\n\n    for layer in base.layers[100:]:\n        layer.trainable = True\n\n    # Model architecture\n    inputs = layers.Input(shape=input_shape)\n    x = base(inputs, training=True)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.3)(x)\n    outputs = layers.Dense(n_classes, activation='softmax')(x)\n    model = Model(inputs, outputs)\n\n    # Compile the model with the correct loss function\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss='sparse_categorical_crossentropy', # The correct loss for 'sparse' mode\n        metrics=['accuracy']\n    )\n    return model\n\nmodel = build_simple_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:46:37.191313Z","iopub.execute_input":"2026-02-23T09:46:37.192115Z","iopub.status.idle":"2026-02-23T09:46:40.580180Z","shell.execute_reply.started":"2026-02-23T09:46:37.192084Z","shell.execute_reply":"2026-02-23T09:46:40.579549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 11: Define Callbacks (with Gentler LR Reduction)\ncheckpoint = ModelCheckpoint(\n    '/kaggle/working/efficientnetB0_bestmodel.keras',\n    monitor='val_accuracy',\n    save_best_only=True,\n    mode='max',\n    verbose=1\n)\n\nearly_stopping = EarlyStopping(\n    monitor='val_accuracy',\n    patience=6, # Stop if no improvement for 5 epochs\n    restore_best_weights=True,\n    mode='max',\n    verbose=1\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,  # Changed to 0.5 for a gentler reduction\n    patience=2,\n    verbose=1,\n    min_lr=1e-7\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:47:17.237123Z","iopub.execute_input":"2026-02-23T09:47:17.237445Z","iopub.status.idle":"2026-02-23T09:47:17.242386Z","shell.execute_reply.started":"2026-02-23T09:47:17.237419Z","shell.execute_reply":"2026-02-23T09:47:17.241559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 12: Train the Model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=30,\n    class_weight=class_weight_dict,\n    callbacks=[checkpoint, early_stopping, reduce_lr],\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T09:47:21.356491Z","iopub.execute_input":"2026-02-23T09:47:21.357181Z","iopub.status.idle":"2026-02-23T11:53:06.188101Z","shell.execute_reply.started":"2026-02-23T09:47:21.357150Z","shell.execute_reply":"2026-02-23T11:53:06.187283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/working'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:27:14.234755Z","iopub.execute_input":"2026-02-23T12:27:14.235414Z","iopub.status.idle":"2026-02-23T12:27:14.240047Z","shell.execute_reply.started":"2026-02-23T12:27:14.235380Z","shell.execute_reply":"2026-02-23T12:27:14.239346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/efficientnetB0_bestmodel.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:27:38.732977Z","iopub.execute_input":"2026-02-23T12:27:38.733340Z","iopub.status.idle":"2026-02-23T12:27:39.266563Z","shell.execute_reply.started":"2026-02-23T12:27:38.733315Z","shell.execute_reply":"2026-02-23T12:27:39.265469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 13: Plot Training History\ndef plot_history(history):\n    \"\"\"Plots accuracy and loss curves for training and validation.\"\"\"\n    fig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\n    # Plot accuracy\n    ax[0].plot(history.history['accuracy'], label='Train Accuracy')\n    ax[0].plot(history.history['val_accuracy'], label='Validation Accuracy')\n    ax[0].set_title('Model Accuracy')\n    ax[0].set_xlabel('Epoch')\n    ax[0].set_ylabel('Accuracy')\n    ax[0].legend()\n\n    # Plot loss\n    ax[1].plot(history.history['loss'], label='Train Loss')\n    ax[1].plot(history.history['val_loss'], label='Validation Loss')\n    ax[1].set_title('Model Loss')\n    ax[1].set_xlabel('Epoch')\n    ax[1].set_ylabel('Loss')\n    ax[1].legend()\n\n    plt.show()\n\nplot_history(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:28:00.953203Z","iopub.execute_input":"2026-02-23T12:28:00.953488Z","iopub.status.idle":"2026-02-23T12:28:01.251760Z","shell.execute_reply.started":"2026-02-23T12:28:00.953464Z","shell.execute_reply":"2026-02-23T12:28:01.250940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ## Cell 14: Evaluate the Model\n# Load the best performing model\nmodel.load_weights('/kaggle/working/efficientnetB0_bestmodel.keras')\n\n# Make predictions on the validation set\npreds = model.predict(val_gen)\npred_classes = np.argmax(preds, axis=1)\n\n# Get true labels directly from the generator for robust evaluation\ntrue_classes = val_gen.classes\n\n\n# Print Classification Report\nprint(\"📊 Classification Report:\\n\")\nprint(classification_report(true_classes, pred_classes, target_names=[str(i) for i in classes]))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:28:33.658405Z","iopub.execute_input":"2026-02-23T12:28:33.659097Z","iopub.status.idle":"2026-02-23T12:30:11.405114Z","shell.execute_reply.started":"2026-02-23T12:28:33.659055Z","shell.execute_reply":"2026-02-23T12:30:11.404261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display Confusion Matrix\ncm = confusion_matrix(true_classes, pred_classes)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=classes, yticklabels=classes)\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-23T12:31:00.433142Z","iopub.execute_input":"2026-02-23T12:31:00.433466Z","iopub.status.idle":"2026-02-23T12:31:00.646109Z","shell.execute_reply.started":"2026-02-23T12:31:00.433437Z","shell.execute_reply":"2026-02-23T12:31:00.645288Z"}},"outputs":[],"execution_count":null}]}