{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":7866129,"sourceType":"datasetVersion","datasetId":4614938},{"sourceId":7869237,"sourceType":"datasetVersion","datasetId":4617269}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'  # Suppress INFO and WARNING messages\nimport tensorflow as tf\n\nprint(\"TensorFlow version:\", tf.__version__)\nprint(\"CUDA available:\", tf.test.is_built_with_cuda())\nprint(\"GPU available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:17:51.142798Z","iopub.execute_input":"2025-06-03T14:17:51.143114Z","iopub.status.idle":"2025-06-03T14:18:04.799771Z","shell.execute_reply.started":"2025-06-03T14:17:51.143084Z","shell.execute_reply":"2025-06-03T14:18:04.798798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.sysconfig.get_build_info()['cuda_version'])\nprint(tf.sysconfig.get_build_info()['cudnn_version'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:08.888501Z","iopub.execute_input":"2025-06-03T14:18:08.888818Z","iopub.status.idle":"2025-06-03T14:18:08.894244Z","shell.execute_reply.started":"2025-06-03T14:18:08.888794Z","shell.execute_reply":"2025-06-03T14:18:08.89335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        tf.config.experimental.set_memory_growth(gpus[0], True)\n        print(\"GPU memory growth set successfully\")\n    except RuntimeError as e:\n        print(e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:13.048174Z","iopub.execute_input":"2025-06-03T14:18:13.04911Z","iopub.status.idle":"2025-06-03T14:18:13.054838Z","shell.execute_reply.started":"2025-06-03T14:18:13.04907Z","shell.execute_reply":"2025-06-03T14:18:13.053967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"TensorFlow version:\", tf.__version__)\nprint(\"CUDA available:\", tf.test.is_built_with_cuda())\nprint(\"GPU available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:16.608071Z","iopub.execute_input":"2025-06-03T14:18:16.608387Z","iopub.status.idle":"2025-06-03T14:18:16.613306Z","shell.execute_reply.started":"2025-06-03T14:18:16.608363Z","shell.execute_reply":"2025-06-03T14:18:16.612489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nimport numpy as np\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models, callbacks, mixed_precision","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:19.872637Z","iopub.execute_input":"2025-06-03T14:18:19.872947Z","iopub.status.idle":"2025-06-03T14:18:21.085443Z","shell.execute_reply.started":"2025-06-03T14:18:19.872922Z","shell.execute_reply":"2025-06-03T14:18:21.084757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"policy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\nprint(\"Mixed precision enabled:\", policy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:27.883235Z","iopub.execute_input":"2025-06-03T14:18:27.884316Z","iopub.status.idle":"2025-06-03T14:18:27.888777Z","shell.execute_reply.started":"2025-06-03T14:18:27.884284Z","shell.execute_reply":"2025-06-03T14:18:27.887863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 8\nEPOCHS = 3\nFROZEN_EPOCHS = 2\n\ntrain_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/trainLabels.csv.zip\"\ntrain_images_dir = \"/kaggle/input/diabetic-retinopathy-train-unzipped/train/\"\ntest_images_dir = \"/kaggle/input/diabetic-retinopathy-test-unzipped/test/\"\nsubmission_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/sampleSubmission.csv.zip\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:31.98388Z","iopub.execute_input":"2025-06-03T14:18:31.984255Z","iopub.status.idle":"2025-06-03T14:18:31.988744Z","shell.execute_reply.started":"2025-06-03T14:18:31.984227Z","shell.execute_reply":"2025-06-03T14:18:31.987922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(train_csv_path)\ndf_train[\"filepath\"] = df_train[\"image\"].apply(lambda x: os.path.join(train_images_dir, f\"{x}.jpeg\"))\n\ntrain_df, val_df = train_test_split(df_train, test_size=0.2, stratify=df_train[\"level\"], random_state=42)\n\ndf_submission = pd.read_csv(submission_csv_path)\ndf_submission[\"filepath\"] = df_submission[\"image\"].apply(lambda x: os.path.join(test_images_dir, f\"{x}.jpeg\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:35.949595Z","iopub.execute_input":"2025-06-03T14:18:35.95017Z","iopub.status.idle":"2025-06-03T14:18:36.166607Z","shell.execute_reply.started":"2025-06-03T14:18:35.950128Z","shell.execute_reply":"2025-06-03T14:18:36.165743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 4. tf.data Pipelines\n# ====================================================\ndef load_image_and_label(filepath, label):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32)\n    return image, label\n\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, 0.1)\n    image = tf.image.random_contrast(image, 0.9, 1.1)\n    return image, label\n\ndef one_hot_encode(image, label):\n    label = tf.one_hot(label, 5)\n    return image, label\n\ndef load_test_image(filepath):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32)\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:40.197079Z","iopub.execute_input":"2025-06-03T14:18:40.197436Z","iopub.status.idle":"2025-06-03T14:18:40.204397Z","shell.execute_reply.started":"2025-06-03T14:18:40.197411Z","shell.execute_reply":"2025-06-03T14:18:40.20348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training set\ndef get_train_ds(df):\n    ds = tf.data.Dataset.from_tensor_slices((df[\"filepath\"].values, df[\"level\"].values))\n    ds = ds.shuffle(len(df))\n    ds = ds.map(load_image_and_label, num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.map(augment, num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.map(one_hot_encode, num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.batch(BATCH_SIZE)\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds.cache()  # Only if fits in RAM\n\nds_train = get_train_ds(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:45.684808Z","iopub.execute_input":"2025-06-03T14:18:45.685142Z","iopub.status.idle":"2025-06-03T14:18:45.984356Z","shell.execute_reply.started":"2025-06-03T14:18:45.685117Z","shell.execute_reply":"2025-06-03T14:18:45.983684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validation set\nds_val = tf.data.Dataset.from_tensor_slices((val_df[\"filepath\"].values, val_df[\"level\"].values))\nds_val = ds_val.map(load_image_and_label, tf.data.AUTOTUNE).map(one_hot_encode, tf.data.AUTOTUNE)\nds_val = ds_val.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n# Test set\nds_test = tf.data.Dataset.from_tensor_slices(df_submission[\"filepath\"].values)\nds_test = ds_test.map(load_test_image, tf.data.AUTOTUNE).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:50.544948Z","iopub.execute_input":"2025-06-03T14:18:50.545558Z","iopub.status.idle":"2025-06-03T14:18:50.62821Z","shell.execute_reply.started":"2025-06-03T14:18:50.545529Z","shell.execute_reply":"2025-06-03T14:18:50.627521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 5. Build Model\n# ====================================================\ndef build_model(name):\n    inputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    if name == \"EfficientNetB2\":\n        preprocess = tf.keras.applications.efficientnet.preprocess_input\n        base = tf.keras.applications.EfficientNetB2(include_top=False, weights=\"imagenet\", input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    elif name == \"ConvNeXtTiny\":\n        preprocess = tf.keras.applications.convnext.preprocess_input\n        base = tf.keras.applications.ConvNeXtTiny(include_top=False, weights=\"imagenet\", input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    \n    x = layers.Lambda(preprocess)(inputs)\n    x = base(x, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)\n    outputs = layers.Dense(5, activation=\"softmax\", dtype=\"float32\")(x)\n    return models.Model(inputs, outputs), base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:54.455451Z","iopub.execute_input":"2025-06-03T14:18:54.456279Z","iopub.status.idle":"2025-06-03T14:18:54.462195Z","shell.execute_reply.started":"2025-06-03T14:18:54.456247Z","shell.execute_reply":"2025-06-03T14:18:54.461264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:18:59.537138Z","iopub.execute_input":"2025-06-03T14:18:59.537484Z","iopub.status.idle":"2025-06-03T14:18:59.955841Z","shell.execute_reply.started":"2025-06-03T14:18:59.537457Z","shell.execute_reply":"2025-06-03T14:18:59.954912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lr_scheduler(epoch, lr):\n    if epoch < FROZEN_EPOCHS:\n        return lr\n    else:\n        return lr * 0.9 if epoch % 2 == 0 else lr\n\nmodel_names = [\"EfficientNetB2\", \"ConvNeXtTiny\"]\nmodels_dict = {}\nval_accuracies = {}\nhistory_dict = {}\n\n# Verify dataset (only if datasets are defined)\ntry:\n    def verify_dataset(ds, name):\n        for images, labels in ds.take(1):\n            print(f\"{name} Dataset - Sample label shape: {labels.shape}, Sample label values: {labels.numpy()}\")\n        return ds\n    verify_dataset(ds_train, \"Train\")\n    verify_dataset(ds_val, \"Validation\")\nexcept NameError:\n    print(\"Warning: ds_train or ds_val not defined. Please run Cells 9 and 10 first.\")\n\nfor name in model_names:\n    print(f\"\\nTraining {name}...\")\n    model, base = build_model(name)\n    \n    checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n        f\"{name}_best_model.keras\", save_best_only=True, monitor=\"val_accuracy\", mode=\"max\"\n    )\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lr_scheduler)\n\n    base.trainable = False\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-3),\n                  loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])\n    history_frozen = model.fit(ds_train, validation_data=ds_val, epochs=FROZEN_EPOCHS,\n                               callbacks=[lr_callback, checkpoint_callback], verbose=2)\n\n    base.trainable = True\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-4),\n                  loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])\n    history_fine_tune = model.fit(ds_train, validation_data=ds_val, epochs=EPOCHS - FROZEN_EPOCHS,\n                                  callbacks=[lr_callback, checkpoint_callback], verbose=2)\n\n    history_dict[name] = {\n        'loss': history_frozen.history['loss'] + history_fine_tune.history['loss'],\n        'accuracy': history_frozen.history['accuracy'] + history_fine_tune.history['accuracy'],\n        'val_loss': history_frozen.history['val_loss'] + history_fine_tune.history['val_loss'],\n        'val_accuracy': history_frozen.history['val_accuracy'] + history_fine_tune.history['val_accuracy']\n    }\n\n    val_loss, val_acc = model.evaluate(ds_val, verbose=0)\n    val_accuracies[name] = val_acc\n    models_dict[name] = model\n    print(f\"{name} validation accuracy: {val_acc:.4f}\")\n\n    model.save(f\"{name}_model.keras\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T14:19:04.833917Z","iopub.execute_input":"2025-06-03T14:19:04.834461Z","iopub.status.idle":"2025-06-03T15:04:56.939678Z","shell.execute_reply.started":"2025-06-03T14:19:04.834425Z","shell.execute_reply":"2025-06-03T15:04:56.934708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_labels_and_predictions(model, ds_val):\n    y_true = []\n    y_pred = []\n    for images, labels in ds_val.unbatch():\n        pred = model.predict(images[tf.newaxis, ...], verbose=0)\n        y_true.append(np.argmax(labels.numpy()))\n        y_pred.append(np.argmax(pred))\n    return np.array(y_true), np.array(y_pred)\n\nfor name in model_names:\n    model = models_dict[name]\n    y_true, y_pred = get_labels_and_predictions(model, ds_val)\n\n    # Confusion Matrix\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=range(5), yticklabels=range(5))\n    plt.title(f'Confusion Matrix for {name}')\n    plt.xlabel('Predicted Label')\n    plt.ylabel('True Label')\n    plt.show()\n\n    # Classification Report\n    report = classification_report(y_true, y_pred, target_names=[f'Level {i}' for i in range(5)])\n    print(f'\\nClassification Report for {name}:\\n{report}')\n\n    # Training/Validation Graph\n    history = history_dict[name]\n    epochs = range(1, EPOCHS + 1)\n    plt.figure(figsize=(12, 5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, history['loss'], label='Training Loss', color='#1E88E5')\n    plt.plot(epochs, history['val_loss'], label='Validation Loss', color='#D81B60')\n    plt.title(f'{name} Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, history['accuracy'], label='Training Accuracy', color='#1E88E5')\n    plt.plot(epochs, history['val_accuracy'], label='Validation Accuracy', color='#D81B60')\n    plt.title(f'{name} Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T15:09:08.339796Z","iopub.execute_input":"2025-06-03T15:09:08.340189Z","iopub.status.idle":"2025-06-03T15:26:02.663467Z","shell.execute_reply.started":"2025-06-03T15:09:08.340159Z","shell.execute_reply":"2025-06-03T15:26:02.661257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 7. Weighted Ensemble on Test Set\n# ====================================================\ntotal_acc = sum(val_accuracies.values())\nweights = {name: acc / total_acc for name, acc in val_accuracies.items()}\nprint(\"Ensemble Weights:\", weights)\n\npreds_list = []\nfor name in model_names:\n    preds = models_dict[name].predict(ds_test, verbose=0)\n    preds_list.append(preds)\n\nensemble_preds = np.zeros_like(preds_list[0])\nfor i, name in enumerate(model_names):\n    ensemble_preds += preds_list[i] * weights[name]\n\nfinal_preds = np.argmax(ensemble_preds, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T15:34:09.505521Z","iopub.execute_input":"2025-06-03T15:34:09.505891Z","iopub.status.idle":"2025-06-03T16:06:23.328675Z","shell.execute_reply.started":"2025-06-03T15:34:09.505863Z","shell.execute_reply":"2025-06-03T16:06:23.327629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Extract weights from the ensemble calculation\nweights = {name: acc / sum(val_accuracies.values()) for name, acc in val_accuracies.items()}\nlabels = list(weights.keys())\nsizes = list(weights.values())\n\n# Create pie chart\nplt.figure(figsize=(6, 6))\nplt.pie(sizes, labels=labels, autopct='%1.1f%%', colors=['#1E88E5', '#D81B60'], startangle=90)\nplt.title('Weighted Ensemble Contributions')\nplt.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 8. Submission\n# ====================================================\ndf_submission[\"level\"] = final_preds\ndf_submission[[\"image\", \"level\"]].to_csv(\"submission.csv\", index=False)\nprint(\"✅ Submission file saved as 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T16:14:18.870115Z","iopub.execute_input":"2025-06-03T16:14:18.871027Z","iopub.status.idle":"2025-06-03T16:14:18.951768Z","shell.execute_reply.started":"2025-06-03T16:14:18.870971Z","shell.execute_reply":"2025-06-03T16:14:18.950851Z"}},"outputs":[],"execution_count":null}]}