{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":7866129,"sourceType":"datasetVersion","datasetId":4614938},{"sourceId":7869237,"sourceType":"datasetVersion","datasetId":4617269}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport cv2\nimport tensorflow as tf\nimport numpy as np\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import layers, models, callbacks, mixed_precision","metadata":{"execution":{"iopub.status.busy":"2024-03-25T17:01:43.893109Z","iopub.execute_input":"2024-03-25T17:01:43.893526Z","iopub.status.idle":"2024-03-25T17:01:44.115232Z","shell.execute_reply.started":"2024-03-25T17:01:43.893493Z","shell.execute_reply":"2024-03-25T17:01:44.114311Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"policy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\nprint(\"Mixed precision enabled:\", policy)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 224\nBATCH_SIZE = 16\nEPOCHS = 5\nFROZEN_EPOCHS = 3\n\ntrain_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/trainLabels.csv.zip\"\ntrain_images_dir = \"/kaggle/input/diabetic-retinopathy-train-unzipped/train/\"\ntest_images_dir = \"/kaggle/input/diabetic-retinopathy-test-unzipped/test/\"\nsubmission_csv_path = \"/kaggle/input/diabetic-retinopathy-detection/sampleSubmission.csv.zip\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(train_csv_path)\ndf_train[\"filepath\"] = df_train[\"image\"].apply(lambda x: os.path.join(train_images_dir, f\"{x}.jpeg\"))\n\ntrain_df, val_df = train_test_split(df_train, test_size=0.2, stratify=df_train[\"level\"], random_state=42)\n\ndf_submission = pd.read_csv(submission_csv_path)\ndf_submission[\"filepath\"] = df_submission[\"image\"].apply(lambda x: os.path.join(test_images_dir, f\"{x}.jpeg\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 4. tf.data Pipelines\n# ====================================================\ndef load_image_and_label(filepath, label):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32)\n    return image, label\n\ndef augment(image, label):\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_brightness(image, 0.1)\n    image = tf.image.random_contrast(image, 0.9, 1.1)\n    return image, label\n\ndef one_hot_encode(image, label):\n    label = tf.one_hot(label, 5)\n    return image, label\n\ndef load_test_image(filepath):\n    image = tf.io.read_file(filepath)\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.image.resize(image, [IMG_SIZE, IMG_SIZE])\n    image = tf.cast(image, tf.float32)\n    return image","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training set\nds_train = tf.data.Dataset.from_tensor_slices((train_df[\"filepath\"].values, train_df[\"level\"].values))\nds_train = ds_train.shuffle(len(train_df)).map(load_image_and_label, tf.data.AUTOTUNE)\nds_train = ds_train.map(augment, tf.data.AUTOTUNE).map(one_hot_encode, tf.data.AUTOTUNE)\nds_train = ds_train.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n# Validation set\nds_val = tf.data.Dataset.from_tensor_slices((val_df[\"filepath\"].values, val_df[\"level\"].values))\nds_val = ds_val.map(load_image_and_label, tf.data.AUTOTUNE).map(one_hot_encode, tf.data.AUTOTUNE)\nds_val = ds_val.batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n\n# Test set\nds_test = tf.data.Dataset.from_tensor_slices(df_submission[\"filepath\"].values)\nds_test = ds_test.map(load_test_image, tf.data.AUTOTUNE).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 5. Build Model\n# ====================================================\ndef build_model(name):\n    inputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\n    if name == \"EfficientNetB2\":\n        preprocess = tf.keras.applications.efficientnet.preprocess_input\n        base = tf.keras.applications.EfficientNetB2(include_top=False, weights=\"imagenet\", input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    elif name == \"ConvNeXtTiny\":\n        preprocess = tf.keras.applications.convnext.preprocess_input\n        base = tf.keras.applications.ConvNeXtTiny(include_top=False, weights=\"imagenet\", input_shape=(IMG_SIZE, IMG_SIZE, 3))\n    \n    x = layers.Lambda(preprocess)(inputs)\n    x = base(x, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)\n    outputs = layers.Dense(5, activation=\"softmax\", dtype=\"float32\")(x)\n    return models.Model(inputs, outputs), base","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 6. Train and Store Models\n# ====================================================\ndef lr_scheduler(epoch, lr):\n    if epoch < FROZEN_EPOCHS:\n        return lr\n    else:\n        return lr * 0.9 if epoch % 3 == 0 else lr\n\nlr_callback = callbacks.LearningRateScheduler(lr_scheduler)\n\nmodel_names = [\"EfficientNetB2\", \"ConvNeXtTiny\"]\nmodels_dict = {}\nval_accuracies = {}\n\nfor name in model_names:\n    print(f\"\\nTraining {name}...\")\n    model, base = build_model(name)\n\n    # Phase 1: Freeze base\n    base.trainable = False\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-3),\n                  loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])\n    model.fit(ds_train, validation_data=ds_val, epochs=FROZEN_EPOCHS, callbacks=[lr_callback], verbose=2)\n\n    # Phase 2: Unfreeze base\n    base.trainable = True\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-4),\n                  loss=\"categorical_crossentropy\", metrics=[\"accuracy\"])\n    model.fit(ds_train, validation_data=ds_val, epochs=EPOCHS - FROZEN_EPOCHS, callbacks=[lr_callback], verbose=2)\n\n    val_loss, val_acc = model.evaluate(ds_val, verbose=0)\n    val_accuracies[name] = val_acc\n    models_dict[name] = model\n    print(f\"{name} validation accuracy: {val_acc:.4f}\")\n\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 7. Weighted Ensemble on Test Set\n# ====================================================\ntotal_acc = sum(val_accuracies.values())\nweights = {name: acc / total_acc for name, acc in val_accuracies.items()}\nprint(\"Ensemble Weights:\", weights)\n\npreds_list = []\nfor name in model_names:\n    preds = models_dict[name].predict(ds_test, verbose=0)\n    preds_list.append(preds)\n\nensemble_preds = np.zeros_like(preds_list[0])\nfor i, name in enumerate(model_names):\n    ensemble_preds += preds_list[i] * weights[name]\n\nfinal_preds = np.argmax(ensemble_preds, axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================\n# 8. Submission\n# ====================================================\ndf_submission[\"level\"] = final_preds\ndf_submission[[\"image\", \"level\"]].to_csv(\"submission.csv\", index=False)\nprint(\"✅ Submission file saved as 'submission.csv'\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}