{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14120161,"sourceType":"datasetVersion","datasetId":8995626}],"dockerImageVersionId":31193,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# CELL 0 — Protobuf compatibility patch (run FIRST)\n# ============================================================\nimport google.protobuf\nfrom google.protobuf import message_factory as _message_factory\n\nprint(\"protobuf version:\", google.protobuf.__version__)\n\n# Some TF 2.x builds expect MessageFactory.GetPrototype, which newer\n# protobuf versions removed. This patch restores a compatible method.\nif not hasattr(_message_factory.MessageFactory, \"GetPrototype\"):\n    def _GetPrototype(self, descriptor):\n        from google.protobuf import message_factory as mf_mod\n        return mf_mod.GetMessageClass(descriptor)\n\n    _message_factory.MessageFactory.GetPrototype = _GetPrototype\n    print(\"Patched MessageFactory.GetPrototype for compatibility.\")\nelse:\n    print(\"MessageFactory already has GetPrototype; no patch needed.\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:07.500699Z","iopub.execute_input":"2025-12-12T10:59:07.500959Z","iopub.status.idle":"2025-12-12T10:59:07.602007Z","shell.execute_reply.started":"2025-12-12T10:59:07.500932Z","shell.execute_reply":"2025-12-12T10:59:07.601210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 1 — Access and Preprocess Data (Q2 setup)\n# ============================================================\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nprint(\"TF version:\", tf.__version__)\n\n# ---- Root directory from competition ----\nroot_dir = \"/kaggle/input/aptos2019-blindness-detection\"\ntrain_img_dir = os.path.join(root_dir, \"train_images\")\ntest_img_dir  = os.path.join(root_dir, \"test_images\")\n\nprint(\"Contents of /kaggle/input:\", os.listdir(\"/kaggle/input\"))\nprint(\"Contents of aptos2019-blindness-detection:\", os.listdir(root_dir))\n\n# ---- Load CSVs ----\ntrain_df = pd.read_csv(os.path.join(root_dir, \"train.csv\"))\ntest_df  = pd.read_csv(os.path.join(root_dir, \"test.csv\"))\n\n# Keep numeric labels for EDA\ntrain_df[\"diagnosis_int\"] = train_df[\"diagnosis\"].copy()\n\n# Build file paths for train and test images\ntrain_df[\"file_path\"] = train_df[\"id_code\"].apply(\n    lambda x: os.path.join(train_img_dir, f\"{x}.png\")\n)\ntest_df[\"file_path\"] = test_df[\"id_code\"].apply(\n    lambda x: os.path.join(test_img_dir, f\"{x}.png\")\n)\n\n# Convert labels to string for flow_from_dataframe (categorical)\ntrain_df[\"diagnosis\"] = train_df[\"diagnosis\"].astype(str)\n\n# Sanity checks: confirm all files exist\nmissing_train_files = train_df[~train_df[\"file_path\"].apply(os.path.exists)]\nmissing_test_files  = test_df[~test_df[\"file_path\"].apply(os.path.exists)]\nprint(f\"Missing training files: {len(missing_train_files)}\")\nprint(f\"Missing test files: {len(missing_test_files)}\")\n\n# ============================================================\n# Shared ImageDataGenerator (rescale + augmentation)\n# Used for: autoencoder pretraining AND classifier training\n# ============================================================\nimg_datagen = ImageDataGenerator(\n    rescale=1.0 / 255.0,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode=\"nearest\",\n    validation_split=0.2  # 20% validation\n)\n\nIMG_SIZE = (224, 224)\nBATCH_SIZE = 32\n\n# --- Classifier generators (supervised) ---\nclf_train_generator = img_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col=\"file_path\",\n    y_col=\"diagnosis\",\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    subset=\"training\"\n)\n\nclf_val_generator = img_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col=\"file_path\",\n    y_col=\"diagnosis\",\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\",\n    subset=\"validation\"\n)\n\nprint(f\"Classifier training samples: {clf_train_generator.samples}\")\nprint(f\"Classifier validation samples: {clf_val_generator.samples}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:07.603293Z","iopub.execute_input":"2025-12-12T10:59:07.603616Z","iopub.status.idle":"2025-12-12T10:59:41.026332Z","shell.execute_reply.started":"2025-12-12T10:59:07.603592Z","shell.execute_reply":"2025-12-12T10:59:41.025692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 2 — Simple visual EDA (class balance + sample images)\n# ============================================================\nimport matplotlib.pyplot as plt\n\n# 1) Class distribution bar plot\nclass_counts = train_df[\"diagnosis_int\"].value_counts().sort_index()\nplt.figure(figsize=(6, 4))\nclass_counts.plot(kind=\"bar\")\nplt.xlabel(\"Diagnosis class\")\nplt.ylabel(\"Count\")\nplt.title(\"Class distribution in training data\")\nplt.show()\n\nprint(\"Class counts:\")\nprint(class_counts)\n\n# 2) Show 3 sample training images with labels\nsample_train = train_df.sample(3, random_state=42)\n\nplt.figure(figsize=(10, 4))\nfor i, row in enumerate(sample_train.itertuples(), 1):\n    img = tf.keras.utils.load_img(row.file_path, target_size=IMG_SIZE)\n    plt.subplot(1, 3, i)\n    plt.imshow(img)\n    plt.axis(\"off\")\n    plt.title(f\"Train\\n{row.id_code}\\nlabel={row.diagnosis_int}\")\nplt.tight_layout()\nplt.show()\n\n# 3) Show 3 sample test images\nsample_test = test_df.sample(3, random_state=42)\n\nplt.figure(figsize=(10, 4))\nfor i, row in enumerate(sample_test.itertuples(), 1):\n    img = tf.keras.utils.load_img(row.file_path, target_size=IMG_SIZE)\n    plt.subplot(1, 3, i)\n    plt.imshow(img)\n    plt.axis(\"off\")\n    plt.title(f\"Test\\n{row.id_code}\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:41.027088Z","iopub.execute_input":"2025-12-12T10:59:41.027326Z","iopub.status.idle":"2025-12-12T10:59:42.791364Z","shell.execute_reply.started":"2025-12-12T10:59:41.027301Z","shell.execute_reply":"2025-12-12T10:59:42.790458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 3 — Autoencoder Data Generators (unsupervised)\n# ============================================================\n# For the autoencoder we only need images (no labels).\n# We reuse img_datagen with the same augmentations/validation split.\n\n# These raw generators return ONLY X (images), no labels.\nae_train_gen_raw = img_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col=\"file_path\",\n    y_col=None,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=None,\n    subset=\"training\",\n    shuffle=True\n)\n\nae_val_gen_raw = img_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col=\"file_path\",\n    y_col=None,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=None,\n    subset=\"validation\",\n    shuffle=False\n)\n\n# Wrap the raw generators so they output (x, x) pairs for autoencoder training\ndef make_autoencoder_generator(raw_gen):\n    while True:\n        batch_x = next(raw_gen)\n        yield (batch_x, batch_x)\n\nae_train_generator = make_autoencoder_generator(ae_train_gen_raw)\nae_val_generator   = make_autoencoder_generator(ae_val_gen_raw)\n\nprint(f\"AE training steps per epoch: {len(ae_train_gen_raw)}\")\nprint(f\"AE validation steps: {len(ae_val_gen_raw)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:42.792851Z","iopub.execute_input":"2025-12-12T10:59:42.793092Z","iopub.status.idle":"2025-12-12T10:59:42.854999Z","shell.execute_reply.started":"2025-12-12T10:59:42.793073Z","shell.execute_reply":"2025-12-12T10:59:42.854238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 4 — Build Convolutional Autoencoder\n# ============================================================\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.optimizers import Adam\n\ninput_shape = (IMG_SIZE[0], IMG_SIZE[1], 3)\ninputs = layers.Input(shape=input_shape)\n\n# ---------- Encoder ----------\nx = layers.Conv2D(32, (3, 3), activation=\"relu\", padding=\"same\")(inputs)\nx = layers.MaxPooling2D((2, 2), padding=\"same\")(x)   # 112x112\n\nx = layers.Conv2D(64, (3, 3), activation=\"relu\", padding=\"same\")(x)\nx = layers.MaxPooling2D((2, 2), padding=\"same\")(x)   # 56x56\n\nx = layers.Conv2D(128, (3, 3), activation=\"relu\", padding=\"same\")(x)\nx = layers.MaxPooling2D((2, 2), padding=\"same\")(x)   # 28x28\n\nx = layers.Conv2D(256, (3, 3), activation=\"relu\", padding=\"same\")(x)\nencoded = layers.MaxPooling2D((2, 2), padding=\"same\", name=\"latent_feature\")(x)  # 14x14x256\n\n# ---------- Decoder ----------\nx = layers.Conv2D(256, (3, 3), activation=\"relu\", padding=\"same\")(encoded)\nx = layers.UpSampling2D((2, 2))(x)   # 28x28\n\nx = layers.Conv2D(128, (3, 3), activation=\"relu\", padding=\"same\")(x)\nx = layers.UpSampling2D((2, 2))(x)   # 56x56\n\nx = layers.Conv2D(64, (3, 3), activation=\"relu\", padding=\"same\")(x)\nx = layers.UpSampling2D((2, 2))(x)   # 112x112\n\nx = layers.Conv2D(32, (3, 3), activation=\"relu\", padding=\"same\")(x)\nx = layers.UpSampling2D((2, 2))(x)   # 224x224\n\ndecoded = layers.Conv2D(3, (3, 3), activation=\"sigmoid\", padding=\"same\")(x)\n\nautoencoder = models.Model(inputs, decoded, name=\"retina_autoencoder\")\n\nautoencoder.compile(\n    optimizer=Adam(learning_rate=1e-3),\n    loss=\"mse\"\n)\n\nautoencoder.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:42.855759Z","iopub.execute_input":"2025-12-12T10:59:42.856071Z","iopub.status.idle":"2025-12-12T10:59:44.681828Z","shell.execute_reply.started":"2025-12-12T10:59:42.856052Z","shell.execute_reply":"2025-12-12T10:59:44.681247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 5 — Train Autoencoder\n# ============================================================\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nae_callbacks = [\n    EarlyStopping(monitor=\"val_loss\", patience=2, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.2, patience=1, verbose=1),\n]\n\nAE_EPOCHS = 5  # keep small for Kaggle runtime\n\nhistory_ae = autoencoder.fit(\n    ae_train_generator,\n    steps_per_epoch=len(ae_train_gen_raw),\n    epochs=AE_EPOCHS,\n    validation_data=ae_val_generator,\n    validation_steps=len(ae_val_gen_raw),\n    callbacks=ae_callbacks\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T10:59:44.682457Z","iopub.execute_input":"2025-12-12T10:59:44.682710Z","iopub.status.idle":"2025-12-12T11:32:43.328127Z","shell.execute_reply.started":"2025-12-12T10:59:44.682693Z","shell.execute_reply":"2025-12-12T11:32:43.327296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 6 — (Optional) Plot Autoencoder Loss Curves\n# ============================================================\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(6, 4))\nplt.plot(history_ae.history[\"loss\"], label=\"train_loss\")\nplt.plot(history_ae.history[\"val_loss\"], label=\"val_loss\")\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"MSE loss\")\nplt.title(\"Autoencoder reconstruction loss\")\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T11:32:43.329368Z","iopub.execute_input":"2025-12-12T11:32:43.329632Z","iopub.status.idle":"2025-12-12T11:32:43.502142Z","shell.execute_reply.started":"2025-12-12T11:32:43.329612Z","shell.execute_reply":"2025-12-12T11:32:43.501603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 7 — Build Classifier Using the Encoder\n# ============================================================\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model\n\nnum_classes = len(clf_train_generator.class_indices)\nprint(\"Classes:\", clf_train_generator.class_indices)\n\n# Extract encoder (from inputs to latent feature layer)\nencoder = Model(\n    inputs=autoencoder.input,\n    outputs=autoencoder.get_layer(\"latent_feature\").output,\n    name=\"retina_encoder\"\n)\n\n# First, freeze encoder to train only classifier head\nencoder.trainable = False\n\n# Classifier model: encoder + global pooling + dense layers\nclf_inputs = layers.Input(shape=input_shape)\nx = encoder(clf_inputs, training=False)\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.4)(x)\nx = Dense(256, activation=\"relu\")(x)\nx = Dropout(0.3)(x)\nclf_outputs = Dense(num_classes, activation=\"softmax\")(x)\n\nclf_model = Model(clf_inputs, clf_outputs, name=\"AE_classifier\")\n\nclf_model.compile(\n    optimizer=Adam(learning_rate=1e-3),\n    loss=\"categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\nclf_model.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T11:32:43.502876Z","iopub.execute_input":"2025-12-12T11:32:43.503190Z","iopub.status.idle":"2025-12-12T11:32:43.557239Z","shell.execute_reply.started":"2025-12-12T11:32:43.503171Z","shell.execute_reply":"2025-12-12T11:32:43.556708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 8 — Train Classifier Head (encoder frozen)\n# ============================================================\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n\ncallbacks_head = [\n    EarlyStopping(monitor=\"val_loss\", patience=2, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.2, patience=1, verbose=1),\n    ModelCheckpoint(\"ae_classifier_head.weights.h5\", monitor=\"val_loss\", save_best_only=True, verbose=1),\n]\n\nHEAD_EPOCHS = 5  # short for assignment\n\nhistory_head = clf_model.fit(\n    clf_train_generator,\n    epochs=HEAD_EPOCHS,\n    validation_data=clf_val_generator,\n    callbacks=callbacks_head\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T11:32:43.558009Z","iopub.execute_input":"2025-12-12T11:32:43.558259Z","iopub.status.idle":"2025-12-12T12:03:43.935198Z","shell.execute_reply.started":"2025-12-12T11:32:43.558236Z","shell.execute_reply":"2025-12-12T12:03:43.934358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 9 — Fine-tune Upper Encoder Layers\n# ============================================================\n# Unfreeze the encoder and fine-tune only the last few blocks.\nencoder.trainable = True\n\n# For simplicity, unfreeze the last N convolutional layers.\n# (You can adjust N to control how much you fine-tune.)\nN_UNFREEZE = 15\nfor layer in encoder.layers[:-N_UNFREEZE]:\n    layer.trainable = False\n\nclf_model.compile(\n    optimizer=Adam(learning_rate=1e-4),\n    loss=\"categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)\n\ncallbacks_ft = [\n    EarlyStopping(monitor=\"val_loss\", patience=2, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.2, patience=1, verbose=1),\n    ModelCheckpoint(\"ae_classifier_finetuned.weights.h5\", monitor=\"val_loss\", save_best_only=True, verbose=1),\n]\n\nFT_EPOCHS = 5  # again short run\n\nhistory_ft = clf_model.fit(\n    clf_train_generator,\n    epochs=FT_EPOCHS,\n    validation_data=clf_val_generator,\n    callbacks=callbacks_ft\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T12:03:43.937352Z","iopub.execute_input":"2025-12-12T12:03:43.937694Z","iopub.status.idle":"2025-12-12T12:35:07.811017Z","shell.execute_reply.started":"2025-12-12T12:03:43.937674Z","shell.execute_reply":"2025-12-12T12:35:07.810349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 10 — Plot Classifier Learning Curves\n# ============================================================\ndef plot_history(hist, title_prefix=\"\"):\n    plt.figure(figsize=(6, 4))\n    plt.plot(hist.history[\"loss\"], label=\"train_loss\")\n    plt.plot(hist.history[\"val_loss\"], label=\"val_loss\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.title(f\"{title_prefix} Loss\")\n    plt.legend()\n    plt.show()\n\n    if \"accuracy\" in hist.history:\n        plt.figure(figsize=(6, 4))\n        plt.plot(hist.history[\"accuracy\"], label=\"train_acc\")\n        plt.plot(hist.history[\"val_accuracy\"], label=\"val_acc\")\n        plt.xlabel(\"Epoch\")\n        plt.ylabel(\"Accuracy\")\n        plt.title(f\"{title_prefix} Accuracy\")\n        plt.legend()\n        plt.show()\n\nplot_history(history_head, \"AE Classifier (Head)\")\nplot_history(history_ft, \"AE Classifier (Fine-tune)\")\n\n# ➜ Add a Markdown cell after this:\n#    - Briefly describe whether val_loss/val_accuracy improved.\n#    - Compare qualitatively with your Q1 EfficientNet model.\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T12:35:07.812053Z","iopub.execute_input":"2025-12-12T12:35:07.812320Z","iopub.status.idle":"2025-12-12T12:35:08.471694Z","shell.execute_reply.started":"2025-12-12T12:35:07.812297Z","shell.execute_reply":"2025-12-12T12:35:08.470916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# CELL 11 — Build Test Generator & Create submission.csv\n# ============================================================\n# For test data we only need images (no labels), same rescale.\ntest_datagen = ImageDataGenerator(rescale=1.0 / 255.0)\n\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    x_col=\"file_path\",\n    y_col=None,\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode=None,\n    shuffle=False\n)\n\nprint(f\"Test samples: {test_generator.samples}\")\n\n# Predict\npred_probs = clf_model.predict(test_generator)\npredicted_classes = tf.argmax(pred_probs, axis=1).numpy()\n\nsubmission_df = pd.DataFrame({\n    \"id_code\": test_df[\"id_code\"],\n    \"diagnosis\": predicted_classes\n})\n\n# VERY IMPORTANT: competition expects exactly this filename\nsubmission_df.to_csv(\"submission.csv\", index=False)\nsubmission_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-12T12:35:08.472663Z","iopub.execute_input":"2025-12-12T12:35:08.472981Z","iopub.status.idle":"2025-12-12T12:36:54.588856Z","shell.execute_reply.started":"2025-12-12T12:35:08.472955Z","shell.execute_reply":"2025-12-12T12:36:54.588197Z"}},"outputs":[],"execution_count":null}]}