{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Important Libraries","metadata":{}},{"cell_type":"code","source":"import os, glob, time\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\n\nfrom sklearn.metrics import cohen_kappa_score, classification_report, confusion_matrix\n\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"3\"\nos.environ[\"XLA_FLAGS\"] = \"--xla_gpu_cuda_data_dir=\"\n# os.environ[\"TF_XLA_FLAGS\"] = \"--tf_xla_auto_jit=0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:50:17.846429Z","iopub.execute_input":"2025-11-19T13:50:17.847206Z","iopub.status.idle":"2025-11-19T13:50:17.853164Z","shell.execute_reply.started":"2025-11-19T13:50:17.847173Z","shell.execute_reply":"2025-11-19T13:50:17.852295Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Access and Preprocess Data","metadata":{}},{"cell_type":"code","source":"\n\n#Add Dataset Path\nroot_dir = '/kaggle/input/aptos2019-blindness-detection'\ntrain_img_dir = os.path.join(root_dir, 'train_images')\ntest_img_dir = os.path.join(root_dir, 'test_images')\n\n## Load the train and test CSV files\ntrain_df = pd.read_csv(os.path.join(root_dir, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(root_dir, 'test.csv'))\n\ntrain_df['file_path'] = train_df['id_code'].apply(lambda x: os.path.join(train_img_dir, f\"{x}.png\"))\ntest_df['file_path'] = test_df['id_code'].apply(lambda x: os.path.join(test_img_dir, f\"{x}.png\"))\n\n# Convert the 'diagnosis' column to string to avoid TypeError with␣ ImageDataGenerator\ntrain_df['diagnosis'] = train_df['diagnosis'].astype(str)\n\ntrain_files = glob.glob(train_img_dir + '/**/*.png', recursive=True)\ntest_files = glob.glob(test_img_dir + '/**/*.png', recursive=True)\n\n# Verify that all paths in 'file_path' exist\nmissing_train_files = train_df[~train_df['file_path'].apply(os.path.exists)]\nif len(missing_train_files) > 0:\n    print(f\"Missing training files: {len(missing_train_files)}\")\n\nprint(\"Train samples:\", len(train_df))\nprint(\"Test samples:\", len(test_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:32:56.066426Z","iopub.execute_input":"2025-11-19T13:32:56.066996Z","iopub.status.idle":"2025-11-19T13:33:04.738825Z","shell.execute_reply.started":"2025-11-19T13:32:56.066974Z","shell.execute_reply":"2025-11-19T13:33:04.738197Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Display Sample Images","metadata":{}},{"cell_type":"code","source":"num_images = 3\nfig, axes = plt.subplots(1, 3, figsize=(12, 12))  # Create a grid of 3x3\n\nfor i, ax in enumerate(axes.flat):\n    if i < num_images:\n        image_path = train_files[i]\n        image = Image.open(image_path)\n        ax.imshow(image, cmap='gray')  # Use cmap='gray' if it's a grayscale image\n        ax.set_title(f'Image {i+1}')\n        ax.axis('off')  # Hide the axes\n    else:\n        ax.axis('off')  # Hide axes for empty subplots\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:04.739495Z","iopub.execute_input":"2025-11-19T13:33:04.739704Z","iopub.status.idle":"2025-11-19T13:33:07.058914Z","shell.execute_reply.started":"2025-11-19T13:33:04.739677Z","shell.execute_reply":"2025-11-19T13:33:07.058187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_images = 3\nfig, axes = plt.subplots(1, 3, figsize=(12, 12))  # Create a grid of 3x3\n\nfor i, ax in enumerate(axes.flat):\n    if i < num_images:\n        image_path = test_files[i]\n        image = Image.open(image_path)\n        ax.imshow(image, cmap='gray')  # Use cmap='gray' if it's a grayscale image\n        ax.set_title(f'Image {i+1}')\n        ax.axis('off')  # Hide the axes\n    else:\n        ax.axis('off')  # Hide axes for empty subplots\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:07.060655Z","iopub.execute_input":"2025-11-19T13:33:07.060856Z","iopub.status.idle":"2025-11-19T13:33:07.63224Z","shell.execute_reply.started":"2025-11-19T13:33:07.060841Z","shell.execute_reply":"2025-11-19T13:33:07.631362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_examples(train_df, test_df, n=3):\n    fig, axes = plt.subplots(2, n, figsize=(4*n, 8))\n    for i, row in enumerate(train_df.sample(n).itertuples()):\n        img = Image.open(row.file_path)\n        axes[0, i].imshow(img)\n        axes[0, i].set_title(f\"Train - label {row.diagnosis}\")\n        axes[0, i].axis('off')\n    for i, row in enumerate(test_df.sample(n).itertuples()):\n        img = Image.open(row.file_path)\n        axes[1, i].imshow(img)\n        axes[1, i].set_title(f\"Test - {row.id_code}\")\n        axes[1, i].axis('off')\n    plt.tight_layout()\n    plt.show()\n\nshow_examples(train_df, test_df, n=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:07.633036Z","iopub.execute_input":"2025-11-19T13:33:07.633234Z","iopub.status.idle":"2025-11-19T13:33:10.939414Z","shell.execute_reply.started":"2025-11-19T13:33:07.633219Z","shell.execute_reply":"2025-11-19T13:33:10.938484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data augmentation and ImageDataGenerator setup\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.resnet.preprocess_input,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    validation_split=0.2 # Use 20% for validation\n    )\n# Training data generator\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical',\n    subset='training' # 80% of data used for training\n    )\n# Validation data generator\nval_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical',\n    subset='validation' # 20% of data used for validation\n    )\n# Test data generator (no labels)\ntest_datagen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.resnet.preprocess_input\n)\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    x_col='file_path',\n    y_col=None,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode=None,\n    shuffle=False\n    )\n# Print summary of each generator for confirmation\nprint(f\"Training samples: {train_generator.samples}\")\nprint(f\"Validation samples: {val_generator.samples}\")\nprint(f\"Test samples: {test_generator.samples}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:10.940397Z","iopub.execute_input":"2025-11-19T13:33:10.9409Z","iopub.status.idle":"2025-11-19T13:33:14.007662Z","shell.execute_reply.started":"2025-11-19T13:33:10.94088Z","shell.execute_reply":"2025-11-19T13:33:14.006936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# cell 3: QWK callback\nclass QWKCallback(tf.keras.callbacks.Callback):\n    def __init__(self, val_generator, patience=0):\n        super().__init__()\n        self.val_gen = val_generator\n        self.best_qwk = -1.0\n        self.history = []\n    def on_epoch_end(self, epoch, logs=None):\n        # predict on val set\n        preds = self.model.predict(self.val_gen, verbose=0)\n        y_pred = np.argmax(preds, axis=1)\n        # true labels: as int (0..4)\n        y_true = self.val_gen.classes \n        qwk = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n        self.history.append(qwk)\n        print(f\" — val_qwk (quadratic) : {qwk:.5f}\")\n        # optional: save best\n        if qwk > self.best_qwk:\n            self.best_qwk = qwk\n            \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:14.008505Z","iopub.execute_input":"2025-11-19T13:33:14.008822Z","iopub.status.idle":"2025-11-19T13:33:14.014634Z","shell.execute_reply.started":"2025-11-19T13:33:14.008793Z","shell.execute_reply":"2025-11-19T13:33:14.013813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. EfficientNetB0 Model","metadata":{}},{"cell_type":"code","source":"\n# import tensorflow as tf\n# from tensorflow.keras.applications import EfficientNetB0\n# from tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.optimizers import Adam\n# # Load the EfficientNetB0 model without the top layer and specify a smaller␣input shape\n# base_model = EfficientNetB0(weights='imagenet', include_top=False,input_shape=(224, 224, 3))\n# base_model.trainable = True\n# # Fine-tune from a specific layer (unfreezing only the last 20 layers)\n# for layer in base_model.layers:\n#         layer.trainable = False\n# for layer in base_model.layers[-20:]:\n#     layer.trainable = True\n# # Add custom layers on top of EfficientNetB0\n# dropout_rate=0.5\n# num_classes=5\n# l2_reg=1e-4\n\n# x = base_model.output\n# x = layers.GlobalAveragePooling2D()(x)\n# x = layers.BatchNormalization()(x)\n# x = layers.Dropout(dropout_rate)(x)\n# x = layers.Dense(128, activation='relu', kernel_regularizer=regularizers.l2(l2_reg))(x)\n# x = layers.BatchNormalization()(x)\n# x = layers.Dropout(dropout_rate/2)(x)\n# outputs = layers.Dense(num_classes, activation='softmax')(x)\n# model = models.Model(inputs=base_model.input, outputs=outputs)\n\n# model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:33:14.015409Z","iopub.execute_input":"2025-11-19T13:33:14.015652Z","iopub.status.idle":"2025-11-19T13:33:17.845757Z","shell.execute_reply.started":"2025-11-19T13:33:14.015638Z","shell.execute_reply":"2025-11-19T13:33:17.845182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # compile + callbacks\n# LR = 1e-4\n# EPOCHS = 15 \n# model.compile(optimizer=Adam(LR), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# qwk_cb = QWKCallback(val_generator)\n# reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1)\n# earlystop = EarlyStopping(monitor='val_loss', patience=7, restore_best_weights=True, verbose=1)\n# checkpoint = ModelCheckpoint('best_effnetb0.h5', monitor='val_loss', save_best_only=True)\n\n# history = model.fit(\n#     train_generator,\n#     validation_data=val_generator,\n#     epochs=EPOCHS,\n#     callbacks=[qwk_cb, reduce_lr, earlystop, checkpoint],\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T13:50:23.184302Z","iopub.execute_input":"2025-11-19T13:50:23.18487Z","iopub.status.idle":"2025-11-19T15:48:35.587293Z","shell.execute_reply.started":"2025-11-19T13:50:23.184842Z","shell.execute_reply":"2025-11-19T15:48:35.586702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # cell 6: plot\n# def plot_history(history):\n#     # history may be Keras History + qwk_cb.history\n#     hist = history.history\n#     epochs = range(1, len(hist['loss'])+1)\n#     plt.figure(figsize=(12,4))\n#     plt.subplot(1,2,1)\n#     plt.plot(epochs, hist['loss'], label='train_loss')\n#     plt.plot(epochs, hist['val_loss'], label='val_loss')\n#     plt.title('Loss')\n#     plt.legend()\n#     plt.subplot(1,2,2)\n#     plt.plot(epochs, hist.get('accuracy', hist.get('acc')), label='train_acc')\n#     plt.plot(epochs, hist['val_accuracy'], label='val_acc')\n#     plt.title('Accuracy')\n#     plt.legend()\n#     plt.show()\n\n# plot_history(history)\n\n# # show qwk per epoch from callback\n# print(\"QWK per epoch:\", qwk_cb.history)\n# plt.plot(range(1, len(qwk_cb.history)+1), qwk_cb.history, marker='o')\n# plt.title('Validation Quadratic Weighted Kappa per epoch')\n# plt.xlabel('Epoch')\n# plt.ylabel('QWK')\n# plt.grid(True)\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T16:02:22.823056Z","iopub.execute_input":"2025-11-19T16:02:22.823805Z","iopub.status.idle":"2025-11-19T16:02:23.324812Z","shell.execute_reply.started":"2025-11-19T16:02:22.82378Z","shell.execute_reply":"2025-11-19T16:02:23.324152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # cell 7: predict & save submission\n# # load best weights if ModelCheckpoint saved\n# model.load_weights('best_effnetb0.h5')\n\n# preds = model.predict(test_generator, verbose=1)\n# pred_labels = np.argmax(preds, axis=1)\n\n# submission = pd.DataFrame({\n#     'id_code': test_df['id_code'],\n#     'diagnosis': pred_labels\n# })\n# # submission.to_csv('submission_effnetb0.csv', index=False)\n# print(\"Saved submission_effnetb0.csv\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## RESNET50","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nbase_model_RNET = ResNet50(weights='imagenet', include_top=False, input_shape=(224,224,3))\n\n# Freeze lower layers \n# for layer in base_model.layers[:-50]:\n#     layer.trainable = False\nfor layer in base_model_RNET.layers:\n        layer.trainable = False\nfor layer in base_model_RNET.layers[-20:]:\n    layer.trainable = True\n# Add custom head\nx = base_model_RNET.output\nx = GlobalAveragePooling2D()(x)\nx = Dropout(0.3)(x)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.3)(x)\noutput = Dense(5, activation='softmax')(x)  # 5 classes: 0-4\n\nmodel_RNET = Model(inputs=base_model_RNET.input, outputs=output)\n\n# Show summary\nmodel_RNET.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T16:05:19.074632Z","iopub.execute_input":"2025-11-19T16:05:19.075345Z","iopub.status.idle":"2025-11-19T16:05:20.213193Z","shell.execute_reply.started":"2025-11-19T16:05:19.075319Z","shell.execute_reply":"2025-11-19T16:05:20.212477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Compile + Callbacks\nLR = 1e-4\nEPOCHS = 15\n\nmodel_RNET.compile(optimizer=Adam(LR), loss='categorical_crossentropy', metrics=['accuracy'])\n\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, verbose=1)\nearlystop = EarlyStopping(monitor='val_loss', patience=7, restore_best_weights=True, verbose=1)\ncheckpoint = ModelCheckpoint('best_resnet50.h5', monitor='val_loss', save_best_only=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T16:05:30.262282Z","iopub.execute_input":"2025-11-19T16:05:30.26296Z","iopub.status.idle":"2025-11-19T16:05:30.272526Z","shell.execute_reply.started":"2025-11-19T16:05:30.262936Z","shell.execute_reply":"2025-11-19T16:05:30.271758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nhistory = model_RNET.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=EPOCHS,\n    callbacks=[reduce_lr, earlystop, checkpoint]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T16:05:32.100437Z","iopub.execute_input":"2025-11-19T16:05:32.101156Z","iopub.status.idle":"2025-11-19T17:36:02.799317Z","shell.execute_reply.started":"2025-11-19T16:05:32.101132Z","shell.execute_reply":"2025-11-19T17:36:02.798499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test set\n\npred_probs = model_RNET.predict(test_generator)\npred_classes = np.argmax(pred_probs, axis=1)\n\n# Prepare submission\nsubmission_df = pd.DataFrame({\n    'id_code': test_df['id_code'],\n    'diagnosis': pred_classes\n})\n\nsubmission_df.to_csv('submission.csv', index=False)\n# print(\"Submission saved as submission_resnet50.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T17:38:05.691851Z","iopub.execute_input":"2025-11-19T17:38:05.69252Z","iopub.status.idle":"2025-11-19T17:39:56.169459Z","shell.execute_reply.started":"2025-11-19T17:38:05.692488Z","shell.execute_reply":"2025-11-19T17:39:56.168618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Autoencoder","metadata":{}},{"cell_type":"code","source":"# # Data augmentation and ImageDataGenerator setup\n# train_datagen = ImageDataGenerator(\n#     preprocessing_function=tf.keras.applications.resnet.preprocess_input,\n#     rotation_range=20,\n#     width_shift_range=0.2,\n#     height_shift_range=0.2,\n#     shear_range=0.2,\n#     zoom_range=0.2,\n#     horizontal_flip=True,\n#     fill_mode='nearest',\n#     validation_split=0.2 # Use 20% for validation\n#     )\n# ## Autoencoder\n# # Training data generator\n# train_generator_ae = train_datagen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='file_path',\n#     y_col='diagnosis',\n#     target_size=(224, 224),\n#     batch_size=32,\n#     class_mode='input',\n#     subset='training' # 80% of data used for training\n#     )\n# # Validation data generator\n# val_generator_ae = train_datagen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='file_path',\n#     y_col='diagnosis',\n#     target_size=(224, 224),\n#     batch_size=32,\n#     class_mode='input',\n#     subset='validation' # 20% of data used for validation\n#     )\n# ## Classification\n# # Training data generator\n# train_generator = train_datagen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='file_path',\n#     y_col='diagnosis',\n#     target_size=(224, 224),\n#     batch_size=32,\n#     class_mode='categorical',\n#     subset='training' # 80% of data used for training\n#     )\n# # Validation data generator\n# val_generator = train_datagen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='file_path',\n#     y_col='diagnosis',\n#     target_size=(224, 224),\n#     batch_size=32,\n#     class_mode='categorical',\n#     subset='validation' # 20% of data used for validation\n#     )\n# # Pretrain the AutoEncoder\n# # Fine-Tune the AutoEncoder and Make Classification\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T08:10:41.479695Z","iopub.execute_input":"2025-11-19T08:10:41.479977Z","iopub.status.idle":"2025-11-19T08:10:41.614771Z","shell.execute_reply.started":"2025-11-19T08:10:41.479946Z","shell.execute_reply":"2025-11-19T08:10:41.613708Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Make Predictions","metadata":{}},{"cell_type":"code","source":"\n# # Create DataFrame with file paths for the test images\n# test_df = pd.DataFrame({\n# 'file_path': [os.path.join(test_img_dir, filename) for filename in os.listdir(test_img_dir) if filename.endswith('.png')]\n# })\n\n# # Add 'id_code' by extracting it from filenames (removing the .png extension)\n# test_df['id_code'] = test_df['file_path'].apply(lambda x: os.path.basename(x).replace('.png', ''))\n\n# # Image preprocessing for test data\n# test_datagen = ImageDataGenerator(preprocessing_function=tf.keras.applications.resnet.preprocess_input)\n# test_generator = test_datagen.flow_from_dataframe(\n#     dataframe=test_df,\n#     x_col='file_path',\n#     target_size=(224, 224),\n#     batch_size=32,\n#     class_mode=None,\n#     shuffle=False\n#     )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T08:10:41.615958Z","iopub.execute_input":"2025-11-19T08:10:41.616342Z","iopub.status.idle":"2025-11-19T08:10:41.65224Z","shell.execute_reply.started":"2025-11-19T08:10:41.616302Z","shell.execute_reply":"2025-11-19T08:10:41.651311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Predict and save to CSV\n# predicted_classes = tf.argmax(base_model.predict(test_generator), axis=1).numpy()\n# submission_df = pd.DataFrame({'id_code': test_df['id_code'], 'diagnosis': predicted_classes})\n# submission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}