{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.369Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_dir = '/kaggle/input/aptos2019-blindness-detection'\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.37Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor dirname, _, filenames in os.walk(root_dir):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**List Directories and printing only first 3 files**","metadata":{}},{"cell_type":"code","source":"import os\n\ndata_dir = \"/kaggle/input/aptos2019-blindness-detection\"\n\nfor dirname, _, filenames in os.walk(data_dir):\n    print(f\"Directory: {dirname}\")\n    for filename in filenames[:3]: \n        print(f\"  File: {filename}\")","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Data Visualization**\n","metadata":{}},{"cell_type":"code","source":"# visualize 3 images from the training set using templete for homework 6\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\ntrain_path = \"/kaggle/input/aptos2019-blindness-detection/train_images\"\ntrain_files = [os.path.join(train_path, f) for f in os.listdir(train_path)[:3]]  \n\nnum_images = 3\nfig, axes = plt.subplots(1, 3, figsize=(12, 12))\n\nfor i, ax in enumerate(axes.flat):\n    if i < num_images:\n        image = Image.open(train_files[i])\n        ax.imshow(image)\n        ax.set_title(f'Train Image {i+1}')\n        ax.axis('off')\n    else:\n        ax.axis('off')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#visualize 3 images from the test set using templete for homework 5\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Set path\ntest_path = \"/kaggle/input/aptos2019-blindness-detection/test_images\"\ntest_files = sorted([os.path.join(test_path, f) for f in os.listdir(test_path) if f.endswith('.png')])[:3]\n\nnum_images = len(test_files)\nfig, axes = plt.subplots(num_images, 1, figsize=(15, 4 * num_images))\n\nif num_images == 1:\n    axes = [axes] \n\nfor i, ax in enumerate(axes):\n    image = Image.open(test_files[i]).convert('L') \n    image_array = np.array(image).flatten() \n\n    ax.plot(image_array) \n    ax.set_title(f'Flattened Pixel Signal - Test Image {i+1}')\n    ax.set_xlabel('Pixel Index')\n    ax.set_ylabel('Pixel Intensity')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Build a CNN Model with Transfer Learning**","metadata":{}},{"cell_type":"markdown","source":"**Define the Data Generators**","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Paths\nroot_dir = '/kaggle/input/aptos2019-blindness-detection'\ntrain_img_dir = os.path.join(root_dir, 'train_images')\ntest_img_dir = os.path.join(root_dir, 'test_images')\n\n# Load the train and test CSV files\ntrain_df = pd.read_csv(os.path.join(root_dir, 'train.csv'))\ntest_df = pd.read_csv(os.path.join(root_dir, 'test.csv'))\n\n# Add 'file_path' column to the dataframe\ntrain_df['file_path'] = train_df['id_code'].apply(lambda x: os.path.join(train_img_dir, f\"{x}.png\"))\ntest_df['file_path'] = test_df['id_code'].apply(lambda x: os.path.join(test_img_dir, f\"{x}.png\"))\n\n# Ensure 'diagnosis' column is treated as a string\ntrain_df['diagnosis'] = train_df['diagnosis'].astype(str)\n\n# Data augmentation for training and validation\ntrain_datagen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    validation_split=0.2  # Use 20% for validation\n)\n\n# Training data generator\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical',\n    subset='training'  # 80% of data used for training\n)\n\n# Validation data generator\nval_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='file_path',\n    y_col='diagnosis',\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode='categorical',\n    subset='validation'  # 20% of data used for validation\n)\n\n# Test data generator (no labels)\ntest_datagen = ImageDataGenerator(\n    preprocessing_function=tf.keras.applications.efficientnet.preprocess_input\n)\ntest_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    x_col='file_path',\n    y_col=None,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode=None,\n    shuffle=False\n)\n\n# Print summary of each generator for confirmation\nprint(f\"Training samples: {train_generator.samples}\")\nprint(f\"Validation samples: {val_generator.samples}\")\nprint(f\"Test samples: {test_generator.samples}\")\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Load EfficientNetB0 and Add Custom Layers**","metadata":{}},{"cell_type":"markdown","source":"**Define EfficientNetB0 Model**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\n\n# Load EfficientNetB0 model with pre-trained weights from ImageNet\nbase_model = EfficientNetB0(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# Freeze all layers initially\nbase_model.trainable = False\n\n# Add custom layers on top of the base model\nx = base_model.output\nx = GlobalAveragePooling2D()(x)  # Reduce the feature map to a single vector\nx = Dropout(0.5)(x)  # Regularization to avoid overfitting\nx = Dense(256, activation='relu')(x)  # Fully connected layer\nx = Dropout(0.5)(x)  # Regularization\npredictions = Dense(5, activation='softmax')(x)  # 5 classes for diagnosis\n\n# Create the model\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\n# Compile the model\nmodel.compile(optimizer=Adam(learning_rate=1e-4), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Print model summary\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train EfficientNetB0\n\neff_history = model.fit(\n    train_generator,\n    epochs=10,\n    validation_data=val_generator\n)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Define ResNet50 Model**","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\n\n# Load ResNet50 model with pre-trained weights\nresnet_base = ResNet50(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# Freeze all layers initially\nresnet_base.trainable = False\n\n# Add custom classification layers\ny = resnet_base.output\ny = GlobalAveragePooling2D()(y)\ny = Dropout(0.5)(y)\ny = Dense(256, activation='relu')(y)\ny = Dropout(0.5)(y)\nresnet_predictions = Dense(5, activation='softmax')(y)\n\n# Create ResNet50 model\nresnet_model = Model(inputs=resnet_base.input, outputs=resnet_predictions)\n\n# Compile the model\nresnet_model.compile(optimizer=Adam(learning_rate=1e-4), loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Print model summary\nresnet_model.summary()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train ResNet50**","metadata":{}},{"cell_type":"code","source":"resnet_history = resnet_model.fit(\n    train_generator,\n    epochs=10,\n    validation_data=val_generator\n)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"** Plot Accuracy Comparison**","metadata":{}},{"cell_type":"code","source":"plt.plot(eff_history.history['val_accuracy'], label='EfficientNetB0')\nplt.plot(resnet_history.history['val_accuracy'], label='ResNet50')\nplt.title('Validation Accuracy Comparison')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Accuracy\nimport matplotlib.pyplot as plt\n\nplt.plot(eff_history.history['val_accuracy'], label='EfficientNetB0')\nplt.plot(resnet_history.history['val_accuracy'], label='ResNet50')\nplt.title('Validation Accuracy Comparison')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Train the Model**","metadata":{}},{"cell_type":"markdown","source":"**Plot Learning Curves**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot accuracy\nplt.plot(history.history['accuracy'], label='Training Accuracy')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\n# Plot loss\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-21T19:47:50.374Z"}},"outputs":[],"execution_count":null}]}