{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":12087253,"sourceType":"datasetVersion","datasetId":7608978},{"sourceId":12094484,"sourceType":"datasetVersion","datasetId":7613658}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import data handling tools\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Import machine learning libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\n\n# TensorFlow & Keras libraries\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping\nimport os\nimport random\n\n# Ignore Warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nprint('Modules loaded')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:05:02.790665Z","iopub.execute_input":"2025-10-29T09:05:02.791652Z","iopub.status.idle":"2025-10-29T09:05:02.925601Z","shell.execute_reply.started":"2025-10-29T09:05:02.791625Z","shell.execute_reply":"2025-10-29T09:05:02.925000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set seeds and deterministic behavior\nnp.random.seed(2025)\ntf.random.set_seed(2025)\nrandom.seed(2025)\nos.environ['PYTHONHASHSEED'] = str(2025)\nos.environ['TF_DETERMINISTIC_OPS'] = '1'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:05:04.498584Z","iopub.execute_input":"2025-10-29T09:05:04.499335Z","iopub.status.idle":"2025-10-29T09:05:04.503416Z","shell.execute_reply.started":"2025-10-29T09:05:04.499306Z","shell.execute_reply":"2025-10-29T09:05:04.502631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\n\n# Paths\ntrain_dir = \"/kaggle/input/aptos2019-blindness-detection/train_images\"\ntrain_csv = \"/kaggle/input/aptos2019-blindness-detection/train.csv\"\n\n# Load CSV\ndf = pd.read_csv(train_csv)\n\n# Image size\nIMG_SIZE = 224\n\n# Arrays to store data\nX = []\ny = []\n\n# Load and preprocess images\nfor i, row in tqdm(df.iterrows(), total=len(df)):\n    img_path = os.path.join(train_dir, row['id_code'] + \".png\")\n    \n    # Load and resize image to (224, 224)\n    img = load_img(img_path, target_size=(IMG_SIZE, IMG_SIZE))\n    \n    # Convert to array and normalize to [0,1]\n    img_array = img_to_array(img) / 255.0\n    \n    X.append(img_array)\n    y.append(row['diagnosis'])\n\n# Convert lists to numpy arrays\nX = np.array(X, dtype=np.float32)\ny = np.array(y, dtype=np.int32)\n\nprint(\"✅ Done loading!\")\nprint(\"X shape:\", X.shape)\nprint(\"y shape:\", y.shape)\nprint(\"Pixel value range:\", X.min(), \"to\", X.max())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:06:19.640364Z","iopub.execute_input":"2025-10-29T09:06:19.641015Z","iopub.status.idle":"2025-10-29T09:13:52.510071Z","shell.execute_reply.started":"2025-10-29T09:06:19.640991Z","shell.execute_reply":"2025-10-29T09:13:52.509319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print the shape of X\nprint(\"Shape of X:\", X.shape)\n# print the shape of y\nprint(\"Shape of y:\", y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T07:47:20.331717Z","iopub.execute_input":"2025-10-29T07:47:20.331978Z","iopub.status.idle":"2025-10-29T07:47:20.336418Z","shell.execute_reply.started":"2025-10-29T07:47:20.331954Z","shell.execute_reply":"2025-10-29T07:47:20.335707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the minimum pixel value in the entire image\nmin_value = np.min(X)\n\n# Get the maximum pixel value in the entire image\nmax_value = np.max(X)\n\nprint(f\"Minimum pixel value: {min_value}\")\nprint(f\"Maximum pixel value: {max_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:14:57.033814Z","iopub.execute_input":"2025-10-29T09:14:57.034146Z","iopub.status.idle":"2025-10-29T09:14:57.867701Z","shell.execute_reply.started":"2025-10-29T09:14:57.034122Z","shell.execute_reply":"2025-10-29T09:14:57.867003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport random\n\n# Number of images to display\nnum_images = 16\nrows, cols = 4, 4\n\n# Randomly sample 16 indices\nrandom_indices = random.sample(range(len(X)), num_images)\n\n# Plotting\nplt.figure(figsize=(12, 12))\nfor i, idx in enumerate(random_indices):\n    plt.subplot(rows, cols, i + 1)\n    plt.imshow(X[idx])\n    plt.title(f\"Label: {y[idx]}\")\n    plt.axis('off')\nplt.suptitle(\"Random 16 Images from Preprocessed Dataset\", fontsize=16)\nplt.tight_layout()\nplt.subplots_adjust(top=0.93)  # Adjust to fit suptitle\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:14:49.326474Z","iopub.execute_input":"2025-10-29T09:14:49.327079Z","iopub.status.idle":"2025-10-29T09:14:50.928075Z","shell.execute_reply.started":"2025-10-29T09:14:49.327027Z","shell.execute_reply":"2025-10-29T09:14:50.926965Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Splitting the Dataset","metadata":{}},{"cell_type":"code","source":"# First split: 80% temp, 20% validation\nX_temp, X_val, y_temp, y_val = train_test_split(\n    X, y, test_size=0.20, random_state=2025, stratify=y\n)\n\n# Second split: from train_val, split into 70% training and 10% testing (i.e., 1/8 for testing)\nX_train, X_test, y_train, y_test = train_test_split(\n    X_temp, y_temp, test_size=1/8, random_state=2025, stratify=y_temp\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:09.698511Z","iopub.execute_input":"2025-10-29T09:15:09.699031Z","iopub.status.idle":"2025-10-29T09:15:10.699068Z","shell.execute_reply.started":"2025-10-29T09:15:09.699007Z","shell.execute_reply":"2025-10-29T09:15:10.698498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\ny_train = to_categorical(y_train, num_classes=5)\ny_val = to_categorical(y_val, num_classes=5)\ny_test = to_categorical(y_test, num_classes=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:12.499295Z","iopub.execute_input":"2025-10-29T09:15:12.499771Z","iopub.status.idle":"2025-10-29T09:15:12.509892Z","shell.execute_reply.started":"2025-10-29T09:15:12.499746Z","shell.execute_reply":"2025-10-29T09:15:12.508922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"y_train shape after one-hot encoding: {y_train.shape}\")\nprint(f\"y_val shape after one-hot encoding: {y_val.shape}\")\nprint(f\"y_test shape after one-hot encoding: {y_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:14.108883Z","iopub.execute_input":"2025-10-29T09:15:14.109500Z","iopub.status.idle":"2025-10-29T09:15:14.113451Z","shell.execute_reply.started":"2025-10-29T09:15:14.109472Z","shell.execute_reply":"2025-10-29T09:15:14.112661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:16.006662Z","iopub.execute_input":"2025-10-29T09:15:16.007178Z","iopub.status.idle":"2025-10-29T09:15:16.270022Z","shell.execute_reply.started":"2025-10-29T09:15:16.007151Z","shell.execute_reply":"2025-10-29T09:15:16.269440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check the number of samples\nprint(\"Number of training samples:\", X_train.shape[0])\nprint(\"Number of validation samples:\", X_val.shape[0])\nprint(\"Number of test samples:\", X_test.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:17.833550Z","iopub.execute_input":"2025-10-29T09:15:17.833818Z","iopub.status.idle":"2025-10-29T09:15:17.837925Z","shell.execute_reply.started":"2025-10-29T09:15:17.833796Z","shell.execute_reply":"2025-10-29T09:15:17.837299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check the shape of samples\nprint(\"Shape of training samples:\", X_train.shape)\nprint(\"Shape of validation samples:\", X_val.shape)\nprint(\"Shape of test samples:\", X_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:19.546084Z","iopub.execute_input":"2025-10-29T09:15:19.546896Z","iopub.status.idle":"2025-10-29T09:15:19.550958Z","shell.execute_reply.started":"2025-10-29T09:15:19.546862Z","shell.execute_reply":"2025-10-29T09:15:19.550303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T09:15:24.138521Z","iopub.execute_input":"2025-10-29T09:15:24.139017Z","iopub.status.idle":"2025-10-29T09:15:24.374787Z","shell.execute_reply.started":"2025-10-29T09:15:24.138995Z","shell.execute_reply":"2025-10-29T09:15:24.374055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.densenet import DenseNet121\n\ndensenet = DenseNet121(\n    weights='/kaggle/input/densenet-bc-121-32-no-top-h5/DenseNet-BC-121-32-no-top.h5',\n    # weights='imagenet',\n    include_top=False,\n    input_shape=(224,224,3)\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# densenet.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T08:18:49.808726Z","iopub.execute_input":"2025-06-10T08:18:49.808988Z","iopub.status.idle":"2025-06-10T08:18:49.812661Z","shell.execute_reply.started":"2025-06-10T08:18:49.808968Z","shell.execute_reply":"2025-06-10T08:18:49.812054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Activation, Dropout, BatchNormalization, GlobalAveragePooling2D, LeakyReLU\nfrom tensorflow.keras import regularizers\ndef build_model():\n    model = Sequential()\n    model.add(densenet)\n    model.add(GlobalAveragePooling2D())\n    model.add(Dropout(0.5))\n    model.add(BatchNormalization()) \n    model.add(Dense(5, activation='softmax', kernel_regularizer=regularizers.l2(0.01)))\n    \n    model.compile(\n        loss='categorical_crossentropy',\n        optimizer=Adam(learning_rate=0.00005),  # Added momentum for better convergence\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T07:48:23.806990Z","iopub.execute_input":"2025-10-29T07:48:23.807697Z","iopub.status.idle":"2025-10-29T07:48:23.812580Z","shell.execute_reply.started":"2025-10-29T07:48:23.807670Z","shell.execute_reply":"2025-10-29T07:48:23.811885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = build_model()\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-29T07:48:26.783460Z","iopub.execute_input":"2025-10-29T07:48:26.783729Z","iopub.status.idle":"2025-10-29T07:48:26.851164Z","shell.execute_reply.started":"2025-10-29T07:48:26.783708Z","shell.execute_reply":"2025-10-29T07:48:26.850619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\n# Early Stopping - monitor val_accuracy\n# es = EarlyStopping(\n#     monitor='val_accuracy', \n#     min_delta=0.001,  # Smaller delta for accuracy (it's a percentage)\n#     patience=5, \n#     verbose=1, \n#     mode='max',  # We want to maximize accuracy\n#     restore_best_weights=True\n# )\n\n# Model Checkpoint - monitor val_accuracy  \nmc = ModelCheckpoint(\n    monitor='val_accuracy', \n    filepath='augment_best_model.keras', \n    verbose=1, \n    save_best_only=True, \n    mode='max'  # We want to maximize accuracy\n)\n\n# Reduce LR on Plateau - monitor val_accuracy\nrlr = ReduceLROnPlateau(\n    monitor='val_accuracy',     # Monitor validation accuracy\n    factor=0.5,                 # Reduce LR by a factor of 2\n    patience=3,                 # Wait for 3 epochs with no improvement\n    verbose=1,                  # Print updates\n    mode='max',                 # We're trying to maximize accuracy\n    min_delta=0.0001,           # Minimum change to qualify as improvement\n    cooldown=1,                 # Wait 1 epoch before resuming normal operation after LR change\n    min_lr=1e-7                 # Do not reduce below this learning rate\n)\n\n# List of Callbacks\ncd = [mc, rlr]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model and store history\nhistory = model.fit(\n    X_train,\n    y_train,\n    validation_data=(X_val, y_val),\n    epochs=50,\n    batch_size=32,\n    callbacks=cd\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate Model","metadata":{}},{"cell_type":"code","source":"import pickle\n\n# Save history to Kaggle's output directory\nwith open('/kaggle/working/augment_model_history.pkl', 'wb') as f:\n    pickle.dump(history.history, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T09:27:11.482479Z","iopub.execute_input":"2025-06-10T09:27:11.48345Z","iopub.status.idle":"2025-06-10T09:27:11.487536Z","shell.execute_reply.started":"2025-06-10T09:27:11.483412Z","shell.execute_reply":"2025-06-10T09:27:11.48692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to plot accuracy history\nimport matplotlib.pyplot as plt\ndef plot_acc_hist(hist):\n    plt.plot(hist.history[\"accuracy\"])\n    plt.plot(hist.history[\"val_accuracy\"])\n    plt.title(\"Model Accuracy\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Train\", \"Validation\"], loc=\"upper left\")\n    plt.show()\n\n# Plot training history\nplot_acc_hist(history)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef plot_loss_hist(hist):\n    plt.plot(hist.history[\"loss\"])\n    plt.plot(hist.history[\"val_loss\"])\n    plt.title(\"Model Loss\")\n    plt.ylabel(\"loss\")\n    plt.xlabel(\"epoch\")\n    plt.legend([\"train\", \"validation\"], loc=\"upper left\")\n    plt.show()\nplot_loss_hist(history)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}