{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 1: Import Libraries and Set Up Paths","metadata":{}},{"cell_type":"code","source":"# --- Block 1: Import Libraries & Define Paths ---\n\n# Core Data Handling & Numerical Operations\nimport pandas as pd\nimport numpy as np\nimport os\n\n# Image Processing & Visualization\nimport cv2\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Deep Learning Framework & Tools\nimport tensorflow as tf\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, BatchNormalization\nfrom tensorflow.keras.applications import DenseNet121\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\n# Model Evaluation\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.utils import class_weight\n\n# --- Define File Paths ---\nBASE_PATH = \"/kaggle/input/aptos2019-blindness-detection/\"\nTRAIN_CSV_PATH = os.path.join(BASE_PATH, \"train.csv\")\nTRAIN_IMG_PATH = os.path.join(BASE_PATH, \"train_images/\")\n\n# Display versions for reproducibility\nprint(\"TensorFlow Version:\", tf.__version__)\nprint(\"Setup Complete. Ready for Block 2.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:26:40.738338Z","iopub.execute_input":"2025-09-15T17:26:40.739309Z","iopub.status.idle":"2025-09-15T17:26:40.745348Z","shell.execute_reply.started":"2025-09-15T17:26:40.739278Z","shell.execute_reply":"2025-09-15T17:26:40.744696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 2: Load and Prepare Data Labels","metadata":{}},{"cell_type":"code","source":"# --- Block 2: Load and Prepare Data Labels ---\n\n# Load the training data CSV file into a pandas DataFrame.\n# A DataFrame is essentially a smart, programmable spreadsheet.\ndf_train = pd.read_csv(TRAIN_CSV_PATH)\n\n# The 'id_code' column in the CSV just has the filename without the extension.\n# We need to add '.png' to each id_code to create the full filename that\n# matches the files in the train_images folder.\n# The .apply() method lets us run a small function on every single row.\ndf_train['id_code'] = df_train['id_code'].apply(lambda x: x + '.png')\n\n# The 'diagnosis' column contains our labels (0, 1, 2, 3, 4).\n# For the ImageDataGenerator we will use later, it's crucial that these\n# labels are treated as categories (strings), not numbers.\ndf_train['diagnosis'] = df_train['diagnosis'].astype(str)\n\n# --- Verification Step ---\n# Let's look at the first 5 rows of our table to verify everything worked.\n# We also print the total number of images we have to train on.\nprint(\"--- Training Data Labels Loaded ---\")\nprint(f\"Total images in the training set: {len(df_train)}\")\nprint(\"\\nFirst 5 entries:\")\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:26:45.903608Z","iopub.execute_input":"2025-09-15T17:26:45.90412Z","iopub.status.idle":"2025-09-15T17:26:45.919977Z","shell.execute_reply.started":"2025-09-15T17:26:45.904089Z","shell.execute_reply":"2025-09-15T17:26:45.919238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 3: Visualize the Class Distribution","metadata":{}},{"cell_type":"code","source":"# --- Block 3: Visualize Class Distribution ---\n\n# First, let's get the counts of each diagnosis.\n# The .value_counts() method is a handy pandas function that counts\n# how many times each unique value appears in the 'diagnosis' column.\n# We use .sort_index() to make sure the classes are in order (0, 1, 2, 3, 4).\nclass_counts = df_train['diagnosis'].value_counts().sort_index()\n\nprint(\"Number of images per class:\")\nprint(class_counts)\n\n# Now, let's create a nice plot to visualize these counts.\nplt.figure(figsize=(10, 6))\nsns.barplot(x=class_counts.index, y=class_counts.values, palette=\"viridis\")\nplt.title('Distribution of Diabetic Retinopathy Classes')\nplt.xlabel('Diagnosis Level')\nplt.ylabel('Number of Images')\nplt.xticks(ticks=[0, 1, 2, 3, 4], labels=['0 - No DR', '1 - Mild', '2 - Moderate', '3 - Severe', '4 - Proliferative'])\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:26:50.44694Z","iopub.execute_input":"2025-09-15T17:26:50.447552Z","iopub.status.idle":"2025-09-15T17:26:50.948235Z","shell.execute_reply.started":"2025-09-15T17:26:50.447506Z","shell.execute_reply":"2025-09-15T17:26:50.94741Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 4: Split the Dataset","metadata":{}},{"cell_type":"code","source":"# --- Block 4: Split the Dataset ---\n\n# We will now split our original, full dataframe into a training set and a validation set.\n# We are using all 3,662 images for maximum performance.\n# test_size=0.2 means 20% of the data will be held back for validation.\n# random_state=42 ensures that we get the exact same split every time we run the code.\n# stratify=df_train['diagnosis'] is the MOST IMPORTANT parameter here. It ensures\n# that the proportion of each class in the training and validation sets is the\n# same as in the original dataset. This is crucial because our data is imbalanced.\ntrain_df, val_df = train_test_split(\n    df_train,\n    test_size=0.2,\n    random_state=42,\n    stratify=df_train['diagnosis']\n)\n\n# --- Verification Step ---\n# Let's check the number of images in our new training and validation sets.\nprint(\"--- Data Split Complete ---\")\nprint(f\"Total training samples: {len(train_df)}\")\nprint(f\"Total validation samples: {len(val_df)}\")\n\nprint(\"\\nDistribution in Training Set:\")\nprint(train_df['diagnosis'].value_counts().sort_index())\n\nprint(\"\\nDistribution in Validation Set:\")\nprint(val_df['diagnosis'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:26:57.704153Z","iopub.execute_input":"2025-09-15T17:26:57.704422Z","iopub.status.idle":"2025-09-15T17:26:57.718454Z","shell.execute_reply.started":"2025-09-15T17:26:57.7044Z","shell.execute_reply":"2025-09-15T17:26:57.717762Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 5: Create Image Data Generators","metadata":{}},{"cell_type":"code","source":"# --- Block 5: Create Image Data Generators ---\n\n# --- Define Image Parameters ---\n# As per our final plan, we use a 224x224 image size. This is crucial for two reasons:\n# 1. It's the standard size for high-performance models like DenseNet, allowing it to see fine details.\n# 2. It will make our final model perfectly compatible with your Streamlit app's pre-processing function.\nIMAGE_SIZE = 224\nBATCH_SIZE = 32 # We will process images in batches of 32\n\n# --- 1. Create the Data Augmentation Generator for Training ---\n# This is our \"secret weapon\". It creates new, slightly modified images from our existing ones in real-time.\n# This prevents the model from just memorizing the training data and forces it to learn the real patterns.\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,          # Normalize pixel values to be between 0 and 1\n    rotation_range=20,       # Randomly rotate images by up to 20 degrees\n    width_shift_range=0.1,   # Randomly shift images horizontally\n    height_shift_range=0.1,  # Randomly shift images vertically\n    shear_range=0.1,         # Apply shearing transformations\n    zoom_range=0.1,          # Randomly zoom in on images\n    horizontal_flip=True,    # Randomly flip images horizontally\n    fill_mode='nearest'      # How to fill in pixels created by rotation or shifting\n)\n\n# --- 2. Create a Simple Generator for Validation ---\n# For our validation (exam) data, we NEVER augment it. We only normalize the pixels.\n# This ensures we get a consistent and honest evaluation of the model's performance.\nval_datagen = ImageDataGenerator(rescale=1./255)\n\n\n# --- 3. Create the Final Generators from our DataFrames ---\n# The .flow_from_dataframe() method connects our pandas DataFrames (the \"map\")\n# to the actual image files on the disk (the \"territory\").\nprint(\"Creating the Training Data Generator (224x224)...\")\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=TRAIN_IMG_PATH,\n    x_col=\"id_code\",          # Column in the dataframe with the filenames\n    y_col=\"diagnosis\",        # Column in the dataframe with the labels\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\"  # We have 5 distinct categories\n)\n\nprint(\"\\nCreating the Validation Data Generator (224x224)...\")\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=TRAIN_IMG_PATH,\n    x_col=\"id_code\",\n    y_col=\"diagnosis\",\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"categorical\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:27:05.675268Z","iopub.execute_input":"2025-09-15T17:27:05.675552Z","iopub.status.idle":"2025-09-15T17:27:16.834004Z","shell.execute_reply.started":"2025-09-15T17:27:05.67553Z","shell.execute_reply":"2025-09-15T17:27:16.833402Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 6: Build the High-Performance Transfer Learning Model","metadata":{}},{"cell_type":"code","source":"# --- Block 6: Build the High-Performance Transfer Learning Model ---\n\n# --- 1. Load the Pre-Trained \"Expert\" Model (DenseNet121) ---\n# We load DenseNet121, which has been pre-trained by experts on the massive ImageNet dataset.\n# `weights='imagenet'` automatically downloads this pre-trained knowledge.\n# `include_top=False` means we are removing its original final layer, because we need to add our own.\nbase_model = DenseNet121(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(IMAGE_SIZE, IMAGE_SIZE, 3)\n)\n\n# --- 2. Freeze the Expert's Knowledge ---\n# We don't want to ruin the valuable knowledge the model already has.\n# So, we \"freeze\" all the layers of the base model to prevent them from changing during our initial training.\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# --- 3. Add Our Own Custom \"Head\" for Retinopathy Classification ---\n# Now, we will add our own new layers on top of the expert base model.\n# These are the only layers that will actually be trained.\nx = base_model.output\nx = GlobalAveragePooling2D()(x) # This layer summarizes the complex features from the base model.\nx = Dense(512, activation='relu')(x)\nx = BatchNormalization()(x)\n# As per the paper's methodology, we add a strong 50% Dropout layer to prevent overfitting.\nx = Dropout(0.5)(x)\n# This is our final output layer, with 5 neurons (one for each class) and a softmax activation.\npredictions = Dense(5, activation='softmax')(x)\n\n# --- 4. Create and Compile the Final Model ---\n# This combines the frozen expert base with our new, trainable head into a single model.\ntransfer_model = Model(inputs=base_model.input, outputs=predictions)\n\n# We use the Adam optimizer with a fine-tuned learning rate.\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.0005)\ntransfer_model.compile(optimizer=optimizer, loss='categorical_crossentropy', metrics=['accuracy'])\n\n# Let's print a summary of our new, powerful model to see its architecture.\nprint(\"--- Transfer Learning Model Summary ---\")\ntransfer_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:27:54.37161Z","iopub.execute_input":"2025-09-15T17:27:54.372302Z","iopub.status.idle":"2025-09-15T17:27:56.437102Z","shell.execute_reply.started":"2025-09-15T17:27:54.37227Z","shell.execute_reply":"2025-09-15T17:27:56.436365Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 7A: Calculate Class Weights","metadata":{}},{"cell_type":"code","source":"# --- Block 7A: Calculate Class Weights ---\n\n# This is our final strategy to fight the data imbalance we saw in Block 3.\n# We give a higher \"weight\" or \"importance\" to the classes with fewer images.\n# This tells the model to pay more attention to the rare classes (like 'Severe') during training.\n\n# Get the true labels from our training dataframe\ny_train_labels = train_df['diagnosis'].astype(int)\n\n# Calculate the weights using sklearn's utility\nclass_weights = class_weight.compute_class_weight(\n    'balanced',\n    classes=np.unique(y_train_labels),\n    y=y_train_labels\n)\n\n# Keras expects the weights in a dictionary format, so we create one.\nclass_weights_dict = dict(enumerate(class_weights))\n\nprint(\"Calculated Class Weights to handle imbalance:\")\nprint(class_weights_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:28:08.844988Z","iopub.execute_input":"2025-09-15T17:28:08.845653Z","iopub.status.idle":"2025-09-15T17:28:08.854747Z","shell.execute_reply.started":"2025-09-15T17:28:08.845619Z","shell.execute_reply":"2025-09-15T17:28:08.853993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 7B: Set Up Callbacks and Train the Model","metadata":{}},{"cell_type":"code","source":"# --- Block 7B: Set Up Callbacks and Train the Model ---\n\n# We will create a list of \"callbacks\" that Keras will use during training.\ncallbacks_list = []\n\n# 1. ModelCheckpoint: The \"Save Your Game\" Assistant\n# This is our most important assistant. It will save the entire model to a file\n# ONLY when its performance on the validation data sets a new \"high score\".\n# --- CHANGE: Saving in .h5 format for compatibility with your Streamlit app ---\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    filepath='best_model.h5',      # The name of the file to save the model to\n    monitor='val_accuracy',        # The metric to watch\n    save_best_only=True,           # Only save if it's the new best score\n    mode='max',                    # We want to maximize accuracy\n    verbose=1                      # Print a message when a new best model is saved\n)\ncallbacks_list.append(checkpoint)\n\n# 2. EarlyStopping: The \"Time Saver\" Assistant\n# This assistant will stop the training automatically if the model's performance\n# doesn't improve for a certain number of epochs (defined by 'patience').\nearly_stopping = tf.keras.callbacks.EarlyStopping(\n    monitor='val_accuracy',\n    patience=10,                   # Wait 10 full epochs for improvement before stopping\n    restore_best_weights=True,     # Automatically load the best model weights at the very end\n    verbose=1\n)\ncallbacks_list.append(early_stopping)\n\n# 3. ReduceLROnPlateau: The \"Smart Nudge\" Assistant\n# If the model's performance plateaus, this assistant will \"nudge\" it by reducing\n# the learning rate, which can help it find a better solution.\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor='val_accuracy',\n    factor=0.2,                    # Reduce learning rate by a factor of 5\n    patience=5,                    # Reduce LR if no improvement for 5 epochs\n    verbose=1,\n    min_lr=0.00001\n)\ncallbacks_list.append(reduce_lr)\n\n\n# --- LAUNCH TRAINING ---\n# We set a high number of epochs, but we fully expect EarlyStopping to find the\n# optimal point and stop the training for us much sooner.\nEPOCHS = 100\n\nprint(\"--- Starting Model Training ---\")\nhistory = transfer_model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    validation_data=val_generator,\n    callbacks=callbacks_list,\n    class_weight=class_weights_dict, # Applying the weights we just calculated\n    steps_per_epoch=len(train_df) // BATCH_SIZE,\n    validation_steps=len(val_df) // BATCH_SIZE\n)\nprint(\"\\n--- Training Complete ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T17:28:20.854548Z","iopub.execute_input":"2025-09-15T17:28:20.854862Z","iopub.status.idle":"2025-09-15T18:28:57.687929Z","shell.execute_reply.started":"2025-09-15T17:28:20.85484Z","shell.execute_reply":"2025-09-15T18:28:57.687009Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Block 8: Comprehensive Model Evaluation","metadata":{}},{"cell_type":"markdown","source":"Block 9: Save and Download Final Model","metadata":{}},{"cell_type":"code","source":"# --- Block 9: Save and Download Final Model ---\n\nfrom IPython.display import FileLink\n\n# The 'best_model' variable still holds our best trained model, which we loaded\n# and evaluated in the previous cell.\n# We will now save this model again, but in the specific .h5 format required by your app.\nH5_MODEL_NAME = \"retina_model_for_app_71.08.h5\"\n\nprint(f\"Saving the final model in .h5 format as: {H5_MODEL_NAME}\")\nbest_model.save(H5_MODEL_NAME)\nprint(\"Save complete.\")\n\n# --- Create a download link for the saved model ---\n# This will generate a blue, clickable link in the output of this cell.\nprint(\"\\nYour final model is ready for download.\")\nprint(\"Click the link below to download your model:\")\nFileLink(H5_MODEL_NAME)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T18:48:49.858118Z","iopub.execute_input":"2025-09-15T18:48:49.858967Z","iopub.status.idle":"2025-09-15T18:48:50.554492Z","shell.execute_reply.started":"2025-09-15T18:48:49.858938Z","shell.execute_reply":"2025-09-15T18:48:50.553759Z"}},"outputs":[],"execution_count":null}]}