{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"91b57367-4587-4b8b-885c-7e610ce760e8","cell_type":"code","source":"\n\n# ## Cell 1: Imports and Setup\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix, cohen_kappa_score\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\nprint(f\"TensorFlow Version: {tf.__version__}\")\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    print(f\"GPU(s) available: {len(gpus)}\")\nelse:\n    print(\"No GPU detected.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"c613f035","cell_type":"code","source":"# ## Cell 2: Configuration\n\n# Paths to your Kaggle data\nTRAIN_CSV = '/kaggle/input/aptos2019-blindness-detection/train.csv'\nTRAIN_IMG_DIR = '/kaggle/input/aptos2019-blindness-detection/train_images'\n\n# Model and training hyperparameters\nIMG_SIZE = (384, 384)\nBATCH_SIZE = 16\nSEED = 42\nEPOCHS = 25  # Increased from 20 to allow more time for augmentation\nNUM_CLASSES = 5\n\n# Set seeds for reproducibility\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"59c80787","cell_type":"code","source":"# ## Cell 3: Load and Prepare Data\n\n# Load the CSV and ensure the diagnosis column is a string for the generator\ndf = pd.read_csv(TRAIN_CSV)\ndf['id_code'] = df['id_code'].astype(str) + '.png'\n\n# This line fixes the error by converting the diagnosis labels to strings\ndf['diagnosis'] = df['diagnosis'].astype(str)\n\nprint(\"Data loaded successfully.\")\nprint(f\"Total samples: {len(df)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"bf63b42f","cell_type":"code","source":"# ## Cell 4: Quick EDA\n# Visualize the class distribution\nplt.figure(figsize=(10, 6))\ndf['diagnosis'].value_counts().sort_index().plot(kind='bar', color='skyblue')\nplt.title('Class Distribution of Diabetic Retinopathy')\nplt.xlabel('Diagnosis Level')\nplt.ylabel('Number of Images')\nplt.xticks(rotation=0)\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"851b7428","cell_type":"code","source":"# ## Cell 5: Train/Validation Split\n# Create a stratified split to maintain class distribution in both sets\ntrain_df, val_df = train_test_split(\n    df,\n    test_size=0.15,\n    random_state=SEED,\n    stratify=df['diagnosis']\n)\n\nprint(f\"Training samples: {len(train_df)}\")\nprint(f\"Validation samples: {len(val_df)}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"4fe1184e","cell_type":"code","source":"# ## Cell 6: Data Generators (with Augmentations)\n\n# Add basic augmentations to the training generator to combat overfitting\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=15,\n    horizontal_flip=True,\n    zoom_range=0.1,\n    width_shift_range=0.1,\n    height_shift_range=0.1\n)\n\n# The validation generator should NOT have augmentations\nval_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=True,\n    seed=SEED\n)\n\nval_gen = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=TRAIN_IMG_DIR,\n    x_col='id_code',\n    y_col='diagnosis',\n    target_size=IMG_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='sparse',\n    shuffle=False\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"238680f6","cell_type":"code","source":"# ## Cell 7: Calculate Class Weights\n# Correctly compute class weights on the integer labels to handle imbalance\nclasses = np.unique(train_df['diagnosis'])\nclass_weights = compute_class_weight(\n    'balanced',\n    classes=classes,\n    y=train_df['diagnosis'].values\n)\nclass_weight_dict = {c: w for c, w in zip(classes, class_weights)}\n\nprint(\"Class weights calculated:\")\nprint(class_weight_dict)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"4be07d15","cell_type":"code","source":"# ## Cell 8: Build the Model\ndef build_simple_model(input_shape=IMG_SIZE + (3,), n_classes=NUM_CLASSES):\n    \"\"\"Builds a simple, robust EfficientNetB0 model for single-phase training.\"\"\"\n    # Base model\n    base = EfficientNetB0(\n        include_top=False,\n        weights='imagenet',\n        input_shape=input_shape\n    )\n    base.trainable = True # Train the whole model\n\n    # Model architecture\n    inputs = layers.Input(shape=input_shape)\n    x = base(inputs, training=True)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.3)(x)\n    outputs = layers.Dense(n_classes, activation='softmax')(x)\n    model = Model(inputs, outputs)\n\n    # Compile the model with the correct loss function\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n        loss='sparse_categorical_crossentropy', # The correct loss for 'sparse' mode\n        metrics=['accuracy']\n    )\n    return model\n\nmodel = build_simple_model()\nmodel.summary()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"f215fac4","cell_type":"code","source":"# ## Cell 9: Define Callbacks (with Gentler LR Reduction)\ncheckpoint = ModelCheckpoint(\n    'best_model.h5',\n    monitor='val_accuracy',\n    save_best_only=True,\n    mode='max',\n    verbose=1\n)\n\nearly_stopping = EarlyStopping(\n    monitor='val_accuracy',\n    patience=5, # Stop if no improvement for 5 epochs\n    restore_best_weights=True,\n    mode='max',\n    verbose=1\n)\n\nreduce_lr = ReduceLROnPlateau(\n    monitor='val_loss',\n    factor=0.5,  # Changed to 0.5 for a gentler reduction\n    patience=2,\n    verbose=1,\n    min_lr=1e-7\n)\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"171f379d","cell_type":"code","source":"# ## Cell 10: Train the Model\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=EPOCHS,\n    class_weight=class_weight_dict,\n    callbacks=[checkpoint, early_stopping, reduce_lr],\n    verbose=1\n)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"9234955d","cell_type":"code","source":"# ## Cell 11: Plot Training History\ndef plot_history(history):\n    \"\"\"Plots accuracy and loss curves for training and validation.\"\"\"\n    fig, ax = plt.subplots(1, 2, figsize=(16, 6))\n\n    # Plot accuracy\n    ax[0].plot(history.history['accuracy'], label='Train Accuracy')\n    ax[0].plot(history.history['val_accuracy'], label='Validation Accuracy')\n    ax[0].set_title('Model Accuracy')\n    ax[0].set_xlabel('Epoch')\n    ax[0].set_ylabel('Accuracy')\n    ax[0].legend()\n\n    # Plot loss\n    ax[1].plot(history.history['loss'], label='Train Loss')\n    ax[1].plot(history.history['val_loss'], label='Validation Loss')\n    ax[1].set_title('Model Loss')\n    ax[1].set_xlabel('Epoch')\n    ax[1].set_ylabel('Loss')\n    ax[1].legend()\n\n    plt.show()\n\nplot_history(history)\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"fe760e26","cell_type":"code","source":"# ## Cell 12: Evaluate the Model\n# Load the best performing model\nmodel.load_weights('best_model.h5')\n\n# Make predictions on the validation set\npreds = model.predict(val_gen)\npred_classes = np.argmax(preds, axis=1)\n\n# Get true labels directly from the generator for robust evaluation\ntrue_classes = val_gen.classes\n\n# Calculate Quadratic Weighted Kappa\nqwk = cohen_kappa_score(true_classes, pred_classes, weights='quadratic')\nprint(f\"\\n📈 Validation Quadratic Weighted Kappa (QWK): {qwk:.4f}\\n\")\n\n# Print Classification Report\nprint(\"📊 Classification Report:\\n\")\nprint(classification_report(true_classes, pred_classes, target_names=[str(i) for i in classes]))\n\n# Display Confusion Matrix\ncm = confusion_matrix(true_classes, pred_classes)\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=classes, yticklabels=classes)\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"13eefbaa-9d59-4dab-b02a-12422ce91604","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}