{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":14774}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:28.351501Z","iopub.execute_input":"2026-10-03T10:56:28.351877Z","iopub.status.idle":"2026-10-03T10:56:37.488163Z","shell.execute_reply.started":"2026-10-03T10:56:28.351842Z","shell.execute_reply":"2026-10-03T10:56:37.4873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.1 — libraries\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\nimport warnings\nwarnings.filterwarnings('ignore')\nsns.set_style('darkgrid')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:37.489559Z","iopub.execute_input":"2026-10-03T10:56:37.490033Z","iopub.status.idle":"2026-10-03T10:56:38.529348Z","shell.execute_reply.started":"2026-10-03T10:56:37.490009Z","shell.execute_reply":"2026-10-03T10:56:38.528546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.2 — dekho files kahan hain\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(dirname, '->', len(filenames), 'files')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:38.530384Z","iopub.execute_input":"2026-10-03T10:56:38.530955Z","iopub.status.idle":"2026-10-03T10:56:41.627089Z","shell.execute_reply.started":"2026-10-03T10:56:38.530919Z","shell.execute_reply":"2026-10-03T10:56:41.626292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.3 — training labels (corrected path)\nBASE = '/kaggle/input/competitions/aptos2019-blindness-detection'\ntrain_df = pd.read_csv(f'{BASE}/train.csv')\n\nprint(\"Shape:\", train_df.shape)\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:41.628097Z","iopub.execute_input":"2026-10-03T10:56:41.628405Z","iopub.status.idle":"2026-10-03T10:56:41.672632Z","shell.execute_reply.started":"2026-10-03T10:56:41.628383Z","shell.execute_reply":"2026-10-03T10:56:41.67205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.4 — grade 0 se 4 tak, kya matlab hai\nlabel_names = {\n    0: 'No DR',\n    1: 'Mild',\n    2: 'Moderate',\n    3: 'Severe',\n    4: 'Proliferative DR'\n}\ntrain_df['diagnosis_name'] = train_df['diagnosis'].map(label_names)\ntrain_df['diagnosis'].value_counts().sort_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:41.674386Z","iopub.execute_input":"2026-10-03T10:56:41.675004Z","iopub.status.idle":"2026-10-03T10:56:41.698331Z","shell.execute_reply.started":"2026-10-03T10:56:41.674974Z","shell.execute_reply":"2026-10-03T10:56:41.697565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.5 — kaunsi class kitni hai\nplt.figure(figsize=(9,5))\norder = [label_names[i] for i in range(5)]\nsns.countplot(x='diagnosis_name', data=train_df, order=order, palette='viridis')\nplt.title('Diabetic Retinopathy — Class Distribution', fontsize=14, weight='bold')\nplt.xlabel('Severity'); plt.ylabel('Number of images')\nplt.xticks(rotation=15)\nplt.show()\n\nprint((train_df['diagnosis_name'].value_counts(normalize=True).round(3) * 100))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:41.699384Z","iopub.execute_input":"2026-10-03T10:56:41.699701Z","iopub.status.idle":"2026-10-03T10:56:41.936494Z","shell.execute_reply.started":"2026-10-03T10:56:41.69967Z","shell.execute_reply":"2026-10-03T10:56:41.935833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.6 — har class ki sample images\ndef load_image(image_id):\n    path = f'{BASE}/train_images/{image_id}.png'\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)   # cv2 BGR deta hai, RGB me badlo\n    return img\n\nfig, axes = plt.subplots(5, 3, figsize=(12, 18))\nfor cls in range(5):\n    sample = train_df[train_df['diagnosis'] == cls].sample(3, random_state=42)\n    for j, (_, row) in enumerate(sample.iterrows()):\n        img = load_image(row['id_code'])\n        axes[cls, j].imshow(img)\n        axes[cls, j].set_title(label_names[cls], fontsize=11)\n        axes[cls, j].axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:41.937359Z","iopub.execute_input":"2026-10-03T10:56:41.937644Z","iopub.status.idle":"2026-10-03T10:56:48.029094Z","shell.execute_reply.started":"2026-10-03T10:56:41.937621Z","shell.execute_reply":"2026-10-03T10:56:48.028101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1.7 — dimensions kitne alag hain\nsizes = []\nfor image_id in train_df['id_code'].sample(30, random_state=1):\n    h, w = load_image(image_id).shape[:2]\n    sizes.append((h, w))\n\nsizes = pd.DataFrame(sizes, columns=['height', 'width'])\nprint(sizes.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:48.030107Z","iopub.execute_input":"2026-10-03T10:56:48.030412Z","iopub.status.idle":"2026-10-03T10:56:51.209062Z","shell.execute_reply.started":"2026-10-03T10:56:48.03039Z","shell.execute_reply":"2026-10-03T10:56:51.208305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2.1 — preprocessing functions\nimport numpy as np\n\ndef crop_image_from_gray(img, tol=7):\n    # retina ke around ka kaala border kaat do\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n    mask = gray > tol\n    if mask.sum() == 0:            # poori image black — waise hi rehne do\n        return img\n    r = img[:, :, 0][np.ix_(mask.any(1), mask.any(0))]\n    g = img[:, :, 1][np.ix_(mask.any(1), mask.any(0))]\n    b = img[:, :, 2][np.ix_(mask.any(1), mask.any(0))]\n    return np.stack([r, g, b], axis=-1)\n\ndef preprocess_image(image_id, img_size=224, sigmaX=10):\n    # load -> crop border -> resize -> Ben Graham -> circle crop\n    img = cv2.imread(f'{BASE}/train_images/{image_id}.png')\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (img_size, img_size))\n    # Ben Graham: lesions ubharte hain + lighting normalize hoti hai\n    img = cv2.addWeighted(img, 4, cv2.GaussianBlur(img, (0, 0), sigmaX), -4, 128)\n    # circle crop: corners ka bacha noise hata do\n    h, w = img.shape[:2]\n    mask = np.zeros((h, w), np.uint8)\n    cv2.circle(mask, (w // 2, h // 2), min(h, w) // 2, 1, -1)\n    return img * mask[..., np.newaxis]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:51.210004Z","iopub.execute_input":"2026-10-03T10:56:51.210355Z","iopub.status.idle":"2026-10-03T10:56:51.218954Z","shell.execute_reply.started":"2026-10-03T10:56:51.210304Z","shell.execute_reply":"2026-10-03T10:56:51.218229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2.2 — before/after comparison\nimport matplotlib.pyplot as plt\n\nsample_ids = train_df.sample(4, random_state=7)['id_code'].values\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\nfor i, image_id in enumerate(sample_ids):\n    raw  = cv2.cvtColor(cv2.imread(f'{BASE}/train_images/{image_id}.png'), cv2.COLOR_BGR2RGB)\n    proc = preprocess_image(image_id)\n    axes[0, i].imshow(raw);  axes[0, i].set_title('Before'); axes[0, i].axis('off')\n    axes[1, i].imshow(proc); axes[1, i].set_title('After');  axes[1, i].axis('off')\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:51.219893Z","iopub.execute_input":"2026-10-03T10:56:51.220312Z","iopub.status.idle":"2026-10-03T10:56:54.98258Z","shell.execute_reply.started":"2026-10-03T10:56:51.220243Z","shell.execute_reply":"2026-10-03T10:56:54.981542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2.3 — ek jagah size define, taaki poore project me consistent rahe\nIMG_SIZE = 224","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:54.983636Z","iopub.execute_input":"2026-10-03T10:56:54.983993Z","iopub.status.idle":"2026-10-03T10:56:54.987988Z","shell.execute_reply.started":"2026-10-03T10:56:54.983964Z","shell.execute_reply":"2026-10-03T10:56:54.987176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2.4 — process + save (ek baar chalega)\nimport os\nfrom tqdm.auto import tqdm\n\nOUT_DIR = '/kaggle/working/train_processed'\nos.makedirs(OUT_DIR, exist_ok=True)\n\nfor image_id in tqdm(train_df['id_code'].values):\n    img = preprocess_image(image_id, img_size=IMG_SIZE)\n    # cv2.imwrite BGR chahta hai, isliye wapas convert\n    cv2.imwrite(f'{OUT_DIR}/{image_id}.png', cv2.cvtColor(img, cv2.COLOR_RGB2BGR))\n\nprint(\"Done! Saved\", len(os.listdir(OUT_DIR)), \"images to\", OUT_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T10:56:54.989021Z","iopub.execute_input":"2026-10-03T10:56:54.98931Z","iopub.status.idle":"2026-10-03T11:06:51.789919Z","shell.execute_reply.started":"2026-10-03T10:56:54.989282Z","shell.execute_reply":"2026-10-03T11:06:51.789306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- reusable figure saver — poore project me kaam aayega ----\nimport os\nos.makedirs('/kaggle/working/figures', exist_ok=True)\n\ndef save_fig(name, dpi=300):\n    plt.savefig(f'/kaggle/working/figures/{name}.png',\n                dpi=dpi, bbox_inches='tight', facecolor='white')\n    print(f\"Saved -> /kaggle/working/figures/{name}.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:06:51.791015Z","iopub.execute_input":"2026-10-03T11:06:51.791352Z","iopub.status.idle":"2026-10-03T11:06:51.796359Z","shell.execute_reply.started":"2026-10-03T11:06:51.79133Z","shell.execute_reply":"2026-10-03T11:06:51.795826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---- Step 2.2b — before/after ko research-quality figure me save ----\nsample_ids = train_df.sample(4, random_state=7)['id_code'].values\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\nfor i, image_id in enumerate(sample_ids):\n    raw  = cv2.cvtColor(cv2.imread(f'{BASE}/train_images/{image_id}.png'), cv2.COLOR_BGR2RGB)\n    proc = preprocess_image(image_id)\n    axes[0, i].imshow(raw);  axes[0, i].set_title('Original', fontsize=13);  axes[0, i].axis('off')\n    axes[1, i].imshow(proc); axes[1, i].set_title('Preprocessed', fontsize=13); axes[1, i].axis('off')\nfig.suptitle('Fundus Preprocessing — Ben Graham Method (sigmaX=10)',\n             fontsize=16, weight='bold', y=1.02)\nplt.tight_layout()\nsave_fig('02_preprocessing_before_after')   # <- yahin save ho gayi\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:06:51.799835Z","iopub.execute_input":"2026-10-03T11:06:51.80027Z","iopub.status.idle":"2026-10-03T11:06:59.404944Z","shell.execute_reply.started":"2026-10-03T11:06:51.80025Z","shell.execute_reply":"2026-10-03T11:06:59.403991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3.1 — stratified split (80/20)\nfrom sklearn.model_selection import train_test_split\n\nPROC_DIR = '/kaggle/working/train_processed'\ntrain_df['path'] = PROC_DIR + '/' + train_df['id_code'] + '.png'\n\ntrain_split, val_split = train_test_split(\n    train_df, test_size=0.20,\n    stratify=train_df['diagnosis'],   # same class ratio dono me\n    random_state=42\n)\nprint(\"Train:\", len(train_split), \" Val:\", len(val_split))\nprint(\"\\nVal class ratio:\")\nprint(val_split['diagnosis'].value_counts(normalize=True).round(3).sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:06:59.406075Z","iopub.execute_input":"2026-10-03T11:06:59.406437Z","iopub.status.idle":"2026-10-03T11:06:59.599146Z","shell.execute_reply.started":"2026-10-03T11:06:59.406412Z","shell.execute_reply":"2026-10-03T11:06:59.598371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3.2 — balanced class weights\nfrom sklearn.utils.class_weight import compute_class_weight\nimport numpy as np\n\nclasses = np.array(sorted(train_split['diagnosis'].unique()))\nweights = compute_class_weight('balanced', classes=classes,\n                               y=train_split['diagnosis'])\nclass_weights = {int(c): float(w) for c, w in zip(classes, weights)}\n\nfor c, w in class_weights.items():\n    print(f\"{label_names[c]:18s} weight = {w:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:06:59.600113Z","iopub.execute_input":"2026-10-03T11:06:59.600632Z","iopub.status.idle":"2026-10-03T11:06:59.608177Z","shell.execute_reply.started":"2026-10-03T11:06:59.600608Z","shell.execute_reply":"2026-10-03T11:06:59.607423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3.3 — tf.data input pipeline\nimport tensorflow as tf\n\nIMG_SIZE   = 224\nBATCH_SIZE = 32\nAUTOTUNE   = tf.data.AUTOTUNE\n\ndef decode_image(path, label):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_png(img, channels=3)\n    img = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n    img = tf.cast(img, tf.float32)          # [0,255], EfficientNet normalize karega\n    return img, label\n\ndef make_ds(df, training=False):\n    ds = tf.data.Dataset.from_tensor_slices((df['path'].values, df['diagnosis'].values))\n    if training:\n        ds = ds.shuffle(len(df), seed=42)\n    ds = ds.map(decode_image, num_parallel_calls=AUTOTUNE)\n    return ds.batch(BATCH_SIZE).prefetch(AUTOTUNE)\n\ntrain_ds = make_ds(train_split, training=True)\nval_ds   = make_ds(val_split,   training=False)\n\nfor imgs, labels in train_ds.take(1):\n    print(\"Batch:\", imgs.shape, \" labels:\", labels.shape)\n    print(\"Pixel range:\", tf.reduce_min(imgs).numpy(), \"to\", tf.reduce_max(imgs).numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:06:59.609269Z","iopub.execute_input":"2026-10-03T11:06:59.610013Z","iopub.status.idle":"2026-10-03T11:07:16.84874Z","shell.execute_reply.started":"2026-10-03T11:06:59.60999Z","shell.execute_reply":"2026-10-03T11:07:16.847923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3.4 — augmentation block + save example figure\nfrom tensorflow.keras import layers\nimport matplotlib.pyplot as plt\n\ndata_augmentation = tf.keras.Sequential([\n    layers.RandomFlip(\"horizontal_and_vertical\"),\n    layers.RandomRotation(0.15),\n    layers.RandomZoom(0.15),\n    layers.RandomBrightness(0.10),\n    layers.RandomContrast(0.10),\n], name=\"data_augmentation\")\n\nsample_img, _ = next(iter(train_ds.take(1)))\nsample_img = sample_img[0]\n\nplt.figure(figsize=(14, 7))\nplt.subplot(2, 4, 1)\nplt.imshow(tf.cast(sample_img, tf.uint8)); plt.axis('off')\nplt.title('Original', weight='bold')\nfor i in range(7):\n    aug = data_augmentation(tf.expand_dims(sample_img, 0), training=True)[0]\n    plt.subplot(2, 4, i + 2)\n    plt.imshow(tf.cast(tf.clip_by_value(aug, 0, 255), tf.uint8)); plt.axis('off')\n    plt.title(f'Augmented {i+1}')\nplt.suptitle('Data Augmentation — random variants of one image',\n             fontsize=15, weight='bold', y=1.0)\nplt.tight_layout()\nsave_fig('03_augmentation_examples')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:16.849922Z","iopub.execute_input":"2026-10-03T11:07:16.85044Z","iopub.status.idle":"2026-10-03T11:07:21.277933Z","shell.execute_reply.started":"2026-10-03T11:07:16.850415Z","shell.execute_reply":"2026-10-03T11:07:21.277042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4.1 — transfer-learning model\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras.applications import EfficientNetB0\n\nNUM_CLASSES = 5\n\ninputs = layers.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\nx = data_augmentation(inputs)                     # augmentation (training-only)\n\nbase_model = EfficientNetB0(include_top=False, weights='imagenet')\nbase_model.trainable = False                      # phase 1: base frozen\n\nx = base_model(x, training=False)                 # BatchNorm inference mode\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dropout(0.4)(x)                         # regularization\noutputs = layers.Dense(NUM_CLASSES, activation='softmax',\n                       kernel_regularizer=regularizers.l2(1e-4))(x)\n\nmodel = models.Model(inputs, outputs)\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-3),\n    loss='sparse_categorical_crossentropy',       # integer labels\n    metrics=['accuracy']\n)\n\ntotal = model.count_params()\ntrainable = int(sum(tf.size(w).numpy() for w in model.trainable_weights))\nprint(f\"Total params:     {total:,}\")\nprint(f\"Trainable params: {trainable:,}   # sirf head — base frozen\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:21.279099Z","iopub.execute_input":"2026-10-03T11:07:21.279384Z","iopub.status.idle":"2026-10-03T11:07:22.939584Z","shell.execute_reply.started":"2026-10-03T11:07:21.279352Z","shell.execute_reply":"2026-10-03T11:07:22.938989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4.2 — sanity check\nfor imgs, labels in val_ds.take(1):\n    preds = model.predict(imgs, verbose=0)\n    print(\"Output shape:\", preds.shape)                 # (32, 5)\n    print(\"Sample probabilities:\", preds[0].round(3))\n    print(\"Sum:\", round(float(preds[0].sum()), 3))      # ~1.0\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:22.9406Z","iopub.execute_input":"2026-10-03T11:07:22.940943Z","iopub.status.idle":"2026-10-03T11:07:26.714338Z","shell.execute_reply.started":"2026-10-03T11:07:22.940919Z","shell.execute_reply":"2026-10-03T11:07:26.713295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4.3 — save architecture diagram\ntry:\n    tf.keras.utils.plot_model(\n        model, to_file='/kaggle/working/figures/04_model_architecture.png',\n        show_shapes=True, dpi=150)\n    print(\"Saved -> figures/04_model_architecture.png\")\nexcept Exception as e:\n    print(\"plot_model skipped (graphviz missing) — optional hai:\", e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:26.715583Z","iopub.execute_input":"2026-10-03T11:07:26.715858Z","iopub.status.idle":"2026-10-03T11:07:26.976453Z","shell.execute_reply.started":"2026-10-03T11:07:26.715822Z","shell.execute_reply":"2026-10-03T11:07:26.975744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5.1 — callbacks\nimport os\nos.makedirs('/kaggle/working/checkpoints', exist_ok=True)\n\nBEST_PATH = '/kaggle/working/checkpoints/best_model.keras'\nLAST_PATH = '/kaggle/working/checkpoints/last_model.keras'\nLOG_PATH  = '/kaggle/working/checkpoints/history_log.csv'\n\ncallbacks = [\n    tf.keras.callbacks.ModelCheckpoint(BEST_PATH, monitor='val_loss',\n                                       save_best_only=True, verbose=1),\n    tf.keras.callbacks.ModelCheckpoint(LAST_PATH),          # har epoch — resume ke liye\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5,\n                                     restore_best_weights=True, verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.3,\n                                         patience=2, min_lr=1e-6, verbose=1),\n    tf.keras.callbacks.CSVLogger(LOG_PATH, append=True),    # history disk pe\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:26.97761Z","iopub.execute_input":"2026-10-03T11:07:26.977908Z","iopub.status.idle":"2026-10-03T11:07:26.983584Z","shell.execute_reply.started":"2026-10-03T11:07:26.977886Z","shell.execute_reply":"2026-10-03T11:07:26.982857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5.2 — resume-aware training\nimport pandas as pd\n\nEPOCHS = 25\n\nif os.path.exists(LAST_PATH):\n    print(\"Checkpoint mila — resume kar rahe hain.\")\n    model = tf.keras.models.load_model(LAST_PATH)\n    initial_epoch = int(pd.read_csv(LOG_PATH)['epoch'].max() + 1)\nelse:\n    print(\"Fresh training start.\")\n    initial_epoch = 0\n\nprint(\"Starting from epoch\", initial_epoch, \"| target\", EPOCHS)\n\nhistory = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS,\n    initial_epoch=initial_epoch,\n    class_weight=class_weights,     # imbalance ka ilaaj\n    callbacks=callbacks,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:07:26.984597Z","iopub.execute_input":"2026-10-03T11:07:26.985268Z","iopub.status.idle":"2026-10-03T11:10:40.686605Z","shell.execute_reply.started":"2026-10-03T11:07:26.985232Z","shell.execute_reply":"2026-10-03T11:10:40.68588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5.3 — training curves from the saved log\nimport matplotlib.pyplot as plt\n\nlog = pd.read_csv(LOG_PATH)\nfig, ax = plt.subplots(1, 2, figsize=(14, 5))\n\nax[0].plot(log['epoch'], log['accuracy'], label='train')\nax[0].plot(log['epoch'], log['val_accuracy'], label='val')\nax[0].set_title('Accuracy'); ax[0].set_xlabel('epoch'); ax[0].legend(); ax[0].grid(alpha=.3)\n\nax[1].plot(log['epoch'], log['loss'], label='train')\nax[1].plot(log['epoch'], log['val_loss'], label='val')\nax[1].set_title('Loss'); ax[1].set_xlabel('epoch'); ax[1].legend(); ax[1].grid(alpha=.3)\n\nfig.suptitle('Training Curves — Phase 1 (frozen base)', fontsize=15, weight='bold')\nplt.tight_layout()\nsave_fig('05_training_curves')\nplt.show()\n\nbest = log.loc[log['val_loss'].idxmin()]\nprint(f\"Best epoch {int(best['epoch'])}: val_acc {best['val_accuracy']:.3f}, val_loss {best['val_loss']:.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:40.688115Z","iopub.execute_input":"2026-10-03T11:10:40.688749Z","iopub.status.idle":"2026-10-03T11:10:41.604837Z","shell.execute_reply.started":"2026-10-03T11:10:40.688726Z","shell.execute_reply":"2026-10-03T11:10:41.603957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6.1 — best model, val predictions\nbest_model = tf.keras.models.load_model('/kaggle/working/checkpoints/best_model.keras')\n\ny_true = np.concatenate([y.numpy() for _, y in val_ds])   # val_ds shuffle nahi hota — order safe\ny_prob = best_model.predict(val_ds, verbose=1)\ny_pred = y_prob.argmax(axis=1)\n\nprint(\"Val samples:\", len(y_true))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:41.605919Z","iopub.execute_input":"2026-10-03T11:10:41.606272Z","iopub.status.idle":"2026-10-03T11:10:49.064714Z","shell.execute_reply.started":"2026-10-03T11:10:41.606251Z","shell.execute_reply":"2026-10-03T11:10:49.063851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6.2 — headline metrics\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score, classification_report\n\nacc = accuracy_score(y_true, y_pred)\nqwk = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\nprint(f\"Accuracy:                 {acc:.4f}\")\nprint(f\"Quadratic Weighted Kappa: {qwk:.4f}\")\nprint()\nprint(classification_report(y_true, y_pred,\n      target_names=[label_names[i] for i in range(5)]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:49.065947Z","iopub.execute_input":"2026-10-03T11:10:49.066295Z","iopub.status.idle":"2026-10-03T11:10:49.092186Z","shell.execute_reply.started":"2026-10-03T11:10:49.06626Z","shell.execute_reply":"2026-10-03T11:10:49.091581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6.3 — confusion matrix heatmap\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\ncm = confusion_matrix(y_true, y_pred)\nnames = [label_names[i] for i in range(5)]\n\nplt.figure(figsize=(8, 6.5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='YlOrRd',\n            xticklabels=names, yticklabels=names)\nplt.xlabel('Predicted'); plt.ylabel('True')\nplt.title('Confusion Matrix — Validation Set', fontsize=14, weight='bold')\nplt.tight_layout()\nsave_fig('06_confusion_matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:49.09293Z","iopub.execute_input":"2026-10-03T11:10:49.093379Z","iopub.status.idle":"2026-10-03T11:10:49.729538Z","shell.execute_reply.started":"2026-10-03T11:10:49.093357Z","shell.execute_reply":"2026-10-03T11:10:49.728925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', '/kaggle/working/checkpoints')\nprint(\"Ready: /kaggle/working/dr_backup.zip — right sidebar Output me isko download karo\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:49.730886Z","iopub.execute_input":"2026-10-03T11:10:49.7311Z","iopub.status.idle":"2026-10-03T11:10:51.389295Z","shell.execute_reply.started":"2026-10-03T11:10:49.731081Z","shell.execute_reply":"2026-10-03T11:10:51.388412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BACKUP — models + figures ek zip me (chhota, fast)\nimport shutil, os\n\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\n\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nsize_kb = os.path.getsize('/kaggle/working/dr_backup.zip') // 1024\nprint(f\"Backup ready: dr_backup.zip ({size_kb} KB)\")\nprint(\"Right sidebar -> Output -> dr_backup.zip -> download\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:51.390351Z","iopub.execute_input":"2026-10-03T11:10:51.390983Z","iopub.status.idle":"2026-10-03T11:10:53.271478Z","shell.execute_reply.started":"2026-10-03T11:10:51.390958Z","shell.execute_reply":"2026-10-03T11:10:53.270817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RESTORE — backup zip se sab wapas\nimport shutil, glob, os\nz = glob.glob('/kaggle/input/**/dr_backup.zip', recursive=True)\nif z:\n    shutil.unpack_archive(z[0], '/kaggle/working')\n    print(\"Restored:\", os.listdir('/kaggle/working'))\nelse:\n    print(\"zip nahi mila — Add Input check karo\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:53.272304Z","iopub.execute_input":"2026-10-03T11:10:53.272597Z","iopub.status.idle":"2026-10-03T11:10:58.468487Z","shell.execute_reply.started":"2026-10-03T11:10:53.272561Z","shell.execute_reply":"2026-10-03T11:10:58.467826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7.1 — unfreeze base, BatchNorm frozen\nft_model = tf.keras.models.load_model('/kaggle/working/checkpoints/best_model.keras')\n\nbase = None\nfor layer in ft_model.layers:\n    if 'efficientnet' in layer.name.lower():\n        base = layer; break\n\nbase.trainable = True\nfor layer in base.layers:\n    if isinstance(layer, tf.keras.layers.BatchNormalization):\n        layer.trainable = False\n\nft_model.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-5),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\nprint(\"Trainable params:\", f\"{int(sum(tf.size(w).numpy() for w in ft_model.trainable_weights)):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:10:58.469583Z","iopub.execute_input":"2026-10-03T11:10:58.470219Z","iopub.status.idle":"2026-10-03T11:11:00.162574Z","shell.execute_reply.started":"2026-10-03T11:10:58.470194Z","shell.execute_reply":"2026-10-03T11:11:00.16183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7.2 — fine-tune callbacks\nFT_BEST = '/kaggle/working/checkpoints/best_finetuned.keras'\nFT_LAST = '/kaggle/working/checkpoints/last_finetuned.keras'\nFT_LOG  = '/kaggle/working/checkpoints/finetune_log.csv'\n\nft_callbacks = [\n    tf.keras.callbacks.ModelCheckpoint(FT_BEST, monitor='val_loss', save_best_only=True, verbose=1),\n    tf.keras.callbacks.ModelCheckpoint(FT_LAST),\n    tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=6, restore_best_weights=True, verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.3, patience=3, min_lr=1e-7, verbose=1),\n    tf.keras.callbacks.CSVLogger(FT_LOG, append=True),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:11:00.163554Z","iopub.execute_input":"2026-10-03T11:11:00.163929Z","iopub.status.idle":"2026-10-03T11:11:00.168969Z","shell.execute_reply.started":"2026-10-03T11:11:00.163907Z","shell.execute_reply":"2026-10-03T11:11:00.168185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7.3 — fine-tune training\nimport pandas as pd, os\nFT_EPOCHS = 40\n\nif os.path.exists(FT_LAST):\n    print(\"Checkpoint mila — resume.\")\n    ft_model = tf.keras.models.load_model(FT_LAST)\n    ft_initial = int(pd.read_csv(FT_LOG)['epoch'].max() + 1)\nelse:\n    ft_initial = 0\n\nprint(\"Fine-tune from epoch\", ft_initial, \"| target\", FT_EPOCHS)\n\nft_history = ft_model.fit(\n    train_ds, validation_data=val_ds,\n    epochs=FT_EPOCHS, initial_epoch=ft_initial,\n    class_weight=class_weights, callbacks=ft_callbacks,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:11:00.170061Z","iopub.execute_input":"2026-10-03T11:11:00.170335Z","iopub.status.idle":"2026-10-03T11:19:28.147609Z","shell.execute_reply.started":"2026-10-03T11:11:00.170306Z","shell.execute_reply":"2026-10-03T11:19:28.146971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7.4 — re-evaluate + compare\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score, classification_report\nft_best = tf.keras.models.load_model(FT_BEST)\ny_pred_ft = ft_best.predict(val_ds, verbose=0).argmax(axis=1)\n\nacc_ft = accuracy_score(y_true, y_pred_ft)\nqwk_ft = cohen_kappa_score(y_true, y_pred_ft, weights='quadratic')\nprint(\"Phase 1 (frozen):    acc %.4f | QWK %.4f\" % (acc, qwk))\nprint(\"Phase 2 (finetuned): acc %.4f | QWK %.4f\" % (acc_ft, qwk_ft))\nprint()\nprint(classification_report(y_true, y_pred_ft, target_names=[label_names[i] for i in range(5)]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:19:28.149052Z","iopub.execute_input":"2026-10-03T11:19:28.149369Z","iopub.status.idle":"2026-10-03T11:19:35.449254Z","shell.execute_reply.started":"2026-10-03T11:19:28.149346Z","shell.execute_reply":"2026-10-03T11:19:35.448549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7.5 — phase 2 research figures\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nft_log = pd.read_csv(FT_LOG)\n\nfig, ax = plt.subplots(1, 2, figsize=(14, 5))\nax[0].plot(ft_log['epoch'], ft_log['accuracy'], label='train')\nax[0].plot(ft_log['epoch'], ft_log['val_accuracy'], label='val')\nax[0].set_title('Accuracy (fine-tuning)'); ax[0].legend(); ax[0].grid(alpha=.3)\nax[1].plot(ft_log['epoch'], ft_log['loss'], label='train')\nax[1].plot(ft_log['epoch'], ft_log['val_loss'], label='val')\nax[1].set_title('Loss (fine-tuning)'); ax[1].legend(); ax[1].grid(alpha=.3)\nfig.suptitle('Training Curves — Phase 2 (fine-tuned)', fontsize=15, weight='bold')\nplt.tight_layout(); save_fig('07_finetune_curves'); plt.show()\n\ncm_ft = confusion_matrix(y_true, y_pred_ft)\nplt.figure(figsize=(8, 6.5))\nsns.heatmap(cm_ft, annot=True, fmt='d', cmap='YlOrRd', xticklabels=names, yticklabels=names)\nplt.xlabel('Predicted'); plt.ylabel('True')\nplt.title('Confusion Matrix — Fine-tuned EfficientNetB0', fontsize=14, weight='bold')\nplt.tight_layout(); save_fig('07_confusion_finetuned'); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:19:35.450305Z","iopub.execute_input":"2026-10-03T11:19:35.450707Z","iopub.status.idle":"2026-10-03T11:19:37.026419Z","shell.execute_reply.started":"2026-10-03T11:19:35.450682Z","shell.execute_reply":"2026-10-03T11:19:37.025855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BACKUP — models + figures ek zip me (chhota, fast)\nimport shutil, os\n\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\n\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nsize_kb = os.path.getsize('/kaggle/working/dr_backup.zip') // 1024\nprint(f\"Backup ready: dr_backup.zip ({size_kb} KB)\")\nprint(\"Right sidebar -> Output -> dr_backup.zip -> download\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:19:37.027214Z","iopub.execute_input":"2026-10-03T11:19:37.027495Z","iopub.status.idle":"2026-10-03T11:19:43.611874Z","shell.execute_reply.started":"2026-10-03T11:19:37.027474Z","shell.execute_reply":"2026-10-03T11:19:43.611153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8.1 — reusable engine\nfrom tensorflow.keras import layers, models, regularizers\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score\nimport pandas as pd, os\n\nRESULTS_CSV = '/kaggle/working/checkpoints/model_results.csv'\n\ndef get_backbone(name):\n    if name == 'ResNet50':\n        from tensorflow.keras.applications import ResNet50 as M\n        from tensorflow.keras.applications.resnet50 import preprocess_input as P\n    elif name == 'DenseNet121':\n        from tensorflow.keras.applications import DenseNet121 as M\n        from tensorflow.keras.applications.densenet import preprocess_input as P\n    elif name == 'EfficientNetB3':\n        from tensorflow.keras.applications import EfficientNetB3 as M\n        from tensorflow.keras.applications.efficientnet import preprocess_input as P\n    else:  # EfficientNetB0\n        from tensorflow.keras.applications import EfficientNetB0 as M\n        from tensorflow.keras.applications.efficientnet import preprocess_input as P\n    return M(include_top=False, weights='imagenet',\n             input_shape=(IMG_SIZE, IMG_SIZE, 3)), P\n\ndef log_result(name, yp):\n    acc = accuracy_score(y_true, yp)\n    qwk = cohen_kappa_score(y_true, yp, weights='quadratic')\n    bacc = accuracy_score((y_true > 0).astype(int), (yp > 0).astype(int))\n    print(f\"{name}: acc {acc:.4f} | QWK {qwk:.4f} | binary {bacc:.4f}\")\n    row = pd.DataFrame([{'model': name, 'accuracy': acc, 'qwk': qwk, 'binary_acc': bacc}])\n    if os.path.exists(RESULTS_CSV):\n        row = pd.concat([pd.read_csv(RESULTS_CSV), row], ignore_index=True).drop_duplicates('model', keep='last')\n    row.to_csv(RESULTS_CSV, index=False)\n\ndef train_backbone(name, head_epochs=12, ft_epochs=25):\n    print(\"\\n===== \" + name + \" =====\")\n    base, preprocess = get_backbone(name)\n    inputs = layers.Input((IMG_SIZE, IMG_SIZE, 3))\n    x = data_augmentation(inputs)\n    x = layers.Lambda(preprocess)(x)          # backbone-specific\n    base.trainable = False\n    x = base(x, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.4)(x)\n    out = layers.Dense(NUM_CLASSES, activation='softmax', kernel_regularizer=regularizers.l2(1e-4))(x)\n    model = models.Model(inputs, out)\n\n    best_path = f'/kaggle/working/checkpoints/best_{name}.keras'\n    def cbs(pat):\n        return [tf.keras.callbacks.ModelCheckpoint(best_path, monitor='val_loss', save_best_only=True),\n                tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=pat, restore_best_weights=True),\n                tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.3, patience=2, min_lr=1e-7)]\n\n    # phase 1: frozen head\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-3), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    model.fit(train_ds, validation_data=val_ds, epochs=head_epochs, class_weight=class_weights, callbacks=cbs(4), verbose=2)\n\n    # phase 2: fine-tune (BN frozen)\n    base.trainable = True\n    for l in base.layers:\n        if isinstance(l, tf.keras.layers.BatchNormalization):\n            l.trainable = False\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    model.fit(train_ds, validation_data=val_ds, epochs=ft_epochs, class_weight=class_weights, callbacks=cbs(6), verbose=2)\n    # evaluate (best weights already in memory)\n    yp = model.predict(val_ds, verbose=0).argmax(axis=1)\n    log_result(name, yp)\n    return model\n\n# B0 (already trained) ka result bhi CSV me daal do\nlog_result('EfficientNetB0', y_pred_ft)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:25:19.775618Z","iopub.execute_input":"2026-10-03T11:25:19.77617Z","iopub.status.idle":"2026-10-03T11:25:19.802745Z","shell.execute_reply.started":"2026-10-03T11:25:19.776142Z","shell.execute_reply":"2026-10-03T11:25:19.802011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8.2 — ResNet50\n_ = train_backbone('ResNet50')\n\n# abhi tak ke results dekho\nimport pandas as pd\nprint(pd.read_csv(RESULTS_CSV).round(4).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:25:37.541402Z","iopub.execute_input":"2026-10-03T11:25:37.541698Z","iopub.status.idle":"2026-10-03T11:35:44.31902Z","shell.execute_reply.started":"2026-10-03T11:25:37.541674Z","shell.execute_reply":"2026-10-03T11:35:44.318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nprint(\"Backup ready:\", os.path.getsize('/kaggle/working/dr_backup.zip')//1024, \"KB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:37:04.789882Z","iopub.execute_input":"2026-10-03T11:37:04.790294Z","iopub.status.idle":"2026-10-03T11:37:24.904532Z","shell.execute_reply.started":"2026-10-03T11:37:04.790267Z","shell.execute_reply":"2026-10-03T11:37:24.903629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom IPython.display import FileLink\n\npath = '/kaggle/working/dr_backup.zip'\nif os.path.exists(path):\n    print(\"Size:\", os.path.getsize(path)//1024, \"KB — neeche link pe click karke download karo\")\n    display(FileLink('dr_backup.zip'))   # /kaggle/working ke andar hai isliye sirf naam\nelse:\n    print(\"zip nahi mila — pehle BACKUP cell chalao\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:45:31.435053Z","iopub.execute_input":"2026-10-03T11:45:31.43561Z","iopub.status.idle":"2026-10-03T11:45:31.44197Z","shell.execute_reply.started":"2026-10-03T11:45:31.435583Z","shell.execute_reply":"2026-10-03T11:45:31.441216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8.3 — DenseNet121\n_ = train_backbone('DenseNet121')\n\nimport pandas as pd\nprint(pd.read_csv(RESULTS_CSV).round(4).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T11:46:11.159873Z","iopub.execute_input":"2026-10-03T11:46:11.160131Z","iopub.status.idle":"2026-10-03T11:56:54.805927Z","shell.execute_reply.started":"2026-10-03T11:46:11.160111Z","shell.execute_reply":"2026-10-03T11:56:54.805232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nprint(\"Backup ready:\", os.path.getsize('/kaggle/working/dr_backup.zip')//1024, \"KB\")\nimport os\nfrom IPython.display import FileLink\n\npath = '/kaggle/working/dr_backup.zip'\nif os.path.exists(path):\n    print(\"Size:\", os.path.getsize(path)//1024, \"KB — neeche link pe click karke download karo\")\n    display(FileLink('dr_backup.zip'))   # /kaggle/working ke andar hai isliye sirf naam\nelse:\n    print(\"zip nahi mila — pehle BACKUP cell chalao\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:00:15.127371Z","iopub.execute_input":"2026-10-03T12:00:15.128022Z","iopub.status.idle":"2026-10-03T12:00:39.649656Z","shell.execute_reply.started":"2026-10-03T12:00:15.127994Z","shell.execute_reply":"2026-10-03T12:00:39.649025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8.4 — EfficientNetB3\n_ = train_backbone('EfficientNetB3')\n\nprint(pd.read_csv(RESULTS_CSV).round(4).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:01:22.564665Z","iopub.execute_input":"2026-10-03T12:01:22.565213Z","iopub.status.idle":"2026-10-03T12:20:50.112027Z","shell.execute_reply.started":"2026-10-03T12:01:22.565184Z","shell.execute_reply":"2026-10-03T12:20:50.111243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nprint(\"Backup ready:\", os.path.getsize('/kaggle/working/dr_backup.zip')//1024, \"KB\")\nimport os\nfrom IPython.display import FileLink\n\npath = '/kaggle/working/dr_backup.zip'\nif os.path.exists(path):\n    print(\"Size:\", os.path.getsize(path)//1024, \"KB — neeche link pe click karke download karo\")\n    display(FileLink('dr_backup.zip'))   # /kaggle/working ke andar hai isliye sirf naam\nelse:\n    print(\"zip nahi mila — pehle BACKUP cell chalao\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:21:50.889483Z","iopub.execute_input":"2026-10-03T12:21:50.889763Z","iopub.status.idle":"2026-10-03T12:22:21.601643Z","shell.execute_reply.started":"2026-10-03T12:21:50.889742Z","shell.execute_reply":"2026-10-03T12:22:21.600682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 9.1 — model comparison figure\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\n\nres = pd.read_csv(RESULTS_CSV).sort_values('qwk', ascending=False).reset_index(drop=True)\nprint(res.round(4).to_string(index=False))\n\nmetrics = ['accuracy', 'qwk', 'binary_acc']\nlabels  = res['model'].values\nx = np.arange(len(labels)); w = 0.25\ncolors = ['#f2b544', '#43b6a6', '#e5544b']\n\nplt.figure(figsize=(11, 6))\nfor i, m in enumerate(metrics):\n    bars = plt.bar(x + (i-1)*w, res[m], w, label=m, color=colors[i])\n    for j, v in enumerate(res[m]):\n        plt.text(x[j] + (i-1)*w, v + 0.012, f'{v:.2f}', ha='center', fontsize=8)\nplt.xticks(x, labels, rotation=10)\nplt.ylim(0, 1.05); plt.ylabel('score')\nplt.title('Model Comparison — APTOS 2019 (validation set)', fontsize=14, weight='bold')\nplt.legend(); plt.grid(axis='y', alpha=.3)\nplt.tight_layout()\nsave_fig('09_model_comparison')\nplt.show()\n\nbest = res.iloc[0]\nBEST_MODEL_NAME = best['model']\nprint(f\"\\nBest by QWK: {BEST_MODEL_NAME}  (QWK {best['qwk']:.4f}, acc {best['accuracy']:.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:24:29.725294Z","iopub.execute_input":"2026-10-03T12:24:29.725697Z","iopub.status.idle":"2026-10-03T12:24:30.417079Z","shell.execute_reply.started":"2026-10-03T12:24:29.725671Z","shell.execute_reply":"2026-10-03T12:24:30.416409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nprint(\"Backup ready:\", os.path.getsize('/kaggle/working/dr_backup.zip')//1024, \"KB\")\nimport os\nfrom IPython.display import FileLink\n\npath = '/kaggle/working/dr_backup.zip'\nif os.path.exists(path):\n    print(\"Size:\", os.path.getsize(path)//1024, \"KB — neeche link pe click karke download karo\")\n    display(FileLink('dr_backup.zip'))   # /kaggle/working ke andar hai isliye sirf naam\nelse:\n    print(\"zip nahi mila — pehle BACKUP cell chalao\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:25:05.344383Z","iopub.execute_input":"2026-10-03T12:25:05.34487Z","iopub.status.idle":"2026-10-03T12:25:40.931258Z","shell.execute_reply.started":"2026-10-03T12:25:05.344834Z","shell.execute_reply":"2026-10-03T12:25:40.93035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 10.1 — grad model from best EfficientNetB3\nimport numpy as np\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nbest_path = '/kaggle/working/checkpoints/best_EfficientNetB3.keras'\ngc_model = tf.keras.models.load_model(\n    best_path, safe_mode=False,\n    custom_objects={'preprocess_input': preprocess_input})   # Lambda fix\n\nbase = [l for l in gc_model.layers if 'efficientnet' in l.name.lower()][0]\nhead = gc_model.layers[gc_model.layers.index(base) + 1:]   # GAP, Dropout, Dense\n\nx = base.output\nfor l in head:\n    x = l(x)\ngrad_model = tf.keras.models.Model(base.input, [base.output, x])\nprint(\"Grad-CAM ready on:\", base.name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:34:42.375748Z","iopub.execute_input":"2026-10-03T12:34:42.376659Z","iopub.status.idle":"2026-10-03T12:34:45.125638Z","shell.execute_reply.started":"2026-10-03T12:34:42.37662Z","shell.execute_reply":"2026-10-03T12:34:45.125014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 10.2 — compute heatmap + overlay\nimport cv2\n\ndef gradcam_heatmap(img_batch):\n    with tf.GradientTape() as tape:\n        conv_out, preds = grad_model(img_batch)\n        idx = tf.argmax(preds[0])\n        channel = preds[:, idx]\n    grads = tape.gradient(channel, conv_out)\n    pooled = tf.reduce_mean(grads, axis=(0, 1, 2))\n    cam = tf.squeeze(conv_out[0] @ pooled[..., None])\n    cam = tf.maximum(cam, 0) / (tf.reduce_max(cam) + 1e-8)\n    return cam.numpy(), int(idx), float(preds[0][idx])\n\ndef overlay_cam(img_uint8, cam, alpha=0.45):\n    hm = cv2.resize(cam, (img_uint8.shape[1], img_uint8.shape[0]))\n    hm = cv2.applyColorMap(np.uint8(255 * hm), cv2.COLORMAP_JET)\n    hm = cv2.cvtColor(hm, cv2.COLOR_BGR2RGB)\n    return cv2.addWeighted(img_uint8, 1 - alpha, hm, alpha, 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:34:50.279257Z","iopub.execute_input":"2026-10-03T12:34:50.28007Z","iopub.status.idle":"2026-10-03T12:34:50.286439Z","shell.execute_reply.started":"2026-10-03T12:34:50.28004Z","shell.execute_reply":"2026-10-03T12:34:50.285608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 10.3 — Grad-CAM figure (explainability)\nsamples = val_split[val_split['diagnosis'] > 0].sample(6, random_state=3)\n\nfig, axes = plt.subplots(2, 6, figsize=(20, 7))\nfor i, (_, row) in enumerate(samples.iterrows()):\n    img = cv2.cvtColor(cv2.imread(row['path']), cv2.COLOR_BGR2RGB)   # processed 224\n    x = tf.cast(img, tf.float32)[None]                               # EfficientNet preprocess = no-op\n    cam, pred, conf = gradcam_heatmap(x)\n    over = overlay_cam(img, cam)\n    axes[0, i].imshow(img);  axes[0, i].axis('off')\n    axes[0, i].set_title(\"True: \" + label_names[row['diagnosis']], fontsize=10)\n    axes[1, i].imshow(over); axes[1, i].axis('off')\n    ok = (pred == row['diagnosis'])\n    axes[1, i].set_title(f\"Pred: {label_names[pred]} ({conf*100:.0f}%)\",\n                         fontsize=10, color='green' if ok else 'red')\nfig.suptitle('Grad-CAM — where EfficientNetB3 looks (red = high attention)',\n             fontsize=15, weight='bold')\nplt.tight_layout(); save_fig('10_gradcam_heatmaps'); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:34:53.909747Z","iopub.execute_input":"2026-10-03T12:34:53.910287Z","iopub.status.idle":"2026-10-03T12:35:01.736583Z","shell.execute_reply.started":"2026-10-03T12:34:53.910259Z","shell.execute_reply":"2026-10-03T12:35:01.734506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil, os\nBK = '/kaggle/working/_backup'\nos.makedirs(BK, exist_ok=True)\nfor src in ['/kaggle/working/checkpoints', '/kaggle/working/figures']:\n    if os.path.exists(src):\n        shutil.copytree(src, f'{BK}/{os.path.basename(src)}', dirs_exist_ok=True)\nshutil.make_archive('/kaggle/working/dr_backup', 'zip', BK)\nprint(\"Backup ready:\", os.path.getsize('/kaggle/working/dr_backup.zip')//1024, \"KB\")\nimport os\nfrom IPython.display import FileLink\n\npath = '/kaggle/working/dr_backup.zip'\nif os.path.exists(path):\n    print(\"Size:\", os.path.getsize(path)//1024, \"KB — neeche link pe click karke download karo\")\n    display(FileLink('dr_backup.zip'))   # /kaggle/working ke andar hai isliye sirf naam\nelse:\n    print(\"zip nahi mila — pehle BACKUP cell chalao\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:35:40.644372Z","iopub.execute_input":"2026-10-03T12:35:40.644885Z","iopub.status.idle":"2026-10-03T12:36:11.244212Z","shell.execute_reply.started":"2026-10-03T12:35:40.64486Z","shell.execute_reply":"2026-10-03T12:36:11.243587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 10.4 — Grad-CAM on NATURAL-color fundus (cleaner figure)\ndef load_natural(image_id, img_size=224):\n    img = cv2.cvtColor(cv2.imread(f'{BASE}/train_images/{image_id}.png'), cv2.COLOR_BGR2RGB)\n    img = crop_image_from_gray(img)\n    img = cv2.resize(img, (img_size, img_size))\n    h, w = img.shape[:2]\n    mask = np.zeros((h, w), np.uint8)\n    cv2.circle(mask, (w//2, h//2), min(h, w)//2, 1, -1)\n    return img * mask[..., np.newaxis]\n\nsamples = val_split[val_split['diagnosis'] > 0].sample(6, random_state=5)\nfig, axes = plt.subplots(2, 6, figsize=(20, 7))\nfor i, (_, row) in enumerate(samples.iterrows()):\n    proc = cv2.cvtColor(cv2.imread(row['path']), cv2.COLOR_BGR2RGB)   # model input\n    nat  = load_natural(row['id_code'])                              # natural color, same geometry\n    cam, pred, conf = gradcam_heatmap(tf.cast(proc, tf.float32)[None])\n    over = overlay_cam(nat, cam, alpha=0.4)\n    axes[0,i].imshow(nat);  axes[0,i].axis('off'); axes[0,i].set_title(\"True: \"+label_names[row['diagnosis']], fontsize=10)\n    ok = pred == row['diagnosis']\n    axes[1,i].imshow(over); axes[1,i].axis('off')\n    axes[1,i].set_title(f\"Pred: {label_names[pred]} ({conf*100:.0f}%)\", fontsize=10, color='green' if ok else 'red')\nfig.suptitle('Grad-CAM on natural fundus — EfficientNetB3', fontsize=15, weight='bold')\nplt.tight_layout(); save_fig('10_gradcam_natural'); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-03T12:38:08.192744Z","iopub.execute_input":"2026-10-03T12:38:08.1934Z","iopub.status.idle":"2026-10-03T12:38:16.889155Z","shell.execute_reply.started":"2026-10-03T12:38:08.193373Z","shell.execute_reply":"2026-10-03T12:38:16.888334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}