{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor root, dirs, files in os.walk('/kaggle/input'):\n    print(root, len(files))\n    break\n!ls /kaggle/input/aptos2019-blindness-detection/","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:43:55.127852Z","iopub.execute_input":"2026-09-22T04:43:55.128113Z","iopub.status.idle":"2026-09-22T04:43:55.248893Z","shell.execute_reply.started":"2026-09-22T04:43:55.128086Z","shell.execute_reply":"2026-09-22T04:43:55.247995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/input'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:44:41.180849Z","iopub.execute_input":"2026-09-22T04:44:41.181291Z","iopub.status.idle":"2026-09-22T04:44:41.186571Z","shell.execute_reply.started":"2026-09-22T04:44:41.181254Z","shell.execute_reply":"2026-09-22T04:44:41.185519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/input/competitions'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:46:50.99189Z","iopub.execute_input":"2026-09-22T04:46:50.992471Z","iopub.status.idle":"2026-09-22T04:46:50.997067Z","shell.execute_reply.started":"2026-09-22T04:46:50.992439Z","shell.execute_reply":"2026-09-22T04:46:50.996442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comp_name = os.listdir('/kaggle/input/competitions')[0]\nprint(comp_name)\nprint(os.listdir(f'/kaggle/input/competitions/{comp_name}'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:47:03.78564Z","iopub.execute_input":"2026-09-22T04:47:03.786056Z","iopub.status.idle":"2026-09-22T04:47:03.791289Z","shell.execute_reply.started":"2026-09-22T04:47:03.785997Z","shell.execute_reply":"2026-09-22T04:47:03.790519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = '/kaggle/input/competitions/aptos2019-blindness-detection'\nprint(os.listdir(BASE))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:47:27.082663Z","iopub.execute_input":"2026-09-22T04:47:27.083332Z","iopub.status.idle":"2026-09-22T04:47:27.08908Z","shell.execute_reply.started":"2026-09-22T04:47:27.083299Z","shell.execute_reply":"2026-09-22T04:47:27.0883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nBASE = '/kaggle/input/competitions/aptos2019-blindness-detection'\ndf = pd.read_csv(f'{BASE}/train.csv')\ndf['path'] = f'{BASE}/train_images/' + df['id_code'] + '.png'\n\ndf_small = df.groupby('diagnosis', group_keys=False).apply(lambda x: x.sample(min(len(x), 150), random_state=42))\nprint(df_small['diagnosis'].value_counts())\n\n# sanity check the first path actually exists\nprint(os.path.exists(df_small['path'].iloc[0]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T04:47:36.535134Z","iopub.execute_input":"2026-09-22T04:47:36.535529Z","iopub.status.idle":"2026-09-22T04:47:36.846489Z","shell.execute_reply.started":"2026-09-22T04:47:36.535499Z","shell.execute_reply":"2026-09-22T04:47:36.845777Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4 — Train/val split + data pipeline","metadata":{}},{"cell_type":"code","source":"for imgs, labels in train_ds.take(1):\n    print(imgs.shape, labels.numpy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:05:47.767987Z","iopub.execute_input":"2026-09-22T05:05:47.768744Z","iopub.status.idle":"2026-09-22T05:05:48.477965Z","shell.execute_reply.started":"2026-09-22T05:05:47.76871Z","shell.execute_reply":"2026-09-22T05:05:48.477308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom sklearn.model_selection import train_test_split\n\ntrain_df, val_df = train_test_split(df_small, test_size=0.2, stratify=df_small['diagnosis'], random_state=42)\n\ndef load_img(path, label):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_png(img, channels=3)\n    img = tf.image.resize(img, [300, 300])\n    img = tf.keras.applications.efficientnet.preprocess_input(img)\n    return img, label\n\ndef make_ds(df, batch=16, shuffle=False):\n    ds = tf.data.Dataset.from_tensor_slices((df['path'].values, df['diagnosis'].values))\n    if shuffle: ds = ds.shuffle(len(df))\n    return ds.map(load_img, num_parallel_calls=tf.data.AUTOTUNE).batch(batch).prefetch(tf.data.AUTOTUNE)\n\ntrain_ds = make_ds(train_df, shuffle=True)\nval_ds = make_ds(val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:05:52.980624Z","iopub.execute_input":"2026-09-22T05:05:52.981134Z","iopub.status.idle":"2026-09-22T05:05:53.021699Z","shell.execute_reply.started":"2026-09-22T05:05:52.981094Z","shell.execute_reply":"2026-09-22T05:05:53.021148Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5 — Build EfficientNet-B3, train a few epochs","metadata":{}},{"cell_type":"code","source":"base = tf.keras.applications.EfficientNetB3(include_top=False, weights='imagenet', input_shape=(300,300,3))\nbase.trainable = False\nmodel = tf.keras.Sequential([\n    base,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(5, activation='softmax')\n])\nmodel.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\nhistory = model.fit(train_ds, validation_data=val_ds, epochs=5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:06:20.611778Z","iopub.execute_input":"2026-09-22T05:06:20.61222Z","iopub.status.idle":"2026-09-22T05:09:30.284461Z","shell.execute_reply.started":"2026-09-22T05:06:20.612189Z","shell.execute_reply":"2026-09-22T05:09:30.28373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 6 — Save training curve","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfig, ax = plt.subplots(1, 2, figsize=(10,4))\nax[0].plot(history.history['loss'], label='train'); ax[0].plot(history.history['val_loss'], label='val')\nax[0].set_title('Loss'); ax[0].legend()\nax[1].plot(history.history['accuracy'], label='train'); ax[1].plot(history.history['val_accuracy'], label='val')\nax[1].set_title('Accuracy'); ax[1].legend()\nplt.savefig('/kaggle/working/training_curve.png', dpi=150, bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:00:45.730382Z","iopub.execute_input":"2026-09-22T05:00:45.731234Z","iopub.status.idle":"2026-09-22T05:00:46.214536Z","shell.execute_reply.started":"2026-09-22T05:00:45.731202Z","shell.execute_reply":"2026-09-22T05:00:46.213799Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 7 — Real predictions on a few images","metadata":{}},{"cell_type":"code","source":"import os\nprint(os.path.exists('/kaggle/working/training_curve.png'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:11:05.817974Z","iopub.execute_input":"2026-09-22T05:11:05.818486Z","iopub.status.idle":"2026-09-22T05:11:05.823621Z","shell.execute_reply.started":"2026-09-22T05:11:05.818456Z","shell.execute_reply":"2026-09-22T05:11:05.822786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nsample = val_df.sample(3)\nfor _, row in sample.iterrows():\n    img, _ = load_img(row['path'], row['diagnosis'])\n    pred = model.predict(np.expand_dims(img, 0))[0]\n    print(row['id_code'], 'Actual:', row['diagnosis'], 'Predicted probs:', np.round(pred, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:11:40.476432Z","iopub.execute_input":"2026-09-22T05:11:40.477242Z","iopub.status.idle":"2026-09-22T05:11:51.118358Z","shell.execute_reply.started":"2026-09-22T05:11:40.477208Z","shell.execute_reply":"2026-09-22T05:11:51.117461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.path.exists('/kaggle/working/training_curve.png'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:12:49.249313Z","iopub.execute_input":"2026-09-22T05:12:49.249776Z","iopub.status.idle":"2026-09-22T05:12:49.254627Z","shell.execute_reply.started":"2026-09-22T05:12:49.249743Z","shell.execute_reply":"2026-09-22T05:12:49.253802Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 8 — Unfreeze the top layers of EfficientNet-B3","metadata":{}},{"cell_type":"code","source":"base.trainable = True\n\n# Freeze everything except the last ~30 layers (partial fine-tuning)\nfor layer in base.layers[:-30]:\n    layer.trainable = False\n\nprint(\"Trainable layers:\", sum(1 for l in base.layers if l.trainable), \"of\", len(base.layers))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:23:49.291366Z","iopub.execute_input":"2026-09-22T05:23:49.29191Z","iopub.status.idle":"2026-09-22T05:23:49.304805Z","shell.execute_reply.started":"2026-09-22T05:23:49.29188Z","shell.execute_reply":"2026-09-22T05:23:49.304178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 9 — Recompile with a much lower learning rate\n","metadata":{}},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:24:17.040207Z","iopub.execute_input":"2026-09-22T05:24:17.040836Z","iopub.status.idle":"2026-09-22T05:24:17.049456Z","shell.execute_reply.started":"2026-09-22T05:24:17.0408Z","shell.execute_reply":"2026-09-22T05:24:17.048739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 10 — Continue training for more epochs","metadata":{}},{"cell_type":"code","source":"history_finetune = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=8\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:24:33.587439Z","iopub.execute_input":"2026-09-22T05:24:33.58828Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 11 — Combine both training histories into one curve","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# merge frozen-phase history with fine-tune-phase history\nfull_loss = history.history['loss'] + history_finetune.history['loss']\nfull_val_loss = history.history['val_loss'] + history_finetune.history['val_loss']\nfull_acc = history.history['accuracy'] + history_finetune.history['accuracy']\nfull_val_acc = history.history['val_accuracy'] + history_finetune.history['val_accuracy']\n\nfig, ax = plt.subplots(1, 2, figsize=(10,4))\nax[0].plot(full_loss, label='train'); ax[0].plot(full_val_loss, label='val')\nax[0].axvline(x=len(history.history['loss'])-0.5, color='gray', linestyle='--', label='fine-tune starts')\nax[0].set_title('Loss'); ax[0].legend()\nax[1].plot(full_acc, label='train'); ax[1].plot(full_val_acc, label='val')\nax[1].axvline(x=len(history.history['accuracy'])-0.5, color='gray', linestyle='--')\nax[1].set_title('Accuracy'); ax[1].legend()\nplt.savefig('/kaggle/working/training_curve_v2.png', dpi=150, bbox_inches='tight')\nplt.show()\n\nprint(\"Final train acc:\", full_acc[-1], \"| Final val acc:\", full_val_acc[-1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:35:48.98652Z","iopub.execute_input":"2026-09-22T05:35:48.98696Z","iopub.status.idle":"2026-09-22T05:35:49.468573Z","shell.execute_reply.started":"2026-09-22T05:35:48.986927Z","shell.execute_reply":"2026-09-22T05:35:49.467873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Final val loss:\", full_val_loss[-1])\nprint(\"Peak val acc:\", max(full_val_acc), \"at epoch\", full_val_acc.index(max(full_val_acc))+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-22T05:38:18.49736Z","iopub.execute_input":"2026-09-22T05:38:18.497783Z","iopub.status.idle":"2026-09-22T05:38:18.502899Z","shell.execute_reply.started":"2026-09-22T05:38:18.497752Z","shell.execute_reply":"2026-09-22T05:38:18.502089Z"}},"outputs":[],"execution_count":null}]}