{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU","colab":{"gpuType":"T4","provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q kaggle","metadata":{"id":"hb4po1CieGqu","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T00:36:22.267272Z","iopub.execute_input":"2026-09-16T00:36:22.268141Z","iopub.status.idle":"2026-09-16T00:36:25.585502Z","shell.execute_reply.started":"2026-09-16T00:36:22.268089Z","shell.execute_reply":"2026-09-16T00:36:25.584713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.listdir(\"/kaggle/input/competitions\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:03:01.919400Z","iopub.execute_input":"2026-09-16T01:03:01.920303Z","iopub.status.idle":"2026-09-16T01:03:01.924485Z","shell.execute_reply.started":"2026-09-16T01:03:01.920269Z","shell.execute_reply":"2026-09-16T01:03:01.923890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:03:17.911515Z","iopub.execute_input":"2026-09-16T01:03:17.912219Z","iopub.status.idle":"2026-09-16T01:03:17.915456Z","shell.execute_reply.started":"2026-09-16T01:03:17.912190Z","shell.execute_reply":"2026-09-16T01:03:17.914774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n\n\nprint(\"Train images:\", len(os.listdir(f\"{DATA_DIR}/train_images\")))\nprint(\"Test images:\", len(os.listdir(f\"{DATA_DIR}/test_images\")))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:03:30.221369Z","iopub.execute_input":"2026-09-16T01:03:30.221805Z","iopub.status.idle":"2026-09-16T01:03:30.397890Z","shell.execute_reply.started":"2026-09-16T01:03:30.221772Z","shell.execute_reply":"2026-09-16T01:03:30.397185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"TensorFlow:\", tf.__version__)\nprint(\"GPU:\", tf.config.list_physical_devices(\"GPU\"))","metadata":{"id":"KRFkNYdG4jnJ","outputId":"987bb097-8159-40a3-8167-bb5c402e36e8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvidia-smi","metadata":{"id":"HBsX5hkJ4tsH","outputId":"6795ae2e-9364-4096-d5a4-f1af8ccdf0fa"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.metrics import f1_score, precision_score, recall_score, accuracy_score\n\nSEED = 42\ntf.random.set_seed(SEED)\nnp.random.seed(SEED)\n\nKAGGLE_DIR = \"/kaggle/input/plant-pathology-2021-fgvc8\"\nLOCAL_DIR = \"/content/plant_pathology\"\nDATA_DIR = KAGGLE_DIR if os.path.exists(KAGGLE_DIR) else LOCAL_DIR\n\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, \"train_images\")\nTEST_IMG_DIR = os.path.join(DATA_DIR, \"test_images\")\nSAMPLE_SUB = os.path.join(DATA_DIR, \"sample_submission.csv\")\n\nIMG_SIZE = 380\nBATCH_SIZE = 16\nEPOCHS_HEAD = 8\nEPOCHS_FINETUNE = 14","metadata":{"id":"yiCg48BtjfXE","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:03:43.987949Z","iopub.execute_input":"2026-09-16T01:03:43.988777Z","iopub.status.idle":"2026-09-16T01:04:00.675807Z","shell.execute_reply.started":"2026-09-16T01:03:43.988749Z","shell.execute_reply":"2026-09-16T01:04:00.675174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/input/competitions\"):\n    if \"train.csv\" in files:\n        print(os.path.join(root, \"train.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:06:25.160179Z","iopub.execute_input":"2026-09-16T01:06:25.161118Z","iopub.status.idle":"2026-09-16T01:06:41.804123Z","shell.execute_reply.started":"2026-09-16T01:06:25.161084Z","shell.execute_reply":"2026-09-16T01:06:41.803358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_CSV = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/train.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:09:13.465414Z","iopub.execute_input":"2026-09-16T01:09:13.465822Z","iopub.status.idle":"2026-09-16T01:09:13.469724Z","shell.execute_reply.started":"2026-09-16T01:09:13.465791Z","shell.execute_reply":"2026-09-16T01:09:13.469017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_CSV)\n\ndf[\"label_list\"] = df[\"labels\"].apply(lambda x: x.split(\" \"))\n\nmlb = MultiLabelBinarizer()\ny = mlb.fit_transform(df[\"label_list\"])\n\nCLASSES = mlb.classes_\nNUM_CLASSES = len(CLASSES)\n\nstrat_key = df[\"labels\"]\n\ndf_labels = pd.DataFrame(y, columns=CLASSES)\n\ndf = pd.concat([df[[\"image\"]], df_labels], axis=1)\n\nprint(df.shape, CLASSES)","metadata":{"id":"JbJwn7Y1jfT2","outputId":"e7aff24b-0341-4072-cd52-b9dd2fec20e8","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:09:15.312853Z","iopub.execute_input":"2026-09-16T01:09:15.313253Z","iopub.status.idle":"2026-09-16T01:09:15.388277Z","shell.execute_reply.started":"2026-09-16T01:09:15.313224Z","shell.execute_reply":"2026-09-16T01:09:15.387599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_IMG_DIR = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/train_images\"\n\nraw = pd.read_csv(TRAIN_CSV)\nstrat_col = raw[\"labels\"]\n\ntry:\n    train_df, val_df = train_test_split(\n        df, test_size=0.2, random_state=SEED, stratify=strat_col\n    )\nexcept ValueError:\n    train_df, val_df = train_test_split(\n        df, test_size=0.2, random_state=SEED\n    )\n\ntrain_paths = (TRAIN_IMG_DIR + \"/\" + train_df[\"image\"]).values\nval_paths = (TRAIN_IMG_DIR + \"/\" + val_df[\"image\"]).values\n\ntrain_labels = train_df[CLASSES].values.astype(\"float32\")\nval_labels = val_df[CLASSES].values.astype(\"float32\")\n\nprint(len(train_paths), len(val_paths))\nprint(val_paths[0])","metadata":{"id":"xyg072-rjfQz","outputId":"43477fed-7266-4a74-b211-a483a75eee39","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:16:54.147449Z","iopub.execute_input":"2026-09-16T01:16:54.147935Z","iopub.status.idle":"2026-09-16T01:16:54.195535Z","shell.execute_reply.started":"2026-09-16T01:16:54.147903Z","shell.execute_reply":"2026-09-16T01:16:54.194893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!nvidia-smi","metadata":{"id":"gQfzFe7l3TyC","outputId":"4e62130e-c93b-4f73-807d-468fa61fe4d8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTOTUNE = tf.data.AUTOTUNE\n\ndef load_image(path, label=None, training=False):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n    if training:\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        img = tf.image.rot90(img, k=tf.random.uniform([], 0, 4, dtype=tf.int32))\n        img = tf.image.random_brightness(img, 0.1)\n        img = tf.image.random_contrast(img, 0.9, 1.1)\n    img = tf.keras.applications.efficientnet.preprocess_input(img)\n    if label is None:\n        return img\n    return img, label\n\ndef make_dataset(paths, labels=None, training=False):\n    if labels is not None:\n        ds = tf.data.Dataset.from_tensor_slices((paths, labels))\n        ds = ds.map(lambda p, l: load_image(p, l, training), num_parallel_calls=AUTOTUNE)\n    else:\n        ds = tf.data.Dataset.from_tensor_slices(paths)\n        ds = ds.map(lambda p: load_image(p, None, training), num_parallel_calls=AUTOTUNE)\n    if training:\n        ds = ds.shuffle(1024, seed=SEED)\n    ds = ds.batch(BATCH_SIZE).prefetch(AUTOTUNE)\n    return ds\n\ntrain_ds = make_dataset(train_paths, train_labels, training=True)\nval_ds = make_dataset(val_paths, val_labels, training=False)","metadata":{"id":"8rXi0gVfjfNe","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:17:06.051519Z","iopub.execute_input":"2026-09-16T01:17:06.052122Z","iopub.status.idle":"2026-09-16T01:17:06.233414Z","shell.execute_reply.started":"2026-09-16T01:17:06.052091Z","shell.execute_reply":"2026-09-16T01:17:06.232853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model = tf.keras.applications.EfficientNetB7(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(IMG_SIZE, IMG_SIZE, 3)\n)\nbase_model.trainable = False\n\ninputs = tf.keras.Input(shape=(IMG_SIZE, IMG_SIZE, 3))\nx = base_model(inputs, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\noutputs = tf.keras.layers.Dense(NUM_CLASSES, activation=\"sigmoid\")(x)\nmodel = tf.keras.Model(inputs, outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n    loss=\"binary_crossentropy\",\n    metrics=[\n        tf.keras.metrics.BinaryAccuracy(name=\"accuracy\"),\n        tf.keras.metrics.Precision(name=\"precision\"),\n        tf.keras.metrics.Recall(name=\"recall\"),\n    ]\n)\nmodel.summary()","metadata":{"id":"1PV7rzdTjfKI","outputId":"dfb13560-d28c-4dd7-a409-d31de71f43d2"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint_head = tf.keras.callbacks.ModelCheckpoint(\n    \"best_head.keras\", monitor=\"val_loss\", save_best_only=True, mode=\"min\"\n)\nearly_stop_head = tf.keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\", patience=3, restore_best_weights=True\n)\n\nhistory_head = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS_HEAD,\n    callbacks=[checkpoint_head, early_stop_head]\n)","metadata":{"id":"tBRxU7fOjfHE","outputId":"dc6d0c5b-5973-4377-da39-e0e1910a803a"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint_head = tf.keras.callbacks.ModelCheckpoint(\n    \"best_head_finetuned.keras\", monitor=\"val_loss\", save_best_only=True, mode=\"min\"\n)\nearly_stop_head = tf.keras.callbacks.EarlyStopping(\n    monitor=\"val_loss\", patience=3, restore_best_weights=True\n)\n\nbase_model.trainable = True\nfor layer in base_model.layers[:-80]:\n    layer.trainable = False\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss=\"binary_crossentropy\",\n    metrics=[\n        tf.keras.metrics.BinaryAccuracy(name=\"accuracy\"),\n        tf.keras.metrics.Precision(name=\"precision\"),\n        tf.keras.metrics.Recall(name=\"recall\"),\n    ]\n)\n\n\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\", factor=0.5, patience=2, min_lr=1e-7\n)\n\nhistory_ft = model.fit(\n    train_ds,\n    validation_data=val_ds,\n    epochs=EPOCHS_FINETUNE,\n    callbacks=[checkpoint_head, early_stop_head, reduce_lr]\n)","metadata":{"id":"wQKYFm-sjfD0","outputId":"d18d8575-bd41-4b5a-ed19-d7e78b466715"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    if \"best_head_finetuned.keras\" in files:\n        MODEL_PATH = os.path.join(root, \"best_head_finetuned.keras\")\n        print(MODEL_PATH)\n        break\n\nmodel = tf.keras.models.load_model(MODEL_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:17:23.054739Z","iopub.execute_input":"2026-09-16T01:17:23.055152Z","iopub.status.idle":"2026-09-16T01:17:28.514369Z","shell.execute_reply.started":"2026-09-16T01:17:23.055123Z","shell.execute_reply":"2026-09-16T01:17:28.513745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_IMG_DIR = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/train_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:14:50.620962Z","iopub.execute_input":"2026-09-16T01:14:50.621250Z","iopub.status.idle":"2026-09-16T01:14:50.625142Z","shell.execute_reply.started":"2026-09-16T01:14:50.621228Z","shell.execute_reply":"2026-09-16T01:14:50.624261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_probs = model.predict(val_ds)\n\ndef optimize_thresholds(y_true, y_prob, num_classes):\n    thresholds = np.full(num_classes, 0.5)\n    grid = np.arange(0.1, 0.91, 0.02)\n    for c in range(num_classes):\n        best_t, best_f1 = 0.5, -1.0\n        for t in grid:\n            pred_c = (y_prob[:, c] >= t).astype(int)\n            f1_c = f1_score(y_true[:, c], pred_c, zero_division=0)\n            if f1_c > best_f1:\n                best_f1, best_t = f1_c, t\n        thresholds[c] = best_t\n    return thresholds\n\nbest_thresholds = optimize_thresholds(val_labels, val_probs, NUM_CLASSES)\nprint(\"Per-class thresholds:\", dict(zip(CLASSES, best_thresholds)))","metadata":{"id":"5gVvQyynje9x","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:17:34.707098Z","iopub.execute_input":"2026-09-16T01:17:34.707371Z","iopub.status.idle":"2026-09-16T01:20:06.828898Z","shell.execute_reply.started":"2026-09-16T01:17:34.707348Z","shell.execute_reply":"2026-09-16T01:20:06.828073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_IMG_DIR = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/train_images\"\n\nprint(\"TRAIN_IMG_DIR =\", TRAIN_IMG_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:20:47.968490Z","iopub.execute_input":"2026-09-16T01:20:47.969140Z","iopub.status.idle":"2026-09-16T01:20:47.973441Z","shell.execute_reply.started":"2026-09-16T01:20:47.969108Z","shell.execute_reply":"2026-09-16T01:20:47.972506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/input/competitions\"):\n    if \"train_images\" in dirs:\n        TRAIN_IMG_DIR = os.path.join(root, \"train_images\")\n        print(\"TRAIN_IMG_DIR =\", TRAIN_IMG_DIR)\n        break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:20:54.591469Z","iopub.execute_input":"2026-09-16T01:20:54.592199Z","iopub.status.idle":"2026-09-16T01:20:54.599890Z","shell.execute_reply.started":"2026-09-16T01:20:54.592167Z","shell.execute_reply":"2026-09-16T01:20:54.599065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_thresholds(y_prob, thresholds):\n    return (y_prob >= thresholds[np.newaxis, :]).astype(int)\n\ndef report_metrics(y_true, y_prob, thresholds, loss_metrics, name):\n    y_pred = apply_thresholds(y_prob, thresholds)\n    loss, acc, prec, rec = loss_metrics[:4]\n    micro_f1 = f1_score(y_true, y_pred, average=\"micro\", zero_division=0)\n    macro_f1 = f1_score(y_true, y_pred, average=\"macro\", zero_division=0)\n    per_label_f1 = f1_score(y_true, y_pred, average=None, zero_division=0)\n    print(f\"{name} loss: {loss:.4f}\")\n    print(f\"{name} accuracy: {acc:.4f}\")\n    print(f\"{name} precision: {prec:.4f}\")\n    print(f\"{name} recall: {rec:.4f}\")\n    print(f\"{name} micro F1: {micro_f1:.4f}\")\n    print(f\"{name} macro F1: {macro_f1:.4f}\")\n    for cls, f1v in zip(CLASSES, per_label_f1):\n        print(f\"  {name} F1[{cls}]: {f1v:.4f}\")\n    return y_pred\n\ntrain_eval_ds = make_dataset(train_paths, train_labels, training=False)\ntrain_probs = model.predict(train_eval_ds)\ntrain_loss_metrics = model.evaluate(train_eval_ds, verbose=0)\nval_loss_metrics = model.evaluate(val_ds, verbose=0)\n\n_ = report_metrics(train_labels, train_probs, best_thresholds, train_loss_metrics, \"Train\")\n_ = report_metrics(val_labels, val_probs, best_thresholds, val_loss_metrics, \"Validation\")","metadata":{"id":"mF3J9isoqXfD","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:21:04.290896Z","iopub.execute_input":"2026-09-16T01:21:04.291220Z","iopub.status.idle":"2026-09-16T01:34:26.616902Z","shell.execute_reply.started":"2026-09-16T01:21:04.291193Z","shell.execute_reply":"2026-09-16T01:34:26.616174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLE_SUB = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/sample_submission.csv\"\nTEST_IMG_DIR = \"/kaggle/input/competitions/plant-pathology-2021-fgvc8/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:38:24.695833Z","iopub.execute_input":"2026-09-16T01:38:24.696090Z","iopub.status.idle":"2026-09-16T01:38:24.700278Z","shell.execute_reply.started":"2026-09-16T01:38:24.696070Z","shell.execute_reply":"2026-09-16T01:38:24.699460Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(SAMPLE_SUB)\n\ntest_paths = (TEST_IMG_DIR + \"/\" + sample_sub[\"image\"]).values\ntest_ds = make_dataset(test_paths, labels=None, training=False)\n\ntest_probs = model.predict(test_ds)\ntest_preds = apply_thresholds(test_probs, best_thresholds)\n\ndef to_label_string(row):\n    labels = [CLASSES[i] for i, v in enumerate(row) if v == 1]\n    if not labels:\n        return \"healthy\"\n    return \" \".join(labels)\n\nsample_sub[\"labels\"] = [to_label_string(r) for r in test_preds]\n\nsample_sub.to_csv(\"submission.csv\", index=False)\n\nsample_sub.head()","metadata":{"id":"-V9xlxqdqa0B","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:38:36.753489Z","iopub.execute_input":"2026-09-16T01:38:36.753934Z","iopub.status.idle":"2026-09-16T01:39:06.352614Z","shell.execute_reply.started":"2026-09-16T01:38:36.753886Z","shell.execute_reply":"2026-09-16T01:39:06.352038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save(\"best_model.keras\")","metadata":{"id":"G_EO3cEkqdv1","trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:39:29.599752Z","iopub.execute_input":"2026-09-16T01:39:29.600158Z","iopub.status.idle":"2026-09-16T01:39:32.920148Z","shell.execute_reply.started":"2026-09-16T01:39:29.600129Z","shell.execute_reply":"2026-09-16T01:39:32.919291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(os.path.exists(\"/kaggle/working/best_model.keras\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:40:26.995443Z","iopub.execute_input":"2026-09-16T01:40:26.996272Z","iopub.status.idle":"2026-09-16T01:40:27.001015Z","shell.execute_reply.started":"2026-09-16T01:40:26.996241Z","shell.execute_reply":"2026-09-16T01:40:27.000099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n\nFileLink(\"/kaggle/working/best_model.keras\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-16T01:41:22.381600Z","iopub.execute_input":"2026-09-16T01:41:22.382107Z","iopub.status.idle":"2026-09-16T01:41:22.388007Z","shell.execute_reply.started":"2026-09-16T01:41:22.382076Z","shell.execute_reply":"2026-09-16T01:41:22.387331Z"}},"outputs":[],"execution_count":null}]}