{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"10a36cd2-35dc-4250-b6bf-b6792eab603c","cell_type":"markdown","source":"# Plant Pathology 2021 — Correct Multi-Label Analysis\n\nThis notebook analyzes the **Plant Pathology 2021** training annotations using the dataset's actual label representation.\n\nImportant:\n- The annotation contains disease labels separated by spaces.\n- `complex` is a **separate label**, not part of `scab frog_eye_leaf_spot complex` as one class.\n- Therefore an annotation such as `scab frog_eye_leaf_spot complex` contains **3 labels**.\n- Multi-label = an image with **2 or more labels**.\n\nThe notebook also exports **every multi-label image** into a ZIP file.","metadata":{}},{"id":"23870498-b067-42ca-a0dc-231dd69cf1e3","cell_type":"code","source":"# ============================================================\n# 1. Imports / configuration\n# ============================================================\nimport os\nimport re\nimport glob\nimport shutil\nimport zipfile\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\npd.set_option(\"display.max_rows\", 200)\npd.set_option(\"display.max_colwidth\", 150)\n\nSEARCH_ROOTS = [\"/kaggle/input\", \"/kaggle/working\", \".\"]\nprint(\"Ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T08:26:30.416582Z","iopub.execute_input":"2026-10-01T08:26:30.417006Z","iopub.status.idle":"2026-10-01T08:26:30.744511Z","shell.execute_reply.started":"2026-10-01T08:26:30.416936Z","shell.execute_reply":"2026-10-01T08:26:30.743677Z"}},"outputs":[],"execution_count":null},{"id":"1a66530b-6b10-4a69-9576-c6568d1d2546","cell_type":"code","source":"# ============================================================\n# 2. Find train.csv\n# ============================================================\ncsv_candidates = []\n\nfor root in SEARCH_ROOTS:\n    if os.path.exists(root):\n        csv_candidates.extend(\n            glob.glob(os.path.join(root, \"**\", \"*.csv\"), recursive=True)\n        )\n\ncsv_candidates = sorted(set(csv_candidates))\n\nprint(\"CSV files found:\")\nfor p in csv_candidates[:100]:\n    print(\" \", p)\n\npreferred = [\n    p for p in csv_candidates\n    if os.path.basename(p).lower() == \"train.csv\"\n    and any(k in p.lower() for k in [\"plant\", \"pathology\", \"fgvc\"])\n]\n\nif not preferred:\n    preferred = [\n        p for p in csv_candidates\n        if os.path.basename(p).lower() == \"train.csv\"\n    ]\n\nif not preferred:\n    raise FileNotFoundError(\n        \"train.csv was not found. Attach Plant Pathology 2021 to the Kaggle notebook.\"\n    )\n\nCSV_PATH = preferred[0]\ndf = pd.read_csv(CSV_PATH)\n\nprint(\"\\nUsing:\", CSV_PATH)\nprint(\"Shape:\", df.shape)\ndisplay(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T08:26:30.746242Z","iopub.execute_input":"2026-10-01T08:26:30.746636Z"}},"outputs":[],"execution_count":null},{"id":"d3963f98-ddde-42e7-85a4-883b75e1e1ed","cell_type":"code","source":"# ============================================================\n# 3. Identify image and annotation columns\n# ============================================================\ncols_lower = {c.lower(): c for c in df.columns}\n\nimage_col = next(\n    (cols_lower[c] for c in\n     [\"image\", \"image_id\", \"filename\", \"file_name\", \"id\"]\n     if c in cols_lower),\n    None\n)\n\nlabel_col = next(\n    (cols_lower[c] for c in\n     [\"labels\", \"label\", \"disease\", \"diseases\", \"target\", \"class\"]\n     if c in cols_lower),\n    None\n)\n\nif image_col is None:\n    image_like = [\n        c for c in df.columns\n        if \"image\" in c.lower() or \"file\" in c.lower()\n    ]\n    if image_like:\n        image_col = image_like[0]\n\nif label_col is None:\n    label_like = [\n        c for c in df.columns\n        if c != image_col and any(\n            k in c.lower() for k in [\"label\", \"disease\", \"class\", \"target\"]\n        )\n    ]\n    if label_like:\n        label_col = label_like[0]\n\nif image_col is None or label_col is None:\n    raise ValueError(f\"Could not identify columns. Found: {list(df.columns)}\")\n\nprint(\"Image column:\", image_col)\nprint(\"Label column:\", label_col)\nprint(\"Rows:\", len(df))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"b37d8257-0e63-467a-a59f-7e5057218eeb","cell_type":"markdown","source":"## Correct label parsing\n\nFor Plant Pathology 2021, labels are individual annotation components.\n\nExamples:\n\n| Annotation | Number of labels |\n|---|---:|\n| `healthy` | 1 |\n| `rust` | 1 |\n| `scab frog_eye_leaf_spot` | 2 |\n| `scab frog_eye_leaf_spot complex` | **3** |\n\n**Do not combine `scab frog_eye_leaf_spot complex` into one label.**\n\nThis is the key correction from the previous notebook.","metadata":{}},{"id":"0ec20216-b45e-4e16-b3f0-301d1a00f6c4","cell_type":"code","source":"# ============================================================\n# 4. Correct PP2021 label parser\n# ============================================================\n\ndef parse_pp2021_labels(value):\n    if pd.isna(value):\n        return []\n\n    s = str(value).strip().lower()\n    s = re.sub(r\"\\s+\", \" \", s)\n\n    if not s:\n        return []\n\n    # PP2021 labels are whitespace-separated annotation components.\n    return s.split()\n\ndf[\"parsed_labels\"] = df[label_col].apply(parse_pp2021_labels)\ndf[\"num_labels\"] = df[\"parsed_labels\"].apply(len)\n\nprint(\"Raw annotations and parsed labels:\")\ndisplay(df[[image_col, label_col, \"parsed_labels\", \"num_labels\"]].head(30))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"1279825a-5287-4e22-9b22-09de259f2c31","cell_type":"code","source":"# ============================================================\n# 5. Core statistics\n# ============================================================\ntotal_rows = len(df)\nunique_images = df[image_col].nunique()\nduplicate_rows = total_rows - unique_images\n\nzero_label = (df[\"num_labels\"] == 0).sum()\nsingle_label = (df[\"num_labels\"] == 1).sum()\nmulti_label = (df[\"num_labels\"] >= 2).sum()\n\nsummary = pd.DataFrame({\n    \"Metric\": [\n        \"Total annotation rows\",\n        \"Unique image IDs\",\n        \"Duplicate image-ID rows\",\n        \"Images with 0 labels\",\n        \"Single-label images\",\n        \"Multi-label images\",\n        \"Multi-label percentage\",\n        \"Number of unique labels\",\n        \"Maximum labels on one image\",\n    ],\n    \"Value\": [\n        total_rows,\n        unique_images,\n        duplicate_rows,\n        zero_label,\n        single_label,\n        multi_label,\n        f\"{100 * multi_label / total_rows:.2f}%\",\n        len(set(label for labels in df[\"parsed_labels\"] for label in labels)),\n        int(df[\"num_labels\"].max()),\n    ]\n})\n\ndisplay(summary)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"9c9c7518-4eb5-45bc-b7ce-e5f93b004476","cell_type":"code","source":"# ============================================================\n# 6. Number of labels per image\n# ============================================================\ndistribution = (\n    df[\"num_labels\"]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"Number_of_labels\")\n    .reset_index(name=\"Images\")\n)\n\ndistribution[\"Percentage\"] = (\n    100 * distribution[\"Images\"] / total_rows\n).round(2)\n\ndisplay(distribution)\n\nplt.figure(figsize=(8, 4))\nplt.bar(\n    distribution[\"Number_of_labels\"].astype(str),\n    distribution[\"Images\"]\n)\nplt.xlabel(\"Number of labels per image\")\nplt.ylabel(\"Images\")\nplt.title(\"Plant Pathology 2021 — Label Count Distribution\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"12d74891-2970-4896-a443-234606348fde","cell_type":"code","source":"# ============================================================\n# 7. Explicit check for 3-label images\n# ============================================================\nthree_label_df = df[df[\"num_labels\"] == 3].copy()\n\nprint(\"Number of 3-label images:\", len(three_label_df))\nprint(\"Percentage of all images:\", f\"{100 * len(three_label_df) / total_rows:.2f}%\")\n\nif len(three_label_df):\n    display(\n        three_label_df[\n            [image_col, label_col, \"parsed_labels\", \"num_labels\"]\n        ].head(50)\n    )\nelse:\n    print(\"No 3-label rows were found in the CSV.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"f60d0d57-bf65-4f43-b88a-bc47b53a9f11","cell_type":"code","source":"# ============================================================\n# 8. Individual label frequencies\n# ============================================================\ncounter = Counter(\n    label\n    for labels in df[\"parsed_labels\"]\n    for label in labels\n)\n\nlabel_freq = (\n    pd.DataFrame(counter.items(), columns=[\"Label\", \"Image_Count\"])\n    .sort_values(\"Image_Count\", ascending=False)\n    .reset_index(drop=True)\n)\n\nlabel_freq[\"Percentage_of_images\"] = (\n    100 * label_freq[\"Image_Count\"] / total_rows\n).round(2)\n\ndisplay(label_freq)\n\nplt.figure(figsize=(10, 5))\nplt.bar(label_freq[\"Label\"], label_freq[\"Image_Count\"])\nplt.xticks(rotation=45, ha=\"right\")\nplt.ylabel(\"Images\")\nplt.title(\"Individual Label Frequency\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"a7b9d971-4b4f-435a-b5c4-765bd50186f3","cell_type":"code","source":"# ============================================================\n# 9. Label combinations\n# ============================================================\ndf[\"label_combination\"] = df[\"parsed_labels\"].apply(\n    lambda x: \" + \".join(x)\n)\n\ncombo_freq = (\n    df[df[\"num_labels\"] >= 2][\"label_combination\"]\n    .value_counts()\n    .rename_axis(\"Label_Combination\")\n    .reset_index(name=\"Images\")\n)\n\nif len(combo_freq):\n    combo_freq[\"Percentage_of_all_images\"] = (\n        100 * combo_freq[\"Images\"] / total_rows\n    ).round(2)\n\n    combo_freq[\"Percentage_of_multilabel_images\"] = (\n        100 * combo_freq[\"Images\"] / multi_label\n    ).round(2)\n\ndisplay(combo_freq)\nprint(\"Unique multi-label combinations:\", len(combo_freq))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"c2ebfa6d-eff2-449d-8405-e1d5b4640a9a","cell_type":"code","source":"# ============================================================\n# 10. Locate actual image files\n# ============================================================\nall_image_files = []\n\nfor root in SEARCH_ROOTS:\n    if os.path.exists(root):\n        for ext in [\n            \"*.jpg\", \"*.jpeg\", \"*.png\",\n            \"*.JPG\", \"*.JPEG\", \"*.PNG\"\n        ]:\n            all_image_files.extend(\n                glob.glob(os.path.join(root, \"**\", ext), recursive=True)\n            )\n\nall_image_files = sorted(set(all_image_files))\n\nprint(\"Image files found:\", len(all_image_files))\n\nimage_map = {}\n\nfor p in all_image_files:\n    image_map[Path(p).name] = p\n    image_map[Path(p).stem] = p\n\ndef find_image_path(image_id):\n    s = str(image_id)\n\n    candidates = [\n        s,\n        Path(s).name,\n        Path(s).stem,\n        s + \".jpg\",\n        s + \".jpeg\",\n        s + \".png\",\n        s + \".JPG\",\n        s + \".JPEG\",\n        s + \".PNG\",\n    ]\n\n    for c in candidates:\n        if c in image_map:\n            return image_map[c]\n\n    return None\n\ndf[\"image_path\"] = df[image_col].apply(find_image_path)\n\nmissing_images = df[\"image_path\"].isna().sum()\n\nprint(\"Matched image files:\", total_rows - missing_images)\nprint(\"Missing image files:\", missing_images)\n\nif missing_images:\n    display(\n        df.loc[\n            df[\"image_path\"].isna(),\n            [image_col, label_col]\n        ].head(20)\n    )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"971c5c54-db56-4845-9440-5aad2737ac27","cell_type":"markdown","source":"## Extract all multi-label images\n\nThe following cell selects **every image with 2 or more labels**.\n\nFor example, if the corrected analysis finds:\n\n- 1,000 images with 2 labels\n- 100 images with 3 labels\n\nthen the ZIP will contain **all 1,100 images**.","metadata":{}},{"id":"99566ecf-00e6-4ed7-9171-9f99eea513a5","cell_type":"code","source":"# ============================================================\n# 11. Create complete multi-label subset\n# ============================================================\nmultilabel_df = df[df[\"num_labels\"] >= 2].copy()\n\nprint(\"Multi-label images:\", len(multilabel_df))\nprint(\n    \"Multi-label percentage:\",\n    f\"{100 * len(multilabel_df) / total_rows:.2f}%\"\n)\n\nprint(\"\\nBreakdown within multi-label subset:\")\ndisplay(\n    multilabel_df[\"num_labels\"]\n    .value_counts()\n    .sort_index()\n    .rename_axis(\"Number_of_labels\")\n    .reset_index(name=\"Images\")\n)\n\ndisplay(\n    multilabel_df[\n        [\n            image_col,\n            label_col,\n            \"parsed_labels\",\n            \"num_labels\",\n            \"label_combination\",\n            \"image_path\",\n        ]\n    ].head(30)\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"78bb7701-5dd7-4a71-8684-6ff6081b7307","cell_type":"code","source":"# ============================================================\n# 12. Export ALL multi-label images to ZIP\n# ============================================================\nOUTPUT_DIR = Path(\n    \"/kaggle/working/plant_pathology_2021_multilabel\"\n)\n\nZIP_PATH = Path(\n    \"/kaggle/working/plant_pathology_2021_multilabel_images.zip\"\n)\n\nMANIFEST_PATH = Path(\n    \"/kaggle/working/plant_pathology_2021_multilabel_manifest.csv\"\n)\n\nif OUTPUT_DIR.exists():\n    shutil.rmtree(OUTPUT_DIR)\n\nif ZIP_PATH.exists():\n    ZIP_PATH.unlink()\n\nif MANIFEST_PATH.exists():\n    MANIFEST_PATH.unlink()\n\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\ncopied = 0\nmissing = 0\nrecords = []\n\nfor _, row in multilabel_df.iterrows():\n\n    src = row[\"image_path\"]\n\n    if not src or not os.path.isfile(src):\n        missing += 1\n        continue\n\n    src = Path(src)\n    dst = OUTPUT_DIR / src.name\n\n    # Handle duplicate filenames safely.\n    if dst.exists():\n        i = 1\n        while dst.exists():\n            dst = OUTPUT_DIR / f\"{src.stem}_{i}{src.suffix}\"\n            i += 1\n\n    shutil.copy2(src, dst)\n    copied += 1\n\n    records.append({\n        \"image_id\": row[image_col],\n        \"original_annotation\": row[label_col],\n        \"parsed_labels\": \" | \".join(row[\"parsed_labels\"]),\n        \"num_labels\": row[\"num_labels\"],\n        \"label_combination\": row[\"label_combination\"],\n        \"copied_filename\": dst.name,\n    })\n\nmanifest = pd.DataFrame(records)\nmanifest.to_csv(MANIFEST_PATH, index=False)\n\nwith zipfile.ZipFile(\n    ZIP_PATH,\n    \"w\",\n    compression=zipfile.ZIP_DEFLATED\n) as zf:\n\n    for file_path in sorted(OUTPUT_DIR.iterdir()):\n        if file_path.is_file():\n            zf.write(\n                file_path,\n                arcname=file_path.name\n            )\n\nprint(\"=\" * 60)\nprint(\"EXPORT COMPLETE\")\nprint(\"=\" * 60)\nprint(\"Multi-label annotations :\", len(multilabel_df))\nprint(\"Images copied           :\", copied)\nprint(\"Images missing          :\", missing)\nprint(\"Output folder           :\", OUTPUT_DIR)\nprint(\"Manifest CSV            :\", MANIFEST_PATH)\nprint(\"ZIP                     :\", ZIP_PATH)\nprint(\n    \"ZIP size                :\",\n    f\"{ZIP_PATH.stat().st_size / (1024**2):.2f} MB\"\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"1e49d738-93de-44b7-880c-2a230f847b7c","cell_type":"code","source":"# ============================================================\n# 13. Verify ZIP\n# ============================================================\nwith zipfile.ZipFile(ZIP_PATH, \"r\") as zf:\n    names = zf.namelist()\n\nprint(\"Files inside ZIP:\", len(names))\nprint(\"First 20:\")\nfor name in names[:20]:\n    print(\" \", name)\n\nassert len(names) == copied\n\nprint(\"\\nZIP verification passed.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"40f72995-3859-427b-b234-9764b17639c8","cell_type":"markdown","source":"## Final outputs\n\nAfter running all cells, Kaggle Output will contain:\n\n- `plant_pathology_2021_multilabel_images.zip`\n  - every image with **2 or more labels**\n- `plant_pathology_2021_multilabel_manifest.csv`\n  - image ID\n  - original annotation\n  - parsed labels\n  - number of labels\n  - label combination\n  - copied filename\n\nThe analysis now explicitly preserves **3-label annotations** instead of incorrectly treating them as a single combined label.","metadata":{}}]}