{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.8.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":124685,"databundleVersionId":14664296,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PlantCLEF 2026 — Test Image EDA\n\nThis notebook performs exploratory data analysis on the PlantCLEF 2025 test dataset:\n- Locates test CSV and image folder automatically\n- Joins quadrat_id to image paths\n- Computes image statistics (resolution, brightness, blur metrics, etc.)\n- Visualizes distributions and displays image grids\n- Saves results to CSV","metadata":{}},{"cell_type":"markdown","source":"## Setup and Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport math\nimport random\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image, ImageOps\nfrom tqdm.auto import tqdm\n\npd.set_option(\"display.max_columns\", 200)\npd.set_option(\"display.width\", 200)\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:37:58.457379Z","iopub.execute_input":"2026-02-07T15:37:58.457834Z","iopub.status.idle":"2026-02-07T15:37:58.464415Z","shell.execute_reply.started":"2026-02-07T15:37:58.457806Z","shell.execute_reply":"2026-02-07T15:37:58.463307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper Functions","metadata":{}},{"cell_type":"code","source":"def find_file(filename: str) -> str:\n    \"\"\"Find a file under /kaggle/input/\"\"\"\n    hits = glob.glob(f\"/kaggle/input/**/{filename}\", recursive=True)\n    if not hits:\n        raise FileNotFoundError(f\"Couldn't find '{filename}' under /kaggle/input/\")\n    hits = sorted(hits, key=len)  # shortest path is usually the intended one\n    return hits[0]\n\ndef find_dir(dirname: str) -> str:\n    \"\"\"Find a directory under /kaggle/input/\"\"\"\n    hits = [p for p in glob.glob(f\"/kaggle/input/**/{dirname}\", recursive=True) if os.path.isdir(p)]\n    if not hits:\n        raise FileNotFoundError(f\"Couldn't find directory '{dirname}' under /kaggle/input/\")\n    # Prefer the deepest folder that actually contains images\n    hits = sorted(hits, key=lambda p: (len(p), p))\n    return hits[0]\n\ndef list_image_files(root: str, exts=(\"jpg\", \"jpeg\", \"png\", \"webp\")):\n    \"\"\"Recursively list all image files in a directory\"\"\"\n    files = []\n    for ext in exts:\n        files.extend(glob.glob(os.path.join(root, f\"**/*.{ext}\"), recursive=True))\n        files.extend(glob.glob(os.path.join(root, f\"**/*.{ext.upper()}\"), recursive=True))\n    return sorted(set(files))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:37:58.466163Z","iopub.execute_input":"2026-02-07T15:37:58.466745Z","iopub.status.idle":"2026-02-07T15:37:58.484433Z","shell.execute_reply.started":"2026-02-07T15:37:58.466708Z","shell.execute_reply":"2026-02-07T15:37:58.483354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Locate Competition Assets","metadata":{}},{"cell_type":"code","source":"test_csv = find_file(\"PlantCLEF2025_test.csv\")\nspecies_csv = None\npseudo_urls_csv = None\n\n# Optional files (won't error if missing)\ntry:\n    species_csv = find_file(\"species_ids.csv\")\nexcept Exception:\n    pass\ntry:\n    pseudo_urls_csv = find_file(\"pseudoquadrats_without_labels_complementary_training_set_urls.csv\")\nexcept Exception:\n    pass\n\nprint(\"Found:\")\nprint(\" - test_csv:\", test_csv)\nif species_csv: \n    print(\" - species_ids.csv:\", species_csv)\nif pseudo_urls_csv: \n    print(\" - pseudo urls:\", pseudo_urls_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:37:58.485648Z","iopub.execute_input":"2026-02-07T15:37:58.486067Z","iopub.status.idle":"2026-02-07T15:38:00.081117Z","shell.execute_reply.started":"2026-02-07T15:37:58.486041Z","shell.execute_reply":"2026-02-07T15:38:00.080269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Explore Test Metadata","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(test_csv, sep=';')\nprint(\"Test metadata:\", test.shape)\ndisplay(test.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.082182Z","iopub.execute_input":"2026-02-07T15:38:00.082439Z","iopub.status.idle":"2026-02-07T15:38:00.102285Z","shell.execute_reply.started":"2026-02-07T15:38:00.082418Z","shell.execute_reply":"2026-02-07T15:38:00.101407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parse date\ntest[\"date\"] = pd.to_datetime(test[\"date\"], errors=\"coerce\")\nprint(\"Test date range:\", test[\"date\"].min(), \"→\", test[\"date\"].max())\nprint(\"Unique authors:\", test[\"author\"].nunique(), \"| Unique licenses:\", test[\"license\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.104578Z","iopub.execute_input":"2026-02-07T15:38:00.105332Z","iopub.status.idle":"2026-02-07T15:38:00.124815Z","shell.execute_reply.started":"2026-02-07T15:38:00.105304Z","shell.execute_reply":"2026-02-07T15:38:00.123626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Find Test Image Folder","metadata":{}},{"cell_type":"code","source":"candidate_dirs = []\nfor d in [\"PlantCLEF2025_test_images\", \"PlantCLEF2025test\", \"PlantCLEF2025_test\", \"PlantCLEF2025test_images\"]:\n    try:\n        candidate_dirs.append(find_dir(d))\n    except Exception:\n        pass\n\n# Deduplicate and expand nested case (folder inside folder with same name)\ncandidate_dirs = sorted(set(candidate_dirs), key=len)\n\ndef pick_image_root(candidates):\n    best = None\n    best_count = -1\n    best_files = None\n    for c in candidates:\n        files = list_image_files(c)\n        # If folder contains a subfolder with same name, test that too\n        nested = os.path.join(c, os.path.basename(c))\n        if os.path.isdir(nested):\n            files_nested = list_image_files(nested)\n            if len(files_nested) > len(files):\n                c = nested\n                files = files_nested\n        if len(files) > best_count:\n            best = c\n            best_count = len(files)\n            best_files = files\n    return best, best_files\n\nif not candidate_dirs:\n    raise RuntimeError(\"Could not find any likely test image directory under /kaggle/input/.\")\nimg_root, img_files = pick_image_root(candidate_dirs)\n\nprint(\"Chosen test image root:\", img_root)\nprint(\"Number of image files found:\", len(img_files))\nprint(\"Example image file:\", img_files[0] if img_files else \"None\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.126254Z","iopub.execute_input":"2026-02-07T15:38:00.126830Z","iopub.status.idle":"2026-02-07T15:38:00.405603Z","shell.execute_reply.started":"2026-02-07T15:38:00.126799Z","shell.execute_reply":"2026-02-07T15:38:00.404637Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Map Quadrat IDs to Image Paths","metadata":{}},{"cell_type":"code","source":"# Build lookup by filename stem\nstem_to_path = {}\nfor p in img_files:\n    stem_to_path[Path(p).stem] = p\n\ndef find_by_substring(quadrat_id: str):\n    \"\"\"Fallback: find any file whose name contains the quadrat_id\"\"\"\n    hits = [p for p in img_files if quadrat_id in Path(p).name]\n    return hits[0] if hits else None\n\ntest[\"image_path\"] = test[\"quadrat_id\"].map(stem_to_path)\nmissing = test[\"image_path\"].isna().sum()\n\nif missing > 0:\n    print(f\"{missing} quadrat_ids didn't match by exact stem. Trying substring fallback...\")\n    for i, row in test.loc[test[\"image_path\"].isna(), [\"quadrat_id\"]].iterrows():\n        test.at[i, \"image_path\"] = find_by_substring(row[\"quadrat_id\"])\n\nmissing2 = test[\"image_path\"].isna().sum()\nprint(\"Matched images:\", len(test) - missing2, \"/\", len(test))\nif missing2 > 0:\n    print(\"Still missing:\", missing2)\n    display(test.loc[test[\"image_path\"].isna(), [\"quadrat_id\"]].head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.406866Z","iopub.execute_input":"2026-02-07T15:38:00.407263Z","iopub.status.idle":"2026-02-07T15:38:00.431158Z","shell.execute_reply.started":"2026-02-07T15:38:00.407225Z","shell.execute_reply":"2026-02-07T15:38:00.430338Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Statistics Functions","metadata":{}},{"cell_type":"code","source":"DOWNSAMPLE_MAX = 384  # keeps some structure, still fast\n\ndef to_small_rgb(img: Image.Image, max_side=DOWNSAMPLE_MAX) -> Image.Image:\n    \"\"\"Convert to RGB and downsample for fast processing\"\"\"\n    img = ImageOps.exif_transpose(img)\n    img = img.convert(\"RGB\")\n    w, h = img.size\n    scale = max(w, h) / max_side\n    if scale > 1:\n        new_w = int(round(w / scale))\n        new_h = int(round(h / scale))\n        img = img.resize((new_w, new_h), resample=Image.BILINEAR)\n    return img\n\ndef luminance_stats(rgb: np.ndarray):\n    \"\"\"Compute brightness statistics from RGB array\"\"\"\n    # rgb uint8 [H,W,3]\n    r = rgb[..., 0].astype(np.float32)\n    g = rgb[..., 1].astype(np.float32)\n    b = rgb[..., 2].astype(np.float32)\n    y = 0.2126 * r + 0.7152 * g + 0.0722 * b  # 0..255 approx\n    return y.mean(), y.std(), (y < 35).mean(), y\n\ndef laplacian_var(gray: np.ndarray):\n    \"\"\"Compute blur proxy using Laplacian variance (higher = sharper)\"\"\"\n    # gray float32 [H,W] (0..255)\n    # Laplacian approx: -4I + N + S + E + W\n    c = gray\n    up = np.roll(c, 1, axis=0)\n    dn = np.roll(c, -1, axis=0)\n    lf = np.roll(c, 1, axis=1)\n    rt = np.roll(c, -1, axis=1)\n    lap = -4.0 * c + up + dn + lf + rt\n    return float(lap.var())\n\ndef compute_image_row(path: str):\n    \"\"\"Compute all statistics for a single image\"\"\"\n    p = Path(path)\n    size_bytes = p.stat().st_size if p.exists() else np.nan\n\n    with Image.open(path) as im:\n        im = ImageOps.exif_transpose(im)\n        w, h = im.size\n        mode = im.mode\n\n        small = to_small_rgb(im, max_side=DOWNSAMPLE_MAX)\n        arr = np.array(small)  # uint8\n        b_mean, b_std, dark_frac, y = luminance_stats(arr)\n        blur = laplacian_var(y.astype(np.float32))\n\n    return {\n        \"width\": int(w),\n        \"height\": int(h),\n        \"megapixels\": float(w * h) / 1e6,\n        \"aspect\": float(w) / float(h) if h else np.nan,\n        \"mode\": mode,\n        \"filesize_mb\": float(size_bytes) / (1024 * 1024),\n        \"brightness_mean\": float(b_mean),\n        \"brightness_std\": float(b_std),\n        \"dark_frac\": float(dark_frac),\n        \"blur_var\": float(blur),\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.432293Z","iopub.execute_input":"2026-02-07T15:38:00.432599Z","iopub.status.idle":"2026-02-07T15:38:00.453027Z","shell.execute_reply.started":"2026-02-07T15:38:00.432560Z","shell.execute_reply":"2026-02-07T15:38:00.451921Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compute Image Statistics","metadata":{}},{"cell_type":"code","source":"# Compute stats for rows with images\nrows = []\nidxs = test.index[test[\"image_path\"].notna()].tolist()\n\nprint(\"Computing image stats (this should be quick for ~2105 images)...\")\nfor i in tqdm(idxs):\n    path = test.at[i, \"image_path\"]\n    try:\n        rows.append((i, compute_image_row(path)))\n    except Exception as e:\n        rows.append((i, {\"error\": str(e)}))\n\nstats_df = pd.DataFrame([r for _, r in rows], index=[i for i, _ in rows])\ntest_eda = test.join(stats_df)\n\nprint(\"\\nEDA table:\", test_eda.shape)\ndisplay(test_eda.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:38:00.454226Z","iopub.execute_input":"2026-02-07T15:38:00.454600Z","iopub.status.idle":"2026-02-07T15:50:18.067288Z","shell.execute_reply.started":"2026-02-07T15:38:00.454574Z","shell.execute_reply":"2026-02-07T15:50:18.066199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Results","metadata":{}},{"cell_type":"code","source":"out_csv = \"/kaggle/working/plantclef_test_image_eda.csv\"\ntest_eda.to_csv(out_csv, index=False)\nprint(\"Saved:\", out_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:18.068624Z","iopub.execute_input":"2026-02-07T15:50:18.069476Z","iopub.status.idle":"2026-02-07T15:50:18.122971Z","shell.execute_reply.started":"2026-02-07T15:50:18.069435Z","shell.execute_reply":"2026-02-07T15:50:18.122059Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization Functions","metadata":{}},{"cell_type":"code","source":"def hist(series, bins=40, title=\"\", xlabel=\"\"):\n    \"\"\"Plot histogram of a series\"\"\"\n    s = series.dropna()\n    plt.figure()\n    plt.hist(s.values, bins=bins)\n    plt.title(title)\n    plt.xlabel(xlabel)\n    plt.ylabel(\"count\")\n    plt.show()\n\ndef bar_top(series, top=20, title=\"\"):\n    \"\"\"Plot bar chart of top values\"\"\"\n    vc = series.value_counts(dropna=False).head(top)\n    plt.figure()\n    vc.plot(kind=\"bar\")\n    plt.title(title)\n    plt.ylabel(\"count\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:18.123941Z","iopub.execute_input":"2026-02-07T15:50:18.124218Z","iopub.status.idle":"2026-02-07T15:50:18.131138Z","shell.execute_reply.started":"2026-02-07T15:50:18.124146Z","shell.execute_reply":"2026-02-07T15:50:18.130288Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Distribution Plots","metadata":{}},{"cell_type":"code","source":"bar_top(test_eda[\"author\"], 20, \"Test quadrats: author distribution\")\nbar_top(test_eda[\"license\"], 20, \"Test quadrats: license distribution\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:18.132219Z","iopub.execute_input":"2026-02-07T15:50:18.132534Z","iopub.status.idle":"2026-02-07T15:50:18.544321Z","shell.execute_reply.started":"2026-02-07T15:50:18.132509Z","shell.execute_reply":"2026-02-07T15:50:18.543357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hist(test_eda[\"width\"], title=\"Image width\", xlabel=\"pixels\")\nhist(test_eda[\"height\"], title=\"Image height\", xlabel=\"pixels\")\nhist(test_eda[\"megapixels\"], title=\"Megapixels per image\", xlabel=\"MP\")\nhist(test_eda[\"aspect\"], title=\"Aspect ratio (width/height)\", xlabel=\"ratio\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:18.545441Z","iopub.execute_input":"2026-02-07T15:50:18.545768Z","iopub.status.idle":"2026-02-07T15:50:19.190586Z","shell.execute_reply.started":"2026-02-07T15:50:18.545742Z","shell.execute_reply":"2026-02-07T15:50:19.189753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hist(test_eda[\"brightness_mean\"], title=\"Mean brightness (luminance)\", xlabel=\"0..255\")\nhist(test_eda[\"dark_frac\"], title=\"Dark pixel fraction (luminance < 35)\", xlabel=\"fraction\")\nhist(test_eda[\"blur_var\"], title=\"Blur proxy (Laplacian variance) — higher is sharper\", xlabel=\"variance\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:19.193535Z","iopub.execute_input":"2026-02-07T15:50:19.193935Z","iopub.status.idle":"2026-02-07T15:50:19.681849Z","shell.execute_reply.started":"2026-02-07T15:50:19.193907Z","shell.execute_reply":"2026-02-07T15:50:19.680904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scatter: width vs height\nplt.figure()\nxy = test_eda[[\"width\",\"height\"]].dropna()\nplt.scatter(xy[\"width\"], xy[\"height\"], s=6)\nplt.title(\"Width vs height\")\nplt.xlabel(\"width\")\nplt.ylabel(\"height\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:19.683015Z","iopub.execute_input":"2026-02-07T15:50:19.683366Z","iopub.status.idle":"2026-02-07T15:50:19.843612Z","shell.execute_reply.started":"2026-02-07T15:50:19.683332Z","shell.execute_reply":"2026-02-07T15:50:19.842707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Grid Visualization","metadata":{}},{"cell_type":"code","source":"def show_grid(paths, titles=None, cols=4, figsize=(14, 10), suptitle=None):\n    \"\"\"Display a grid of images\"\"\"\n    paths = [p for p in paths if isinstance(p, str) and os.path.exists(p)]\n    n = len(paths)\n    if n == 0:\n        print(\"No images to display.\")\n        return\n    rows = int(math.ceil(n / cols))\n    plt.figure(figsize=figsize)\n    for i, p in enumerate(paths, 1):\n        with Image.open(p) as im:\n            im = ImageOps.exif_transpose(im)\n            plt.subplot(rows, cols, i)\n            plt.imshow(im)\n            plt.axis(\"off\")\n            if titles is not None:\n                plt.title(titles[i-1], fontsize=9)\n    if suptitle:\n        plt.suptitle(suptitle)\n    plt.tight_layout()\n    plt.show()\n\ndef show_extremes(df, col, n=8, ascending=True, title=\"\"):\n    \"\"\"Show extreme values for a given column\"\"\"\n    sub = df[df[\"image_path\"].notna() & df[col].notna()].sort_values(col, ascending=ascending).head(n)\n    titles = [f\"{qid}\\n{col}={val:.2f}\" for qid, val in zip(sub[\"quadrat_id\"], sub[col])]\n    show_grid(sub[\"image_path\"].tolist(), titles=titles, suptitle=title, figsize=(14, 8))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:19.845015Z","iopub.execute_input":"2026-02-07T15:50:19.845392Z","iopub.status.idle":"2026-02-07T15:50:19.854847Z","shell.execute_reply.started":"2026-02-07T15:50:19.845353Z","shell.execute_reply":"2026-02-07T15:50:19.853649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Random Sample","metadata":{}},{"cell_type":"code","source":"sample = test_eda[test_eda[\"image_path\"].notna()].sample(n=min(12, len(test_eda)), random_state=SEED)\nshow_grid(\n    sample[\"image_path\"].tolist(),\n    titles=sample[\"quadrat_id\"].tolist(),\n    suptitle=\"Random test quadrat samples\",\n    figsize=(14, 10)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:19.856013Z","iopub.execute_input":"2026-02-07T15:50:19.856281Z","iopub.status.idle":"2026-02-07T15:50:28.770492Z","shell.execute_reply.started":"2026-02-07T15:50:19.856257Z","shell.execute_reply":"2026-02-07T15:50:28.769380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Extreme Cases: Brightness","metadata":{}},{"cell_type":"code","source":"show_extremes(test_eda, \"brightness_mean\", n=8, ascending=True,  title=\"Darkest images (by mean luminance)\")\nshow_extremes(test_eda, \"brightness_mean\", n=8, ascending=False, title=\"Brightest images (by mean luminance)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:28.771642Z","iopub.execute_input":"2026-02-07T15:50:28.772088Z","iopub.status.idle":"2026-02-07T15:50:38.875440Z","shell.execute_reply.started":"2026-02-07T15:50:28.772054Z","shell.execute_reply":"2026-02-07T15:50:38.874032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Extreme Cases: Blur/Sharpness","metadata":{}},{"cell_type":"code","source":"show_extremes(test_eda, \"blur_var\", n=8, ascending=True,  title=\"Most blurry (lowest Laplacian variance)\")\nshow_extremes(test_eda, \"blur_var\", n=8, ascending=False, title=\"Sharpest (highest Laplacian variance)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:50:38.876810Z","iopub.execute_input":"2026-02-07T15:50:38.877075Z","iopub.status.idle":"2026-02-07T15:50:47.999683Z","shell.execute_reply.started":"2026-02-07T15:50:38.877044Z","shell.execute_reply":"2026-02-07T15:50:47.997281Z"}},"outputs":[],"execution_count":null}]}