{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":124685,"databundleVersionId":14664296}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-09T23:54:15.964871Z","iopub.execute_input":"2026-04-09T23:54:15.965261Z","iopub.status.idle":"2026-04-09T23:54:22.363481Z","shell.execute_reply.started":"2026-04-09T23:54:15.96523Z","shell.execute_reply":"2026-04-09T23:54:22.361752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image\nimport random\n\n# --- Update this to match the actual competition slug on Kaggle ---\n# After you \"Add Data\" in your Kaggle notebook, check the path:\nBASE_DIR = Path(\"/kaggle/input/competitions/plantclef-2026/\")  # adjust if slug differs\n\n# List everything at the top level\nfor p in sorted(BASE_DIR.iterdir()):\n    kind = \"DIR\" if p.is_dir() else \"FILE\"\n    size = p.stat().st_size if p.is_file() else \"\"\n    print(f\"  {kind}  {p.name}  {size}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-09T23:57:17.546174Z","iopub.execute_input":"2026-04-09T23:57:17.54724Z","iopub.status.idle":"2026-04-09T23:57:17.561449Z","shell.execute_reply.started":"2026-04-09T23:57:17.547192Z","shell.execute_reply":"2026-04-09T23:57:17.560445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the official species list\nspecies_path = BASE_DIR / \"species_ids.csv\"\nif species_path.exists():\n    species_df = pd.read_csv(species_path)\n    print(f\"Total species: {len(species_df)}\")\n    print(f\"Columns: {list(species_df.columns)}\")\n    species_df.head(10)\nelse:\n    print(f\"species_ids.csv not found at {species_path} — check BASE_DIR\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-09T23:57:31.429394Z","iopub.execute_input":"2026-04-09T23:57:31.430176Z","iopub.status.idle":"2026-04-09T23:57:31.456529Z","shell.execute_reply.started":"2026-04-09T23:57:31.430136Z","shell.execute_reply":"2026-04-09T23:57:31.45562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training metadata — images must be downloaded externally (~160-281 GB)\ntrain_meta_path = BASE_DIR / \"PlantCLEF2024_single_plant_training_metadata.csv\"\n\n#train_meta_path= \"/kaggle/input/competitions/plantclef-2026/PlantCLEF2024_single_plant_training_metadata.csv\"\nif train_meta_path.exists():\n    train_df = pd.read_csv(train_meta_path, sep=\";\")\n    print(f\"Training metadata rows : {len(train_df):,}\")\n    print(f\"Columns               : {list(train_df.columns)}\")\n    print(f\"Unique species (approx): {train_df['species_id'].nunique() if 'species_id' in train_df.columns else 'N/A'}\")\n    train_df.head()\nelse:\n    print(f\"Training metadata CSV not found at {train_meta_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:00:33.707095Z","iopub.execute_input":"2026-04-10T00:00:33.70792Z","iopub.status.idle":"2026-04-10T00:00:56.716022Z","shell.execute_reply.started":"2026-04-10T00:00:33.707886Z","shell.execute_reply":"2026-04-10T00:00:56.714828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of images per species\nif 'train_df' in dir() and 'species_id' in train_df.columns:\n    imgs_per_species = train_df['species_id'].value_counts()\n    \n    fig, axes = plt.subplots(1, 2, figsize=(14, 4))\n    \n    axes[0].hist(imgs_per_species.values, bins=80, edgecolor='black', alpha=0.7)\n    axes[0].set_xlabel(\"Images per species\")\n    axes[0].set_ylabel(\"Number of species\")\n    axes[0].set_title(\"Distribution of images per species\")\n    \n    axes[1].hist(imgs_per_species.values, bins=80, edgecolor='black', alpha=0.7, log=True)\n    axes[1].set_xlabel(\"Images per species\")\n    axes[1].set_ylabel(\"Number of species (log)\")\n    axes[1].set_title(\"Same — log scale\")\n    \n    plt.tight_layout()\n    plt.show()\n    \n    print(f\"Min images/species : {imgs_per_species.min()}\")\n    print(f\"Median             : {imgs_per_species.median():.0f}\")\n    print(f\"Max                : {imgs_per_species.max()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:01:18.756624Z","iopub.execute_input":"2026-04-10T00:01:18.757148Z","iopub.status.idle":"2026-04-10T00:01:19.640003Z","shell.execute_reply.started":"2026-04-10T00:01:18.757111Z","shell.execute_reply":"2026-04-10T00:01:19.639192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test metadata\ntest_meta_path = BASE_DIR / \"PlantCLEF2025_test.csv\"\nif test_meta_path.exists():\n    test_df = pd.read_csv(test_meta_path)\n    print(f\"Test metadata rows: {len(test_df)}\")\n    print(f\"Columns           : {list(test_df.columns)}\")\n    test_df.head()\nelse:\n    print(f\"Test metadata CSV not found at {test_meta_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:01:53.057821Z","iopub.execute_input":"2026-04-10T00:01:53.05815Z","iopub.status.idle":"2026-04-10T00:01:53.081456Z","shell.execute_reply.started":"2026-04-10T00:01:53.058121Z","shell.execute_reply":"2026-04-10T00:01:53.08036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Browse test images\ntest_img_dir = BASE_DIR / \"PlantCLEF2025_test_images\" / \"PlantCLEF2025_test_images\"\n\n\n\nprint(f\"Test images found: {len(test_images)}\")\nif not test_img_dir.exists():\n    # Try alternate naming\n    candidates = [d for d in BASE_DIR.iterdir() if d.is_dir() and 'test' in d.name.lower()]\n    test_img_dir = candidates[0] if candidates else test_img_dir\n\nif test_img_dir.exists():\n    test_images = sorted(test_img_dir.glob(\"*.jpg\")) + \\\n              sorted(test_img_dir.glob(\"*.JPG\")) + \\\n              sorted(test_img_dir.glob(\"*.png\"))\n    print(f\"Test images found: {len(test_images)}\")\n    \n    # Show a sample grid\n    sample = random.sample(test_images, min(9, len(test_images)))\n    fig, axes = plt.subplots(3, 3, figsize=(14, 14))\n    for ax, img_path in zip(axes.flat, sample):\n        img = Image.open(img_path)\n        ax.imshow(img)\n        ax.set_title(img_path.name, fontsize=8)\n        ax.axis(\"off\")\n    for ax in axes.flat[len(sample):]:\n        ax.axis(\"off\")\n    plt.suptitle(\"Sample Test Quadrat Images\", fontsize=14)\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(f\"Test image directory not found. Available dirs: {[d.name for d in BASE_DIR.iterdir() if d.is_dir()]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:06:02.741286Z","iopub.execute_input":"2026-04-10T00:06:02.742069Z","iopub.status.idle":"2026-04-10T00:06:10.495278Z","shell.execute_reply.started":"2026-04-10T00:06:02.742034Z","shell.execute_reply":"2026-04-10T00:06:10.493589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check image resolutions in the test set\nif test_img_dir.exists() and len(test_images) > 0:\n    widths, heights = [], []\n    for p in test_images[:200]:  # sample first 200 for speed\n        with Image.open(p) as im:\n            w, h = im.size\n            widths.append(w)\n            heights.append(h)\n    \n    fig, ax = plt.subplots(figsize=(7, 5))\n    ax.scatter(widths, heights, alpha=0.4, s=10)\n    ax.set_xlabel(\"Width (px)\")\n    ax.set_ylabel(\"Height (px)\")\n    ax.set_title(\"Test Image Resolutions\")\n    plt.tight_layout()\n    plt.show()\n    \n    print(f\"Width  — min: {min(widths)}, max: {max(widths)}, median: {np.median(widths):.0f}\")\n    print(f\"Height — min: {min(heights)}, max: {max(heights)}, median: {np.median(heights):.0f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:06:21.852159Z","iopub.execute_input":"2026-04-10T00:06:21.852507Z","iopub.status.idle":"2026-04-10T00:06:25.340406Z","shell.execute_reply.started":"2026-04-10T00:06:21.852477Z","shell.execute_reply":"2026-04-10T00:06:25.339403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pseudo_path = BASE_DIR / \"pseudoquadrats_without_labels_complementary_training_set_urls.csv\"\nif pseudo_path.exists():\n    pseudo_df = pd.read_csv(pseudo_path)\n    print(f\"Pseudo-quadrat URLs: {len(pseudo_df):,}\")\n    print(f\"Columns: {list(pseudo_df.columns)}\")\n    pseudo_df.head()\nelse:\n    print(f\"Pseudo-quadrat CSV not found at {pseudo_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:06:32.931847Z","iopub.execute_input":"2026-04-10T00:06:32.932179Z","iopub.status.idle":"2026-04-10T00:06:33.902322Z","shell.execute_reply.started":"2026-04-10T00:06:32.932152Z","shell.execute_reply":"2026-04-10T00:06:33.901421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for bundled model weights\nmodel_dirs = [d for d in BASE_DIR.iterdir() if d.is_dir() and 'dinov2' in d.name.lower()]\nif model_dirs:\n    for md in model_dirs:\n        files = list(md.rglob(\"*\"))\n        total_mb = sum(f.stat().st_size for f in files if f.is_file()) / 1e6\n        print(f\"{md.name}  —  {len(files)} files, {total_mb:.1f} MB\")\nelse:\n    print(\"No pre-trained model directories found in the dataset.\")\n    print(\"You can add them from the competition's Models tab or download from Zenodo.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:06:41.714745Z","iopub.execute_input":"2026-04-10T00:06:41.715093Z","iopub.status.idle":"2026-04-10T00:06:41.727954Z","shell.execute_reply.started":"2026-04-10T00:06:41.715065Z","shell.execute_reply":"2026-04-10T00:06:41.726709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"=\" * 60)\nprint(\"PlantCLEF 2026 Dataset Summary\")\nprint(\"=\" * 60)\nif 'species_df' in dir():\n    print(f\"  Species             : {len(species_df):,}\")\nif 'train_df' in dir():\n    print(f\"  Training metadata   : {len(train_df):,} rows\")\n    if 'species_id' in train_df.columns:\n        print(f\"  Training species    : {train_df['species_id'].nunique():,}\")\nif 'test_df' in dir():\n    print(f\"  Test metadata       : {len(test_df):,} rows\")\nif test_img_dir.exists():\n    print(f\"  Test images on disk : {len(test_images):,}\")\nif 'pseudo_df' in dir():\n    print(f\"  Pseudo-quadrat URLs : {len(pseudo_df):,}\")\nprint(\"=\" * 60)\nprint(\"\\nNOTE: Training images (~160-281 GB) must be downloaded\")\nprint(\"externally via the links in the competition description.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T00:07:09.733246Z","iopub.execute_input":"2026-04-10T00:07:09.733841Z","iopub.status.idle":"2026-04-10T00:07:09.759998Z","shell.execute_reply.started":"2026-04-10T00:07:09.733804Z","shell.execute_reply":"2026-04-10T00:07:09.758666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}