{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":89850,"databundleVersionId":11256103,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nimport requests\nimport PIL\nfrom io import BytesIO","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:08:45.356943Z","iopub.execute_input":"2025-03-20T15:08:45.357325Z","iopub.status.idle":"2025-03-20T15:08:46.657836Z","shell.execute_reply.started":"2025-03-20T15:08:45.357285Z","shell.execute_reply":"2025-03-20T15:08:46.656706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = Path(\"/kaggle/input/plantclef-2025\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:08:46.659141Z","iopub.execute_input":"2025-03-20T15:08:46.659805Z","iopub.status.idle":"2025-03-20T15:08:46.664497Z","shell.execute_reply.started":"2025-03-20T15:08:46.659752Z","shell.execute_reply":"2025-03-20T15:08:46.663383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Data: Individual plant images\nThe csv file uses `;` as seperators","metadata":{}},{"cell_type":"code","source":"df_trn = pd.read_csv(data_dir/\"PlantCLEF2024_single_plant_training_metadata.csv\", sep=\";\", low_memory=False)\nprint (f\"Shape: {df_trn.shape}\")\ndf_trn.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:08:46.667260Z","iopub.execute_input":"2025-03-20T15:08:46.667673Z","iopub.status.idle":"2025-03-20T15:09:10.688266Z","shell.execute_reply.started":"2025-03-20T15:08:46.667628Z","shell.execute_reply":"2025-03-20T15:09:10.687142Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The training dataset consists of **image URLs** along with various metadata, including:  \n- **`species_id`** – Unique identifier for the plant species. **(target)** \n- **`organ`** – The part of the plant captured in the image (e.g., leaf, flower, fruit).  \n- **`author`** – The contributor who provided the image.  \n- **`latitude` / `longitude` / `altitude`** – Geolocation data of the observation.  \n- **`license`** – The usage rights associated with the image.  \n- **`dataset` / `publisher`** – The source of the data.  \n- **`references`** – Links to external resources related to the observation. ","metadata":{}},{"cell_type":"markdown","source":"Images belonging to the same `species_id` are stored together in consecutive rows within the dataframe according to the comp's description. ","metadata":{}},{"cell_type":"markdown","source":"## Learn tags\nThey also seperate train/validation/test data for us.","metadata":{}},{"cell_type":"code","source":"learn_tag_counts = df_trn['learn_tag'].value_counts()\nlearn_tag_counts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:10.689994Z","iopub.execute_input":"2025-03-20T15:09:10.690288Z","iopub.status.idle":"2025-03-20T15:09:10.771785Z","shell.execute_reply.started":"2025-03-20T15:09:10.690264Z","shell.execute_reply":"2025-03-20T15:09:10.770562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for tag, count in learn_tag_counts.items():\n    percent = (count / len(df_trn)) * 100\n    print(f\"{tag}: {percent:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:10.773017Z","iopub.execute_input":"2025-03-20T15:09:10.773494Z","iopub.status.idle":"2025-03-20T15:09:10.796417Z","shell.execute_reply.started":"2025-03-20T15:09:10.773418Z","shell.execute_reply":"2025-03-20T15:09:10.795169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_trn.groupby(\"learn_tag\")[\"species_id\"].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:10.797564Z","iopub.execute_input":"2025-03-20T15:09:10.797878Z","iopub.status.idle":"2025-03-20T15:09:10.939990Z","shell.execute_reply.started":"2025-03-20T15:09:10.797853Z","shell.execute_reply":"2025-03-20T15:09:10.938804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Display sample images","metadata":{}},{"cell_type":"code","source":"sampled_data = {}\nfor tag in [\"train\", \"val\", \"test\"]:\n    sampled_data[tag] = df_trn[df_trn[\"learn_tag\"] == tag].sample(5)\n\nfig, axes = plt.subplots(nrows=3, ncols=5, figsize=(15, 9))\nfig.subplots_adjust(hspace=0.5)\n\nfor row, (tag, data) in enumerate(sampled_data.items()):\n    for col, (url, species_id) in enumerate(zip(data[\"url\"], data[\"species_id\"])):\n        ax = axes[row, col]  # Lấy ô tương ứng\n        try:\n            response = requests.get(url, timeout=5)\n            img = PIL.Image.open(BytesIO(response.content))\n            ax.imshow(img)\n            ax.set_title(f\"{tag}\\nSpecies: {species_id}\", fontsize=10)\n            ax.axis(\"off\")\n        except Exception as e:\n            ax.set_title(f\"{tag}\\nError\", fontsize=10)\n            ax.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:10.941106Z","iopub.execute_input":"2025-03-20T15:09:10.941516Z","iopub.status.idle":"2025-03-20T15:09:26.262165Z","shell.execute_reply.started":"2025-03-20T15:09:10.941477Z","shell.execute_reply":"2025-03-20T15:09:26.260622Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Which organs are shown in the pictures?","metadata":{}},{"cell_type":"code","source":"organ_counts = df_trn[\"organ\"].value_counts()\n\nplt.figure(figsize=(6, 6))\nplt.pie(organ_counts, labels=organ_counts.index, autopct=\"%1.1f%%\", startangle=140, colors=plt.cm.Paired.colors)\n\nplt.title(\"Organ Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:26.263217Z","iopub.execute_input":"2025-03-20T15:09:26.263548Z","iopub.status.idle":"2025-03-20T15:09:26.492601Z","shell.execute_reply.started":"2025-03-20T15:09:26.263520Z","shell.execute_reply":"2025-03-20T15:09:26.491145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## What family do those plants belong to?","metadata":{}},{"cell_type":"code","source":"species_counts = df_trn['species'].value_counts()\ngenus_counts = df_trn['genus'].value_counts()\nfamily_counts = df_trn['family'].value_counts()\n\nprint(f\"There are {len(species_counts)} species, {len(genus_counts)} genuses and {len(family_counts)} families\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:26.493901Z","iopub.execute_input":"2025-03-20T15:09:26.494213Z","iopub.status.idle":"2025-03-20T15:09:26.726089Z","shell.execute_reply.started":"2025-03-20T15:09:26.494187Z","shell.execute_reply":"2025-03-20T15:09:26.725004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_species = species_counts.nlargest(10)\ntop_genuses = genus_counts.nlargest(10)\ntop_families = family_counts.nlargest(10)\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\naxes[0].pie(top_species, labels=top_species.index, autopct='%1.1f%%', startangle=90)\naxes[0].set_title(\"Top 10 popular species\")\n\naxes[1].pie(top_genuses, labels=top_genuses.index, autopct='%1.1f%%', startangle=90)\naxes[1].set_title(\"Top 10 popular genuses\")\n\naxes[2].pie(top_genuses, labels=top_genuses.index, autopct='%1.1f%%', startangle=90)\naxes[2].set_title(\"Top 10 popular families\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:26.727258Z","iopub.execute_input":"2025-03-20T15:09:26.727701Z","iopub.status.idle":"2025-03-20T15:09:27.280041Z","shell.execute_reply.started":"2025-03-20T15:09:26.727645Z","shell.execute_reply":"2025-03-20T15:09:27.278851Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Data: Vegetation quadrat images","metadata":{}},{"cell_type":"code","source":"df_tst = pd.read_csv(data_dir/\"PlantCLEF2025_test.csv\", sep=';')\ndf_tst.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:27.281077Z","iopub.execute_input":"2025-03-20T15:09:27.281355Z","iopub.status.idle":"2025-03-20T15:09:27.300661Z","shell.execute_reply.started":"2025-03-20T15:09:27.281332Z","shell.execute_reply":"2025-03-20T15:09:27.299307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Display sample images","metadata":{}},{"cell_type":"code","source":"img_dir = Path(data_dir/\"PlantCLEF2025_test_images/PlantCLEF2025_test_images\")\nimg_files = [f for f in os.listdir(img_dir)]\nsample_images = random.sample(img_files, 5)\n\nfig, axes = plt.subplots(1, len(sample_images), figsize=(15, 5))\nfor ax, img_name in zip(axes, sample_images):\n    img_path = os.path.join(img_dir, img_name)\n    img = PIL.Image.open(img_path)\n    ax.imshow(img)\n    ax.set_title(img_name[:20])\n    ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:27.304535Z","iopub.execute_input":"2025-03-20T15:09:27.304868Z","iopub.status.idle":"2025-03-20T15:09:33.349008Z","shell.execute_reply.started":"2025-03-20T15:09:27.304840Z","shell.execute_reply":"2025-03-20T15:09:33.347358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Unlike standard image classification tasks where a model predicts a single label for each image, this problem requires identifying **multiple species** from a quadrat.","metadata":{}},{"cell_type":"markdown","source":"## Authors\nThere are only a few authors of the public test images","metadata":{}},{"cell_type":"code","source":"author_counts = df_tst['author'].value_counts()\n\nplt.figure(figsize=(6, 6))\nplt.pie(author_counts, labels=author_counts.index, autopct=\"%1.1f%%\", startangle=140, colors=plt.cm.Paired.colors)\n\nplt.title(\"Authors\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:33.351065Z","iopub.execute_input":"2025-03-20T15:09:33.351502Z","iopub.status.idle":"2025-03-20T15:09:33.668648Z","shell.execute_reply.started":"2025-03-20T15:09:33.351463Z","shell.execute_reply":"2025-03-20T15:09:33.667542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Do test authors appear in train?","metadata":{}},{"cell_type":"code","source":"train_authors = set(df_trn[\"author\"])\ntest_authors = set(df_tst[\"author\"])\n\ncommon_authors = test_authors.intersection(train_authors)\nprint(f\"Authors appeared in both train and test: {len(common_authors)} / {len(test_authors)}\")\nprint(list(common_authors))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:33.669804Z","iopub.execute_input":"2025-03-20T15:09:33.670099Z","iopub.status.idle":"2025-03-20T15:09:33.959742Z","shell.execute_reply.started":"2025-03-20T15:09:33.670075Z","shell.execute_reply":"2025-03-20T15:09:33.958433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_common_authors = df_trn[df_trn[\"author\"].isin(common_authors)]\nspecies_per_author_train = train_common_authors.groupby(\"author\")[\"species_id\"].nunique()\n\nprint(\"The amount of species each authot in common authors took pictures:\")\nprint(species_per_author_train.sort_values(ascending=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:33.960868Z","iopub.execute_input":"2025-03-20T15:09:33.961199Z","iopub.status.idle":"2025-03-20T15:09:34.113145Z","shell.execute_reply.started":"2025-03-20T15:09:33.961162Z","shell.execute_reply":"2025-03-20T15:09:34.111796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I wonder if these authors only photographed the same plant species in the test set as in the train set...","metadata":{}},{"cell_type":"markdown","source":"# No Labels Complementary Train Data\nAccording to the comp's desc, *the motivation behind this second dataset is to help models better adapt to multi-species vegetation quadrat images.*","metadata":{}},{"cell_type":"code","source":"df_com_trn = pd.read_csv(data_dir/\"pseudoquadrats_without_labels_complementary_training_set_urls.csv\")\nprint(df_com_trn.shape)\ndf_com_trn.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:34.114607Z","iopub.execute_input":"2025-03-20T15:09:34.115062Z","iopub.status.idle":"2025-03-20T15:09:34.753826Z","shell.execute_reply.started":"2025-03-20T15:09:34.115020Z","shell.execute_reply":"2025-03-20T15:09:34.752748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_data = df_com_trn.sample(5)\n\nfig, axes = plt.subplots(1, len(sample_data), figsize=(15, 5))\n\nfor ax, url in zip(axes, sample_data.iloc[:, 0]):\n    try:\n        response = requests.get(url, timeout=5)\n        img = PIL.Image.open(BytesIO(response.content))\n        ax.imshow(img)\n        ax.axis(\"off\")\n    except Exception as e:\n        ax.set_title(f\"Error: {e}\")\n        ax.axis(\"off\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T15:09:34.754959Z","iopub.execute_input":"2025-03-20T15:09:34.755385Z","iopub.status.idle":"2025-03-20T15:09:42.617170Z","shell.execute_reply.started":"2025-03-20T15:09:34.755344Z","shell.execute_reply":"2025-03-20T15:09:42.615859Z"}},"outputs":[],"execution_count":null}]}