{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":89850,"databundleVersionId":11256103,"sourceType":"competition"},{"sourceId":228781,"sourceType":"modelInstanceVersion","modelInstanceId":195042,"modelId":216938}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction\nThe purpose of this notebook is to show how to load the Dinov2 finetunes model provided by the organisers of the PlantCLEF2025 competition and infer on a random test image.","metadata":{}},{"cell_type":"markdown","source":"- hi guys, this is a copy of the official starter notebook\n- but I've added comments and explicitly printed some intermediate steps\n- I'm new to computer vision so these steps help me understand better, just sharing for others too!","metadata":{}},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\n# library for pre-trained cv models \nimport timm                              \nimport torch","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:27.173895Z","iopub.execute_input":"2025-03-26T15:30:27.174254Z","iopub.status.idle":"2025-03-26T15:30:27.178503Z","shell.execute_reply.started":"2025-03-26T15:30:27.174211Z","shell.execute_reply":"2025-03-26T15:30:27.177523Z"}},"outputs":[],"execution_count":15},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Load the csv data\ndf_species_ids = pd.read_csv('/kaggle/input/plantclef-2025/species_ids.csv')\ndf_metadata = pd.read_csv('/kaggle/input/plantclef-2025/PlantCLEF2024_single_plant_training_metadata.csv', sep=';', dtype={'partner': str})\nid_to_species = df_metadata[['species_id', 'species']].drop_duplicates().set_index('species_id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:27.179738Z","iopub.execute_input":"2025-03-26T15:30:27.180044Z","iopub.status.idle":"2025-03-26T15:30:39.021299Z","shell.execute_reply.started":"2025-03-26T15:30:27.180017Z","shell.execute_reply":"2025-03-26T15:30:39.020235Z"}},"outputs":[],"execution_count":16},{"cell_type":"markdown","source":"# Load random image","metadata":{}},{"cell_type":"code","source":"# Load a specific image\nimg = Image.open('/kaggle/input/plantclef-2025/PlantCLEF2025_test_images/PlantCLEF2025_test_images/GUARDEN-CBNMed-30-4-16-3-20240428.jpg')\nplt.imshow(img)\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:39.023345Z","iopub.execute_input":"2025-03-26T15:30:39.023597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.size","metadata":{"trusted":true,"execution":{"iopub.status.idle":"2025-03-26T15:30:39.617978Z","shell.execute_reply.started":"2025-03-26T15:30:39.6129Z","shell.execute_reply":"2025-03-26T15:30:39.616906Z"}},"outputs":[{"execution_count":18,"output_type":"execute_result","data":{"text/plain":"(2274, 2274)"},"metadata":{}}],"execution_count":18},{"cell_type":"markdown","source":"- that partiular image size is 2274 x 2274 pixels","metadata":{}},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"# Set the device to 'cuda' (GPU) for training or inference if a GPU is available\ndevice = torch.device('cuda')\n\n# Create the Vision Transformer model (ViT) with specific configuration\nmodel = timm.create_model(\n    # Model architecture (predefined ViT model)\n    'vit_base_patch14_reg4_dinov2.lvd142m',  \n    # Do not load the default pretrained weights, model weights initialised randomly\n    pretrained=False,      \n    # Set the number of output classes based on the length of the species IDs\n    num_classes=len(df_species_ids),        \n    checkpoint_path='/kaggle/input/dinov2_patch14_reg4_onlyclassifier_then_all/pytorch/default/3/model_best.pth.tar'  # Path to the pre-trained model checkpoint\n)\n\n# Move the model to the specified device (GPU if available)\nmodel = model.to(device)\n\n# Set the model to evaluation mode (important for inference to disable certain layers like dropout)\nmodel = model.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:39.619035Z","iopub.execute_input":"2025-03-26T15:30:39.619332Z","iopub.status.idle":"2025-03-26T15:30:42.032426Z","shell.execute_reply.started":"2025-03-26T15:30:39.619302Z","shell.execute_reply":"2025-03-26T15:30:42.031481Z"}},"outputs":[],"execution_count":19},{"cell_type":"markdown","source":"# Load Model Configurations","metadata":{}},{"cell_type":"code","source":"# Get model-specific configuration, including normalization and resizing transforms\ndata_config = timm.data.resolve_model_data_config(model)\ndata_config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:42.033385Z","iopub.execute_input":"2025-03-26T15:30:42.03364Z","iopub.status.idle":"2025-03-26T15:30:42.038915Z","shell.execute_reply.started":"2025-03-26T15:30:42.033582Z","shell.execute_reply":"2025-03-26T15:30:42.038201Z"}},"outputs":[{"execution_count":20,"output_type":"execute_result","data":{"text/plain":"{'input_size': (3, 518, 518),\n 'interpolation': 'bicubic',\n 'mean': (0.485, 0.456, 0.406),\n 'std': (0.229, 0.224, 0.225),\n 'crop_pct': 1.0,\n 'crop_mode': 'center'}"},"metadata":{}}],"execution_count":20},{"cell_type":"markdown","source":"- model expects input size 3 x 518 x 518, 3 color channels and size of 518 x 518\n- `bicubic` is an interpolation method to resize images. considers the nearest 16 pixels to compute the new values\n- often used for high quality resizing","metadata":{}},{"cell_type":"markdown","source":"- `crop_pct` is the percentage of the image to crop when performing data augmentation\n- 1.0 means no cropping","metadata":{}},{"cell_type":"markdown","source":"- center means the image will be cropped from the center","metadata":{}},{"cell_type":"markdown","source":"# Apply transformations","metadata":{}},{"cell_type":"code","source":"# Create the necessary transformation (resize, normalize) for the input image based on the model's config\ntransforms = timm.data.create_transform(**data_config, is_training=False)\ntransforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:42.03973Z","iopub.execute_input":"2025-03-26T15:30:42.039959Z","iopub.status.idle":"2025-03-26T15:30:42.052724Z","shell.execute_reply.started":"2025-03-26T15:30:42.039924Z","shell.execute_reply":"2025-03-26T15:30:42.05207Z"}},"outputs":[{"execution_count":21,"output_type":"execute_result","data":{"text/plain":"Compose(\n    Resize(size=518, interpolation=bicubic, max_size=None, antialias=True)\n    CenterCrop(size=(518, 518))\n    MaybeToTensor()\n    Normalize(mean=tensor([0.4850, 0.4560, 0.4060]), std=tensor([0.2290, 0.2240, 0.2250]))\n)"},"metadata":{}}],"execution_count":21},{"cell_type":"markdown","source":"- resize to shorter size 518 pixels\n- center crop to 518 x 518\n- converts images to tensor if it already isnt\n- normalised using specific mean and std","metadata":{}},{"cell_type":"markdown","source":"# Extract Top 5 Predictions","metadata":{}},{"cell_type":"code","source":"# Perform inference without updating gradients (to save memory)\nwith torch.no_grad():\n    # Check if an image is provided (img is not None)\n    if img != None:\n        # Apply transformations to the image (resize, normalize) and add a batch dimension\n        img = transforms(img).unsqueeze(0)  # unsqueeze adds a batch dimension (turning it into a batch of 1)\n        \n        # Move the image tensor to the specified device (GPU or CPU)\n        img = img.to(device)\n\n        # Perform the forward pass to get predictions\n        output = model(img)  # Run the model on the image\n\n        # Get the top 5 predictions (probabilities and class indices)\n        top5_probabilities, top5_class_indices = torch.topk(output.softmax(dim=1), k=5)\n        \n        # Move the top 5 probabilities and class indices to CPU and convert to numpy arrays for further handling\n        top5_probabilities = top5_probabilities.cpu().detach().numpy()\n        top5_class_indices = top5_class_indices.cpu().detach().numpy()\n    \n        # Loop through the top 5 predictions and print species ID, species name, and probability\n        for proba, cid in zip(top5_probabilities[0], top5_class_indices[0]):\n            # Map class index to species ID\n            species_id = df_species_ids.iloc[cid].item()\n            \n            # Map species ID to species name using the 'id_to_species' mapping\n            species = id_to_species.loc[species_id].item()\n\n            # Print the species ID, name, and prediction probability\n            print(species_id, species, proba)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-26T15:30:42.053644Z","iopub.execute_input":"2025-03-26T15:30:42.053931Z","iopub.status.idle":"2025-03-26T15:30:42.259962Z","shell.execute_reply.started":"2025-03-26T15:30:42.053902Z","shell.execute_reply":"2025-03-26T15:30:42.259232Z"}},"outputs":[{"name":"stdout","text":"1394664 Anthemis tomentosa L. 0.10699336\n1398319 Anthemis ruthenica M.Bieb. 0.06801872\n1357220 Anthemis cretica L. 0.06224221\n1357227 Anthemis maritima L. 0.045665238\n1357236 Anthemis secundiramea Biv. 0.038899273\n","output_type":"stream"}],"execution_count":22},{"cell_type":"markdown","source":"- Highest classification score is only 10%, pretty bad","metadata":{}},{"cell_type":"markdown","source":"As observed from the predictions, since the model is trained on a mono-label dataset, two predicted species could both be present in the quadrat, or the model might be uncertain between them.","metadata":{}}]}