{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91448,"databundleVersionId":12156235,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"data_path = '/kaggle/input/fungi-clef-2025/'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:38:51.736356Z","iopub.execute_input":"2025-10-08T00:38:51.736550Z","iopub.status.idle":"2025-10-08T00:38:51.743001Z","shell.execute_reply.started":"2025-10-08T00:38:51.736533Z","shell.execute_reply":"2025-10-08T00:38:51.742288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install git+https://github.com/mlfoundations/open_clip.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:38:51.743821Z","iopub.execute_input":"2025-10-08T00:38:51.744085Z","iopub.status.idle":"2025-10-08T00:40:28.652928Z","shell.execute_reply.started":"2025-10-08T00:38:51.744060Z","shell.execute_reply":"2025-10-08T00:40:28.652122Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport yaml\nfrom pathlib import Path\nfrom types import SimpleNamespace\nimport argparse\n\nimport numpy as np\nimport pandas as pd\nimport torch\n# import faiss\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom torchvision import transforms as tfms\nimport torchvision.transforms as T\nimport open_clip\n\nfrom typing import Sequence, Tuple, Any, Dict, List, Optional, Union\nimport importlib","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:40:28.655113Z","iopub.execute_input":"2025-10-08T00:40:28.655365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class FungiTastic(torch.nn.Module):\n    \"\"\"\n    Dataset class for the FewShot subset of the Danish Fungi dataset (size 300, closed-set).\n\n    This dataset loader supports training, validation, and testing splits, and provides\n    convenient access to images, class IDs, and file paths. It also supports optional\n    image transformations.\n    \"\"\"\n\n    SPLIT2STR = {'train': 'Train', 'val': 'Val', 'test': 'Test'}\n\n    def __init__(self, root: str, split: str = 'val', transform=None):\n        \"\"\"\n        Initializes the FungiTastic dataset.\n\n        Args:\n            root (str): The root directory of the dataset.\n            split (str, optional): The dataset split to use. Must be one of {'train', 'val', 'test'}.\n                Defaults to 'val'.\n            transform (callable, optional): Optional transform to be applied on a sample.\n        \"\"\"\n        super().__init__()\n        self.split = split\n        self.transform = transform\n        self.df = self._get_df(root, split)\n\n        assert \"image_path\" in self.df\n        if self.split != 'test':\n            assert \"category_id\" in self.df\n            self.n_classes = len(self.df['category_id'].unique())\n            self.category_id2label = {\n                k: v[0] for k, v in self.df.groupby('category_id')['species'].unique().to_dict().items()\n            }\n            self.label2category_id = {\n                v: k for k, v in self.category_id2label.items()\n            }\n\n    def add_embeddings(self, embeddings: pd.DataFrame):\n        \"\"\"\n        Updates the dataset instance with new embeddings.\n\n        Args:\n            embeddings (pd.DataFrame): A DataFrame containing an 'embedding' column.\n                                       It must align with `self.df` in terms of indexing.\n        \"\"\"\n        assert isinstance(embeddings, pd.DataFrame), \"Embeddings must be a pandas DataFrame.\"\n        assert \"embedding\" in embeddings.columns, \"Embeddings DataFrame must have an 'embedding' column.\"\n        assert len(embeddings) == len(self.df), \"Embeddings must match dataset length.\"\n\n        self.df = pd.merge(self.df, embeddings, on=\"filename\", how=\"inner\")\n\n    def get_embeddings_for_class(self, id):\n        # return the embeddings for class class_idx\n        class_idxs = self.df[self.df['category_id'] == id].index\n        return self.df.iloc[class_idxs]['embedding']\n    \n    @staticmethod\n    def _get_df(data_path: str, split: str) -> pd.DataFrame:\n        \"\"\"\n        Loads the dataset metadata as a pandas DataFrame.\n\n        Args:\n            data_path (str): The root directory where the dataset is stored.\n            split (str): The dataset split to load. Must be one of {'train', 'val', 'test'}.\n\n        Returns:\n            pd.DataFrame: A DataFrame containing metadata and file paths for the split.\n        \"\"\"\n        df_path = os.path.join(\n            data_path,\n            \"metadata\",\n            \"FungiTastic-FewShot\",\n            f\"FungiTastic-FewShot-{FungiTastic.SPLIT2STR[split]}.csv\"\n        )\n        df = pd.read_csv(df_path)\n        df[\"image_path\"] = df.filename.apply(\n            lambda x: os.path.join(data_path, \"FungiTastic-FewShot\", split, '300p', x)\n        )\n        return df\n\n    def __getitem__(self, idx: int):\n        \"\"\"\n        Retrieves a single data sample by index.\n    \n        Args:\n            idx (int): Index of the sample to retrieve.\n            ret_image (bool, optional): Whether to explicitly return the image. Defaults to False.\n    \n        Returns:\n            tuple:\n                - If embeddings exist: (image?, embedding, category_id, file_path)\n                - If no embeddings: (image, category_id, file_path) (original version)\n        \"\"\"\n        file_path = self.df[\"image_path\"].iloc[idx].replace('FungiTastic-FewShot', 'images/FungiTastic-FewShot')\n    \n        if self.split != 'test':\n            category_id = self.df[\"category_id\"].iloc[idx]\n        else:\n            category_id = None\n    \n        image = Image.open(file_path)\n    \n        if self.transform:\n            image = self.transform(image)\n    \n        # Check if embeddings exist\n        if \"embedding\" in self.df.columns:\n            emb = torch.tensor(self.df.iloc[idx]['embedding'], dtype=torch.float32).squeeze()\n        else:\n            emb = None  # No embeddings available\n    \n\n        return image, category_id, file_path, emb\n\n\n    def __len__(self):\n        \"\"\"\n        Returns the number of samples in the dataset.\n        \"\"\"\n        return len(self.df)\n\n    def get_class_id(self, idx: int) -> int:\n        \"\"\"\n        Returns the class ID of a specific sample.\n        \"\"\"\n        return self.df[\"category_id\"].iloc[idx]\n\n    def show_sample(self, idx: int) -> None:\n        \"\"\"\n        Displays a sample image along with its class name and index.\n        \"\"\"\n        image, category_id, _, _ = self.__getitem__(idx)\n        class_name = self.category_id2label[category_id]\n\n        plt.imshow(image)\n        plt.title(f\"Class: {class_name}; id: {idx}\")\n        plt.axis('off')\n        plt.show()\n\n    def get_category_idxs(self, category_id: int) -> List[int]:\n        \"\"\"\n        Retrieves all indexes for a given category ID.\n        \"\"\"\n        return self.df[self.df.category_id == category_id].index.tolist()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BioCLIP(torch.nn.Module):\n\n    def __init__(self, device):\n        \"\"\"\n        Initialize the BioCLIP feature extractor.\n        \"\"\"\n        super(BioCLIP, self).__init__()\n        self.device = device\n        self.model = None\n        self.processor = None\n        self.size = (224, 224)\n\n    def load(self):\n        \"\"\"\n        Load the BioCLIP model and its associated image processor.\n        \n        The model is loaded from the Hugging Face Hub and moved to the specified device.\n        \"\"\"\n        self.model, _, self.processor = open_clip.create_model_and_transforms('hf-hub:imageomics/bioclip')\n        self.model.to(self.device)\n\n    def extract_features(self, image):\n        \"\"\"\n        Extract normalized feature embeddings from a given image.\n\n        Args:\n            image (PIL.Image.Image): The input image from which to extract features.\n\n        Returns:\n            torch.Tensor: A normalized feature embedding vector for the input image.\n\n        Raises:\n            ValueError: If the model has not been loaded prior to calling this method.\n        \"\"\"\n        if self.model is None:\n            raise ValueError('Model not loaded')\n        image = image.resize(self.size, Image.BICUBIC)\n        image_tensor_proc = self.processor(image)[None]\n        features = self.model.encode_image(image_tensor_proc.to(self.device))\n        return self.normalize_embedding(features)\n\n    @staticmethod\n    def normalize_embedding(embs):\n        \"\"\"\n        Normalize the embedding vectors to have unit length.\n\n        Args:\n            embs (torch.Tensor): The raw embedding vectors.\n\n        Returns:\n            torch.Tensor: L2-normalized embedding vectors.\n        \"\"\"\n        return torch.nn.functional.normalize(embs.float(), dim=1, p=2)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = BioCLIP(device=device)\nmodel.load()\nmodel.eval()\n\ndef generate_embeddings(dataset):\n\n    idxs = np.arange(len(dataset))\n    im_names, embs = [], []\n    for idx in tqdm(idxs):\n        im, label, file_path, _ = dataset[idx]\n\n        with torch.no_grad():\n            feat = model.extract_features(im)\n            \n        # feat_quant = model.quantize_normalized_embedding(feat)\n\n        im_names.append(os.path.basename(file_path))\n        embs.append(feat.detach().cpu().numpy())\n\n    embeddings = pd.DataFrame({'filename': im_names, 'embedding': embs})\n\n    return embeddings","metadata":{"trusted":true,"execution":{"iopub.status.idle":"2025-10-08T00:40:53.879063Z","shell.execute_reply.started":"2025-10-08T00:40:48.331724Z","shell.execute_reply":"2025-10-08T00:40:53.878488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Load the datasets\n\ntrain_dataset = FungiTastic(root=data_path, split='train', transform=None)\ntest_dataset = FungiTastic(root=data_path, split='test', transform=None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:40:53.879807Z","iopub.execute_input":"2025-10-08T00:40:53.880012Z","iopub.status.idle":"2025-10-08T00:40:54.146286Z","shell.execute_reply.started":"2025-10-08T00:40:53.879995Z","shell.execute_reply":"2025-10-08T00:40:54.145705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embeddings = generate_embeddings(train_dataset)\ntrain_dataset.add_embeddings(embeddings)\nembeddings = generate_embeddings(test_dataset)\ntest_dataset.add_embeddings(embeddings)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:40:54.146917Z","iopub.execute_input":"2025-10-08T00:40:54.147113Z","iopub.status.idle":"2025-10-08T00:44:31.949546Z","shell.execute_reply.started":"2025-10-08T00:40:54.147097Z","shell.execute_reply":"2025-10-08T00:44:31.948716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df=train_dataset.df\ntest_df=test_dataset.df\ntest_df = test_df.drop_duplicates(subset=['observationID'], keep='first')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:44:31.951672Z","iopub.execute_input":"2025-10-08T00:44:31.952132Z","iopub.status.idle":"2025-10-08T00:44:31.964199Z","shell.execute_reply.started":"2025-10-08T00:44:31.952114Z","shell.execute_reply":"2025-10-08T00:44:31.963307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['embedding_1d'] = train_df['embedding'].apply(lambda x: list(np.array(x).flatten()))\ntest_df['embedding_1d'] = test_df['embedding'].apply(lambda x: list(np.array(x).flatten()))\ntrain_df.drop('embedding',axis=1,inplace=True)\ntest_df.drop('embedding',axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:44:31.965095Z","iopub.execute_input":"2025-10-08T00:44:31.965423Z","iopub.status.idle":"2025-10-08T00:44:32.628525Z","shell.execute_reply.started":"2025-10-08T00:44:31.965395Z","shell.execute_reply":"2025-10-08T00:44:32.627757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ntrain_df['category_iden']=le.fit_transform(train_df['category_id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:44:58.796092Z","iopub.execute_input":"2025-10-08T00:44:58.796408Z","iopub.status.idle":"2025-10-08T00:44:58.802032Z","shell.execute_reply.started":"2025-10-08T00:44:58.796387Z","shell.execute_reply":"2025-10-08T00:44:58.801230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import recall_score, make_scorer\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:47:45.495469Z","iopub.execute_input":"2025-10-08T00:47:45.496128Z","iopub.status.idle":"2025-10-08T00:47:46.087803Z","shell.execute_reply.started":"2025-10-08T00:47:45.496107Z","shell.execute_reply":"2025-10-08T00:47:46.087247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:47:51.304956Z","iopub.execute_input":"2025-10-08T00:47:51.305654Z","iopub.status.idle":"2025-10-08T00:47:51.309079Z","shell.execute_reply.started":"2025-10-08T00:47:51.305631Z","shell.execute_reply":"2025-10-08T00:47:51.308350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf_model = RandomForestClassifier(n_estimators=100, random_state=42,verbose=2)\ncv_scores = cross_val_score(\n    estimator=rf_model,\n    X=df_expanded,\n    y=target,\n    cv=skf,          # Use the StratifiedKFold splitter\n    scoring='f1',    # Choose an appropriate metric\n    n_jobs=-1        # Use all processors\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:20:06.394796Z","iopub.execute_input":"2025-10-08T01:20:06.395464Z","iopub.status.idle":"2025-10-08T01:24:59.866914Z","shell.execute_reply.started":"2025-10-08T01:20:06.395436Z","shell.execute_reply":"2025-10-08T01:24:59.865923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Individual Fold F1-Scores: {cv_scores}\")\nprint(f\"Mean F1-Score: {np.mean(cv_scores):.4f}\")\nprint(f\"Standard Deviation of F1-Scores: {np.std(cv_scores):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:20:01.847866Z","iopub.status.idle":"2025-10-08T01:20:01.848204Z","shell.execute_reply.started":"2025-10-08T01:20:01.848025Z","shell.execute_reply":"2025-10-08T01:20:01.848039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:25:24.917678Z","iopub.execute_input":"2025-10-08T01:25:24.918384Z","iopub.status.idle":"2025-10-08T01:25:24.923767Z","shell.execute_reply.started":"2025-10-08T01:25:24.918361Z","shell.execute_reply":"2025-10-08T01:25:24.923149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols=['eventDate', 'year', 'month', 'day', 'habitat', 'countryCode',\n       'hasCoordinate', 'substrate', 'latitude', 'longitude', 'coorUncert',\n       'observationID', 'region', 'district', 'filename', 'metaSubstrate',\n       'elevation', 'landcover', 'biogeographicalRegion', 'image_path',\n       'embedding_1d','category_id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T00:53:47.119656Z","iopub.execute_input":"2025-10-08T00:53:47.120206Z","iopub.status.idle":"2025-10-08T00:53:47.123665Z","shell.execute_reply.started":"2025-10-08T00:53:47.120185Z","shell.execute_reply":"2025-10-08T00:53:47.122807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['embedding_1d'].tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:08:24.665188Z","iopub.execute_input":"2025-10-08T01:08:24.665761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_df['embedding_1d'].apply(pd.Series)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:25:37.841772Z","iopub.execute_input":"2025-10-08T01:25:37.842352Z","iopub.status.idle":"2025-10-08T01:25:39.565494Z","shell.execute_reply.started":"2025-10-08T01:25:37.842325Z","shell.execute_reply":"2025-10-08T01:25:39.564879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y=train_df['category_id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:26:50.912135Z","iopub.execute_input":"2025-10-08T01:26:50.912436Z","iopub.status.idle":"2025-10-08T01:26:50.916200Z","shell.execute_reply.started":"2025-10-08T01:26:50.912417Z","shell.execute_reply":"2025-10-08T01:26:50.915342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score\nimport numpy as np\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nfold = 1\nf1_scores = []\n\nfor train_idx, val_idx in skf.split(X, y):\n    print(f\"\\n=== Fold {fold} ===\")\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n    rf_model = RandomForestClassifier(\n        n_estimators=10,\n        random_state=42,\n        verbose=2,  # logs for training each tree\n        n_jobs=-1\n    )\n\n    rf_model.fit(X_train, y_train)\n    y_pred = rf_model.predict(X_val)\n    f1 = f1_score(y_val, y_pred,average='macro')\n    print(f\"Fold {fold} F1-Score: {f1:.4f}\")\n    f1_scores.append(f1)\n    fold += 1\n\nprint(f\"\\nMean F1-Score: {np.mean(f1_scores):.4f}\")\nprint(f\"Std F1-Score: {np.std(f1_scores):.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T01:44:27.357625Z","iopub.execute_input":"2025-10-08T01:44:27.357891Z","iopub.status.idle":"2025-10-08T02:05:02.881015Z","shell.execute_reply.started":"2025-10-08T01:44:27.357872Z","shell.execute_reply":"2025-10-08T02:05:02.880126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred.shape\ny_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-08T02:06:33.070923Z","iopub.execute_input":"2025-10-08T02:06:33.071207Z","iopub.status.idle":"2025-10-08T02:06:33.076147Z","shell.execute_reply.started":"2025-10-08T02:06:33.071187Z","shell.execute_reply":"2025-10-08T02:06:33.075438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}