{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":89850,"databundleVersionId":11256103,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from typing import List, Set\nimport pandas as pd\nimport csv\nimport random\n\ndef f1_score_per_image(true_labels: Set[int], pred_labels: Set[int]) -> float:\n    \"\"\"\n    Compute the F1 score for a single image.\n    true_labels: set of ground truth species (e.g., {1, 4, 9})\n    pred_labels: set of predicted species (e.g., {1, 4, 5})\n    \"\"\"\n    tp = len(true_labels & pred_labels)  # True Positives\n    fp = len(pred_labels - true_labels)  # False Positives\n    fn = len(true_labels - pred_labels)  # False Negatives\n\n    if tp + fp == 0:\n        precision = 0.0\n    else:\n        precision = tp / (tp + fp)\n\n    if tp + fn == 0:\n        recall = 0.0\n    else:\n        recall = tp / (tp + fn)\n\n    if precision + recall == 0:\n        return 0.0\n\n    return 2 * precision * recall / (precision + recall)\n\n\ndef macro_f1_per_transect(\n    transect_true: List[Set[int]],\n    transect_pred: List[Set[int]]\n) -> float:\n    \"\"\"\n    Compute the average F1 score for a single transect.\n    transect_true: list of sets with ground truth labels for each image in the transect\n    transect_pred: list of sets with predicted labels for each image in the transect\n    \"\"\"\n    assert len(transect_true) == len(transect_pred)\n    scores = [\n        f1_score_per_image(t, p)\n        for t, p in zip(transect_true, transect_pred)\n    ]\n    return sum(scores) / len(scores)\n\n\ndef final_macro_f1_score(\n    all_true: List[List[Set[int]]],\n    all_pred: List[List[Set[int]]]\n) -> float:\n    \"\"\"\n    Compute the final score: the average F1 score across all transects.\n    all_true: list of transects, each being a list of sets of ground truth labels\n    all_pred: list of transects, each being a list of sets of predicted labels\n    \"\"\"\n    assert len(all_true) == len(all_pred)\n    transect_scores = [\n        macro_f1_per_transect(true, pred)\n        for true, pred in zip(all_true, all_pred)\n    ]\n    return sum(transect_scores) / len(transect_scores)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:22.227325Z","iopub.execute_input":"2025-04-30T10:47:22.227958Z","iopub.status.idle":"2025-04-30T10:47:24.146527Z","shell.execute_reply.started":"2025-04-30T10:47:22.227919Z","shell.execute_reply":"2025-04-30T10:47:24.145458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# example\ntrue = [\n    [ {1, 2}, {3} ],            # transekt 1\n    [ {4, 5}, {6, 7, 8} ]       # transekt 2\n]\n\npred = [\n    [ {1, 2}, {2, 3} ],         # transekt 1\n    [ {4}, {6, 9} ]             # transekt 2\n]\n\nscore = final_macro_f1_score(true, pred)\nprint(f\"Final macro-averaged F1 score: {score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:24.148844Z","iopub.execute_input":"2025-04-30T10:47:24.149421Z","iopub.status.idle":"2025-04-30T10:47:24.156209Z","shell.execute_reply.started":"2025-04-30T10:47:24.149380Z","shell.execute_reply":"2025-04-30T10:47:24.155013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/plantclef-2025/PlantCLEF2025_test.csv', sep = ';')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:24.157385Z","iopub.execute_input":"2025-04-30T10:47:24.157777Z","iopub.status.idle":"2025-04-30T10:47:24.222566Z","shell.execute_reply.started":"2025-04-30T10:47:24.157747Z","shell.execute_reply":"2025-04-30T10:47:24.221563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:24.223656Z","iopub.execute_input":"2025-04-30T10:47:24.224022Z","iopub.status.idle":"2025-04-30T10:47:24.257967Z","shell.execute_reply.started":"2025-04-30T10:47:24.223996Z","shell.execute_reply":"2025-04-30T10:47:24.257000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species = pd.read_csv('/kaggle/input/plantclef-2025/species_ids.csv')\nspecies","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:24.260338Z","iopub.execute_input":"2025-04-30T10:47:24.260699Z","iopub.status.idle":"2025-04-30T10:47:24.282135Z","shell.execute_reply.started":"2025-04-30T10:47:24.260671Z","shell.execute_reply":"2025-04-30T10:47:24.281054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_species = list(species['species_id'].unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:47:24.283311Z","iopub.execute_input":"2025-04-30T10:47:24.283688Z","iopub.status.idle":"2025-04-30T10:47:24.294260Z","shell.execute_reply.started":"2025-04-30T10:47:24.283657Z","shell.execute_reply":"2025-04-30T10:47:24.293120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def random_species_subset(species_pool, min_n=1, max_n=5):\n    n = random.randint(min_n, max_n)\n    return random.sample(species_pool, k=n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:48:12.945462Z","iopub.execute_input":"2025-04-30T10:48:12.946674Z","iopub.status.idle":"2025-04-30T10:48:12.951754Z","shell.execute_reply.started":"2025-04-30T10:48:12.946631Z","shell.execute_reply":"2025-04-30T10:48:12.950414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_run = pd.DataFrame()\n\ndf_run['quadrat_id']=test[\"quadrat_id\"]\n\ndf_run['species_ids'] = df_run['quadrat_id'].apply(lambda _: random_species_subset(all_species))\ndf_run","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:48:13.685246Z","iopub.execute_input":"2025-04-30T10:48:13.685997Z","iopub.status.idle":"2025-04-30T10:48:13.710923Z","shell.execute_reply.started":"2025-04-30T10:48:13.685963Z","shell.execute_reply":"2025-04-30T10:48:13.710014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_run.to_csv(\"submission.csv\", sep=',', index=False, quoting=csv.QUOTE_ALL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:48:16.922750Z","iopub.execute_input":"2025-04-30T10:48:16.923528Z","iopub.status.idle":"2025-04-30T10:48:16.937571Z","shell.execute_reply.started":"2025-04-30T10:48:16.923495Z","shell.execute_reply":"2025-04-30T10:48:16.936583Z"}},"outputs":[],"execution_count":null}]}