{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     print(dirname)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:43.004529Z","iopub.execute_input":"2025-06-22T10:44:43.005646Z","iopub.status.idle":"2025-06-22T10:44:44.442080Z","shell.execute_reply.started":"2025-06-22T10:44:43.005597Z","shell.execute_reply":"2025-06-22T10:44:44.441160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy_df = pd.read_csv(\"/kaggle/input/birdclef-2025/taxonomy.csv\")\ntaxonomy_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:44.443568Z","iopub.execute_input":"2025-06-22T10:44:44.444013Z","iopub.status.idle":"2025-06-22T10:44:44.485587Z","shell.execute_reply.started":"2025-06-22T10:44:44.443981Z","shell.execute_reply":"2025-06-22T10:44:44.484524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:51:54.526698Z","iopub.execute_input":"2025-06-22T10:51:54.527039Z","iopub.status.idle":"2025-06-22T10:51:54.671758Z","shell.execute_reply.started":"2025-06-22T10:51:54.527013Z","shell.execute_reply":"2025-06-22T10:51:54.670653Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA for train.csv","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    print(f\"{col}: {df[col].nunique()} unique values\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:45.791934Z","iopub.execute_input":"2025-06-22T10:44:45.792392Z","iopub.status.idle":"2025-06-22T10:44:45.848172Z","shell.execute_reply.started":"2025-06-22T10:44:45.792353Z","shell.execute_reply":"2025-06-22T10:44:45.847264Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## latitude and longitude","metadata":{}},{"cell_type":"code","source":"print(\"Missing latitude:\", df[\"latitude\"].isna().sum())\nprint(\"Missing longitude:\", df[\"longitude\"].isna().sum())\nmissing_geo = df[\"latitude\"].isna().mean() * 100\nprint(f\"{missing_geo:.2f}% of recordings are missing location data.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:47.747188Z","iopub.execute_input":"2025-06-22T10:44:47.747473Z","iopub.status.idle":"2025-06-22T10:44:47.766616Z","shell.execute_reply.started":"2025-06-22T10:44:47.747453Z","shell.execute_reply":"2025-06-22T10:44:47.765629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"has_location\"] = df[\"latitude\"].notna()\nplt.figure(figsize=(10, 6))\nsns.scatterplot(data=df[df[\"has_location\"]], x=\"longitude\", y=\"latitude\", s=10, alpha=0.5)\nplt.title(\"Geographic Distribution of Recordings\")\nplt.xlabel(\"Longitude\")\nplt.ylabel(\"Latitude\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:47.767542Z","iopub.execute_input":"2025-06-22T10:44:47.767833Z","iopub.status.idle":"2025-06-22T10:44:48.034844Z","shell.execute_reply.started":"2025-06-22T10:44:47.767808Z","shell.execute_reply":"2025-06-22T10:44:48.033883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[[\"latitude\", \"longitude\"]].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:44:48.036604Z","iopub.execute_input":"2025-06-22T10:44:48.036918Z","iopub.status.idle":"2025-06-22T10:44:48.059402Z","shell.execute_reply.started":"2025-06-22T10:44:48.036893Z","shell.execute_reply":"2025-06-22T10:44:48.058533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rounded_locs = df[df[\"has_location\"]].copy()\nrounded_locs[\"lat_rounded\"] = rounded_locs[\"latitude\"].round(1)\nrounded_locs[\"lon_rounded\"] = rounded_locs[\"longitude\"].round(1)\n\nlocation_counts = rounded_locs.groupby([\"lat_rounded\", \"lon_rounded\"]).size().reset_index(name=\"count\")\n\nsns.histplot(location_counts[\"count\"], bins=30, log_scale=(False, True))\nplt.title(\"Distribution of Recordings per Geographic Region\")\nplt.xlabel(\"Number of Recordings (per ~10km grid)\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:45:09.143500Z","iopub.execute_input":"2025-06-22T10:45:09.144360Z","iopub.status.idle":"2025-06-22T10:45:09.677383Z","shell.execute_reply.started":"2025-06-22T10:45:09.144322Z","shell.execute_reply":"2025-06-22T10:45:09.676407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n- About 2.83% of recordings are missing latitude/longitude.\n- Most recordings are concentrated in the tropical Americas, consistent with the BirdCLEF focus.","metadata":{}},{"cell_type":"markdown","source":"## primary_label","metadata":{}},{"cell_type":"code","source":"print(\"Number of unique primary labels:\", df['primary_label'].nunique())\n\n# Top and bottom species by frequency\nprimary_counts = df['primary_label'].value_counts()\nprint(\"\\nTop 5 most frequent species:\")\nprint(primary_counts.head())\n\nprint(\"\\nBottom 5 least frequent species:\")\nprint(primary_counts.tail())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:51:49.454746Z","iopub.execute_input":"2025-06-22T10:51:49.455399Z","iopub.status.idle":"2025-06-22T10:51:49.469745Z","shell.execute_reply.started":"2025-06-22T10:51:49.455356Z","shell.execute_reply":"2025-06-22T10:51:49.468566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Species with fewer than 10 samples:\", (primary_counts < 10).sum())\nprint(\"Species with fewer than 50 samples:\", (primary_counts < 50).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:53:09.272221Z","iopub.execute_input":"2025-06-22T10:53:09.273148Z","iopub.status.idle":"2025-06-22T10:53:09.278798Z","shell.execute_reply.started":"2025-06-22T10:53:09.273121Z","shell.execute_reply":"2025-06-22T10:53:09.277808Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n- There are 206 unique species.\n\n- It’s heavily imbalanced — e.g., 78 species have fewer than 50 samples.\n\n- Top species have 600–990 clips, while rare ones have just 2.\n\n- -> This suggests that class balancing techniques will be important for model robustness.","metadata":{}},{"cell_type":"code","source":"import ast\n\ndf['secondary_labels'] = df['secondary_labels'].fillna(\"[]\")\ndf['secondary_labels'] = df['secondary_labels'].apply(ast.literal_eval)\n\nprint(\"Parsed secondary_labels (first 5 rows):\")\nprint(df['secondary_labels'].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:55:17.787042Z","iopub.execute_input":"2025-06-22T10:55:17.787388Z","iopub.status.idle":"2025-06-22T10:55:17.992478Z","shell.execute_reply.started":"2025-06-22T10:55:17.787363Z","shell.execute_reply":"2025-06-22T10:55:17.991282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['n_secondary'] = df['secondary_labels'].apply(len)\n\nprint(\"\\nDistribution of number of secondary labels per clip:\")\nprint(df['n_secondary'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:55:28.443896Z","iopub.execute_input":"2025-06-22T10:55:28.444325Z","iopub.status.idle":"2025-06-22T10:55:28.461443Z","shell.execute_reply.started":"2025-06-22T10:55:28.444299Z","shell.execute_reply":"2025-06-22T10:55:28.460431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_secondaries = set([s for sublist in df['secondary_labels'] for s in sublist])\nall_primaries = set(df['primary_label'].unique())\n\nonly_in_secondary = all_secondaries - all_primaries\n\nprint(\"\\nSpecies that appear only as secondary labels:\", len(only_in_secondary))\nprint(\"Examples:\", list(only_in_secondary)[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T10:55:37.543814Z","iopub.execute_input":"2025-06-22T10:55:37.544115Z","iopub.status.idle":"2025-06-22T10:55:37.557362Z","shell.execute_reply.started":"2025-06-22T10:55:37.544093Z","shell.execute_reply":"2025-06-22T10:55:37.556616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights:\n- Most clips (27,744) have exactly 1 secondary label, a few clips have 2+ secondaries, but it’s rare.\n\n- We checked whether secondary_labels adds any new species: Nope, the only thing in secondary_labels not in primary_label was '', which is just a placeholder.\n\n- -> The task is effectively single-label classification using primary_label only. secondary_labels adds very little modeling value and can be ignored.","metadata":{}},{"cell_type":"markdown","source":"## Type","metadata":{}},{"cell_type":"code","source":"print(\"Unique values in 'type':\", df['type'].unique()[:20])\nprint(\"\\nClip count per type:\")\nprint(df['type'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:05:41.559903Z","iopub.execute_input":"2025-06-22T11:05:41.560341Z","iopub.status.idle":"2025-06-22T11:05:41.570841Z","shell.execute_reply.started":"2025-06-22T11:05:41.560315Z","shell.execute_reply":"2025-06-22T11:05:41.569665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert the stringified list into a real Python list.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:06:49.675968Z","iopub.execute_input":"2025-06-22T11:06:49.676412Z","iopub.status.idle":"2025-06-22T11:06:49.680502Z","shell.execute_reply.started":"2025-06-22T11:06:49.676365Z","shell.execute_reply":"2025-06-22T11:06:49.679499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ast\nfrom collections import Counter\n\ndf['type_parsed'] = df['type'].apply(ast.literal_eval)\n\ntype_counter = Counter([label for sublist in df['type_parsed'] for label in sublist if label.strip()])\n\nprint(\"Top 15 most common individual 'type' labels:\")\nfor label, count in type_counter.most_common(15):\n    print(f\"{label}: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:06:58.894838Z","iopub.execute_input":"2025-06-22T11:06:58.895159Z","iopub.status.idle":"2025-06-22T11:06:59.230901Z","shell.execute_reply.started":"2025-06-22T11:06:58.895136Z","shell.execute_reply":"2025-06-22T11:06:59.230144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels, counts = zip(*type_counter.most_common(15))\nplt.figure(figsize=(10, 5))\nplt.barh(labels[::-1], counts[::-1])\nplt.xlabel(\"Count\")\nplt.title(\"Top 15 Individual 'Type' Labels\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:07:38.689550Z","iopub.execute_input":"2025-06-22T11:07:38.690499Z","iopub.status.idle":"2025-06-22T11:07:38.930112Z","shell.execute_reply.started":"2025-06-22T11:07:38.690466Z","shell.execute_reply":"2025-06-22T11:07:38.929257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df['rating'].describe())\nprint(\"\\nUnique rating values and their counts:\")\nprint(df['rating'].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:11:44.077832Z","iopub.execute_input":"2025-06-22T11:11:44.078131Z","iopub.status.idle":"2025-06-22T11:11:44.091316Z","shell.execute_reply.started":"2025-06-22T11:11:44.078111Z","shell.execute_reply":"2025-06-22T11:11:44.090432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rating_counts = df['rating'].value_counts().sort_index()\n\nrating_df = pd.DataFrame(list(rating_counts.items()), columns=[\"Rating\", \"Count\"])\n\nplt.figure(figsize=(10, 6))\nplt.bar(rating_df[\"Rating\"], rating_df[\"Count\"])\nplt.xlabel(\"Rating\")\nplt.ylabel(\"Number of Clips\")\nplt.title(\"Distribution of Clip Ratings\")\nplt.xticks(rating_df[\"Rating\"])\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-22T11:14:42.947617Z","iopub.execute_input":"2025-06-22T11:14:42.947928Z","iopub.status.idle":"2025-06-22T11:14:43.182377Z","shell.execute_reply.started":"2025-06-22T11:14:42.947904Z","shell.execute_reply":"2025-06-22T11:14:43.181238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}