{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n# <div align=\"center\"> BirdCLEF+ 2025 \n    \n\n<div align=\"center\"> <img src=\"https://www.kaggle.com/competitions/91844/images/header\" width=\"500\"></div><br>\n\n\n### **From the competition material, we learn the following**:<br>\n**Goal**: Apply your machine-learning expertise to identify under-studied species based on their acoustic signatures in the 🌱🇨🇴[Silencio Natural Reserve](https://www.fundacionbiodiversa.org/fundacion2024/information/)🌱🇨🇴.  Specifically, you'll develop computational methods to process continuous audio data and recognize species from different taxonomic groups by their sounds, thereby helping the local ecological restoration projects.\n\n**What to predict**: For each row_id, you should predict the probability that a given species was present. There is one column per species. Each row covers a five-second window of audio.\n\n**Evaluation**: The evaluation metric for this contest is a version of <ins>macro-averaged ROC-AUC</ins> that skips classes which have no true positive labels.\n","metadata":{}},{"cell_type":"markdown","source":"# 📚 Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport soundfile as sf\nimport librosa\nfrom pathlib import Path\nfrom matplotlib import pyplot as plt\nimport IPython.display as ipd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.709404Z","iopub.execute_input":"2025-03-16T12:45:43.709724Z","iopub.status.idle":"2025-03-16T12:45:43.716112Z","shell.execute_reply.started":"2025-03-16T12:45:43.709698Z","shell.execute_reply":"2025-03-16T12:45:43.714971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"code","source":"# Reading in the competition material\ntrain =  pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\ntaxonomy =  pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.717732Z","iopub.execute_input":"2025-03-16T12:45:43.718187Z","iopub.status.idle":"2025-03-16T12:45:43.864385Z","shell.execute_reply.started":"2025-03-16T12:45:43.718150Z","shell.execute_reply":"2025-03-16T12:45:43.863301Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### train.csv","metadata":{}},{"cell_type":"code","source":"print(f\"This dataset has {train.shape[0]} rows and {train.shape[1]} columns.\")\nprint(f\"There are {train.isna().sum().sum()} NA's in the dataset.\")\n\n# quick look at the data\ntrain.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.865341Z","iopub.execute_input":"2025-03-16T12:45:43.865650Z","iopub.status.idle":"2025-03-16T12:45:43.903444Z","shell.execute_reply.started":"2025-03-16T12:45:43.865622Z","shell.execute_reply":"2025-03-16T12:45:43.902233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# a closer look reveals that our NAs are confined only to the latitude & longitude columns\ntrain[['latitude','longitude']].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.904679Z","iopub.execute_input":"2025-03-16T12:45:43.905145Z","iopub.status.idle":"2025-03-16T12:45:43.916557Z","shell.execute_reply.started":"2025-03-16T12:45:43.905101Z","shell.execute_reply":"2025-03-16T12:45:43.915381Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### We will look at the submission format later, but we can examine the primary_label (and scientific_name) columns to understand the breadth of the predictions.  These columns represent the individual species we need to make predictions for.","metadata":{}},{"cell_type":"code","source":"print(f\"There are {train['primary_label'].nunique()} primary labels corresponding to {train['scientific_name'].nunique()} scientific names.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.918185Z","iopub.execute_input":"2025-03-16T12:45:43.918734Z","iopub.status.idle":"2025-03-16T12:45:43.945973Z","shell.execute_reply.started":"2025-03-16T12:45:43.918685Z","shell.execute_reply":"2025-03-16T12:45:43.944792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Let's look at the filename column and compare it to the number of .ogg files in the ","metadata":{}},{"cell_type":"code","source":"print(f\"There are {train['filename'].nunique()} unique filenames in train.\")\n# This 28564 matches the number of rows in train.csv\n\n# iterate through /train_audio file structure and count .ogg files\ntrain_ogg_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input/birdclef-2025/train_audio'):\n    for filename in filenames:\n        train_ogg_files.append(os.path.join(dirname, filename))\nprint(f\"There are {len(train_ogg_files)} .ogg files in train_audio\")\n\n# So there is one file in /train_audio that corresponds to the filename in train.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:43.947117Z","iopub.execute_input":"2025-03-16T12:45:43.947532Z","iopub.status.idle":"2025-03-16T12:45:57.445577Z","shell.execute_reply.started":"2025-03-16T12:45:43.947467Z","shell.execute_reply":"2025-03-16T12:45:57.444249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Lastly, let's examine how many sound samples per species we have","metadata":{}},{"cell_type":"code","source":"train.groupby('primary_label')['filename'].count()\n\n# We can see we have a vastly different number of sound samples per species, with the max being 990!","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:57.446593Z","iopub.execute_input":"2025-03-16T12:45:57.446945Z","iopub.status.idle":"2025-03-16T12:45:57.459165Z","shell.execute_reply.started":"2025-03-16T12:45:57.446910Z","shell.execute_reply":"2025-03-16T12:45:57.458067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### taxonomy.csv","metadata":{}},{"cell_type":"code","source":"print(f\"This dataset has {taxonomy.shape[0]} rows and {taxonomy.shape[1]} columns.\")\nprint(f\"There are {taxonomy.isna().sum().sum()} NA's in the dataset.\")\n\n# quick look at the data\ntaxonomy.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:57.460533Z","iopub.execute_input":"2025-03-16T12:45:57.460878Z","iopub.status.idle":"2025-03-16T12:45:57.481551Z","shell.execute_reply.started":"2025-03-16T12:45:57.460840Z","shell.execute_reply":"2025-03-16T12:45:57.480366Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### We see some columns we are familiar with, such as primary_label and scientific_name.  Let's look at class_name","metadata":{}},{"cell_type":"code","source":"taxonomy.groupby('class_name')['primary_label'].count().plot.pie(x='class_name',figsize=(5,5),title='Taxonomy class_name breakout',ylabel='')\n\n# We can see from this pie chart that the majority of the sound samples involve Aves","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:57.482504Z","iopub.execute_input":"2025-03-16T12:45:57.482890Z","iopub.status.idle":"2025-03-16T12:45:57.627190Z","shell.execute_reply.started":"2025-03-16T12:45:57.482856Z","shell.execute_reply":"2025-03-16T12:45:57.626099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### train_audio","metadata":{}},{"cell_type":"markdown","source":"#### The train_audio directory contains the .ogg files for our training data.  Earlier we compiled all the .ogg files in that directory:","metadata":{}},{"cell_type":"code","source":"print(f\"There are {len(train_ogg_files)} .ogg files in train_audio\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:57.628223Z","iopub.execute_input":"2025-03-16T12:45:57.628568Z","iopub.status.idle":"2025-03-16T12:45:57.634554Z","shell.execute_reply.started":"2025-03-16T12:45:57.628539Z","shell.execute_reply":"2025-03-16T12:45:57.633017Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Let's pull a single .ogg file out, visualize & listen to it.","metadata":{}},{"cell_type":"code","source":"def print_plot_play(x, Fs, text=''):\n    \"\"\"\n    1. Prints information about an audio singal, \n    2. plots the waveform\n    3. Creates player\n    \"\"\"\n    print('%s Fs = %d, x.shape = %s, x.dtype = %s' % (text, Fs, x.shape, x.dtype))\n    plt.figure(figsize=(8, 2))\n    plt.plot(x, color='gray')\n    plt.xlim([0, x.shape[0]])\n    plt.xlabel('Time (samples)')\n    plt.ylabel('Amplitude')\n    plt.tight_layout()\n    plt.show()\n    ipd.display(ipd.Audio(data=x, rate=Fs))\n\ndef print_spectral_centroids(x, Fs, text=''):\n    \"\"\"\n    1. Prints spectral centroid of audio signal\n    \"\"\"\n    print('%s Fs = %d, x.shape = %s, x.dtype = %s' % (text, Fs, x.shape, x.dtype))\n    plt.plot(librosa.feature.spectral_centroid(y=x, sr=Fs)[0])\n    plt.xlabel('Frame number')\n    plt.ylabel('frequency (Hz)')\n    plt.title('Spectral centroids')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:59:34.636763Z","iopub.execute_input":"2025-03-16T12:59:34.637203Z","iopub.status.idle":"2025-03-16T12:59:34.645998Z","shell.execute_reply.started":"2025-03-16T12:59:34.637161Z","shell.execute_reply":"2025-03-16T12:59:34.644635Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### This sample sounds peaceful!","metadata":{}},{"cell_type":"code","source":"# load the audio\nx, Fs = librosa.load('/kaggle/input/birdclef-2025/train_audio/126247/iNat1109254.ogg', sr=None)\n\n# print spectral centroid\nprint_spectral_centroids(x,Fs,'OGG file:')\n\n# print waveform & play audio\nprint_plot_play(x=x, Fs=Fs, text='OGG file: ')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:59:36.446967Z","iopub.execute_input":"2025-03-16T12:59:36.447423Z","iopub.status.idle":"2025-03-16T12:59:37.152282Z","shell.execute_reply.started":"2025-03-16T12:59:36.447381Z","shell.execute_reply":"2025-03-16T12:59:37.150428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### This sample is something different!","metadata":{}},{"cell_type":"code","source":"# load the audio\nx, Fs = librosa.load('/kaggle/input/birdclef-2025/train_audio/1139490/CSA36385.ogg', sr=None)\n\n# print spectral centroid\nprint_spectral_centroids(x,Fs,'OGG file:')\n\n# print waveform & play audio\nprint_plot_play(x=x, Fs=Fs, text='OGG file: ')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T13:00:07.861527Z","iopub.execute_input":"2025-03-16T13:00:07.861948Z","iopub.status.idle":"2025-03-16T13:00:09.410578Z","shell.execute_reply.started":"2025-03-16T13:00:07.861916Z","shell.execute_reply":"2025-03-16T13:00:09.407469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### As has been mentioned in a discussion post [here](https://www.kaggle.com/competitions/birdclef-2025/discussion/567551), some of the audio samples contain recordings of the individuals documenting their sampling efforts.  A notebook [here](https://www.kaggle.com/code/kdmitrie/bc25-separation-voice-from-data) has discussed how to remove those.","metadata":{}},{"cell_type":"markdown","source":"### train_soundscapes","metadata":{}},{"cell_type":"markdown","source":"#### The structure of this directory mimics what will be in test_soundscapes.  From the <ins>competition material</ins>, when we submit a notebook, test_soundscapes: <br><br><center>```will be populated with approximately 700 recordings to be used for scoring.```</center>","metadata":{}},{"cell_type":"markdown","source":"#### We will read train_soundscapes just like we will the test_soundscapes, as recommended in [this starter notebook](https://www.kaggle.com/code/stefankahl/birdclef-2025-sample-submission).","metadata":{}},{"cell_type":"code","source":"soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\nsoundscapes = [os.path.join(soundscape_path, afile) for afile in sorted(os.listdir(soundscape_path)) if afile.endswith('.ogg')]\nif len(soundscapes) == 0:\n    # not submission\n    soundscape_path = '/kaggle/input/birdclef-2025/train_soundscapes/'\n    soundscapes = [os.path.join(soundscape_path, afile) for afile in sorted(os.listdir(soundscape_path)) if afile.endswith('.ogg')]\n    soundscapes = soundscapes[0:700] # simulate processing time of test if this is not a submission\n    \nprint(f\"There are {len(soundscapes)} ogg files in soundscapes.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:59.113916Z","iopub.execute_input":"2025-03-16T12:45:59.114245Z","iopub.status.idle":"2025-03-16T12:45:59.145668Z","shell.execute_reply.started":"2025-03-16T12:45:59.114214Z","shell.execute_reply":"2025-03-16T12:45:59.144622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Class labels from train audio\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\ncols = [\"row_id\"] + class_labels\n\nsubmission = pd.DataFrame(columns=['row_id'] + class_labels)\n\nfor soundscape in soundscapes:\n\n    # Load audio\n    sig, rate = librosa.load(path=soundscape, sr=None)\n    \n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n        \n    # Make predictions for each chunk\n    for i, chunk in enumerate(chunks):\n        \n        # Get row id  (soundscape id + end time of 5s chunk)      \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n        \n        # Different from the starter notebook referenced above, we will fill predictions with '.5' for now\n        ### Placeholder for inference ###\n        scores = [.5]*(len(class_labels))\n        \n        # Append to predictions as new row\n        new_row = pd.DataFrame([[row_id] + (scores)], columns=['row_id'] + class_labels)\n        submission = pd.concat([submission, new_row], axis=0, ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:45:59.146902Z","iopub.execute_input":"2025-03-16T12:45:59.147226Z","iopub.status.idle":"2025-03-16T12:47:59.356548Z","shell.execute_reply.started":"2025-03-16T12:45:59.147196Z","shell.execute_reply":"2025-03-16T12:47:59.355207Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🤞 Submission\n","metadata":{}},{"cell_type":"markdown","source":"#### Let's look at the submission format","metadata":{}},{"cell_type":"code","source":"sample_submission =  pd.read_csv('/kaggle/input/birdclef-2025/sample_submission.csv')\nprint(f\"This dataset has {sample_submission.shape[0]} rows and {sample_submission.shape[1]} columns.\")\n\n# quick look at the data\nsample_submission.head(3)\n\n# Most importantly is we need all 207 columns, and need to build our rows in 5 second increments for each file in the *_soundscapes directory","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:47:59.357685Z","iopub.execute_input":"2025-03-16T12:47:59.358061Z","iopub.status.idle":"2025-03-16T12:47:59.390858Z","shell.execute_reply.started":"2025-03-16T12:47:59.358019Z","shell.execute_reply":"2025-03-16T12:47:59.389727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Sample submission shape is {sample_submission.shape}\")\nprint(f\"Submission shape is {submission.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:47:59.391784Z","iopub.execute_input":"2025-03-16T12:47:59.392072Z","iopub.status.idle":"2025-03-16T12:47:59.398413Z","shell.execute_reply.started":"2025-03-16T12:47:59.392045Z","shell.execute_reply":"2025-03-16T12:47:59.396575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:47:59.399403Z","iopub.execute_input":"2025-03-16T12:47:59.399767Z","iopub.status.idle":"2025-03-16T12:47:59.437858Z","shell.execute_reply.started":"2025-03-16T12:47:59.399732Z","shell.execute_reply":"2025-03-16T12:47:59.435707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-16T12:47:59.439345Z","iopub.execute_input":"2025-03-16T12:47:59.440022Z","iopub.status.idle":"2025-03-16T12:47:59.879759Z","shell.execute_reply.started":"2025-03-16T12:47:59.439961Z","shell.execute_reply":"2025-03-16T12:47:59.878243Z"}},"outputs":[],"execution_count":null}]}