{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **Notebook Overview**\n\nIn this notebook we attempt to clean the BirdCLEF training audio files.\n\n**Major issue:** The audio files contains species acoustic sounds along with human annotations.\n\n**Proposed solution:** We attempt to identify the speech segments using state of the art Voice Activity Detection(VAD) models like pyannote/segmentation model. Using that we extract the Non speech segments(NSS).\n\nNSS format : Dictionary with keys \"start\" and \"end\", where \"start\" and \"end\" contains the corresponding timestamps(in seconds) where there's no speech. \n\nThe resulting file is of .npy format. Where each row represents each audio file. The columns are \"primary_label\", \"file_name\" and \"NSS_segments\" respectively.\n\nNotebook creted by : Divyaprakash Rathinasabapathy. [LinkedIn](http://https://www.linkedin.com/in/divyaprakash-rathinasabapathy/)","metadata":{}},{"cell_type":"code","source":"# Library installations\n\n! pip install pyannote.audio","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Library imports\nimport pandas as pd\nimport numpy as np\nimport torch, torchaudio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:35:58.121664Z","iopub.execute_input":"2025-04-23T05:35:58.121945Z","iopub.status.idle":"2025-04-23T05:35:58.125787Z","shell.execute_reply.started":"2025-04-23T05:35:58.121924Z","shell.execute_reply":"2025-04-23T05:35:58.125061Z"}},"outputs":[],"execution_count":48},{"cell_type":"code","source":"# Reading the csv files\nmetadata_df = pd.read_csv(r\"/kaggle/input/birdclef-2025/train.csv\")\ntaxanomy = pd.read_csv(r\"/kaggle/input/birdclef-2025/taxonomy.csv\")\n\n# Merging them using \"primary_label\"\ndf_to_merge = taxanomy[[\"primary_label\", \"class_name\"]]\nmetadata_df = pd.merge(metadata_df, df_to_merge, how = \"inner\", on = \"primary_label\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:35:59.329814Z","iopub.execute_input":"2025-04-23T05:35:59.330387Z","iopub.status.idle":"2025-04-23T05:35:59.436359Z","shell.execute_reply.started":"2025-04-23T05:35:59.330362Z","shell.execute_reply":"2025-04-23T05:35:59.435581Z"}},"outputs":[],"execution_count":49},{"cell_type":"code","source":"# Reformatting paths\naudio_df = metadata_df[[\"primary_label\", \"filename\", \"class_name\"]]\naudio_df = audio_df.copy()\naudio_df.loc[:, \"path\"] = audio_df[\"filename\"].apply(lambda x: str(\"/kaggle/input/birdclef-2025/train_audio/\" + x))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:36:01.132913Z","iopub.execute_input":"2025-04-23T05:36:01.133267Z","iopub.status.idle":"2025-04-23T05:36:01.149582Z","shell.execute_reply.started":"2025-04-23T05:36:01.133242Z","shell.execute_reply":"2025-04-23T05:36:01.148831Z"}},"outputs":[],"execution_count":50},{"cell_type":"markdown","source":"### **Voice activity detection**\n\nThe audio training samples contains a lot of noise in the form of human annotators who speak after the species sound. Let us first segregate them seperately from species sounds to make the dataset cleaner. We will use pyannote's segmentation model to do this.","metadata":{}},{"cell_type":"code","source":"# Importing the pyannote segmentation model\nfrom pyannote.audio import Model\nfrom kaggle_secrets import UserSecretsClient\n\nuser_secrets = UserSecretsClient()\nHF_TOKEN = user_secrets.get_secret(\"HF_TOKEN\")\n\n\npyannote_segmenter = Model.from_pretrained(\"pyannote/segmentation\", \n                                           use_auth_token=HF_TOKEN\n                                          )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Defining the pipeline with hyperparameters for VAD\nfrom pyannote.audio.pipelines import VoiceActivityDetection\n\npipeline = VoiceActivityDetection(segmentation=pyannote_segmenter)\n\nHYPER_PARAMETERS = {\n  # onset/offset activation thresholds\n  \"onset\": 0.5, \"offset\": 0.5,\n  # remove speech regions shorter than that many seconds.\n  \"min_duration_on\": 0.0,\n  # fill non-speech regions shorter than that many seconds.\n  \"min_duration_off\": 0.0\n}\n\npipeline.instantiate(HYPER_PARAMETERS)\npipeline.to(torch.device(\"cuda\"))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let's create a Pytorch dataloader to compute non voice activity segments in batches.\n\nfrom torch.utils.data import DataLoader, Dataset\n\nclass AudioFilesDataset(Dataset):\n    def __init__(self, df):\n        self.df = df\n\n    def __len__(self):\n        return len(self.df)\n\n    def get_nss(self, SPEECH_SEGMENTS, waveform, sr):\n\n        \"\"\"\n        This function takes the speech segments annotations and original audio \n        waveform as inputs and computes non speech segment annotations.\n        \"\"\"\n        \n        NSS = { \"start\": [], \"end\": [] }\n        audio_length = waveform.shape[1]/sr\n        if SPEECH_SEGMENTS[\"start\"]:\n            if SPEECH_SEGMENTS[\"start\"][0] > 0:\n                NSS[\"start\"].append(0)\n                NSS[\"end\"].append(SPEECH_SEGMENTS[\"start\"][0])\n            \n            for i in range(len(SPEECH_SEGMENTS[\"start\"])-1):\n                NSS[\"start\"].append(SPEECH_SEGMENTS[\"end\"][i])\n                NSS[\"end\"].append(SPEECH_SEGMENTS[\"start\"][i+1])\n            \n            if SPEECH_SEGMENTS[\"end\"][-1] < audio_length:\n                NSS[\"start\"].append(SPEECH_SEGMENTS[\"end\"][-1])\n                NSS[\"end\"].append(audio_length)\n\n        return NSS       \n\n    def get_vad(self, path):\n\n        \"\"\"\n        This function takes the path of the audio file as an input and computes\n        the speech segments annotations using the VAD pipeline of pyannote's\n        segmentation model.\n        \"\"\"\n        \n        with torch.no_grad():\n            vad = pipeline(path)\n\n        SPEECH_SEGMENTS = {\"start\" : [], \"end\" : []}\n\n        for segment in vad.get_timeline():\n            SPEECH_SEGMENTS[\"start\"].append(segment.start)\n            SPEECH_SEGMENTS[\"end\"].append(segment.end)\n\n        return SPEECH_SEGMENTS         \n        \n    def __getitem__(self, idx):\n        \n        audio_path = self.df[\"path\"].iloc[idx]\n        waveform, sr = torchaudio.load(audio_path)\n        SPEECH_SEGMENTS = self.get_vad(audio_path)\n        NSS = self.get_nss(SPEECH_SEGMENTS, waveform, sr)\n        \n        return NSS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T03:59:48.386467Z","iopub.execute_input":"2025-04-23T03:59:48.386949Z","iopub.status.idle":"2025-04-23T03:59:48.394295Z","shell.execute_reply.started":"2025-04-23T03:59:48.386925Z","shell.execute_reply":"2025-04-23T03:59:48.393654Z"}},"outputs":[],"execution_count":38},{"cell_type":"code","source":"audio_dataset = AudioFilesDataset(audio_df)\naudio_loader = DataLoader(audio_dataset, batch_size = 1, shuffle = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T03:59:51.100326Z","iopub.execute_input":"2025-04-23T03:59:51.100596Z","iopub.status.idle":"2025-04-23T03:59:51.104859Z","shell.execute_reply.started":"2025-04-23T03:59:51.100576Z","shell.execute_reply":"2025-04-23T03:59:51.104137Z"}},"outputs":[],"execution_count":39},{"cell_type":"code","source":"nss_segments_list = []\nfor _, batch in enumerate(audio_loader):\n    nss_segments_list.append(batch)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_df[\"NSS_segments\"] = nss_segments_list\naudio_df = audio_df[[\"primary_label\", \"filename\", \"NSS_segments\"]]\n\n# Reformating the annotations from tensor form to normal python list format\naudio_df[\"NSS_segments\"] = audio_df[\"NSS_segments\"].apply(lambda x: {\n    \"start\" : [i.item() for i in x[\"start\"]],\n    \"end\" : [i.item() for i in x[\"end\"]],\n} )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:38:38.330726Z","iopub.execute_input":"2025-04-23T05:38:38.331405Z","iopub.status.idle":"2025-04-23T05:38:38.341877Z","shell.execute_reply.started":"2025-04-23T05:38:38.331376Z","shell.execute_reply":"2025-04-23T05:38:38.341189Z"}},"outputs":[],"execution_count":57},{"cell_type":"code","source":"# Saving the file as .npy file to preserve data types\nnss_segment_annotations = audio_df.to_records(index=False)\nnp.save(\"nss_annotations.npy\", nss_segment_annotations)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:44:52.767176Z","iopub.execute_input":"2025-04-23T05:44:52.76777Z","iopub.status.idle":"2025-04-23T05:44:52.816728Z","shell.execute_reply.started":"2025-04-23T05:44:52.767747Z","shell.execute_reply":"2025-04-23T05:44:52.816209Z"}},"outputs":[],"execution_count":64},{"cell_type":"code","source":"sample = np.load(r\"/kaggle/working/nss_annotations.npy\", allow_pickle = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T05:51:11.517873Z","iopub.execute_input":"2025-04-23T05:51:11.518534Z","iopub.status.idle":"2025-04-23T05:51:11.910886Z","shell.execute_reply.started":"2025-04-23T05:51:11.518506Z","shell.execute_reply":"2025-04-23T05:51:11.910313Z"}},"outputs":[],"execution_count":65}]}