{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Some recordings contain human voice\n\nAs @martinapreusse [said](https://www.kaggle.com/competitions/birdclef-2025/discussion/567551#3148665), \n> All recordings of the author Fabio A. Sarria-S contain a human voice.\n\nLet's look at them and delete all unnecessary sounds!","metadata":{}},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:01.715241Z","iopub.execute_input":"2025-03-17T09:12:01.715695Z","iopub.status.idle":"2025-03-17T09:12:05.641574Z","shell.execute_reply.started":"2025-03-17T09:12:01.715652Z","shell.execute_reply":"2025-03-17T09:12:05.640495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Separate Fabio's recordings","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\nfabio = df[df.author == 'Fabio A. Sarria-S'].copy()\n\nprint(f'We have {len(fabio)} Fabio\\'s recordings in total')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:05.642965Z","iopub.execute_input":"2025-03-17T09:12:05.643441Z","iopub.status.idle":"2025-03-17T09:12:05.831092Z","shell.execute_reply.started":"2025-03-17T09:12:05.643411Z","shell.execute_reply":"2025-03-17T09:12:05.830252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Plot the avarage energy of the sound - we can clearly see the pattern","metadata":{}},{"cell_type":"code","source":"N = 10\nchunk_len = 0.05 # Chunk len in seconds\n\nfig, ax = plt.subplots(nrows=N)\nfig.set_size_inches((24, 3 * N))\nfor n in range(N):\n    # Load the data\n    rec = fabio.iloc[n]\n    wav, sr = librosa.load(f'/kaggle/input/birdclef-2025/train_audio/{rec.filename}')\n\n    # Calculate the sound power\n    power = wav ** 2\n    \n    # Split the data into chunks and sum the energy in every chunk\n    chunk = int(chunk_len * sr)\n    \n    pad = int(np.ceil(len(power) / chunk) * chunk - len(power))\n    power = np.pad(power, (0, pad))\n    power = power.reshape((-1, chunk)).sum(axis=1)\n\n    t = np.arange(len(power)) * chunk_len\n    ax[n].plot(t, 10 * np.log10(power))\n    ax[n].set_xlim([0, 20])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:05.832822Z","iopub.execute_input":"2025-03-17T09:12:05.833092Z","iopub.status.idle":"2025-03-17T09:12:25.226668Z","shell.execute_reply.started":"2025-03-17T09:12:05.833061Z","shell.execute_reply":"2025-03-17T09:12:25.225547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The idea is to identify the points where the line intersects -50dB level the first and the second time","metadata":{}},{"cell_type":"code","source":"level = -50\nN = 1\nresult = fabio['filename'].to_frame()\n\nfor n, rec in fabio.iterrows():\n    # Load the data\n    wav, sr = librosa.load(f'/kaggle/input/birdclef-2025/train_audio/{rec.filename}')\n\n    # Calculate the sound power\n    power = wav ** 2\n    \n    # Split the data into chunks and sum the energy in every chunk\n    chunk = int(chunk_len * sr)\n    \n    pad = int(np.ceil(len(power) / chunk) * chunk - len(power))\n    power = np.pad(power, (0, pad))\n    power = power.reshape((-1, chunk)).sum(axis=1)\n\n    power_dB = 10 * np.log10(power)\n    x = power_dB - level\n    intersections = np.where(x[:-1] * x[1:] < 0)[0]\n    a, b = intersections[:2]\n    result.loc[result['filename'] == rec.filename, 'start'] = a * chunk_len\n    result.loc[result['filename'] == rec.filename, 'stop'] = b * chunk_len","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:25.228005Z","iopub.execute_input":"2025-03-17T09:12:25.228567Z","iopub.status.idle":"2025-03-17T09:12:30.481657Z","shell.execute_reply.started":"2025-03-17T09:12:25.228522Z","shell.execute_reply":"2025-03-17T09:12:30.480595Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Display the result and check that it's reasonable","metadata":{}},{"cell_type":"code","source":"result.to_csv('fabio.csv', index=False)\ndisplay(result)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:30.482745Z","iopub.execute_input":"2025-03-17T09:12:30.483024Z","iopub.status.idle":"2025-03-17T09:12:30.512632Z","shell.execute_reply.started":"2025-03-17T09:12:30.483000Z","shell.execute_reply":"2025-03-17T09:12:30.511688Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Using VAD while exploring the other records. Quite good!","metadata":{}},{"cell_type":"code","source":"torch.set_num_threads(1)\nmodel, (get_speech_timestamps, _, read_audio, _, _) = torch.hub.load(repo_or_dir='snakers4/silero-vad', model='silero_vad')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:30.513519Z","iopub.execute_input":"2025-03-17T09:12:30.513852Z","iopub.status.idle":"2025-03-17T09:12:32.097686Z","shell.execute_reply.started":"2025-03-17T09:12:30.513815Z","shell.execute_reply":"2025-03-17T09:12:32.096887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import IPython.display as ipd\n\ndf = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\ndf = df[df.collection == 'CSA'].copy()\n\nauthor_map = {\n    'Paula Caycedo-Rosales | Juan-Pablo López': 'Paula Caycedo-Rosales',\n    'Ana María Ospina-Larrea | Daniela Murillo': 'Ana María Ospina-Larrea',\n    'Eliana Barona-Cortés | Daniela García-Cobos': 'Eliana Barona-Cortés',\n    'Eliana Barona- Cortés': 'Eliana Barona-Cortés',\n    'Alexandra Butrago-Cardona': 'Alexandra Buitrago-Cardona',\n    'Diego A Gómez-Morales': 'Diego A. Gomez-Morales',\n}\nauthor_map_func = lambda x: author_map[x] if x in author_map.keys() else x\n\ndf.author = df.author.map(author_map_func)\nauthors = sorted(df.author.unique())\n\n# Here, I limit the output to 3 authors. Otherwise, the webpage becomes too heavy to load.\n# If your are interested, please check the previous version of the notebook!\nfor author in authors[:3]:\n    selection = df[df.author == author].copy()\n    print(f'We have {len(selection)} recordings by {author} in total')\n    \n    N = len(selection)\n    chunk_len = 0.1 # Chunk len in seconds\n    \n    for n in range(N):\n        # Load the data\n        rec = selection.iloc[n]\n        fname = f'/kaggle/input/birdclef-2025/train_audio/{rec.filename}'\n        wav, sr = librosa.load(fname)\n    \n        # Calculate the sound power\n        power = wav ** 2\n        \n        # Split the data into chunks and sum the energy in every chunk\n        chunk = int(chunk_len * sr)\n        \n        pad = int(np.ceil(len(power) / chunk) * chunk - len(power))\n        power = np.pad(power, (0, pad))\n        power = power.reshape((-1, chunk)).sum(axis=1)\n\n        speech_timestamps = get_speech_timestamps(torch.Tensor(wav), model)\n        segmentation = np.zeros_like(wav)\n        for st in speech_timestamps:\n            segmentation[st['start']: st['end']] = 20\n    \n        fig = plt.figure(figsize=(24, 3))\n        fig.suptitle(f'{rec.filename} by {rec.author}')\n        \n        t = np.arange(len(power)) * chunk_len\n        plt.plot(t, 10 * np.log10(power), 'b')\n        \n        t = np.arange(len(segmentation)) / sr\n        plt.plot(t, segmentation, 'r')        \n        plt.show()\n        \n        display(ipd.Audio(fname))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:32.098753Z","iopub.execute_input":"2025-03-17T09:12:32.099125Z","iopub.status.idle":"2025-03-17T09:13:11.623398Z","shell.execute_reply.started":"2025-03-17T09:12:32.099088Z","shell.execute_reply":"2025-03-17T09:13:11.622049Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Now, is there any voice in the train soundscapes?","metadata":{}},{"cell_type":"markdown","source":"As @tomok1 [proposed](https://www.kaggle.com/competitions/birdclef-2025/discussion/568886#3154723), the default threshold of 0.5 may be lowered to 0.4.","metadata":{}},{"cell_type":"code","source":"from glob import glob\nimport pickle\n\nfiles = sorted(glob('/kaggle/input/birdclef-2025/train_audio/*/*.ogg'))\nvoice_data = {}\nfor fname in files:\n    wav = read_audio(fname)\n    speech_timestamps = get_speech_timestamps(wav, model, return_seconds=True, threshold=0.5)\n    if len(speech_timestamps):\n        voice_data[fname] = speech_timestamps\n\n        with open('train_voice_summary.txt', 'a') as f:\n            f.write(f'{fname}\\n')\n            \nwith open('train_voice_data.pkl', 'wb') as f:\n    pickle.dump(voice_data, f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob\nimport pickle\n\nfiles = sorted(glob('/kaggle/input/birdclef-2025/train_soundscapes/*.ogg'))\nvoice_data = {}\nfor fname in files:\n    wav = read_audio(fname)\n    speech_timestamps = get_speech_timestamps(wav, model, return_seconds=True, threshold=0.5) # default threshold\n    if len(speech_timestamps):\n        voice_data[fname] = speech_timestamps\n\n        with open('ss_voice_summary.txt', 'a') as f:\n            f.write(f'{fname}\\n')\n            \nwith open('ss_voice_data.pkl', 'wb') as f:\n    pickle.dump(voice_data, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:13:36.944678Z","iopub.execute_input":"2025-03-17T09:13:36.945013Z","iopub.status.idle":"2025-03-17T09:13:46.350985Z","shell.execute_reply.started":"2025-03-17T09:13:36.944986Z","shell.execute_reply":"2025-03-17T09:13:46.350170Z"}},"outputs":[],"execution_count":null}]}