{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945,"isSourceIdPinned":false}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Some recordings contain human voice\n\nAs @martinapreusse [said](https://www.kaggle.com/competitions/birdclef-2025/discussion/567551#3148665), \n> All recordings of the author Fabio A. Sarria-S contain a human voice.\n\nLet's look at them and delete all unnecessary sounds!","metadata":{}},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:01.715241Z","iopub.execute_input":"2025-03-17T09:12:01.715695Z","iopub.status.idle":"2025-03-17T09:12:05.641574Z","shell.execute_reply.started":"2025-03-17T09:12:01.715652Z","shell.execute_reply":"2025-03-17T09:12:05.640495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Using VAD while exploring the records. Quite good!","metadata":{}},{"cell_type":"code","source":"torch.set_num_threads(1)\nmodel, (get_speech_timestamps, _, read_audio, _, _) = torch.hub.load(repo_or_dir='snakers4/silero-vad', model='silero_vad')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:12:30.513519Z","iopub.execute_input":"2025-03-17T09:12:30.513852Z","iopub.status.idle":"2025-03-17T09:12:32.097686Z","shell.execute_reply.started":"2025-03-17T09:12:30.513815Z","shell.execute_reply":"2025-03-17T09:12:32.096887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Is there any voice in the train soundscapes?","metadata":{}},{"cell_type":"markdown","source":"As @tomok1 [proposed](https://www.kaggle.com/competitions/birdclef-2025/discussion/568886#3154723), the default threshold of 0.5 may be lowered to 0.4.","metadata":{}},{"cell_type":"code","source":"from glob import glob\nimport pickle\n\nfiles = sorted(glob('/kaggle/input/competitions/birdclef-2026/train_audio/*/*.ogg'))\nvoice_data = {}\nfor fname in files:\n    wav = read_audio(fname)\n    speech_timestamps = get_speech_timestamps(wav, model, return_seconds=True, threshold=0.5)\n    if len(speech_timestamps):\n        voice_data[fname] = speech_timestamps\n\n        with open('train_voice_summary.txt', 'a') as f:\n            f.write(f'{fname}\\n')\n            \nwith open('train_voice_data.pkl', 'wb') as f:\n    pickle.dump(voice_data, f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob\nimport pickle\n\nfiles = sorted(glob('/kaggle/input/competitions/birdclef-2026/train_soundscapes/*.ogg'))\n\nvoice_data = {}\nfor fname in files:\n    wav = read_audio(fname)\n    speech_timestamps = get_speech_timestamps(wav, model, return_seconds=True, threshold=0.5) # default threshold\n    if len(speech_timestamps):\n        voice_data[fname] = speech_timestamps\n\n        with open('ss_voice_summary.txt', 'a') as f:\n            f.write(f'{fname}\\n')\n            \nwith open('ss_voice_data.pkl', 'wb') as f:\n    pickle.dump(voice_data, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-17T09:13:36.944678Z","iopub.execute_input":"2025-03-17T09:13:36.945013Z","iopub.status.idle":"2025-03-17T09:13:46.350985Z","shell.execute_reply.started":"2025-03-17T09:13:36.944986Z","shell.execute_reply":"2025-03-17T09:13:46.35017Z"}},"outputs":[],"execution_count":null}]}