{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I have used just like \"CLEF: Dataset Enhancer [Part I]\" Silero VAD model. Which helped a lot! So far it is a best model I have tried for this purpose.\nNow I have made few tries and just do not get the I dea","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        pass\n        \nprint(\"Load process is finished\")\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T09:05:09.771643Z","iopub.execute_input":"2025-05-25T09:05:09.772031Z","iopub.status.idle":"2025-05-25T09:05:28.255918Z","shell.execute_reply.started":"2025-05-25T09:05:09.772006Z","shell.execute_reply":"2025-05-25T09:05:28.254923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch\nimport os\nfrom pathlib import Path\nimport shutil\nimport soundfile as sf\nimport torchaudio\nimport random\nimport IPython.display as ipd\nimport tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T09:05:31.012859Z","iopub.execute_input":"2025-05-25T09:05:31.013186Z","iopub.status.idle":"2025-05-25T09:05:37.711964Z","shell.execute_reply.started":"2025-05-25T09:05:31.013161Z","shell.execute_reply":"2025-05-25T09:05:37.711137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T09:05:41.556620Z","iopub.execute_input":"2025-05-25T09:05:41.557225Z","iopub.status.idle":"2025-05-25T09:05:41.934286Z","shell.execute_reply.started":"2025-05-25T09:05:41.557189Z","shell.execute_reply":"2025-05-25T09:05:41.933052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport librosa\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport IPython.display as ipd\nimport pandas as pd\nimport soundfile as sf\n\n# === Load VAD ===\ntorch.set_num_threads(1)\nmodel, (get_speech_timestamps, _, _, _, _) = torch.hub.load(\n    repo_or_dir='snakers4/silero-vad', \n    model='silero_vad',\n    trust_repo=True\n)\n\n# === Settings ===\nAUDIO_BASE_DIR = Path(\"/kaggle/input/birdclef-2025/train_audio\")\nAUTHOR_FILTER = \"Fabio A. Sarria-S\"\nCHUNK_LEN = 0.1  # seconds\nNUM_AUDIO_TO_DISPLAY = 26  # display limit\n\n# === Load your dataframe ===\n# train_df = pd.read_csv(\"path/to/train.csv\")  # Uncomment and adjust if needed\nfabio_df = train_df[train_df['author'] == AUTHOR_FILTER].copy()\nfabio_df_len = len(fabio_df)\nfabio_df.reset_index(drop=True, inplace=True)\n\n\nNUM_AUDIO_TO_DISPLAY = min(NUM_AUDIO_TO_DISPLAY, fabio_df_len)\nprint(f\"Total files to be displayed: {NUM_AUDIO_TO_DISPLAY}\")\n# === Iterate and visualize ===\nfor i, row in fabio_df.iterrows():\n    file_path = AUDIO_BASE_DIR / row['filename']\n    if not file_path.exists():\n        print(f\"Missing file: {file_path}\")\n        continue\n\n    # Load audio\n    wav_full, sr = librosa.load(file_path, sr=16000)\n    wav = wav_full.copy()\n\n    # Detect speech\n    speech_timestamps = get_speech_timestamps(torch.Tensor(wav), model)\n\n    # Truncate audio from start of first detected speech\n    if speech_timestamps:\n        first_start = speech_timestamps[0]['start']\n        wav = wav[:first_start]\n\n    # Compute audio power in dB (for detection only)\n    chunk = int(CHUNK_LEN * sr)\n    power = wav ** 2\n    pad = int(np.ceil(len(power) / chunk) * chunk - len(power))\n    power = np.pad(power, (0, pad))\n    power_chunks = power.reshape((-1, chunk)).sum(axis=1)\n    power_db = 10 * np.log10(power_chunks + 1e-6)\n\n    # Use mean of valid chunks instead of max to define threshold\n    mean_db = np.mean(power_db)\n    threshold = mean_db - 10  # 10 dB below mean plateau\n    valid_mask = power_db >= threshold\n\n    if np.any(valid_mask):\n        start_chunk = np.argmax(valid_mask)\n        end_chunk = len(valid_mask) - np.argmax(valid_mask[::-1])\n        power_db_trimmed = power_db[start_chunk:end_chunk]\n\n        # Detect trailing silence using rolling window on power_db\n        window_size = 3\n        silence_threshold = threshold\n        binary_silence = (power_db_trimmed < silence_threshold).astype(int)\n        rolling = np.convolve(binary_silence, np.ones(window_size, dtype=int), mode='valid')\n\n        silence_start_rel = np.argmax(rolling == window_size)\n        if rolling[silence_start_rel] == window_size:\n            end_chunk = start_chunk + silence_start_rel\n\n        start_sample = start_chunk * chunk\n        end_sample = end_chunk * chunk\n        \n        start_time_sec = round(start_sample / sr, 1)\n        end_time_sec = round(end_sample / sr, 1)\n        \n        start_chunk = int(start_sample / chunk)\n        start_chunk = int(end_sample / chunk)\n        \n        wav = wav_full[start_sample:end_sample]  # Slice original waveform cleanly\n        power_db = power_db[start_chunk:end_chunk]\n\n        print(f\"ROI start: {start_time_sec:.2f}s, end: {end_time_sec:.2f}s\")\n\n\n    # Create dummy segmentation (voice already removed)\n    segmentation = np.zeros_like(wav)\n\n    # limit number of displayed\n    print(f\"Showing file {i+1}\")\n    if i < NUM_AUDIO_TO_DISPLAY:\n        # Plot\n        fig = plt.figure(figsize=(24, 3))\n        fig.suptitle(f\"{row['filename']} by {row['author']}\")\n        t_power = np.arange(len(power_db)) * CHUNK_LEN\n        plt.plot(t_power, power_db, 'b', label='Audio Power (dB)')\n        t_seg = np.arange(len(segmentation)) / sr\n        plt.plot(t_seg, segmentation, 'r', label='Voice Detection')\n        plt.xlabel('Time (s)')\n        plt.ylabel('Amplitude (dB) / Voice Detection')\n        plt.legend()\n        plt.show()\n\n        # Play audio\n        display(ipd.Audio(wav, rate=sr))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T09:07:06.512837Z","iopub.execute_input":"2025-05-25T09:07:06.513194Z","iopub.status.idle":"2025-05-25T09:08:21.603715Z","shell.execute_reply.started":"2025-05-25T09:07:06.513166Z","shell.execute_reply":"2025-05-25T09:08:21.602787Z"}},"outputs":[],"execution_count":null}]}