{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":234320181,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Human Voice Removal Caution Around Ruther1\n\nThis notebook demonstrates that some caution should be taken around removal of human audio outright as some species can easily be misclassified as human voices. When evaluating the species I noticed that my generated melspecs for this species were very tiny as I had originally clipped the human voice elements. Upon further review I noticed human voices were not present for this species.\n\nThis useful notebook was used here for the human voice detection: https://www.kaggle.com/code/kdmitrie/bc25-separation-voice-from-data\n\nIn general it's really helpful but I wanted to note to exercise caution given that. Some users reported looking for sounds with 50% or less human voices and utilizing the melspec containing that vs filtering the human voices outright. Perhaps an approach where that is done first but if the sound file does not contain 50% of non human voices falling back to the first 5s as an approach would get around this issue.","metadata":{}},{"cell_type":"code","source":"import os\nimport pickle\n\n# Path to the pickle in your Kaggle input (read‑only!)\npkl_path = \"/kaggle/input/bc25-separation-voice-from-data/train_voice_data.pkl\"\n\nif not os.path.isfile(pkl_path):\n    raise FileNotFoundError(f\"No such file: {pkl_path}\")\n\n# Open in binary‑read mode and load\nwith open(pkl_path, \"rb\") as f:\n    train_voice_data = pickle.load(f)\n\nprint(f\"Loaded object of type {type(train_voice_data)}\")\ntry:\n    print(f\"Contains {len(train_voice_data)} entries\")\nexcept Exception:\n    pass","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-18T14:27:11.313522Z","iopub.execute_input":"2025-05-18T14:27:11.313815Z","iopub.status.idle":"2025-05-18T14:27:11.326951Z","shell.execute_reply.started":"2025-05-18T14:27:11.313797Z","shell.execute_reply":"2025-05-18T14:27:11.326085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Audio, display\n\nruther_records = [\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC512534.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC257300.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC681215.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC257299.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC257303.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC657821.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC178070.ogg\",\n    \"/kaggle/input/birdclef-2025/train_audio/ruther1/XC504229.ogg\"\n]\n\nfor rec_path in ruther_records:\n    metadata = train_voice_data.get(rec_path, {})\n    print(f\"{rec_path}: {metadata}\")\n    display(Audio(rec_path, autoplay=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-18T14:31:14.959180Z","iopub.execute_input":"2025-05-18T14:31:14.959402Z","iopub.status.idle":"2025-05-18T14:31:15.027073Z","shell.execute_reply.started":"2025-05-18T14:31:14.959387Z","shell.execute_reply":"2025-05-18T14:31:15.026418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}