{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nimport librosa\n\nnp.random.seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T16:24:20.952551Z","iopub.execute_input":"2025-03-29T16:24:20.952957Z","iopub.status.idle":"2025-03-29T16:24:20.957243Z","shell.execute_reply.started":"2025-03-29T16:24:20.952922Z","shell.execute_reply":"2025-03-29T16:24:20.956176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT = '/kaggle/input/birdclef-2025/'\nsub = pd.read_csv('/kaggle/input/birdclef-2025/sample_submission.csv')\ntrain = pd.read_csv(ROOT+'train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T15:40:32.604535Z","iopub.execute_input":"2025-03-29T15:40:32.604872Z","iopub.status.idle":"2025-03-29T15:40:32.803877Z","shell.execute_reply.started":"2025-03-29T15:40:32.604846Z","shell.execute_reply":"2025-03-29T15:40:32.803027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nrow_id: A slug of soundscape_[soundscape_id]_[end_time] for the prediction; e.g.,\nSegment 00:15-00:20 of 1-minute test soundscape soundscape_12345.ogg has row ID soundscape_12345_20.\n'''\n\n\ndef generate_sub():\n    classes = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n    df = pd.DataFrame(columns = ['row_id'] + classes)\n    row_id = []\n    aux_count = 0\n    for f in os.listdir(ROOT+'test_soundscapes'):   \n        if os.path.isfile(os.path.join(ROOT+'test_soundscapes', f)):\n            for i in range(1, 60//5+1):\n                row_id.append('soundscape_'+f.split('.')[0]+f'_{i*5}')\n                df.loc[aux_count, 'row_id'] = row_id[-1]\n                df.loc[aux_count, classes[0]:] = [np.random.rand(len(classes)) for _ in classes]\n                aux_count += 1\n                \n    df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T17:52:00.971477Z","iopub.execute_input":"2025-03-29T17:52:00.971924Z","iopub.status.idle":"2025-03-29T17:52:00.980638Z","shell.execute_reply.started":"2025-03-29T17:52:00.971883Z","shell.execute_reply":"2025-03-29T17:52:00.979435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nfrom Stefan Kahl's Notebook\nhttps://www.kaggle.com/code/stefankahl/birdclef-2025-sample-submission/notebook\n'''\n\nimport os\nimport librosa\nimport numpy as np\nimport pandas as pd\n\n\n# Set seed\nnp.random.seed(42)\n\n# Class labels from train audio\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n\n# List of test soundscapes (only visible during submission)\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n\n# Open each soundscape and make predictions for 5-second segments\n# Use pandas df with 'row_id' plus class labels as columns\npredictions = pd.DataFrame(columns=['row_id'] + class_labels)\nfor soundscape in test_soundscapes:\n\n    # Load audio\n    sig, rate = librosa.load(path=soundscape, sr=None)\n\n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n        \n    # Make predictions for each chunk\n    for i, chunk in enumerate(chunks):\n        \n        # Get row id  (soundscape id + end time of 5s chunk)      \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n        \n        # Make prediction (let's use random scores for now)\n        # scores = model.predict...\n        scores = np.random.rand(len(class_labels))\n        \n        # Append to predictions as new row\n        new_row = pd.DataFrame([[row_id] + list(scores)], columns=['row_id'] + class_labels)\n        predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n        \n# Save prediction as csv\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T17:52:01.986404Z","iopub.execute_input":"2025-03-29T17:52:01.986843Z","iopub.status.idle":"2025-03-29T17:52:02.030813Z","shell.execute_reply.started":"2025-03-29T17:52:01.986809Z","shell.execute_reply":"2025-03-29T17:52:02.029619Z"}},"outputs":[],"execution_count":null}]}