{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Creating Random Soundscapes\n\nAs the test dataset is composed of soundscapes containing multiple birds, it seems natural to want to train a multilabel classifier using soundscapes generated from the training dataset. This function will overlay the audio from `max_birds` training samples and compute the MFCC values for the new soundscape. The function will output the 40 MFCC values which can be used for inputs and a sparse dataframe (i.e. some rows containing Nones) with the labels from the training samples used. \n\nNote that the MFCC values are not the same as the raw spectrograms used in the baseline solution for a CNN. However, these coefficients have shown good performance for audio classification in other applications, so they should be usable for this problem. \n\nSave and Run All this notebook to generate your own output or use the csv's saved in the \"Outputs\" tab if you like my configuration. Note that this notebook takes around 2 hours to complete (or can generate around 10k soundscapes per hour).","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\nimport os\n\nfrom tqdm import tqdm\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-30T12:16:36.192986Z","iopub.execute_input":"2024-04-30T12:16:36.193956Z","iopub.status.idle":"2024-04-30T12:16:36.674130Z","shell.execute_reply.started":"2024-04-30T12:16:36.193916Z","shell.execute_reply":"2024-04-30T12:16:36.672967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_soundscapes(data_path, metadata_path, max_birds = 5, num_samples=5000):\n    features = []\n    labels = []\n\n    metadata = pd.read_csv(metadata_path)\n    metadata_s = metadata.sample(num_samples)\n\n    counter = 0\n    \n    for index, row in tqdm(metadata_s.iterrows(), total=num_samples):\n        file_path = os.path.join(data_path, row['filename'])\n\n        randoms = np.random.randint(0, max_birds)\n\n        # Load the audio file and resample it\n        target_sr = 22050\n        audio, sample_rate = librosa.load(file_path, mono=False, sr=None)\n        \n        total_len = len(audio) / sample_rate\n        desired_samples = 5 * sample_rate\n        if len(audio) - desired_samples - 1 <= 0: continue\n        \n        rand_start = np.random.randint(0, len(audio) - desired_samples - 1)\n\n        audio = audio[rand_start : rand_start + desired_samples]\n        \n        extras = ((len(metadata)-1)*np.random.rand(randoms)).round()\n\n        s_labels = [x.replace(\"[\", \"\").replace(\"]\", \"\").replace(\"'\", \"\") if x != \"[]\" else None for x in row[\"secondary_labels\"].split(\",\")]\n        if s_labels[0] == None: s_labels = []\n        \n        temp_labels = [row[\"primary_label\"]] + s_labels\n        for r in extras:\n            temp_audio, sr = librosa.load(os.path.join(data_path, metadata[\"filename\"].iloc[int(r)]), mono=False, sr=None)\n            \n            desired_samples = 5 * sr\n            if len(temp_audio) - desired_samples - 1 <= 0: continue\n            rand_start = np.random.randint(0, len(temp_audio) - desired_samples - 1)\n            \n            temp_audio = temp_audio[rand_start : rand_start + desired_samples]\n            \n            if len(temp_audio) > len(audio):\n                audio += temp_audio[:len(audio)]\n            elif len(temp_audio) == len(audio):\n                audio += temp_audio\n            else:\n                audio = temp_audio + audio[:len(temp_audio)]\n            \n            s_labels = [x.replace(\"[\", \"\").replace(\"]\", \"\").replace(\"'\", \"\") if x != \"[]\" else None for x in metadata[\"secondary_labels\"].iloc[int(r)].split(\",\")]\n            if s_labels[0] == None: s_labels = []\n            \n            temp_labels += [metadata[\"primary_label\"].iloc[int(r)]] + s_labels\n        \n        # Extract MFCC features\n        mfccs = librosa.feature.mfcc(y=audio, sr=target_sr, n_mfcc=40)\n        mfccs_scaled = np.mean(mfccs.T, axis=0)\n\n        # Append features and labels\n        features.append(mfccs_scaled)\n        labels.append(temp_labels)\n\n        if counter % 1000 == 0:\n            print(f\"Generating sample {counter}/{num_samples}\")\n        counter += 1\n        \n    return pd.DataFrame(features), pd.DataFrame(labels)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:39:33.884283Z","iopub.execute_input":"2024-04-30T12:39:33.884678Z","iopub.status.idle":"2024-04-30T12:39:33.900952Z","shell.execute_reply.started":"2024-04-30T12:39:33.884644Z","shell.execute_reply":"2024-04-30T12:39:33.899744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")\ntrain_meta.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:39:34.812414Z","iopub.execute_input":"2024-04-30T12:39:34.812819Z","iopub.status.idle":"2024-04-30T12:39:34.936885Z","shell.execute_reply.started":"2024-04-30T12:39:34.812786Z","shell.execute_reply":"2024-04-30T12:39:34.935740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features, labels = load_soundscapes(\"/kaggle/input/birdclef-2024/train_audio\", \"/kaggle/input/birdclef-2024/train_metadata.csv\", max_birds=5, num_samples=20000)","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:39:36.058726Z","iopub.execute_input":"2024-04-30T12:39:36.059135Z","iopub.status.idle":"2024-04-30T12:43:05.696552Z","shell.execute_reply.started":"2024-04-30T12:39:36.059100Z","shell.execute_reply":"2024-04-30T12:43:05.692883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(features)\nprint(labels)\n\nfeatures.to_csv(\"train_features_mfcc.csv\")\nlabels.to_csv(\"train_labels_mfcc.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-30T12:46:21.139401Z","iopub.execute_input":"2024-04-30T12:46:21.139938Z","iopub.status.idle":"2024-04-30T12:46:21.211399Z","shell.execute_reply.started":"2024-04-30T12:46:21.139879Z","shell.execute_reply":"2024-04-30T12:46:21.210238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}