{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2024\n> The objective of this competition is to identify under-studied Indian bird species by their calls.\n\n<div align=\"center\">\n  <img src=\"https://i.ibb.co/47F4P9R/birdclef2024.png\" alt=\"BirdCLEF 2024\">\n</div>\n","metadata":{}},{"cell_type":"markdown","source":"# Approach\nIn the dataset we are provided with audio recording of bird sounds.\n1. Generate spectrograms from audio signals.\n2. Augment the generated spectrograms.\n3. Train a model on the spectrogram data.\n4. Save the trained model. as in the submission file we cannot use GPU so training needs to be done seperately.\n5. Create a seperate notebook that loads the trained model and generates submission.","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore') # to ignore warnings\nimport os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa # library for audio analysis\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-02T05:43:42.376380Z","iopub.execute_input":"2024-05-02T05:43:42.377945Z","iopub.status.idle":"2024-05-02T05:44:06.031604Z","shell.execute_reply.started":"2024-05-02T05:43:42.377881Z","shell.execute_reply":"2024-05-02T05:44:06.030289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration setup","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 0\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10\n    preset = 'efficientnetv2_b2_imagenet'\n    \n    # Data augmentation parameters\n    augment=True\n\n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"execution":{"iopub.status.busy":"2024-05-02T05:44:06.033407Z","iopub.execute_input":"2024-05-02T05:44:06.034029Z","iopub.status.idle":"2024-05-02T05:44:06.063608Z","shell.execute_reply.started":"2024-05-02T05:44:06.033976Z","shell.execute_reply":"2024-05-02T05:44:06.062536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/birdclef-2024\"","metadata":{"execution":{"iopub.status.busy":"2024-05-02T06:04:43.937876Z","iopub.execute_input":"2024-05-02T06:04:43.939364Z","iopub.status.idle":"2024-05-02T06:04:43.946764Z","shell.execute_reply.started":"2024-05-02T06:04:43.939309Z","shell.execute_reply":"2024-05-02T06:04:43.945106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_taxonomy = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")\ntrain_metadata = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-05-02T05:44:20.010008Z","iopub.execute_input":"2024-05-02T05:44:20.010425Z","iopub.status.idle":"2024-05-02T05:44:20.304475Z","shell.execute_reply.started":"2024-05-02T05:44:20.010382Z","shell.execute_reply":"2024-05-02T05:44:20.303609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_taxonomy.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T05:44:21.593757Z","iopub.execute_input":"2024-05-02T05:44:21.595256Z","iopub.status.idle":"2024-05-02T05:44:21.625719Z","shell.execute_reply.started":"2024-05-02T05:44:21.595195Z","shell.execute_reply":"2024-05-02T05:44:21.624251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T05:44:23.304495Z","iopub.execute_input":"2024-05-02T05:44:23.305807Z","iopub.status.idle":"2024-05-02T05:44:23.334936Z","shell.execute_reply.started":"2024-05-02T05:44:23.305757Z","shell.execute_reply":"2024-05-02T05:44:23.333277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modifying Metadata for easier processing","metadata":{}},{"cell_type":"code","source":"train_metadata['file_path'] = path + '/train_audio/' + train_metadata.filename\ntrain_metadata['target'] = train_metadata.primary_label.map(CFG.name2label)\ntrain_metadata['file_name'] = train_metadata.file_path.map(lambda x: x.split('/')[-1])\ntrain_metadata['xc_id'] = train_metadata.file_path.map(lambda x: x.split('/')[-1].split('.')[0])\ndisplay(train_metadata.file_path[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-02T06:31:08.938555Z","iopub.execute_input":"2024-05-02T06:31:08.939045Z","iopub.status.idle":"2024-05-02T06:31:09.011988Z","shell.execute_reply.started":"2024-05-02T06:31:08.939007Z","shell.execute_reply":"2024-05-02T06:31:09.009203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath): \n    audio, sr = librosa.load(filepath)\n    return audio, sr\ndef get_spectrogram(audio): # generates spectrograms from audio\n    spec = librosa.feature.melspectrogram(y=audio, \n                                   sr=CFG.sample_rate, \n                                   n_mels=256,\n                                   n_fft=2048,\n                                   hop_length=512,\n                                   fmax=CFG.fmax,\n                                   fmin=CFG.fmin,\n                                   )\n    spec = librosa.power_to_db(spec, ref=1.0)\n    min_ = spec.min()\n    max_ = spec.max()\n    if max_ != min_:\n        spec = (spec - min_)/(max_ - min_)\n    return spec\n\ndef display_audio(row):\n    caption = f'Id: {row.filename} | Name: {row.common_name} | Sci.Name: {row.scientific_name} | Rating: {row.rating}'\n    audio, sr = load_audio(row.file_path)\n    audio = audio[:CFG.audio_len]\n    spec = get_spectrogram(audio)\n    print(\"# Audio:\")\n    display(ipd.Audio(audio, rate=CFG.sample_rate))\n    print('# Visualization:')\n    fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n    fig.suptitle(caption)\n    lid.waveshow(audio,\n                 sr=CFG.sample_rate,\n                 ax=ax[0],\n                 color= cmap(0.1))\n    lid.specshow(spec, \n                 sr = CFG.sample_rate, \n                 hop_length=512,\n                 n_fft=2048,\n                 fmin=CFG.fmin,\n                 fmax=CFG.fmax,\n                 x_axis = 'time', \n                 y_axis = 'mel',\n                 cmap = 'coolwarm',\n                 ax=ax[1])\n    ax[0].set_xlabel('');\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T06:31:12.602070Z","iopub.execute_input":"2024-05-02T06:31:12.602555Z","iopub.status.idle":"2024-05-02T06:31:12.618287Z","shell.execute_reply.started":"2024-05-02T06:31:12.602519Z","shell.execute_reply":"2024-05-02T06:31:12.616822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = train_metadata.iloc[0]\ndisplay_audio(row)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T06:31:16.171723Z","iopub.execute_input":"2024-05-02T06:31:16.173362Z","iopub.status.idle":"2024-05-02T06:31:26.206534Z","shell.execute_reply.started":"2024-05-02T06:31:16.173310Z","shell.execute_reply":"2024-05-02T06:31:26.205484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}