{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945},{"sourceType":"modelInstanceVersion","sourceId":6130,"databundleVersionId":7429420,"modelInstanceId":4598,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6120,"databundleVersionId":7429401,"modelInstanceId":4603,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6128,"databundleVersionId":7429417,"modelInstanceId":4596,"modelId":2797},{"sourceType":"kernelVersion","sourceId":311807047}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2025 with KerasCV and Keras","metadata":{}},{"cell_type":"markdown","source":"This notebook reports the process of training a Deep Learning model to recognize bird species by their songs (audio data). Specifically, this notebook uses the EfficientNetV2 backbone from KerasCV on the competition dataset. It also shows how to convert audio data to mel-spectrograms using Keras. The original notebook is https://www.kaggle.com/code/awsaf49/birdclef24-kerascv-starter-train, which was adapted to the current competition. ","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:08.858545Z","iopub.execute_input":"2026-04-16T17:16:08.859395Z","iopub.status.idle":"2026-04-16T17:16:08.866508Z","shell.execute_reply.started":"2026-04-16T17:16:08.859354Z","shell.execute_reply":"2026-04-16T17:16:08.865860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Library Version\nprint(\"TensorFlow:\", tf.__version__)\nprint(\"Keras:\", keras.__version__)\nprint(\"KerasCV:\", keras_cv.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:08.867970Z","iopub.execute_input":"2026-04-16T17:16:08.868309Z","iopub.status.idle":"2026-04-16T17:16:08.886328Z","shell.execute_reply.started":"2026-04-16T17:16:08.868282Z","shell.execute_reply":"2026-04-16T17:16:08.885557Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 5 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10\n    preset = 'efficientnetv2_b0'\n    \n    # Data augmentation parameters\n    augment=True\n\n    # Class Labels for BirdCLEF 25\n    class_names = sorted(os.listdir('/kaggle/input/competitions/birdclef-2026/train_audio'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:08.887437Z","iopub.execute_input":"2026-04-16T17:16:08.887807Z","iopub.status.idle":"2026-04-16T17:16:08.902948Z","shell.execute_reply.started":"2026-04-16T17:16:08.887760Z","shell.execute_reply":"2026-04-16T17:16:08.902185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reproducibility\nSets value for random seed to produce similar result in each run.","metadata":{}},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:08.904617Z","iopub.execute_input":"2026-04-16T17:16:08.905040Z","iopub.status.idle":"2026-04-16T17:16:08.917553Z","shell.execute_reply.started":"2026-04-16T17:16:08.905014Z","shell.execute_reply":"2026-04-16T17:16:08.916819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data ","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/competitions/birdclef-2026\"\n\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['filepath'] = BASE_PATH + '/train_audio/' + df.filename\ndf['target'] = df.primary_label.map(CFG.name2label)\ndf['filename'] = df.filepath.map(lambda x: x.split('/')[-1])\ndf['xc_id'] = df.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# Display rwos\ndf.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:08.918561Z","iopub.execute_input":"2026-04-16T17:16:08.918940Z","iopub.status.idle":"2026-04-16T17:16:09.166481Z","shell.execute_reply.started":"2026-04-16T17:16:08.918916Z","shell.execute_reply":"2026-04-16T17:16:09.165731Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Split\nFollowing code will split the data into folds using target stratification.\nNote: Some classess have too few samples thus not each fold contains all the classes.","metadata":{}},{"cell_type":"code","source":"# Import required packages\nfrom sklearn.model_selection import train_test_split\n\ntrain_df, valid_df = train_test_split(df, test_size=0.2)\n\nprint(f\"Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:09.167685Z","iopub.execute_input":"2026-04-16T17:16:09.168125Z","iopub.status.idle":"2026-04-16T17:16:09.195475Z","shell.execute_reply.started":"2026-04-16T17:16:09.168094Z","shell.execute_reply":"2026-04-16T17:16:09.194649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Decoders\n\nThe following code will decode the raw audio from .ogg file and also decode the spectrogram from the audio file. Additionally, we will apply Z-Score standardization and Min-Max normalization to ensure consistent inputs to the model.","metadata":{}},{"cell_type":"code","source":"def get_audio(path):\n    def load_audio(p):\n        audio, _ = librosa.load(p.decode(), sr=CFG.sample_rate, mono=True)\n        return audio.astype(\"float32\")\n    \n    audio = tf.numpy_function(load_audio, [path], tf.float32)\n    audio.set_shape([None])  # important\n    return audio\n\n# Decodes Audio\ndef build_decoder(with_labels=True, dim=1024):\n    def crop_or_pad(audio, target_len, pad_mode=\"constant\"):\n        audio_len = tf.shape(audio)[0]\n        diff_len = abs(\n            target_len - audio_len\n        )  # find difference between target and audio length\n        if audio_len < target_len:  # do padding if audio length is shorter\n            pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            pad2 = diff_len - pad1\n            audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=pad_mode)\n        elif audio_len > target_len:  # do cropping if audio length is larger\n            idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            audio = audio[idx : (idx + target_len)]\n        return tf.reshape(audio, [target_len])\n\n    def apply_preproc(spec):\n        # Standardize\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize using Min-Max\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(\n            tf.math.equal(max_val - min_val, 0),\n            spec - min_val,\n            (spec - min_val) / (max_val - min_val),\n        )\n        return spec\n\n    def get_target(target):\n        target = tf.reshape(target, [1])\n        target = tf.cast(tf.one_hot(target, CFG.num_classes), tf.float32)\n        target = tf.reshape(target, [CFG.num_classes])\n        return target\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Crop or pad audio to keep a fixed length\n        audio = crop_or_pad(audio, dim)\n        # Audio to Spectrogram\n        spec = keras.layers.MelSpectrogram(\n            num_mel_bins=CFG.img_size[0],\n            fft_length=CFG.nfft,\n            sequence_stride=CFG.hop_length,\n            sampling_rate=CFG.sample_rate,\n        )(audio)\n        # Apply normalization and standardization\n        spec = apply_preproc(spec)\n        # Spectrogram to 3 channel image (for imagenet)\n        spec = tf.tile(spec[..., None], [1, 1, 3])\n        spec = tf.reshape(spec, [*CFG.img_size, 3])\n        return spec\n\n    def decode_with_labels(path, label):\n        label = get_target(label)\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:09.196544Z","iopub.execute_input":"2026-04-16T17:16:09.196902Z","iopub.status.idle":"2026-04-16T17:16:09.209268Z","shell.execute_reply.started":"2026-04-16T17:16:09.196865Z","shell.execute_reply":"2026-04-16T17:16:09.208125Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Augmenters\n\nFollowing code will apply augmentations to spectrogram data. In this notebook, we will use MixUp, CutOut (TimeMasking and FreqMasking) from KerasCV.\nNote that, these augmentations will be applied to batch of spectrograms rather than single spectrograms.","metadata":{}},{"cell_type":"code","source":"def build_augmenter():\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=0.4),\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.12)), # time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # freq-masking\n    ]\n    \n    def augment(img, label):\n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:09.211736Z","iopub.execute_input":"2026-04-16T17:16:09.212103Z","iopub.status.idle":"2026-04-16T17:16:09.225610Z","shell.execute_reply.started":"2026-04-16T17:16:09.212076Z","shell.execute_reply":"2026-04-16T17:16:09.224846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Pipeline\n\nFollowing code builds the complete pipeline of the data flow. It uses tf.data.Dataset for data processing. Here are some cool features of tf.data,\n\n- We can build complex input pipelines from simple, reusable pieces usingtf.data API . For example, the pipeline for an audio model might aggregate data from files in a distributed file system, apply random transformation/augmentation to each audio/spectrogram, and merge randomly selected data into a batch for training.\n- Moreover tf.data API provides a tf.data.Dataset feature that represents a sequence of components where each component comprises one or more pieces. For instance, in an audio pipeline, a component might be a single training example, with a pair of tensor pieces representing the audio and its label.\n\nCheck out this [doc](http://www.tensorflow.org/guide/data) if you want to learn more about tf.data.","metadata":{}},{"cell_type":"code","source":"def build_dataset(paths, labels=None, batch_size=32, \n                  decode_fn=None, augment_fn=None, cache=True,\n                  augment=False, shuffle=2048):\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None, dim=CFG.audio_len)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter()\n        \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths,) if labels is None else (paths, labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache() if cache else ds\n    if shuffle:\n        opt = tf.data.Options()\n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=True)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:09.226603Z","iopub.execute_input":"2026-04-16T17:16:09.226951Z","iopub.status.idle":"2026-04-16T17:16:09.239680Z","shell.execute_reply.started":"2026-04-16T17:16:09.226926Z","shell.execute_reply":"2026-04-16T17:16:09.238715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build Train and Valid Dataloaders","metadata":{}},{"cell_type":"code","source":"# Train\ntrain_paths = train_df.filepath.values\ntrain_labels = train_df.target.values\ntrain_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n                         shuffle=True, augment=CFG.augment)\n\n# Valid\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.target.values\nvalid_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n                         shuffle=False, augment=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:09.241260Z","iopub.execute_input":"2026-04-16T17:16:09.241510Z","iopub.status.idle":"2026-04-16T17:16:10.458572Z","shell.execute_reply.started":"2026-04-16T17:16:09.241485Z","shell.execute_reply":"2026-04-16T17:16:10.457894Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling \n\nBuilding a model for an audio recognition task with spectrograms as input is quite straightforward, as it is very similar to image classification. This is because the shape of spectrogram data is very similar to image data. In this notebook, to perform the audio recognition task, we will utilize the EfficientNetV2 ImageNet-pretrained model as the backbone. Even though this backbone is pretrained with ImageNet data instead of spectrogram data, we can leverage transfer learning to adapt it to our spectrogram-based task.\n\nNote that we can train our model on any duration audio file (here we are using 10 seconds), but we will always infer on 5-second audio files (as per competition rules). To facilitate this, we have set the model input shape to (None, None, 3), which will allow us to have variable-length input during training and inference.\n\nIn case you are wondering, Why not train and infer on both 5-second? In the train data, we have long audio files, but we are not sure which part of the audio contains the labeled bird's song. In other words, this is weakly labeled. To ensure the provided label is accurately suited to the audio, we are using a larger audio size than 5 seconds. You are welcome to try out different audio lengths for training.","metadata":{}},{"cell_type":"code","source":"# Create an input layer for the model\ninp = keras.layers.Input(shape=(None, None, 3))\n# Pretrained backbone\nbackbone = keras_cv.models.EfficientNetV2Backbone.from_preset(\n    CFG.preset,\n)\nout = keras_cv.models.ImageClassifier(\n    backbone=backbone,\n    num_classes=CFG.num_classes,\n    name=\"classifier\"\n)(inp)\n# Build model\nmodel = keras.models.Model(inputs=inp, outputs=out)\n# Compile model with optimizer, loss and metrics\nmodel.compile(optimizer=\"adam\",\n              loss=keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n              metrics=[keras.metrics.AUC(name='auc')],\n             )\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:10.459410Z","iopub.execute_input":"2026-04-16T17:16:10.459660Z","iopub.status.idle":"2026-04-16T17:16:38.437832Z","shell.execute_reply.started":"2026-04-16T17:16:10.459621Z","shell.execute_reply":"2026-04-16T17:16:38.437115Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LR Schedule \n\nLearning Rate scheduler for transfer learning. The learning rate starts from lr_start, then decreases to alr_min using different methods namely,\n- step: Reduce lr step wise like stair.\n- cos: Follow Cosine graph to reduce lr.\n- exp: Reduce lr exponentially.","metadata":{}},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=8, mode='cos', epochs=10, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 8e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback\n\nlr_cb = get_lr_callback(CFG.batch_size, plot=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:38.438769Z","iopub.execute_input":"2026-04-16T17:16:38.439365Z","iopub.status.idle":"2026-04-16T17:16:38.447589Z","shell.execute_reply.started":"2026-04-16T17:16:38.439318Z","shell.execute_reply":"2026-04-16T17:16:38.446835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Checkpoint","metadata":{}},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.weights.h5\",\n                                         monitor='val_auc',\n                                         save_best_only=True,\n                                         save_weights_only=True,\n                                         mode='max')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:38.448796Z","iopub.execute_input":"2026-04-16T17:16:38.449172Z","iopub.status.idle":"2026-04-16T17:16:38.465750Z","shell.execute_reply.started":"2026-04-16T17:16:38.449132Z","shell.execute_reply":"2026-04-16T17:16:38.464631Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    validation_data=valid_ds, \n    epochs=CFG.epochs,\n    callbacks=[lr_cb, ckpt_cb], \n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T17:16:38.467068Z","iopub.execute_input":"2026-04-16T17:16:38.467964Z","execution_failed":"2026-04-16T17:43:52.782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Result Summary","metadata":{}},{"cell_type":"code","source":"best_epoch = np.argmax(history.history[\"val_auc\"])\nbest_score = history.history[\"val_auc\"][best_epoch]\nprint('>>> Best AUC: ', best_score)\nprint('>>> Best Epoch: ', best_epoch+1)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-04-16T17:43:52.783Z"}},"outputs":[],"execution_count":null}]}