{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceType":"competition","sourceId":25954,"databundleVersionId":2091745},{"sourceType":"competition","sourceId":33246,"databundleVersionId":3221581},{"sourceType":"competition","sourceId":44224,"databundleVersionId":5188730},{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821},{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945},{"sourceType":"modelInstanceVersion","sourceId":6130,"databundleVersionId":7429420,"modelInstanceId":4598,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6128,"databundleVersionId":7429417,"modelInstanceId":4596,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6120,"databundleVersionId":7429401,"modelInstanceId":4603,"modelId":2797},{"sourceType":"kernelVersion","sourceId":311807047},{"sourceType":"kernelVersion","sourceId":315847830}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Fine-tune on 2026 training data","metadata":{}},{"cell_type":"markdown","source":"After acquiring general bird acoustic knowledge, the model was fine-tuned using the BirdCLEF 2026 training data. This stage aims to adapt the model to the specific species distribution and recording characteristics of the 2026 challenge. The training data consists of the labeled recordings provided in train.csv together with the available soundscape annotations.\n","metadata":{}},{"cell_type":"code","source":"!pip install librosa","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:00:17.566085Z","iopub.execute_input":"2026-05-06T19:00:17.566238Z","iopub.status.idle":"2026-05-06T19:00:27.730633Z","shell.execute_reply.started":"2026-05-06T19:00:17.566220Z","shell.execute_reply":"2026-05-06T19:00:27.729614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"jax\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nimport gc\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nfrom sklearn.model_selection import train_test_split\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:00:27.731161Z","iopub.execute_input":"2026-05-06T19:00:27.731333Z","iopub.status.idle":"2026-05-06T19:01:12.058131Z","shell.execute_reply.started":"2026-05-06T19:00:27.731312Z","shell.execute_reply":"2026-05-06T19:01:12.057120Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 5 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10\n    preset = 'efficientnetv2_b0'\n    \n    # Data augmentation parameters\n    augment=True\n\n    # Class Labels for BirdCLEF 25\n    class_names = sorted(os.listdir('/kaggle/input/competitions/birdclef-2026/train_audio'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.063478Z","iopub.execute_input":"2026-05-06T19:01:12.063656Z","iopub.status.idle":"2026-05-06T19:01:12.087940Z","shell.execute_reply.started":"2026-05-06T19:01:12.063641Z","shell.execute_reply":"2026-05-06T19:01:12.087109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.088477Z","iopub.execute_input":"2026-05-06T19:01:12.088649Z","iopub.status.idle":"2026-05-06T19:01:12.091623Z","shell.execute_reply.started":"2026-05-06T19:01:12.088634Z","shell.execute_reply":"2026-05-06T19:01:12.090891Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Load Dataset BirdCLEF 2026\n\nBASE_PATH = \"/kaggle/input/competitions/birdclef-2026\"\n\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['filepath'] = BASE_PATH + '/train_audio/' + df.filename\ndf['target'] = df.primary_label.map(CFG.name2label)\ndf['filename'] = df.filepath.map(lambda x: x.split('/')[-1])\ndf['xc_id'] = df.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# Display rwos\ndf.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.092141Z","iopub.execute_input":"2026-05-06T19:01:12.092302Z","iopub.status.idle":"2026-05-06T19:01:12.370850Z","shell.execute_reply.started":"2026-05-06T19:01:12.092288Z","shell.execute_reply":"2026-05-06T19:01:12.369911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Split\nFollowing code will split the data into folds using target stratification.\nNote: Some classess have too few samples thus not each fold contains all the classes.","metadata":{}},{"cell_type":"code","source":"# Split training data and validation data \n\ntrain_df, valid_df = train_test_split(df, test_size=0.2)\n\nprint(f\"Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.371431Z","iopub.execute_input":"2026-05-06T19:01:12.371615Z","iopub.status.idle":"2026-05-06T19:01:12.394793Z","shell.execute_reply.started":"2026-05-06T19:01:12.371582Z","shell.execute_reply":"2026-05-06T19:01:12.393944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Decoders\n\nThe following code will decode the raw audio from .ogg file and also decode the spectrogram from the audio file. Additionally, we will apply Z-Score standardization (centering to mean 0, standard deviation 1) and Min-Max normalization (scaling to a 0–1 range) to ensure consistent inputs to the model.","metadata":{}},{"cell_type":"code","source":"def get_audio(path):\n    def load_audio(p):\n        audio, _ = librosa.load(p.decode(), sr=CFG.sample_rate, mono=True)\n        return audio.astype(\"float32\")\n    \n    audio = tf.numpy_function(load_audio, [path], tf.float32)\n    audio.set_shape([None])  # important\n    return audio\n\n# Decodes Audio\ndef build_decoder(with_labels=True, dim=1024):\n    def crop_or_pad(audio, target_len, pad_mode=\"constant\"):\n        audio_len = tf.shape(audio)[0]\n        diff_len = abs(\n            target_len - audio_len\n        )  # find difference between target and audio length\n        if audio_len < target_len:  # do padding if audio length is shorter\n            pad1 = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            pad2 = diff_len - pad1\n            audio = tf.pad(audio, paddings=[[pad1, pad2]], mode=pad_mode)\n        elif audio_len > target_len:  # do cropping if audio length is larger\n            idx = tf.random.uniform([], maxval=diff_len, dtype=tf.int32)\n            audio = audio[idx : (idx + target_len)]\n        return tf.reshape(audio, [target_len])\n\n    def apply_preproc(spec):\n        # Standardize\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize using Min-Max\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(\n            tf.math.equal(max_val - min_val, 0),\n            spec - min_val,\n            (spec - min_val) / (max_val - min_val),\n        )\n        return spec\n\n    def get_target(target):\n        target = tf.reshape(target, [1])\n        target = tf.cast(tf.one_hot(target, CFG.num_classes), tf.float32)\n        target = tf.reshape(target, [CFG.num_classes])\n        return target\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Crop or pad audio to keep a fixed length\n        audio = crop_or_pad(audio, dim)\n        # Audio to Spectrogram\n        spec = keras.layers.MelSpectrogram(\n            num_mel_bins=CFG.img_size[0],\n            fft_length=CFG.nfft,\n            sequence_stride=CFG.hop_length,\n            sampling_rate=CFG.sample_rate,\n        )(audio)\n        # Apply normalization and standardization\n        spec = apply_preproc(spec)\n        # Spectrogram to 3 channel image (for imagenet)\n        spec = tf.tile(spec[..., None], [1, 1, 3])\n        spec = tf.reshape(spec, [*CFG.img_size, 3])\n        return spec\n\n    def decode_with_labels(path, label):\n        label = get_target(label)\n        return decode(path), label\n\n    return decode_with_labels if with_labels else decode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.395430Z","iopub.execute_input":"2026-05-06T19:01:12.395607Z","iopub.status.idle":"2026-05-06T19:01:12.403030Z","shell.execute_reply.started":"2026-05-06T19:01:12.395579Z","shell.execute_reply":"2026-05-06T19:01:12.402192Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Augmenters\n\nFollowing code will apply augmentations to spectrogram data. In this notebook, we will use MixUp, CutOut (TimeMasking and FreqMasking) from KerasCV.\nNote that, these augmentations will be applied to batch of spectrograms rather than single spectrograms.","metadata":{}},{"cell_type":"code","source":"def build_augmenter():\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=0.4),\n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.12)), # time-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # freq-masking\n    ]\n    \n    def augment(img, label):\n        data = {\"images\":img, \"labels\":label}\n        for augmenter in augmenters:\n            if tf.random.uniform([]) < 0.35:\n                data = augmenter(data, training=True)\n        return data[\"images\"], data[\"labels\"]\n    \n    return augment","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.403585Z","iopub.execute_input":"2026-05-06T19:01:12.403771Z","iopub.status.idle":"2026-05-06T19:01:12.417436Z","shell.execute_reply.started":"2026-05-06T19:01:12.403755Z","shell.execute_reply":"2026-05-06T19:01:12.416684Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Pipeline\n\nFollowing code builds the complete pipeline of the data flow. It uses tf.data.Dataset for data processing.","metadata":{}},{"cell_type":"code","source":"def build_dataset(paths, labels=None, batch_size=32, \n                  decode_fn=None, augment_fn=None, cache=False,\n                  augment=False, shuffle=2048):\n\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None, dim=CFG.audio_len)\n\n    if augment_fn is None:\n        augment_fn = build_augmenter()\n        \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths,) if labels is None else (paths, labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache() if cache else ds\n    if shuffle:\n        opt = tf.data.Options()\n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=True)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.417910Z","iopub.execute_input":"2026-05-06T19:01:12.418062Z","iopub.status.idle":"2026-05-06T19:01:12.426273Z","shell.execute_reply.started":"2026-05-06T19:01:12.418048Z","shell.execute_reply":"2026-05-06T19:01:12.425504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build Train and Valid Dataloaders","metadata":{}},{"cell_type":"code","source":"# Building Dataset for BirdCLEF 2026\n\n# Train\ntrain_paths = train_df.filepath.values\ntrain_labels = train_df.target.values\ntrain_ds = build_dataset(train_paths, train_labels, batch_size=CFG.batch_size,\n                         shuffle=True, augment=CFG.augment)\n\n# Valid\nvalid_paths = valid_df.filepath.values\nvalid_labels = valid_df.target.values\nvalid_ds = build_dataset(valid_paths, valid_labels, batch_size=CFG.batch_size,\n                         shuffle=False, augment=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:12.426795Z","iopub.execute_input":"2026-05-06T19:01:12.426958Z","iopub.status.idle":"2026-05-06T19:01:20.332538Z","shell.execute_reply.started":"2026-05-06T19:01:12.426943Z","shell.execute_reply":"2026-05-06T19:01:20.331300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"def build_model():\n\n    # Create an input layer for the model\n    inp = keras.layers.Input(shape=(None, None, 3))\n    # Pretrained backbone\n    backbone = keras_cv.models.EfficientNetV2Backbone.from_preset(\n        CFG.preset,\n    )\n    out = keras_cv.models.ImageClassifier(\n        backbone=backbone,\n        num_classes=CFG.num_classes,\n        name=\"classifier\"\n    )(inp)\n    # Build model\n    model = keras.models.Model(inputs=inp, outputs=out)\n    # Compile model with optimizer, loss and metrics\n    model.compile(optimizer=\"adam\",\n                  loss=keras.losses.CategoricalCrossentropy(label_smoothing=0.02),\n                  metrics=[keras.metrics.AUC(name='auc')],\n                 )\n    model.summary()\n    return model ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:20.333173Z","iopub.execute_input":"2026-05-06T19:01:20.333343Z","iopub.status.idle":"2026-05-06T19:01:20.337235Z","shell.execute_reply.started":"2026-05-06T19:01:20.333326Z","shell.execute_reply":"2026-05-06T19:01:20.336467Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LR Schedule \n\nLearning Rate scheduler for transfer learning. The learning rate starts from lr_start, then decreases to alr_min using different methods namely,\n- step: Reduce lr step wise like stair.\n- cos: Follow Cosine graph to reduce lr.\n- exp: Reduce lr exponentially.","metadata":{}},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=8, mode='cos', epochs=10, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 8e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback\n\nlr_cb = get_lr_callback(CFG.batch_size, plot=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:20.337797Z","iopub.execute_input":"2026-05-06T19:01:20.337949Z","iopub.status.idle":"2026-05-06T19:01:20.356950Z","shell.execute_reply.started":"2026-05-06T19:01:20.337935Z","shell.execute_reply":"2026-05-06T19:01:20.356076Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Checkpoint","metadata":{}},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.weights.h5\",\n                                         monitor='val_auc',\n                                         save_best_only=True,\n                                         save_weights_only=True,\n                                         mode='max')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:20.357403Z","iopub.execute_input":"2026-05-06T19:01:20.357552Z","iopub.status.idle":"2026-05-06T19:01:20.370497Z","shell.execute_reply.started":"2026-05-06T19:01:20.357537Z","shell.execute_reply":"2026-05-06T19:01:20.369624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"def train_model(train_ds, valid_ds, weights_path=None, epochs=CFG.epochs):\n    model = build_model()\n    \n    if weights_path is not None:\n        model.load_weights(weights_path)\n\n    history = model.fit(\n        train_ds,\n        validation_data=valid_ds,\n        epochs=epochs,\n        callbacks=[lr_cb, ckpt_cb],\n        verbose=1\n    )\n    \n    return model, history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:20.370975Z","iopub.execute_input":"2026-05-06T19:01:20.371126Z","iopub.status.idle":"2026-05-06T19:01:20.380401Z","shell.execute_reply.started":"2026-05-06T19:01:20.371113Z","shell.execute_reply":"2026-05-06T19:01:20.379626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Stage 2: Fine-tune on 2026 training data","metadata":{}},{"cell_type":"code","source":"# Clear memory\ngc.collect()\n\ntry:\n    # Try TPU\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"Running on TPU\")\n\nexcept ValueError:\n    # Fallback\n    strategy = tf.distribute.get_strategy()\n    print(\"Running on CPU/GPU\")\n\n# Use the strategy (TPU or not)\nwith strategy.scope():\n    model_stage2, history = train_model(\n        train_ds,\n        valid_ds,\n        weights_path=\"/kaggle/input/notebooks/martinaolivato/stage-1-multi-stage-pretraining-kerascv/stage1.weights.h5\",\n        epochs=CFG.epochs\n    )\n# Load BEST weights\nmodel_stage2.load_weights(\"best_model.weights.h5\")\n\n# Save for next stage\nmodel_stage2.save_weights(\"/kaggle/working/stage2.weights.h5\")\n\n# Save the model \nmodel_stage2.save(\"/kaggle/working/stage2_model.keras\")\n\n# Save metadata\nnp.save(\"/kaggle/working/class_names2.npy\", CFG.class_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:01:20.380874Z","iopub.execute_input":"2026-05-06T19:01:20.381023Z","iopub.status.idle":"2026-05-06T19:15:30.422760Z","shell.execute_reply.started":"2026-05-06T19:01:20.381009Z","shell.execute_reply":"2026-05-06T19:15:30.421458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Result Summary","metadata":{}},{"cell_type":"code","source":"best_epoch = np.argmax(history.history[\"val_auc\"])\nbest_score = history.history[\"val_auc\"][best_epoch]\nprint('>>> Best AUC: ', best_score)\nprint('>>> Best Epoch: ', best_epoch+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-06T19:15:30.423239Z","iopub.execute_input":"2026-05-06T19:15:30.423426Z","iopub.status.idle":"2026-05-06T19:15:30.427508Z","shell.execute_reply.started":"2026-05-06T19:15:30.423408Z","shell.execute_reply":"2026-05-06T19:15:30.426573Z"}},"outputs":[],"execution_count":null}]}