{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceType":"competition","sourceId":25954,"databundleVersionId":2091745},{"sourceType":"competition","sourceId":33246,"databundleVersionId":3221581},{"sourceType":"competition","sourceId":44224,"databundleVersionId":5188730},{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821},{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945},{"sourceType":"modelInstanceVersion","sourceId":6120,"databundleVersionId":7429401,"modelInstanceId":4603,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6130,"databundleVersionId":7429420,"modelInstanceId":4598,"modelId":2797},{"sourceType":"modelInstanceVersion","sourceId":6128,"databundleVersionId":7429417,"modelInstanceId":4596,"modelId":2797},{"sourceType":"kernelVersion","sourceId":317192684}],"dockerImageVersionId":31402,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Weakly Supervised SED Training","metadata":{}},{"cell_type":"markdown","source":"A Sound Event Detection (SED) approach attempts to preserve temporal information and identify bird activity within smaller regions of the spectrogram before producing a final clip-level prediction. The idea is to introduce a SED head on top of the EfficientNet backbone to better capture localized bird vocalizations within spectrograms.\n","metadata":{}},{"cell_type":"code","source":"!pip install librosa","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:01.406598Z","iopub.execute_input":"2026-06-01T18:17:01.407454Z","iopub.status.idle":"2026-06-01T18:17:05.585044Z","shell.execute_reply.started":"2026-06-01T18:17:01.407420Z","shell.execute_reply":"2026-06-01T18:17:05.584018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom tensorflow.keras import layers\n\nimport numpy as np \nimport pandas as pd\n\nimport gc\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nfrom sklearn.model_selection import train_test_split\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:05.586700Z","iopub.execute_input":"2026-06-01T18:17:05.587274Z","iopub.status.idle":"2026-06-01T18:17:25.267321Z","shell.execute_reply.started":"2026-06-01T18:17:05.587243Z","shell.execute_reply":"2026-06-01T18:17:25.266687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 64\n    \n    # Audio duration, sample rate, and length\n    duration = 5 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    epochs = 10\n    preset = 'efficientnetv2_b0'\n    \n    # Data augmentation parameters\n    augment=True\n\n    # Class Labels for BirdCLEF 25\n    class_names = sorted(os.listdir('/kaggle/input/competitions/birdclef-2026/train_audio'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}\n    input_shape = (128, 384, 3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.268186Z","iopub.execute_input":"2026-06-01T18:17:25.268799Z","iopub.status.idle":"2026-06-01T18:17:25.281178Z","shell.execute_reply.started":"2026-06-01T18:17:25.268771Z","shell.execute_reply":"2026-06-01T18:17:25.280586Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.283010Z","iopub.execute_input":"2026-06-01T18:17:25.283306Z","iopub.status.idle":"2026-06-01T18:17:25.298985Z","shell.execute_reply.started":"2026-06-01T18:17:25.283284Z","shell.execute_reply":"2026-06-01T18:17:25.298171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data ","metadata":{}},{"cell_type":"code","source":"# Load Dataset BirdCLEF 2026\n\nBASE_PATH = \"/kaggle/input/competitions/birdclef-2026\"\n\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['filepath'] = BASE_PATH + '/train_audio/' + df.filename\ndf['target'] = df.primary_label.map(CFG.name2label)\ndf['filename'] = df.filepath.map(lambda x: x.split('/')[-1])\ndf['xc_id'] = df.filepath.map(lambda x: x.split('/')[-1].split('.')[0])\n\n# Display rwos\ndf.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.299830Z","iopub.execute_input":"2026-06-01T18:17:25.300295Z","iopub.status.idle":"2026-06-01T18:17:25.608314Z","shell.execute_reply.started":"2026-06-01T18:17:25.300256Z","shell.execute_reply":"2026-06-01T18:17:25.607588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Split\nFollowing code will split the data into folds using target stratification.","metadata":{}},{"cell_type":"code","source":"# Count samples per class\nclass_counts = df[\"target\"].value_counts()\n\n# Keep only classes with >= 2 samples\nvalid_classes = class_counts[class_counts >= 2].index\n\n# Filter dataframe\ndf = df[df[\"target\"].isin(valid_classes)]\n\nprint(\"Remaining samples:\", len(df))\nprint(\"Remaining classes:\", df[\"target\"].nunique())\n\ntrain_df, valid_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df[\"target\"],\n    random_state=CFG.seed\n)\n\nprint(len(train_df), len(valid_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.609241Z","iopub.execute_input":"2026-06-01T18:17:25.609626Z","iopub.status.idle":"2026-06-01T18:17:25.658607Z","shell.execute_reply.started":"2026-06-01T18:17:25.609587Z","shell.execute_reply":"2026-06-01T18:17:25.657985Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Audio Loader","metadata":{}},{"cell_type":"code","source":"def load_audio(path):\n\n    audio, _ = librosa.load(\n        path,\n        sr=CFG.sample_rate,\n        mono=True\n    )\n\n    if len(audio) < CFG.audio_len:\n\n        pad = CFG.audio_len - len(audio)\n\n        audio = np.pad(audio, (0, pad))\n\n    else:\n\n        audio = audio[:CFG.audio_len]\n\n    return audio.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.659438Z","iopub.execute_input":"2026-06-01T18:17:25.659677Z","iopub.status.idle":"2026-06-01T18:17:25.664424Z","shell.execute_reply.started":"2026-06-01T18:17:25.659656Z","shell.execute_reply":"2026-06-01T18:17:25.663563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Spectrogram Function","metadata":{}},{"cell_type":"code","source":"mel_layer = keras.layers.MelSpectrogram(\n    num_mel_bins=CFG.img_size[0],\n    fft_length=CFG.nfft,\n    sequence_stride=CFG.hop_length,\n    sampling_rate=CFG.sample_rate,\n)\n\ndef create_spectrogram(audio):\n\n    # STFT\n    spec = tf.signal.stft(\n        audio,\n        frame_length=1024,\n        frame_step=512,\n        fft_length=1024\n    )\n\n    # Magnitude\n    spec = tf.abs(spec)\n\n    # Log transform (VERY IMPORTANT)\n    spec = tf.math.log(spec + 1e-6)\n\n    # Add channel dimension\n    spec = tf.expand_dims(spec, axis=-1)\n\n    # Resize\n    spec = tf.image.resize(\n        spec,\n        (128, 384)\n    )\n\n    # Convert to RGB\n    spec = tf.image.grayscale_to_rgb(spec)\n\n    return spec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.665395Z","iopub.execute_input":"2026-06-01T18:17:25.666147Z","iopub.status.idle":"2026-06-01T18:17:25.682535Z","shell.execute_reply.started":"2026-06-01T18:17:25.666092Z","shell.execute_reply":"2026-06-01T18:17:25.681767Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Decoders","metadata":{}},{"cell_type":"code","source":"def decode_audio(path, label):\n\n    def _load(path_):\n\n        audio = load_audio(path_.decode())\n\n        return audio\n\n    audio = tf.numpy_function(\n        _load,\n        [path],\n        tf.float32\n    )\n\n    audio.set_shape([CFG.audio_len])\n\n    # Prevent exploding audio\n    audio = tf.clip_by_value(audio, -1.0, 1.0)\n\n    spec = create_spectrogram(audio)\n\n    # Remove NaNs/Infs\n    spec = tf.where(tf.math.is_nan(spec), 0.0, spec)\n    spec = tf.where(tf.math.is_inf(spec), 0.0, spec)\n\n    # Normalize safely\n    spec = tf.clip_by_value(spec, -10.0, 10.0)\n\n    label = tf.one_hot(\n        label,\n        CFG.num_classes\n    )\n\n    return spec, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.683401Z","iopub.execute_input":"2026-06-01T18:17:25.683589Z","iopub.status.idle":"2026-06-01T18:17:25.690800Z","shell.execute_reply.started":"2026-06-01T18:17:25.683573Z","shell.execute_reply":"2026-06-01T18:17:25.690137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Pipeline\n\nFollowing code builds the complete pipeline of the data flow. It uses tf.data.Dataset for data processing.","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.AUTOTUNE\n\ndef build_dataset(paths, labels, shuffle=True):\n\n    ds = tf.data.Dataset.from_tensor_slices(\n        (paths, labels)\n    )\n\n    ds = ds.map(\n        decode_audio,\n        num_parallel_calls=AUTO\n    )\n\n    if shuffle:\n\n        ds = ds.shuffle(\n            2048,\n            seed=CFG.seed\n        )\n\n    ds = ds.batch(CFG.batch_size)\n\n    ds = ds.prefetch(AUTO)\n\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.693622Z","iopub.execute_input":"2026-06-01T18:17:25.693837Z","iopub.status.idle":"2026-06-01T18:17:25.703341Z","shell.execute_reply.started":"2026-06-01T18:17:25.693817Z","shell.execute_reply":"2026-06-01T18:17:25.702624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build Train and Valid Dataloaders","metadata":{}},{"cell_type":"code","source":"train_paths = train_df[\"filepath\"].values\ntrain_labels = train_df[\"target\"].values\n\nvalid_paths = valid_df[\"filepath\"].values\nvalid_labels = valid_df[\"target\"].values\n\ntrain_ds = (\n    tf.data.Dataset.from_tensor_slices(\n        (train_paths, train_labels)\n    )\n    .map(decode_audio, num_parallel_calls=tf.data.AUTOTUNE)\n    .shuffle(1024)\n    .batch(CFG.batch_size)\n    .prefetch(tf.data.AUTOTUNE)\n)\n\nvalid_ds = (\n    tf.data.Dataset.from_tensor_slices(\n        (valid_paths, valid_labels)\n    )\n    .map(decode_audio, num_parallel_calls=tf.data.AUTOTUNE)\n    .batch(CFG.batch_size)\n    .prefetch(tf.data.AUTOTUNE)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:25.704106Z","iopub.execute_input":"2026-06-01T18:17:25.704446Z","iopub.status.idle":"2026-06-01T18:17:27.711959Z","shell.execute_reply.started":"2026-06-01T18:17:25.704412Z","shell.execute_reply":"2026-06-01T18:17:27.711471Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Build SED Model ","metadata":{}},{"cell_type":"code","source":"def build_sed_model():\n\n    inputs = keras.Input(shape= CFG.input_shape)\n\n    # Backbone\n    backbone = keras.applications.EfficientNetB0(\n    include_top=False,\n    weights=None,\n    input_tensor=inputs)\n\n    x = backbone.output\n    # Feature map (shape: (batch, time, channels))\n    x = layers.Conv2D(\n        512,\n        3,\n        padding=\"same\",\n        activation=\"relu\"\n    )(x)\n\n    framewise = layers.Conv2D(\n        CFG.num_classes,\n        1,\n        activation=\"sigmoid\"\n    )(x)\n\n    clipwise = layers.GlobalAveragePooling2D()(framewise)\n\n    model = keras.Model(inputs, clipwise)\n    return model\n\n# Build Model\n\nmodel = build_sed_model()\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:27.712748Z","iopub.execute_input":"2026-06-01T18:17:27.712942Z","iopub.status.idle":"2026-06-01T18:17:29.900926Z","shell.execute_reply.started":"2026-06-01T18:17:27.712922Z","shell.execute_reply":"2026-06-01T18:17:29.900339Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"# Load Stage 3 Weights\n\nWEIGHTS_PATH = \"/kaggle/input/notebooks/martinaolivato/stage-3-multi-stage-pretraining-kerascv/stage3.weights.h5\"\n\nmodel.load_weights(\n    WEIGHTS_PATH,\n    skip_mismatch=True\n)\n\nprint(\"Stage 3 weights loaded.\")\n\n# Compile\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(1e-4),\n    loss=\"binary_crossentropy\",\n    metrics=[keras.metrics.AUC(name=\"AUC\")]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:29.901855Z","iopub.execute_input":"2026-06-01T18:17:29.902186Z","iopub.status.idle":"2026-06-01T18:17:30.240462Z","shell.execute_reply.started":"2026-06-01T18:17:29.902161Z","shell.execute_reply":"2026-06-01T18:17:30.239824Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Callbacks","metadata":{}},{"cell_type":"code","source":"callbacks = [\n\n    keras.callbacks.ModelCheckpoint(\n        \"stage4_best.weights.h5\",\n        monitor=\"val_auc\",\n        mode=\"max\",\n        save_best_only=True,\n        save_weights_only=True\n    ),\n\n    keras.callbacks.EarlyStopping(\n        monitor=\"val_auc\",\n        mode=\"max\",\n        patience=5,\n        restore_best_weights=True\n    ),\n\n    keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_auc\",\n        mode=\"max\",\n        factor=0.5,\n        patience=2\n    )\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:30.241308Z","iopub.execute_input":"2026-06-01T18:17:30.241618Z","iopub.status.idle":"2026-06-01T18:17:30.246865Z","shell.execute_reply.started":"2026-06-01T18:17:30.241596Z","shell.execute_reply":"2026-06-01T18:17:30.245859Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"# Clear memory\ngc.collect()\n\nhistory = model.fit(\n    train_ds,\n    validation_data=valid_ds,\n    epochs=CFG.epochs,\n    callbacks=callbacks\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T18:17:30.247981Z","iopub.execute_input":"2026-06-01T18:17:30.248295Z","iopub.status.idle":"2026-06-01T20:19:57.220567Z","shell.execute_reply.started":"2026-06-01T18:17:30.248264Z","shell.execute_reply":"2026-06-01T20:19:57.219589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Best wieghts\n\nmodel.load_weights(\"/kaggle/working/stage4_best.weights.h5\")\n\nmodel.save_weights(\"/kaggle/working/stage4_sed.weights.h5\")\n\nmodel.save(\"/kaggle/working/stage4_sed_model.keras\")\n\nprint(\"SED model saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T20:19:57.221782Z","iopub.execute_input":"2026-06-01T20:19:57.222077Z","iopub.status.idle":"2026-06-01T20:20:00.184265Z","shell.execute_reply.started":"2026-06-01T20:19:57.222046Z","shell.execute_reply":"2026-06-01T20:20:00.183476Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Inference on Soundscapes","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(valid_ds)\n\nprint(predictions.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-01T20:20:00.185236Z","iopub.execute_input":"2026-06-01T20:20:00.185593Z","execution_failed":"2026-06-01T20:31:32.705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Result Summary","metadata":{}},{"cell_type":"code","source":"best_epoch = np.argmax(history.history[\"val_AUC\"])\nbest_score = history.history[\"val_AUC\"][best_epoch]\nprint('>>> Best AUC: ', best_score)\nprint('>>> Best Epoch: ', best_epoch+1)","metadata":{"trusted":true,"execution":{"execution_failed":"2026-06-01T20:31:32.706Z"}},"outputs":[],"execution_count":null}]}