{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":12090287,"sourceType":"datasetVersion","datasetId":7610998}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 🌳 Notebook Highlights & Acknowledgments\nI'm excited to share the achievements of this notebook and acknowledge the resources that made it possible.\n\n* 🚀 High-Performance Results: This single-model approach achieves a private score of 0.901. Furthermore, by using a model ensemble, the score is boosted to an impressive 0.909.\n\n* 🙏 Acknowledgments: A special thank you to the author of this fantastic notebook: \"[Bird2025 single SED model inference LB 0.857](https://www.kaggle.com/code/i2nfinit3y/bird2025-single-sed-model-inference-lb-0-857)\". The model architecture provided was fundamental to the success of this project.\n\n* 💡 Key Improvement: Pseudo-Labeling: Integrating pseudo-labeling proved to be highly beneficial. This technique expanded the training dataset by 12%, leading to a significant performance increase of over 1% on the final score.\n\nI hope you find this notebook insightful!","metadata":{}},{"cell_type":"code","source":"import glob\nimport os\nimport random\nimport sys\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport timm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as torchdata\nfrom torchaudio.transforms import AmplitudeToDB, MelSpectrogram\nfrom tqdm.auto import tqdm\nimport glob\nimport concurrent.futures\nimport shutil\nimport albumentations as A\nimport torchaudio\nfrom typing import Union\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.571823Z","iopub.execute_input":"2025-06-07T15:43:45.572219Z","iopub.status.idle":"2025-06-07T15:43:45.579703Z","shell.execute_reply.started":"2025-06-07T15:43:45.572189Z","shell.execute_reply":"2025-06-07T15:43:45.578646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/birdclef-2025/sample_submission.csv\")\ntarget_columns_ = sub.columns.tolist()\ntarget_columns = sub.columns.tolist()[1:]\nnum_classes = len(target_columns)\n\nTOTAL_SECONDS_CHUNKS = 12\ntest_path = \"/kaggle/input/birdclef-2025/test_soundscapes/\"\nfiles = glob.glob(f'{test_path}*')\nif len(files) == 1:\n    TOTAL_SECONDS_CHUNKS = 2\n\nseconds = [i for i in range(5, (TOTAL_SECONDS_CHUNKS*5) + 5, 5)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.581343Z","iopub.execute_input":"2025-06-07T15:43:45.581651Z","iopub.status.idle":"2025-06-07T15:43:45.609744Z","shell.execute_reply.started":"2025-06-07T15:43:45.581631Z","shell.execute_reply":"2025-06-07T15:43:45.608791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_path = \"/kaggle/input/birdclef-2025/test_soundscapes/\"\n\nfiles = glob.glob(f'{test_path}*')\nif len(files) == 1:\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1442779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446379.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1146779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1426779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1441779.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446179.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446719.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446771.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1446789.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_audio/bubcur1/XC10728.ogg', '/kaggle/working/soundscape_1448779.ogg')\n    test_path = \"/kaggle/working/\"\n    \nprint (test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.610762Z","iopub.execute_input":"2025-06-07T15:43:45.611074Z","iopub.status.idle":"2025-06-07T15:43:45.637576Z","shell.execute_reply.started":"2025-06-07T15:43:45.611032Z","shell.execute_reply":"2025-06-07T15:43:45.636605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mel_spec_params = {\n    \"sample_rate\": 32000,\n    \"n_mels\": 128,\n    \"f_min\": 20,\n    \"f_max\": 16000,\n    \"n_fft\": 2048,\n    \"hop_length\": 512,\n    \"normalized\": True,\n    \"center\" : True,\n    \"pad_mode\" : \"constant\",\n    \"norm\" : \"slaney\",\n    \"onesided\" : True,\n    \"mel_scale\" : \"slaney\"\n}\ntop_db = 80","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.639475Z","iopub.execute_input":"2025-06-07T15:43:45.639806Z","iopub.status.idle":"2025-06-07T15:43:45.644716Z","shell.execute_reply.started":"2025-06-07T15:43:45.639787Z","shell.execute_reply":"2025-06-07T15:43:45.643820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_melspec(X, eps=1e-6):\n    mean = X.mean((1, 2), keepdim=True)\n    std = X.std((1, 2), keepdim=True)\n    Xstd = (X - mean) / (std + eps)\n\n    norm_min, norm_max = (\n        Xstd.min(-1)[0].min(-1)[0],\n        Xstd.max(-1)[0].max(-1)[0],\n    )\n    fix_ind = (norm_max - norm_min) > eps * torch.ones_like(\n        (norm_max - norm_min)\n    )\n    V = torch.zeros_like(Xstd)\n    if fix_ind.sum():\n        V_fix = Xstd[fix_ind]\n        norm_max_fix = norm_max[fix_ind, None, None]\n        norm_min_fix = norm_min[fix_ind, None, None]\n        V_fix = torch.max(\n            torch.min(V_fix, norm_max_fix),\n            norm_min_fix,\n        )\n        V_fix = (V_fix - norm_min_fix) / (norm_max_fix - norm_min_fix)\n        V[fix_ind] = V_fix\n    return V","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.645583Z","iopub.execute_input":"2025-06-07T15:43:45.645955Z","iopub.status.idle":"2025-06-07T15:43:45.657108Z","shell.execute_reply.started":"2025-06-07T15:43:45.645861Z","shell.execute_reply":"2025-06-07T15:43:45.656172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_val = A.Compose([\n    A.Resize(256, 256),\n    A.Normalize()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.658097Z","iopub.execute_input":"2025-06-07T15:43:45.658412Z","iopub.status.idle":"2025-06-07T15:43:45.678498Z","shell.execute_reply.started":"2025-06-07T15:43:45.658389Z","shell.execute_reply":"2025-06-07T15:43:45.677483Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TestDataset(torchdata.Dataset):\n    def __init__(self, \n                 df: pd.DataFrame, \n                 clip: np.ndarray,\n                ):\n        \n        self.df = df\n        self.clip = clip\n        self.mel_transform = torchaudio.transforms.MelSpectrogram(**mel_spec_params)\n        self.db_transform = torchaudio.transforms.AmplitudeToDB(stype='power', top_db=top_db)\n        self.transform = transforms_val\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx: int):\n\n        sample = self.df.loc[idx, :]\n        row_id = sample.row_id\n\n        end_seconds = int(sample.seconds)\n        start_seconds = int(end_seconds - 5)\n        \n        wave = self.clip[:, 32000 * start_seconds : 32000 * end_seconds]\n        \n        mel_spectrogram = normalize_melspec(self.db_transform(self.mel_transform(wave)))\n        mel_spectrogram = mel_spectrogram * 255\n        mel_spectrogram = mel_spectrogram.expand(3, -1, -1).permute(1, 2, 0).numpy()\n        \n        res = self.transform(image=mel_spectrogram)\n        spec = res['image'].astype(np.float32)\n        spec = spec.transpose(2, 0, 1)\n        \n        return {\n            \"row_id\": row_id,\n            \"wave\": spec,\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.679257Z","iopub.execute_input":"2025-06-07T15:43:45.679536Z","iopub.status.idle":"2025-06-07T15:43:45.688036Z","shell.execute_reply.started":"2025-06-07T15:43:45.679518Z","shell.execute_reply":"2025-06-07T15:43:45.687182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_power_to_low_ranked_cols(\n    p: np.ndarray,\n    top_k: int = 30,\n    exponent: Union[int, float] = 2,\n    inplace: bool = True\n) -> np.ndarray:\n    if not inplace:\n        p = p.copy()\n\n    # Identify columns whose max value ranks below `top_k`\n    tail_cols = np.argsort(-p.max(axis=0))[top_k:]\n\n    # Apply the power transformation to those columns\n    p[:, tail_cols] = p[:, tail_cols] ** exponent\n    return p\n\ndef prediction_for_clip(audio_path):\n    wav, org_sr = torchaudio.load(audio_path, normalize=True)\n    clip = torchaudio.functional.resample(wav, orig_freq=org_sr, new_freq=32000)\n    \n    name_ = os.path.basename(audio_path).split(\".ogg\")[0]\n    row_ids_for_df = [f\"{name_}_{s}\" for s in seconds]\n\n    test_df = pd.DataFrame({\"row_id\": row_ids_for_df, \"seconds\": seconds})\n    \n    dataset = TestDataset(df=test_df, clip=clip) # Uses global mel_spec_params, etc.\n    loader = torchdata.DataLoader(dataset, batch_size=16, num_workers=os.cpu_count(), shuffle=False, pin_memory=True)\n\n    # Store {row_id: averaged_probas_array}\n    model_averaged_probas_dict = {}\n\n    for batch_data in loader:\n        batch_row_ids = batch_data['row_id']\n        wave_tensor = batch_data['wave'].to(device)\n        \n        batch_model_outputs = []\n        with torch.no_grad():\n            for i, model_instance in enumerate(models):\n                \n                logits = torch.logit(model_instance.infer(wave_tensor))    \n\n                p10 = logits.quantile(0.10, dim=1, keepdim=True)   # (B, 1)\n                p90 = logits.quantile(0.90, dim=1, keepdim=True)   # (B, 1)\n            \n                clipped_logits = logits.clamp(min=p10, max=p90)\n            \n                probs_orig   = logits.sigmoid()           # σ(logits)\n                probs_clip   = clipped_logits.sigmoid()   # σ(clipped_logits)\n                probs_blend  = (0.5 * (probs_orig + probs_clip)).cpu().numpy()\n            \n                # probs_blend = apply_power_to_low_ranked_cols(probs_orig.cpu().numpy(), top_k=30, exponent=2)\n                batch_model_outputs.append(probs_blend)\n        \n        # Average predictions across models for this batch\n        # Shape: (num_models, batch_size, num_classes) -> (batch_size, num_classes)\n        averaged_batch_probas = np.mean(batch_model_outputs, axis=0)\n        \n        for i, r_id in enumerate(batch_row_ids):\n            model_averaged_probas_dict[str(r_id)] = averaged_batch_probas[i]\n\n    final_predictions_for_clip = {}\n    num_chunks = len(seconds)\n\n    for idx, current_sec in enumerate(seconds):\n        current_row_id = f\"{name_}_{current_sec}\"\n        current_probas = model_averaged_probas_dict.get(current_row_id, np.zeros(len(target_columns)))\n\n        # Left neighbor (boundary: use current if first)\n        left_sec = seconds[max(0, idx - 1)]\n        left_row_id = f\"{name_}_{left_sec}\"\n        left_probas = model_averaged_probas_dict.get(left_row_id, current_probas) # Default to current if missing\n        if idx == 0: # Explicit boundary for first element\n            left_probas = current_probas\n\n        # Right neighbor (boundary: use current if last)\n        right_sec = seconds[min(num_chunks - 1, idx + 1)]\n        right_row_id = f\"{name_}_{right_sec}\"\n        right_probas = model_averaged_probas_dict.get(right_row_id, current_probas) # Default to current if missing\n        if idx == num_chunks - 1: # Explicit boundary for last element\n            right_probas = current_probas\n        \n        # Apply weights: 0.3 * left + 0.4 * current + 0.3 * right\n        smoothed_probas = 0.2 * left_probas + 0.6 * current_probas + 0.2 * right_probas\n        \n        final_predictions_for_clip[current_row_id] = {}\n        for i, label_name in enumerate(target_columns):\n            final_predictions_for_clip[current_row_id][label_name] = smoothed_probas[i]\n            \n    return final_predictions_for_clip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.773419Z","iopub.execute_input":"2025-06-07T15:43:45.773756Z","iopub.status.idle":"2025-06-07T15:43:45.788563Z","shell.execute_reply.started":"2025-06-07T15:43:45.773733Z","shell.execute_reply":"2025-06-07T15:43:45.787567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def init_layer(layer):\n    nn.init.xavier_uniform_(layer.weight)\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\n\ndef init_bn(bn):\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.0)\n\n\nclass AttBlockV2(nn.Module):\n    def __init__(self,\n                 in_features: int,\n                 out_features: int,\n                 activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n        return x, norm_att, cla\n\n    def nonlinear_transform(self, x):\n        if self.activation == 'linear':\n            return x\n        elif self.activation == 'sigmoid':\n            return torch.sigmoid(x)\n\nclass TimmSED(nn.Module):\n    def __init__(self, backbone, pretrained, in_chans=3):\n        super().__init__()\n        \n        self.num_classes = num_classes\n        \n        # BatchNorm for mel spectrograms\n        self.bn0 = nn.BatchNorm2d(256)\n        \n        # Create backbone model\n        self.backbone = timm.create_model(\n            backbone,\n            pretrained=pretrained,\n            in_chans=in_chans,\n            drop_rate=0.2,\n            drop_path_rate=0.2,\n            # cache_dir=cache_dir\n        )\n        \n        # Remove classifier layers to get features\n        layers = list(self.backbone.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n        \n        # Get backbone output features\n        if \"efficientnet\" in backbone:\n            backbone_out = self.backbone.classifier.in_features\n        elif \"eca\" in backbone:\n            backbone_out = self.backbone.head.fc.in_features\n        elif \"res\" in backbone:\n            backbone_out = self.backbone.fc.in_features\n        else:\n            backbone_out = self.backbone.num_features\n            \n        self.fc1 = nn.Linear(backbone_out, backbone_out, bias=True)\n        self.att_block = AttBlockV2(backbone_out, self.num_classes, activation=\"linear\")\n        \n        self.init_weight()\n\n    def init_weight(self):\n        init_bn(self.bn0)\n        init_layer(self.fc1)\n\n    def extract_feature(self, x):\n        # x shape: (batch_size, channels, freq, time)\n        x = x.permute((0, 1, 3, 2))  # (batch_size, channels, time, freq)\n        frames_num = x.shape[2]\n        \n        x = x.transpose(1, 3)  # (batch_size, freq, time, channels)\n        x = self.bn0(x)\n        x = x.transpose(1, 3)  # (batch_size, channels, time, freq)\n        \n        x = x.transpose(2, 3)  # (batch_size, channels, freq, time)\n        \n        # Encoder forward\n        x = self.encoder(x)\n        \n        # Global average pooling on frequency dimension\n        x = torch.mean(x, dim=2)  # (batch_size, channels, time)\n        \n        # Channel smoothing\n        x1 = F.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = F.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n        \n        x = F.dropout(x, p=0.5, training=self.training)\n        x = x.transpose(1, 2)  # (batch_size, time, channels)\n        x = F.relu_(self.fc1(x))\n        x = x.transpose(1, 2)  # (batch_size, channels, time)\n        x = F.dropout(x, p=0.5, training=self.training)\n        \n        return x, frames_num\n\n    def forward(self, x):\n        x, frames_num = self.extract_feature(x)\n        \n        clipwise_output, norm_att, segmentwise_output = self.att_block(x)\n        \n        return clipwise_output\n\n    def infer(self, x, tta_delta=2, infer_duration=5, train_duration=5):\n        \n        x, _ = self.extract_feature(x)\n        time_att = torch.tanh(self.att_block.att(x))\n        feat_time = x.size(-1)\n        \n        start = feat_time / 2 - feat_time * (infer_duration / train_duration) / 2\n        end = start + feat_time * (infer_duration / train_duration)\n        start = int(start)\n        end = int(end)\n        \n        pred = self.attention_infer(start, end, x, time_att)\n        \n        start_minus = max(0, start - tta_delta)\n        end_minus = end - tta_delta\n        pred_minus = self.attention_infer(start_minus, end_minus, x, time_att)\n        \n        start_plus = start + tta_delta\n        end_plus = min(feat_time, end + tta_delta)\n        pred_plus = self.attention_infer(start_plus, end_plus, x, time_att)\n        \n        pred = 0.5 * pred + 0.25 * pred_minus + 0.25 * pred_plus\n        return pred\n        \n    def attention_infer(self, start, end, x, time_att):\n        feat = x[:, :, start:end]\n        framewise_pred = torch.sigmoid(self.att_block.cla(feat))\n        framewise_pred_max = framewise_pred.max(dim=2)[0]\n        return framewise_pred_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.790320Z","iopub.execute_input":"2025-06-07T15:43:45.790649Z","iopub.status.idle":"2025-06-07T15:43:45.811244Z","shell.execute_reply.started":"2025-06-07T15:43:45.790620Z","shell.execute_reply":"2025-06-07T15:43:45.810312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cpu\"\nmodel_paths = [\n    \"/kaggle/input/bird2025-great-models/exp8_modelsoup_1.bin\",\n    # \"/kaggle/input/bird2025-great-models/exp8_modelsoup_2.bin\",\n    \"/kaggle/input/bird2025-great-models/exp8_modelsoup_3.bin\",\n    \n]\nmodels = []\nfor i, model_path in enumerate(model_paths):\n    try:\n        print(f\"Loading model: {model_path}\")\n        checkpoint = torch.load(model_path, map_location=torch.device(device), weights_only=False)\n        \n        model = TimmSED(backbone='eca_nfnet_l0', pretrained=False)\n        model.load_state_dict(checkpoint['state_dict'])\n        model = model.to(device)\n        model.eval()\n        model.zero_grad()\n        \n        models.append(model)\n    except Exception as e:\n        print(f\"Error loading model {path}: {e}\")\n\nif not models:\n    raise ValueError(\"No models were loaded. Please check model_paths.\")\n\nprint(f\"Total models loaded: {len(models)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:45.812148Z","iopub.execute_input":"2025-06-07T15:43:45.812389Z","iopub.status.idle":"2025-06-07T15:43:48.474696Z","shell.execute_reply.started":"2025-06-07T15:43:45.812372Z","shell.execute_reply":"2025-06-07T15:43:48.473945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    \n    all_audios = list(glob.glob(f'{test_path}*.ogg'))\n    with concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n        dicts = list(executor.map(prediction_for_clip, all_audios))\n    \n    prediction_dicts = {}\n    for d in dicts:\n        prediction_dicts.update(d)\n        \n    submission = pd.DataFrame.from_dict(prediction_dicts, \"index\").rename_axis(\"row_id\").reset_index()\n    submission.to_csv(\"submission.csv\", index=False)\n    print (\"Done\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T15:43:48.476266Z","iopub.execute_input":"2025-06-07T15:43:48.477031Z","iopub.status.idle":"2025-06-07T15:43:54.725803Z","shell.execute_reply.started":"2025-06-07T15:43:48.477005Z","shell.execute_reply":"2025-06-07T15:43:54.724911Z"}},"outputs":[],"execution_count":null}]}