{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11828260,"sourceType":"datasetVersion","datasetId":7430593},{"sourceId":11839829,"sourceType":"datasetVersion","datasetId":7438816},{"sourceId":11867185,"sourceType":"datasetVersion","datasetId":7457365},{"sourceId":11922453,"sourceType":"datasetVersion","datasetId":7495659},{"sourceId":411821,"sourceType":"modelInstanceVersion","modelInstanceId":336229,"modelId":357232}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":27.447062,"end_time":"2025-03-12T14:13:11.647927","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-03-12T14:12:44.200865","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission1(convnext_BD)》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"from torch import nn\nimport torch\nimport torchaudio\nimport librosa\n\nfrom torchvision import transforms\nfrom torchaudio.transforms import Resample\nimport datasets\nimport warnings\nimport pandas as pd\nimport json\nimport os\nimport glob\nfrom typing import Dict, Optional\nimport random\nimport numpy as np\nimport datasets\nimport torch\nfrom torch import nn\nfrom transformers import AutoConfig, ConvNextForImageClassification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:07:53.351904Z","iopub.execute_input":"2025-08-21T11:07:53.352288Z","iopub.status.idle":"2025-08-21T11:08:22.953424Z","shell.execute_reply.started":"2025-08-21T11:07:53.352248Z","shell.execute_reply":"2025-08-21T11:08:22.952130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    train_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    seed = 42\n    train_csv = \"/kaggle/input/birdclef-2025/train.csv\"\n    train_soundscapes = \"/kaggle/input/birdclef-2025/train_soundscapes\"\n    test_soundscapes = \"/kaggle/input/birdclef-2025/test_soundscapes/\"\n    sample_submission_csv = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\n    num_classes = 206\n    submission_mode = len(glob.glob(\"/kaggle/input/birdclef-2025/train_soundscapes\")) > 0\n\n\ncheckpoint_path = \"/kaggle/input/convnext_db_new/tensorflow2/default/1/birdset_epoch1.pt\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:22.954770Z","iopub.execute_input":"2025-08-21T11:08:22.955612Z","iopub.status.idle":"2025-08-21T11:08:22.963137Z","shell.execute_reply.started":"2025-08-21T11:08:22.955563Z","shell.execute_reply":"2025-08-21T11:08:22.961968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_seed(seed: int=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic=True\n    torch.backends.cudnn.benchmark=False\n\n    print(f\"Setting seed : {seed}\")\nset_seed()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:22.964370Z","iopub.execute_input":"2025-08-21T11:08:22.964838Z","iopub.status.idle":"2025-08-21T11:08:23.015960Z","shell.execute_reply.started":"2025-08-21T11:08:22.964768Z","shell.execute_reply":"2025-08-21T11:08:23.014673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torchaudio\nfrom torchaudio.transforms import Resample\n\ndef preprocess_audio(audio_path, window_size=5, sample_rate=32000):\n    audio, sr = torchaudio.load(audio_path, normalize=True)\n    if sr != sample_rate:\n        audio = Resample(orig_freq=sr, new_freq=sample_rate)(audio)\n\n    window_size_samples = window_size * sample_rate\n    one_sec_samples = sample_rate\n\n    total_samples = audio.size(1)\n    num_segments = (total_samples) // window_size_samples  # ceil division\n\n    segments = {}\n    for i in range(num_segments):\n        start = i * window_size_samples\n        end = start + window_size_samples\n        segment = audio[:, start:end]\n\n        if segment.size(1) < window_size_samples:\n            # Get last 1 second\n            repeat_part = segment[:, -one_sec_samples:] if segment.size(1) >= one_sec_samples else segment\n            repeats_needed = (window_size_samples - segment.size(1) + one_sec_samples - 1) // one_sec_samples\n            repeated = repeat_part.repeat(1, repeats_needed)\n            segment = torch.cat([segment, repeated], dim=1)[:, :window_size_samples]  # Trim extra if needed\n\n        end_second = (i + 1) * window_size\n        filename_key = f\"{os.path.splitext(os.path.basename(audio_path))[0]}_{end_second}\"\n        segment =  segment.mean(dim=0, keepdim=True)\n        segments[filename_key] = segment.squeeze(-1)\n\n    return segments","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.017185Z","iopub.execute_input":"2025-08-21T11:08:23.017624Z","iopub.status.idle":"2025-08-21T11:08:23.029323Z","shell.execute_reply.started":"2025-08-21T11:08:23.017580Z","shell.execute_reply":"2025-08-21T11:08:23.028117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PowerToDB(nn.Module):\n    def __init__(self, ref=1.0, amin=1e-10, top_db=80.0):\n        super(PowerToDB, self).__init__()\n        # Initialize parameters\n        self.ref = ref\n        self.amin = amin\n        self.top_db = top_db\n\n    def forward(self, S):\n        # Convert S to a PyTorch tensor if it is not already\n        S = torch.as_tensor(S, dtype=torch.float32)\n\n        if self.amin <= 0:\n            raise ValueError(\"amin must be strictly positive\")\n\n        if torch.is_complex(S):\n            warnings.warn(\n                \"power_to_db was called on complex input so phase \"\n                \"information will be discarded. To suppress this warning, \"\n                \"call power_to_db(S.abs()**2) instead.\",\n                stacklevel=2,\n            )\n            magnitude = S.abs()\n        else:\n            magnitude = S\n\n        # Check if ref is a callable function or a scalar\n        if callable(self.ref):\n            ref_value = self.ref(magnitude)\n        else:\n            ref_value = torch.abs(torch.tensor(self.ref, dtype=S.dtype))\n\n        # Compute the log spectrogram\n        log_spec = 10.0 * torch.log10(\n            torch.maximum(magnitude, torch.tensor(self.amin, device=magnitude.device))\n        )\n        log_spec -= 10.0 * torch.log10(\n            torch.maximum(ref_value, torch.tensor(self.amin, device=magnitude.device))\n        )\n\n        # Apply top_db threshold if necessary\n        if self.top_db is not None:\n            if self.top_db < 0:\n                raise ValueError(\"top_db must be non-negative\")\n            log_spec = torch.maximum(log_spec, log_spec.max() - self.top_db)\n\n        return log_spec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.030764Z","iopub.execute_input":"2025-08-21T11:08:23.031107Z","iopub.status.idle":"2025-08-21T11:08:23.060318Z","shell.execute_reply.started":"2025-08-21T11:08:23.031076Z","shell.execute_reply":"2025-08-21T11:08:23.058926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ConvNextClassifier(nn.Module):\n    \"\"\"\n    ConvNext model for audio classification.\n    \"\"\"\n\n    def __init__(\n        self,\n        num_channels: int = 1,\n        num_classes: Optional[int] = None,\n        checkpoint: Optional[str] = None,\n        local_checkpoint: Optional[str] = None,\n        cache_dir: Optional[str] = None,\n        pretrain_info= None #PretrainInfoConfig = None,\n    ):\n        \"\"\"\n        Note: Either num_classes or pretrain_info must be given\n        Args:\n            num_channels: Number of input channels.\n            checkpoint: huggingface checkpoint path of any model of correct type\n            num_classes: number of classification heads to be used in the model\n            local_checkpoint: local path to checkpoint file\n            cache_dir: specified cache dir to save model files at\n            pretrain_info: hf_path and hf_name of info will be used to infer if num_classes is None\n        \"\"\"\n        super().__init__()\n\n        if pretrain_info:\n            self.hf_path = pretrain_info.hf_path\n            self.hf_name = (\n                pretrain_info.hf_name\n                if not pretrain_info.hf_pretrain_name\n                else pretrain_info.hf_pretrain_name\n            )\n            self.num_classes = len(\n                datasets.load_dataset_builder(self.hf_path, self.hf_name)\n                .info.features[\"ebird_code\"]\n                .names\n            )\n        else:\n            self.hf_path = None\n            self.hf_name = None\n            self.num_classes = num_classes\n\n        self.num_channels = num_channels\n        self.checkpoint = checkpoint\n        self.local_checkpoint = local_checkpoint\n        self.cache_dir = cache_dir\n\n        self.model = None\n\n        self._initialize_model()\n\n    def _initialize_model(self):\n        \"\"\"Initializes the ConvNext model based on specified attributes.\"\"\"\n\n        adjusted_state_dict = None\n\n        if self.checkpoint:\n            if self.local_checkpoint:\n                state_dict = torch.load(self.local_checkpoint)[\"state_dict\"]\n\n                # Update this part to handle the necessary key replacements\n                adjusted_state_dict = {}\n                for key, value in state_dict.items():\n                    # Handle 'model.model.' prefix\n                    new_key = key.replace(\"model.model.\", \"\")\n\n                    # Handle 'model._orig_mod.model.' prefix\n                    new_key = new_key.replace(\"model._orig_mod.model.\", \"\")\n\n                    # Assign the adjusted key\n                    adjusted_state_dict[new_key] = value\n\n            self.model = ConvNextForImageClassification.from_pretrained(\n                self.checkpoint,\n                num_labels=self.num_classes,\n                num_channels=self.num_channels,\n                cache_dir=self.cache_dir,\n                state_dict=adjusted_state_dict,\n                ignore_mismatched_sizes=True,\n            )\n        else:\n            config = AutoConfig.from_pretrained(\n                \"/kaggle/input/fb/pytorch/default/1/convnext-base-224-22k\",\n                num_labels=self.num_classes,\n                num_channels=self.num_channels,\n            )\n            self.model = ConvNextForImageClassification(config)\n\n    def forward(\n        self, input_values: torch.Tensor, labels: Optional[torch.Tensor] = None\n    ) -> torch.Tensor:\n        \"\"\"\n        Defines the forward pass of the ConvNext model.\n\n        Args:\n            input_values (torch.Tensor): An input batch.\n            labels (Optional[torch.Tensor]): The corresponding labels. Default is None.\n\n        Returns:\n            torch.Tensor: The output of the ConvNext model.\n        \"\"\"\n        output = self.model(input_values)\n        logits = output.logits\n\n        return logits\n\n    @torch.inference_mode()\n    def get_logits(self, dataloader, device):\n        pass\n\n    @torch.inference_mode()\n    def get_probas(self, dataloader, device):\n        pass\n\n    @torch.inference_mode()\n    def get_representations(self, dataloader, device):\n        pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.063268Z","iopub.execute_input":"2025-08-21T11:08:23.063639Z","iopub.status.idle":"2025-08-21T11:08:23.096887Z","shell.execute_reply.started":"2025-08-21T11:08:23.063605Z","shell.execute_reply":"2025-08-21T11:08:23.095631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ConvNextBirdSet(nn.Module):\n    \"\"\"\n    BirdSet ConvNext model trained on BirdSet XCL dataset.\n    The model expects a raw 1 channel 5s waveform with sample rate of 32kHz as an input.\n    Its preprocess function will:\n        - convert the waveform to a spectrogram: n_fft: 1024, hop_length: 320, power: 2.0\n        - melscale the spectrogram: n_mels: 128, n_stft: 513\n        - dbscale with top_db: 80\n        - normalize the spectrogram mean: -4.268, std: 4.569 (from esc-50)\n    \"\"\"\n\n    def __init__(\n        self,\n        PowerToDB,\n        num_classes=9736,\n    ):\n        super().__init__()\n        self.model = ConvNextClassifier(\n            checkpoint=\"/kaggle/input/convnext-xcl-transformers-model\",\n            num_classes=num_classes,\n        )\n        self.spectrogram_converter = torchaudio.transforms.Spectrogram(\n            n_fft=1024, hop_length=320, power=2.0\n        )\n        self.mel_converter = torchaudio.transforms.MelScale(\n            n_mels=128, n_stft=513, sample_rate=32_000\n        )\n        self.normalizer = transforms.Normalize((-4.268,), (4.569,))\n        self.powerToDB = PowerToDB(top_db=80)\n        self.config = self.model.model.config\n\n    def preprocess(self, waveform: torch.Tensor):\n        # convert waveform to spectrogram\n        spectrogram = self.spectrogram_converter(waveform)\n        spectrogram = spectrogram.to(torch.float32)\n        melspec = self.mel_converter(spectrogram)\n        dbscale = self.powerToDB(melspec)\n        normalized_dbscale = self.normalizer(dbscale)\n        # add dimension 3 from left\n        normalized_dbscale = normalized_dbscale.unsqueeze(-3)\n\n        return normalized_dbscale\n\n    def forward(self, input: torch.Tensor):\n        # spectrogram = self.preprocess(waveform)\n        return self.model(input)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.101654Z","iopub.execute_input":"2025-08-21T11:08:23.102209Z","iopub.status.idle":"2025-08-21T11:08:23.120683Z","shell.execute_reply.started":"2025-08-21T11:08:23.102132Z","shell.execute_reply":"2025-08-21T11:08:23.119420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_dir = glob.glob(Config.test_soundscapes + \"/*.ogg\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.121843Z","iopub.execute_input":"2025-08-21T11:08:23.122201Z","iopub.status.idle":"2025-08-21T11:08:23.165393Z","shell.execute_reply.started":"2025-08-21T11:08:23.122166Z","shell.execute_reply":"2025-08-21T11:08:23.164257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdsetModule(torch.nn.Module):\n    def __init__(self,PowerToDB,num_classes=206):\n        super().__init__()\n        self.model = ConvNextBirdSet(PowerToDB, num_classes=num_classes)\n\n    def forward(self, x):\n        preprocessed = self.model.preprocess(x)\n        return self.model(preprocessed)\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = BirdsetModule(PowerToDB, num_classes=206).to(device)\nmodel.load_state_dict(torch.load(checkpoint_path, map_location=device))\nmodel.eval()\nmodel.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:23.166449Z","iopub.execute_input":"2025-08-21T11:08:23.166894Z","iopub.status.idle":"2025-08-21T11:08:29.986862Z","shell.execute_reply.started":"2025-08-21T11:08:23.166808Z","shell.execute_reply":"2025-08-21T11:08:29.985695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def inference(model, audio_tensor, device):\n    model.eval()\n    with torch.no_grad():\n        audio_tensor = audio_tensor.to(device)\n        logits = model(audio_tensor)  # add batch dim\n        probs = torch.softmax(logits, dim=1)\n        #top_probs, top_indices = torch.topk(probs, k=top_k)\n        return probs.squeeze(0).tolist() #top_indices.squeeze(0).tolist(), top_probs.squeeze(0).tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:29.987934Z","iopub.execute_input":"2025-08-21T11:08:29.988346Z","iopub.status.idle":"2025-08-21T11:08:29.993533Z","shell.execute_reply.started":"2025-08-21T11:08:29.988312Z","shell.execute_reply":"2025-08-21T11:08:29.992298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/input/mapping/map.txt\",\"r\") as file:\n    content = file.read()\nprint(\"Taking \",audio_dir)\nmap_text = json.loads(content)\nprobabilities=[]\nrows = []\nfor index, audio_path in enumerate(audio_dir):\n\n    audio_tensor_dict = preprocess_audio(audio_path)\n    for key, value in audio_tensor_dict.items():\n        probs = inference(model,value,device)\n        probabilities.append(probs)\n        rows.append(key)\ncolumns = list(map_text.values())\ndf = pd.DataFrame(probabilities,columns=columns)\ndf.insert(0,\"row_id\",rows)\nprint(\"Writing Submission...\")\ndf.to_csv(\"submission0.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:29.994495Z","iopub.execute_input":"2025-08-21T11:08:29.994896Z","iopub.status.idle":"2025-08-21T11:08:30.042571Z","shell.execute_reply.started":"2025-08-21T11:08:29.994858Z","shell.execute_reply":"2025-08-21T11:08:30.041661Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission2(nfnet)》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom typing import Union\n\ndef apply_power_to_low_ranked_cols(\n    p: np.ndarray,\n    top_k: int = 30,\n    exponent: Union[int, float] = 2,\n    inplace: bool = True\n) -> np.ndarray:\n    \"\"\"\n    Rank columns by their column‑wise maximum and raise every column whose\n    rank falls below `top_k` to a given power.\n\n    Parameters\n    ----------\n    p : np.ndarray\n        A 2‑D array of shape **(n_chunks, n_classes)**.\n\n        - **n_chunks** is the number of fixed‑length time chunks obtained\n          after slicing the input audio (or other sequential data).  \n          *Example:* In the BirdCLEF `test_soundscapes` set, each file is\n          60 s long. If you extract non‑overlapping 5 s windows,  \n          `n_chunks = 60 s / 5 s = 12`.\n        - **n_classes** is the number of classes being predicted.\n        - Each element `p[i, j]` is the score or probability of class *j*\n          in chunk *i*.\n\n    top_k : int, default=35\n        The highest‑ranked columns (by their maximum value) that remain\n        unchanged.\n\n    exponent : int or float, default=1.5\n        The power applied to the selected low‑ranked columns  \n        (e.g. `2` squares, `0.5` takes the square root, `3` cubes).\n\n    inplace : bool, default=True\n        If `True`, modify `p` in place.  \n        If `False`, operate on a copy and leave the original array intact.\n\n    Returns\n    -------\n    np.ndarray\n        The transformed array. It is the same object as `p` when\n        `inplace=True`; otherwise, it is a new array.\n\n    \"\"\"\n    if not inplace:\n        p = p.copy()\n\n    # Identify columns whose max value ranks below `top_k`\n    tail_cols = np.argsort(-p.max(axis=0))[top_k:]\n\n    # Apply the power transformation to those columns\n    p[:, tail_cols] = p[:, tail_cols] ** exponent\n    return p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:30.043662Z","iopub.execute_input":"2025-08-21T11:08:30.044026Z","iopub.status.idle":"2025-08-21T11:08:30.050744Z","shell.execute_reply.started":"2025-08-21T11:08:30.043979Z","shell.execute_reply":"2025-08-21T11:08:30.049611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport timm\nimport torch.nn.functional as F\nimport torchaudio\nimport torchaudio.transforms as AT\nfrom contextlib import contextmanager\nimport concurrent.futures","metadata":{"papermill":{"duration":12.984639,"end_time":"2025-03-12T14:13:00.145177","exception":false,"start_time":"2025-03-12T14:12:47.160538","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:30.052173Z","iopub.execute_input":"2025-08-21T11:08:30.052571Z","iopub.status.idle":"2025-08-21T11:08:33.456144Z","shell.execute_reply.started":"2025-08-21T11:08:30.052540Z","shell.execute_reply":"2025-08-21T11:08:33.455139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_audio_dir = '/kaggle/input/birdclef-2025/test_soundscapes'\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nfile_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n\ndebug = False\nprint('Debug mode:', debug)\nprint('Number of test soundscapes:', len(file_list))","metadata":{"papermill":{"duration":0.105385,"end_time":"2025-03-12T14:13:00.253425","exception":false,"start_time":"2025-03-12T14:13:00.14804","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:33.457364Z","iopub.execute_input":"2025-08-21T11:08:33.457780Z","iopub.status.idle":"2025-08-21T11:08:33.466042Z","shell.execute_reply.started":"2025-08-21T11:08:33.457739Z","shell.execute_reply":"2025-08-21T11:08:33.464764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wav_sec = 5\nsample_rate = 32000\nmin_segment = sample_rate*wav_sec\n\nclass_labels = sorted(os.listdir('../input/birdclef-2025/train_audio/'))\n\nn_fft=1024\nwin_length=1024\nhop_length=512\nf_min=50\nf_max=16000\nn_mels=128\n\nmel_spectrogram = AT.MelSpectrogram(\n    sample_rate=sample_rate,\n    n_fft=n_fft,\n    win_length=win_length,\n    hop_length=hop_length,\n    center=True,\n    f_min=f_min,\n    f_max=f_max,\n    pad_mode=\"reflect\",\n    power=2.0,\n    norm='slaney',\n    n_mels=n_mels,\n    mel_scale=\"htk\",\n    # normalized=True\n)\n\ndef normalize_std(spec, eps=1e-6):\n    mean = torch.mean(spec)\n    std = torch.std(spec)\n    return torch.where(std == 0, spec-mean, (spec - mean) / (std+eps))\n\ndef audio_to_mel(filepath=None):\n    waveform, sample_rate = torchaudio.load(filepath,backend=\"soundfile\")\n    len_wav = waveform.shape[1]\n    waveform = waveform[0,:].reshape(1, len_wav) # stereo->mono mono->mono\n    PREDS = []\n    for i in range(12):\n        waveform2 = waveform[:,i*sample_rate*5:i*sample_rate*5+sample_rate*5]\n        melspec = mel_spectrogram(waveform2)\n        melspec = torch.log(melspec+1e-6)\n        melspec = normalize_std(melspec)\n        melspec = torch.unsqueeze(melspec, dim=0)\n        \n        PREDS.append(melspec)\n    return torch.vstack(PREDS)","metadata":{"papermill":{"duration":0.144235,"end_time":"2025-03-12T14:13:00.400505","exception":false,"start_time":"2025-03-12T14:13:00.25627","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:33.467051Z","iopub.execute_input":"2025-08-21T11:08:33.467352Z","iopub.status.idle":"2025-08-21T11:08:33.511705Z","shell.execute_reply.started":"2025-08-21T11:08:33.467324Z","shell.execute_reply":"2025-08-21T11:08:33.510526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def init_layer(layer):\n    nn.init.xavier_uniform_(layer.weight)\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\n\ndef init_bn(bn):\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.0)\n\n\ndef init_weights(model):\n    classname = model.__class__.__name__\n    if classname.find(\"Conv2d\") != -1:\n        nn.init.xavier_uniform_(model.weight, gain=np.sqrt(2))\n        model.bias.data.fill_(0)\n    elif classname.find(\"BatchNorm\") != -1:\n        model.weight.data.normal_(1.0, 0.02)\n        model.bias.data.fill_(0)\n    elif classname.find(\"GRU\") != -1:\n        for weight in model.parameters():\n            if len(weight.size()) > 1:\n                nn.init.orghogonal_(weight.data)\n    elif classname.find(\"Linear\") != -1:\n        model.weight.data.normal_(0, 0.01)\n        model.bias.data.zero_()\n\n\ndef interpolate(x, ratio):\n    (batch_size, time_steps, classes_num) = x.shape\n    upsampled = x[:, :, None, :].repeat(1, 1, ratio, 1)\n    upsampled = upsampled.reshape(batch_size, time_steps * ratio, classes_num)\n    return upsampled\n\n\ndef pad_framewise_output(framewise_output, frames_num):\n    output = F.interpolate(\n        framewise_output.unsqueeze(1),\n        size=(frames_num, framewise_output.size(2)),\n        align_corners=True,\n        mode=\"bilinear\").squeeze(1)\n\n    return output\n\n\nclass AttBlockV2(nn.Module):\n    def __init__(self,\n                 in_features: int,\n                 out_features: int,\n                 activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n        return x, norm_att, cla\n\n    def nonlinear_transform(self, x):\n        if self.activation == 'linear':\n            return x\n        elif self.activation == 'sigmoid':\n            return torch.sigmoid(x)\n\n\nclass TimmSED(nn.Module):\n    def __init__(self, base_model_name: str, pretrained=False, num_classes=24, in_channels=1, n_mels=24):\n        super().__init__()\n\n        self.bn0 = nn.BatchNorm2d(n_mels)\n\n        base_model = timm.create_model(\n            base_model_name, pretrained=pretrained, in_chans=in_channels)\n        layers = list(base_model.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n\n        in_features = base_model.num_features\n\n        self.fc1 = nn.Linear(in_features, in_features, bias=True)\n        self.att_block2 = AttBlockV2(\n            in_features, num_classes, activation=\"sigmoid\")\n\n        self.init_weight()\n\n    def init_weight(self):\n        init_bn(self.bn0)\n        init_layer(self.fc1)\n        \n\n    def forward(self, input_data):\n        x = input_data.transpose(2,3)\n        x = torch.cat((x,x,x),1)\n\n        x = x.transpose(2, 3)\n\n        x = self.encoder(x)\n        \n        x = torch.mean(x, dim=2)\n\n        x1 = F.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = F.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n\n        x = x.transpose(1, 2)\n        x = F.relu_(self.fc1(x))\n        x = x.transpose(1, 2)\n\n        (clipwise_output, norm_att, segmentwise_output) = self.att_block2(x)\n        logit = torch.sum(norm_att * self.att_block2.cla(x), dim=2)\n\n        output_dict = {\n            'logit': logit,\n        }\n\n        return output_dict","metadata":{"papermill":{"duration":2.175154,"end_time":"2025-03-12T14:13:02.578522","exception":false,"start_time":"2025-03-12T14:13:00.403368","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:33.512848Z","iopub.execute_input":"2025-08-21T11:08:33.513266Z","iopub.status.idle":"2025-08-21T11:08:33.531932Z","shell.execute_reply.started":"2025-08-21T11:08:33.513225Z","shell.execute_reply":"2025-08-21T11:08:33.530665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model_name='eca_nfnet_l0'\npretrained=False\nin_channels=3\n\nMODELS = [f'/kaggle/input/birdclef-2025-sed-models-p/sed{i}.pth' for i in range(3)]\n\nMODELS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:33.532925Z","iopub.execute_input":"2025-08-21T11:08:33.533264Z","iopub.status.idle":"2025-08-21T11:08:33.553403Z","shell.execute_reply.started":"2025-08-21T11:08:33.533237Z","shell.execute_reply":"2025-08-21T11:08:33.552443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = []\nfor path in MODELS:\n    model = TimmSED(base_model_name=base_model_name,\n               pretrained=pretrained,\n               num_classes=len(class_labels),\n               in_channels=in_channels,\n               n_mels=n_mels);\n    model.load_state_dict(torch.load(path, weights_only=True, map_location=torch.device('cpu')))\n    model.eval();\n    models.append(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:33.554415Z","iopub.execute_input":"2025-08-21T11:08:33.554772Z","iopub.status.idle":"2025-08-21T11:08:37.960630Z","shell.execute_reply.started":"2025-08-21T11:08:33.554744Z","shell.execute_reply":"2025-08-21T11:08:37.959583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prediction(afile):    \n    global pred\n    path = test_audio_dir + afile + '.ogg'\n    with torch.inference_mode():\n        sig = audio_to_mel(path)\n        outputs = None\n        for model in models:\n            model.eval()\n            p = model(sig)\n            p = torch.sigmoid(p['logit']).detach().cpu().numpy() \n            p = apply_power_to_low_ranked_cols(p, top_k=30,exponent=2)\n            if outputs is None: outputs = p\n            else: outputs += p\n            \n        outputs /= len(models)\n        chunks = [[] for i in range(12)]\n        for i in range(len(chunks)):        \n            chunk_end_time = (i + 1) * 5\n            row_id = afile + '_' + str(chunk_end_time)\n            pred['row_id'].append(row_id)\n            bird_no = 0\n            for bird in class_labels:         \n                pred[bird].append(outputs[i,bird_no])\n                bird_no += 1\n        gc.collect()","metadata":{"papermill":{"duration":0.011209,"end_time":"2025-03-12T14:13:02.593243","exception":false,"start_time":"2025-03-12T14:13:02.582034","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:37.964585Z","iopub.execute_input":"2025-08-21T11:08:37.964935Z","iopub.status.idle":"2025-08-21T11:08:37.971901Z","shell.execute_reply.started":"2025-08-21T11:08:37.964899Z","shell.execute_reply":"2025-08-21T11:08:37.970895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = {'row_id': []}\nfor species_code in class_labels:\n    pred[species_code] = []\n    \nstart = time.time()\nwith concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor:\n    _ = list(executor.map(prediction, file_list))\nend_t = time.time()\n\nif debug == True:\n    print(700*(end_t - start)/60/debug_num)","metadata":{"papermill":{"duration":6.823541,"end_time":"2025-03-12T14:13:09.419521","exception":false,"start_time":"2025-03-12T14:13:02.59598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:37.973420Z","iopub.execute_input":"2025-08-21T11:08:37.973824Z","iopub.status.idle":"2025-08-21T11:08:38.022544Z","shell.execute_reply.started":"2025-08-21T11:08:37.973789Z","shell.execute_reply":"2025-08-21T11:08:38.021562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = pd.DataFrame(pred, columns = ['row_id'] + class_labels) \ndisplay(results.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:38.023580Z","iopub.execute_input":"2025-08-21T11:08:38.023968Z","iopub.status.idle":"2025-08-21T11:08:38.064780Z","shell.execute_reply.started":"2025-08-21T11:08:38.023912Z","shell.execute_reply":"2025-08-21T11:08:38.063773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results.to_csv(\"submission1.csv\", index=False)    \n\nsub = pd.read_csv('submission1.csv')\ncols = sub.columns[1:]\ngroups = sub['row_id'].str.rsplit('_', n=1).str[0]\ngroups = groups.values\nfor group in np.unique(groups):\n    sub_group = sub[group == groups]\n    predictions = sub_group[cols].values\n    new_predictions = predictions.copy()\n    for i in range(1, predictions.shape[0]-1):\n        new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n    new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n    new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n    sub_group[cols] = new_predictions\n    sub[group == groups] = sub_group\nsub.to_csv(\"submission1.csv\", index=False)\n\n\nif debug:\n    display(results)","metadata":{"papermill":{"duration":0.097214,"end_time":"2025-03-12T14:13:09.519812","exception":false,"start_time":"2025-03-12T14:13:09.422598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:38.065775Z","iopub.execute_input":"2025-08-21T11:08:38.066116Z","iopub.status.idle":"2025-08-21T11:08:38.100500Z","shell.execute_reply.started":"2025-08-21T11:08:38.066089Z","shell.execute_reply":"2025-08-21T11:08:38.099422Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Finaly Blending》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"# ------------------------------------------- #\n# [IMPORTANT]\n# * Blending Weight\n# ------------------------------------------- #\nsub_w = [0.25, 0.75]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:38.101500Z","iopub.execute_input":"2025-08-21T11:08:38.101774Z","iopub.status.idle":"2025-08-21T11:08:38.120072Z","shell.execute_reply.started":"2025-08-21T11:08:38.101751Z","shell.execute_reply":"2025-08-21T11:08:38.118906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load target list and prepare column names\nlist_TARGETs = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\nlist_targets_0 = [f'{TARGET} 0' for TARGET in list_TARGETs]\nlist_targets_1 = [f'{TARGET} 1' for TARGET in list_TARGETs]\n\n# Load both predictions\ndf0 = pd.read_csv(\"/kaggle/working/submission0.csv\")\ndf1 = pd.read_csv(\"/kaggle/working/submission1.csv\")\n\n# Rename columns to distinguish models\ndf0 = df0.rename(columns={TARGET : f'{TARGET} 0' for TARGET in list_TARGETs})\ndf1 = df1.rename(columns={TARGET : f'{TARGET} 1' for TARGET in list_TARGETs})\n\n# Merge on row_id\ndfs = pd.merge(df0, df1, on='row_id')\n\n# Compute blended predictions all at once (avoids fragmentation)\nblended_columns = {\n    TARGET: dfs[f'{TARGET} 0'] * sub_w[0] + dfs[f'{TARGET} 1'] * sub_w[1]\n    for TARGET in list_TARGETs\n}\n\nblended_df = pd.DataFrame(blended_columns)\n\n# Final DataFrame: row_id + blended target columns\nfinal_df = pd.concat([dfs[['row_id']], blended_df], axis=1)\n\n# Save\nfinal_df.to_csv(\"submission.csv\", index=False)\nprint(\"DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T11:08:38.121181Z","iopub.execute_input":"2025-08-21T11:08:38.121587Z","iopub.status.idle":"2025-08-21T11:08:38.264868Z","shell.execute_reply.started":"2025-08-21T11:08:38.121547Z","shell.execute_reply":"2025-08-21T11:08:38.263542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}