{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11828260,"sourceType":"datasetVersion","datasetId":7430593},{"sourceId":11867185,"sourceType":"datasetVersion","datasetId":7457365}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":27.447062,"end_time":"2025-03-12T14:13:11.647927","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-03-12T14:12:44.200865","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\nOnly Submission(LoadLocalTrainModel)\n</b></h1> ","metadata":{}},{"cell_type":"markdown","source":"### **ℹ️INFO**\n* This notebook is an weighted blend(nfnet * 0.6 + convnextv2 * 0.4).\n    * **GreatWork LB.850(nfnet)** https://www.kaggle.com/code/myso1987/post-processing-with-power-adjustment-for-low-rank\n    * **MyLocalTrainModel(convnextv2)** https://www.kaggle.com/datasets/hideyukizushi/bird25-d-330v2-ppv15-convnextv2-nano/data\n\n### **ℹ️WeightedBlend**\n* In CV tasks, it is common to adopt diverse backbones to ensure robustness, and I am sharing this in this competition because it is connected to the LB boost.\n* P.S. Adjusting the blending weights should make it fit LB better and improve the score. In that case, it is redundant, so I recommend submitting in a private notebook rather than a public notebook.\n\n### **ℹ️Appendix**\n* My old LB.829 notebook from this competition\n* https://www.kaggle.com/code/hideyukizushi/bird25-onlyinf-v2-s-focallossbce-cv-962-lb-829\n\n","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission1(convnextv2)》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nimport logging\nimport time\nimport math\nimport cv2\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport timm\nfrom tqdm.auto import tqdm\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.258371Z","iopub.execute_input":"2025-05-19T08:05:02.259183Z","iopub.status.idle":"2025-05-19T08:05:02.267364Z","shell.execute_reply.started":"2025-05-19T08:05:02.259112Z","shell.execute_reply":"2025-05-19T08:05:02.266088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n \n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    \n    # ------------------------------------------- #\n    # [IMPORTANT]\n    # * Melspectrogram & Audio Params\n    # ------------------------------------------- #\n    FS = 32000  \n    WINDOW_SIZE = 5\n    N_FFT = 2048\n    HOP_LENGTH = 512\n    N_MELS = 512\n    FMIN = 20\n    FMAX = 16000\n    TARGET_SHAPE = (256, 256)\n\n    # ------------------------------------------- #\n    # * Model def\n    # ------------------------------------------- #\n    model_path = '/kaggle/input/bird25-d-330v2-ppv15-convnextv2-nano'\n    model_name = 'convnextv2_nano.fcmae_ft_in22k_in1k'\n    use_specific_folds = True\n    folds = [0,1]\n    in_channels = 1\n    device = 'cpu'  \n    \n    # Inference parameters\n    batch_size = 16\n    use_tta = False  \n    tta_count = 3\n    threshold = 0.5\n\n    # util\n    debug = False\n    debug_count = 3\n\ncfg = CFG()\n\nprint(f\"Using device: {cfg.device}\")\nprint(f\"Loading taxonomy data...\")\ntaxonomy_df = pd.read_csv(cfg.taxonomy_csv)\nspecies_ids = taxonomy_df['primary_label'].tolist()\nnum_classes = len(species_ids)\nprint(f\"Number of classes: {num_classes}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.268816Z","iopub.execute_input":"2025-05-19T08:05:02.269231Z","iopub.status.idle":"2025-05-19T08:05:02.307602Z","shell.execute_reply.started":"2025-05-19T08:05:02.269187Z","shell.execute_reply":"2025-05-19T08:05:02.306337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6):\n        super(GeM, self).__init__()\n        self.p = nn.Parameter(torch.ones(1)*p)\n        self.eps = eps\n\n    def forward(self, x):\n        return self.gem(x, p=self.p, eps=self.eps)\n\n    def gem(self, x, p=3, eps=1e-6):\n        return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1./p)\n\n    def __repr__(self):\n        return self.__class__.__name__ + \\\n                '(' + 'p=' + '{:.4f}'.format(self.p.data.tolist()[0]) + \\\n                ', ' + 'eps=' + str(self.eps) + ')'\nclass BirdCLEFModel(nn.Module):\n    def __init__(self, cfg, num_classes):\n        super().__init__()\n        self.cfg = cfg\n        \n        self.backbone = timm.create_model(\n            cfg.model_name,\n            pretrained=False,  \n            in_chans=cfg.in_channels,\n            drop_rate=0.0,    \n            drop_path_rate=0.0\n        )\n        \n        if 'efficientnet' in cfg.model_name:\n            backbone_out = self.backbone.classifier.in_features\n            self.backbone.classifier = nn.Identity()\n        elif 'resnet' in cfg.model_name:\n            backbone_out = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        else:\n            backbone_out = self.backbone.get_classifier().in_features\n            self.backbone.reset_classifier(0, '')\n        \n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        self.feat_dim = backbone_out\n        self.classifier = nn.Linear(backbone_out, num_classes)\n        \n    def forward(self, x):\n        features = self.backbone(x)\n        \n        if isinstance(features, dict):\n            features = features['features']\n            \n        if len(features.shape) == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        \n        logits = self.classifier(features)\n        return logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.309209Z","iopub.execute_input":"2025-05-19T08:05:02.309556Z","iopub.status.idle":"2025-05-19T08:05:02.324806Z","shell.execute_reply.started":"2025-05-19T08:05:02.30953Z","shell.execute_reply":"2025-05-19T08:05:02.32327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio2melspec(audio_data, cfg):\n    \"\"\"Convert audio data to mel spectrogram\"\"\"\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        power=2.0,\n        pad_mode=\"reflect\",\n        norm='slaney',\n        htk=True,\n        center=True,\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    \n    return mel_spec_norm\n\ndef process_audio_segment(audio_data, cfg):\n    \"\"\"Process audio segment to get mel spectrogram\"\"\"\n    if len(audio_data) < cfg.FS * cfg.WINDOW_SIZE:\n        audio_data = np.pad(audio_data, \n                          (0, cfg.FS * cfg.WINDOW_SIZE - len(audio_data)), \n                          mode='constant')\n    \n    mel_spec = audio2melspec(audio_data, cfg)\n    \n    # Resize if needed\n    if mel_spec.shape != cfg.TARGET_SHAPE:\n        mel_spec = cv2.resize(mel_spec, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n        \n    return mel_spec.astype(np.float32)\n    \ndef find_model_files(cfg):\n    \"\"\"\n    Find all .pth model files in the specified model directory\n    \"\"\"\n    model_files = []\n    \n    model_dir = Path(cfg.model_path)\n    \n    for path in model_dir.glob('**/*.pth'):\n        model_files.append(str(path))\n    \n    return model_files\n\ndef load_models(cfg, num_classes):\n    \"\"\"\n    Load all found model files and prepare them for ensemble\n    \"\"\"\n    models = []\n    \n    model_files = find_model_files(cfg)\n    \n    if not model_files:\n        print(f\"Warning: No model files found under {cfg.model_path}!\")\n        return models\n    \n    print(f\"Found a total of {len(model_files)} model files.\")\n    \n    if cfg.use_specific_folds:\n        filtered_files = []\n        for fold in cfg.folds:\n            fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n            filtered_files.extend(fold_files)\n        model_files = filtered_files\n        print(f\"Using {len(model_files)} model files for the specified folds ({cfg.folds}).\")\n    \n    for model_path in model_files:\n        try:\n            print(f\"Loading model: {model_path}\")\n            checkpoint = torch.load(model_path, map_location=torch.device(cfg.device))\n            \n            model = BirdCLEFModel(cfg, num_classes)\n            model.load_state_dict(checkpoint['model_state_dict'])\n            model = model.to(cfg.device)\n            model.eval()\n            \n            models.append(model)\n        except Exception as e:\n            print(f\"Error loading model {model_path}: {e}\")\n    \n    return models\n\ndef predict_on_spectrogram(audio_path, models, cfg, species_ids):\n    \"\"\"Process a single audio file and predict species presence for each 5-second segment\"\"\"\n    predictions = []\n    row_ids = []\n    soundscape_id = Path(audio_path).stem\n    \n    try:\n        print(f\"Processing {soundscape_id}\")\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS)\n        \n        total_segments = int(len(audio_data) / (cfg.FS * cfg.WINDOW_SIZE))\n        \n        for segment_idx in range(total_segments):\n            start_sample = segment_idx * cfg.FS * cfg.WINDOW_SIZE\n            end_sample = start_sample + cfg.FS * cfg.WINDOW_SIZE\n            segment_audio = audio_data[start_sample:end_sample]\n            \n            end_time_sec = (segment_idx + 1) * cfg.WINDOW_SIZE\n            row_id = f\"{soundscape_id}_{end_time_sec}\"\n            row_ids.append(row_id)\n\n            if cfg.use_tta:\n                all_preds = []\n                \n                for tta_idx in range(cfg.tta_count):\n                    mel_spec = process_audio_segment(segment_audio, cfg)\n                    mel_spec = apply_tta(mel_spec, tta_idx)\n\n                    mel_spec = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    mel_spec = mel_spec.to(cfg.device)\n\n                    if len(models) == 1:\n                        with torch.no_grad():\n                            outputs = models[0](mel_spec)\n                            probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                            all_preds.append(probs)\n                    else:\n                        segment_preds = []\n                        for model in models:\n                            with torch.no_grad():\n                                outputs = model(mel_spec)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n                        \n                        avg_preds = np.mean(segment_preds, axis=0)\n                        all_preds.append(avg_preds)\n\n                final_preds = np.mean(all_preds, axis=0)\n            else:\n                mel_spec = process_audio_segment(segment_audio, cfg)\n                \n                mel_spec = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                mel_spec = mel_spec.to(cfg.device)\n                \n                if len(models) == 1:\n                    with torch.no_grad():\n                        outputs = models[0](mel_spec)\n                        final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                else:\n                    segment_preds = []\n                    for model in models:\n                        with torch.no_grad():\n                            outputs = model(mel_spec)\n                            probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                            segment_preds.append(probs)\n\n                    final_preds = np.mean(segment_preds, axis=0)\n                    \n            predictions.append(final_preds)\n            \n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n    \n    return row_ids, predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.469531Z","iopub.execute_input":"2025-05-19T08:05:02.469966Z","iopub.status.idle":"2025-05-19T08:05:02.492876Z","shell.execute_reply.started":"2025-05-19T08:05:02.469923Z","shell.execute_reply":"2025-05-19T08:05:02.491616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_tta(spec, tta_idx):\n    \"\"\"Apply test-time augmentation\"\"\"\n    if tta_idx == 0:\n        # Original spectrogram\n        return spec\n    elif tta_idx == 1:\n        # Time shift (horizontal flip)\n        return np.flip(spec, axis=1)\n    elif tta_idx == 2:\n        # Frequency shift (vertical flip)\n        return np.flip(spec, axis=0)\n    else:\n        return spec\n\ndef run_inference(cfg, models, species_ids):\n    \"\"\"Run inference on all test soundscapes\"\"\"\n    test_files = list(Path(cfg.test_soundscapes).glob('*.ogg'))\n    \n    if cfg.debug:\n        print(f\"Debug mode enabled, using only {cfg.debug_count} files\")\n        test_files = test_files[:cfg.debug_count]\n    \n    print(f\"Found {len(test_files)} test soundscapes\")\n\n    all_row_ids = []\n    all_predictions = []\n\n    for audio_path in tqdm(test_files):\n        row_ids, predictions = predict_on_spectrogram(str(audio_path), models, cfg, species_ids)\n        all_row_ids.extend(row_ids)\n        all_predictions.extend(predictions)\n    \n    return all_row_ids, all_predictions\n\ndef create_submission(row_ids, predictions, species_ids, cfg):\n    \"\"\"Create submission dataframe\"\"\"\n    print(\"Creating submission dataframe...\")\n\n    submission_dict = {'row_id': row_ids}\n    \n    for i, species in enumerate(species_ids):\n        submission_dict[species] = [pred[i] for pred in predictions]\n\n    submission_df = pd.DataFrame(submission_dict)\n\n    submission_df.set_index('row_id', inplace=True)\n\n    sample_sub = pd.read_csv(cfg.submission_csv, index_col='row_id')\n\n    missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n    if missing_cols:\n        print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n        for col in missing_cols:\n            submission_df[col] = 0.0\n\n    submission_df = submission_df[sample_sub.columns]\n\n    submission_df = submission_df.reset_index()\n    \n    return submission_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.494223Z","iopub.execute_input":"2025-05-19T08:05:02.494587Z","iopub.status.idle":"2025-05-19T08:05:02.519687Z","shell.execute_reply.started":"2025-05-19T08:05:02.494542Z","shell.execute_reply":"2025-05-19T08:05:02.518352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    start_time = time.time()\n    print(\"Starting BirdCLEF-2025 inference...\")\n    print(f\"TTA enabled: {cfg.use_tta} (variations: {cfg.tta_count if cfg.use_tta else 0})\")\n\n    models = load_models(cfg, num_classes)\n    \n    if not models:\n        print(\"No models found! Please check model paths.\")\n        return\n    \n    print(f\"Model usage: {'Single model' if len(models) == 1 else f'Ensemble of {len(models)} models'}\")\n\n    row_ids, predictions = run_inference(cfg, models, species_ids)\n\n    submission_df = create_submission(row_ids, predictions, species_ids, cfg)\n\n    submission_path = 'submission1.csv'\n    submission_df.to_csv(submission_path, index=False)\n    print(f\"Submission saved to {submission_path}\")\n    \n    end_time = time.time()\n    print(f\"Inference completed in {(end_time - start_time)/60:.2f} minutes\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:02.522049Z","iopub.execute_input":"2025-05-19T08:05:02.522504Z","iopub.status.idle":"2025-05-19T08:05:04.0711Z","shell.execute_reply.started":"2025-05-19T08:05:02.522462Z","shell.execute_reply":"2025-05-19T08:05:04.069941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv('submission1.csv')\ncols = sub.columns[1:]\ngroups = sub['row_id'].str.rsplit('_', n=1).str[0]\ngroups = groups.values\nfor group in np.unique(groups):\n    sub_group = sub[group == groups]\n    predictions = sub_group[cols].values\n    new_predictions = predictions.copy()\n    for i in range(1, predictions.shape[0]-1):\n        new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n    new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n    new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n    sub_group[cols] = new_predictions\n    sub[group == groups] = sub_group\nsub.to_csv(\"submission1.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.072812Z","iopub.execute_input":"2025-05-19T08:05:04.07314Z","iopub.status.idle":"2025-05-19T08:05:04.108643Z","shell.execute_reply.started":"2025-05-19T08:05:04.073112Z","shell.execute_reply":"2025-05-19T08:05:04.107393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission2(nfnet)》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom typing import Union\n\ndef apply_power_to_low_ranked_cols(\n    p: np.ndarray,\n    top_k: int = 30,\n    exponent: Union[int, float] = 2,\n    inplace: bool = True\n) -> np.ndarray:\n    \"\"\"\n    Rank columns by their column‑wise maximum and raise every column whose\n    rank falls below `top_k` to a given power.\n\n    Parameters\n    ----------\n    p : np.ndarray\n        A 2‑D array of shape **(n_chunks, n_classes)**.\n\n        - **n_chunks** is the number of fixed‑length time chunks obtained\n          after slicing the input audio (or other sequential data).  \n          *Example:* In the BirdCLEF `test_soundscapes` set, each file is\n          60 s long. If you extract non‑overlapping 5 s windows,  \n          `n_chunks = 60 s / 5 s = 12`.\n        - **n_classes** is the number of classes being predicted.\n        - Each element `p[i, j]` is the score or probability of class *j*\n          in chunk *i*.\n\n    top_k : int, default=30\n        The highest‑ranked columns (by their maximum value) that remain\n        unchanged.\n\n    exponent : int or float, default=2\n        The power applied to the selected low‑ranked columns  \n        (e.g. `2` squares, `0.5` takes the square root, `3` cubes).\n\n    inplace : bool, default=True\n        If `True`, modify `p` in place.  \n        If `False`, operate on a copy and leave the original array intact.\n\n    Returns\n    -------\n    np.ndarray\n        The transformed array. It is the same object as `p` when\n        `inplace=True`; otherwise, it is a new array.\n\n    \"\"\"\n    if not inplace:\n        p = p.copy()\n\n    # Identify columns whose max value ranks below `top_k`\n    tail_cols = np.argsort(-p.max(axis=0))[top_k:]\n\n    # Apply the power transformation to those columns\n    p[:, tail_cols] = p[:, tail_cols] ** exponent\n    return p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.110014Z","iopub.execute_input":"2025-05-19T08:05:04.110484Z","iopub.status.idle":"2025-05-19T08:05:04.117843Z","shell.execute_reply.started":"2025-05-19T08:05:04.110449Z","shell.execute_reply":"2025-05-19T08:05:04.116484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport timm\nimport torch.nn.functional as F\nimport torchaudio\nimport torchaudio.transforms as AT\nfrom contextlib import contextmanager\nimport concurrent.futures","metadata":{"papermill":{"duration":12.984639,"end_time":"2025-03-12T14:13:00.145177","exception":false,"start_time":"2025-03-12T14:12:47.160538","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.119263Z","iopub.execute_input":"2025-05-19T08:05:04.119939Z","iopub.status.idle":"2025-05-19T08:05:04.182152Z","shell.execute_reply.started":"2025-05-19T08:05:04.119896Z","shell.execute_reply":"2025-05-19T08:05:04.18075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_audio_dir = '../input/birdclef-2025/test_soundscapes/'\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nfile_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n\ndebug = False\nprint('Debug mode:', debug)\nprint('Number of test soundscapes:', len(file_list))","metadata":{"papermill":{"duration":0.105385,"end_time":"2025-03-12T14:13:00.253425","exception":false,"start_time":"2025-03-12T14:13:00.14804","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.183292Z","iopub.execute_input":"2025-05-19T08:05:04.183676Z","iopub.status.idle":"2025-05-19T08:05:04.211676Z","shell.execute_reply.started":"2025-05-19T08:05:04.183649Z","shell.execute_reply":"2025-05-19T08:05:04.210333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wav_sec = 5\nsample_rate = 32000\nmin_segment = sample_rate*wav_sec\n\nclass_labels = sorted(os.listdir('../input/birdclef-2025/train_audio/'))\n\nn_fft=1024\nwin_length=1024\nhop_length=512\nf_min=50\nf_max=16000\nn_mels=128\n\nmel_spectrogram = AT.MelSpectrogram(\n    sample_rate=sample_rate,\n    n_fft=n_fft,\n    win_length=win_length,\n    hop_length=hop_length,\n    center=True,\n    f_min=f_min,\n    f_max=f_max,\n    pad_mode=\"reflect\",\n    power=2.0,\n    norm='slaney',\n    n_mels=n_mels,\n    mel_scale=\"htk\",\n    # normalized=True\n)\n\ndef normalize_std(spec, eps=1e-6):\n    mean = torch.mean(spec)\n    std = torch.std(spec)\n    return torch.where(std == 0, spec-mean, (spec - mean) / (std+eps))\n\ndef audio_to_mel(filepath=None):\n    waveform, sample_rate = torchaudio.load(filepath,backend=\"soundfile\")\n    len_wav = waveform.shape[1]\n    waveform = waveform[0,:].reshape(1, len_wav) # stereo->mono mono->mono\n    PREDS = []\n    for i in range(12):\n        waveform2 = waveform[:,i*sample_rate*5:i*sample_rate*5+sample_rate*5]\n        melspec = mel_spectrogram(waveform2)\n        melspec = torch.log(melspec+1e-6)\n        melspec = normalize_std(melspec)\n        melspec = torch.unsqueeze(melspec, dim=0)\n        \n        PREDS.append(melspec)\n    return torch.vstack(PREDS)","metadata":{"papermill":{"duration":0.144235,"end_time":"2025-03-12T14:13:00.400505","exception":false,"start_time":"2025-03-12T14:13:00.25627","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.215928Z","iopub.execute_input":"2025-05-19T08:05:04.21635Z","iopub.status.idle":"2025-05-19T08:05:04.237828Z","shell.execute_reply.started":"2025-05-19T08:05:04.216318Z","shell.execute_reply":"2025-05-19T08:05:04.236665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def init_layer(layer):\n    nn.init.xavier_uniform_(layer.weight)\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\n\ndef init_bn(bn):\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.0)\n\n\ndef init_weights(model):\n    classname = model.__class__.__name__\n    if classname.find(\"Conv2d\") != -1:\n        nn.init.xavier_uniform_(model.weight, gain=np.sqrt(2))\n        model.bias.data.fill_(0)\n    elif classname.find(\"BatchNorm\") != -1:\n        model.weight.data.normal_(1.0, 0.02)\n        model.bias.data.fill_(0)\n    elif classname.find(\"GRU\") != -1:\n        for weight in model.parameters():\n            if len(weight.size()) > 1:\n                nn.init.orghogonal_(weight.data)\n    elif classname.find(\"Linear\") != -1:\n        model.weight.data.normal_(0, 0.01)\n        model.bias.data.zero_()\n\n\ndef interpolate(x, ratio):\n    (batch_size, time_steps, classes_num) = x.shape\n    upsampled = x[:, :, None, :].repeat(1, 1, ratio, 1)\n    upsampled = upsampled.reshape(batch_size, time_steps * ratio, classes_num)\n    return upsampled\n\n\ndef pad_framewise_output(framewise_output, frames_num):\n    output = F.interpolate(\n        framewise_output.unsqueeze(1),\n        size=(frames_num, framewise_output.size(2)),\n        align_corners=True,\n        mode=\"bilinear\").squeeze(1)\n\n    return output\n\n\nclass AttBlockV2(nn.Module):\n    def __init__(self,\n                 in_features: int,\n                 out_features: int,\n                 activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n        return x, norm_att, cla\n\n    def nonlinear_transform(self, x):\n        if self.activation == 'linear':\n            return x\n        elif self.activation == 'sigmoid':\n            return torch.sigmoid(x)\n\n\nclass TimmSED(nn.Module):\n    def __init__(self, base_model_name: str, pretrained=False, num_classes=24, in_channels=1, n_mels=24):\n        super().__init__()\n\n        self.bn0 = nn.BatchNorm2d(n_mels)\n\n        base_model = timm.create_model(\n            base_model_name, pretrained=pretrained, in_chans=in_channels)\n        layers = list(base_model.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n\n        in_features = base_model.num_features\n\n        self.fc1 = nn.Linear(in_features, in_features, bias=True)\n        self.att_block2 = AttBlockV2(\n            in_features, num_classes, activation=\"sigmoid\")\n\n        self.init_weight()\n\n    def init_weight(self):\n        init_bn(self.bn0)\n        init_layer(self.fc1)\n        \n\n    def forward(self, input_data):\n        x = input_data.transpose(2,3)\n        x = torch.cat((x,x,x),1)\n\n        x = x.transpose(2, 3)\n\n        x = self.encoder(x)\n        \n        x = torch.mean(x, dim=2)\n\n        x1 = F.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = F.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n\n        x = x.transpose(1, 2)\n        x = F.relu_(self.fc1(x))\n        x = x.transpose(1, 2)\n\n        (clipwise_output, norm_att, segmentwise_output) = self.att_block2(x)\n        logit = torch.sum(norm_att * self.att_block2.cla(x), dim=2)\n\n        output_dict = {\n            'logit': logit,\n        }\n\n        return output_dict","metadata":{"papermill":{"duration":2.175154,"end_time":"2025-03-12T14:13:02.578522","exception":false,"start_time":"2025-03-12T14:13:00.403368","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.239823Z","iopub.execute_input":"2025-05-19T08:05:04.240263Z","iopub.status.idle":"2025-05-19T08:05:04.264976Z","shell.execute_reply.started":"2025-05-19T08:05:04.24022Z","shell.execute_reply":"2025-05-19T08:05:04.26348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model_name='eca_nfnet_l0'\npretrained=False\nin_channels=3\n\nMODELS = [f'/kaggle/input/birdclef-2025-sed-models-p/sed{i}.pth' for i in range(3)]\n\nMODELS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.266534Z","iopub.execute_input":"2025-05-19T08:05:04.267121Z","iopub.status.idle":"2025-05-19T08:05:04.293715Z","shell.execute_reply.started":"2025-05-19T08:05:04.267083Z","shell.execute_reply":"2025-05-19T08:05:04.292547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = []\nfor path in MODELS:\n    model = TimmSED(base_model_name=base_model_name,\n               pretrained=pretrained,\n               num_classes=len(class_labels),\n               in_channels=in_channels,\n               n_mels=n_mels);\n    model.load_state_dict(torch.load(path, weights_only=True, map_location=torch.device('cpu')))\n    model.eval();\n    models.append(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:04.294981Z","iopub.execute_input":"2025-05-19T08:05:04.295367Z","iopub.status.idle":"2025-05-19T08:05:06.153157Z","shell.execute_reply.started":"2025-05-19T08:05:04.295323Z","shell.execute_reply":"2025-05-19T08:05:06.151698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prediction(afile):    \n    global pred\n    path = test_audio_dir + afile + '.ogg'\n    with torch.inference_mode():\n        sig = audio_to_mel(path)\n        outputs = None\n        for model in models:\n            model.eval()\n            p = model(sig)\n            p = torch.sigmoid(p['logit']).detach().cpu().numpy() \n            p = apply_power_to_low_ranked_cols(p, top_k=30,exponent=2)\n            if outputs is None: outputs = p\n            else: outputs += p\n            \n        outputs /= len(models)\n        chunks = [[] for i in range(12)]\n        for i in range(len(chunks)):        \n            chunk_end_time = (i + 1) * 5\n            row_id = afile + '_' + str(chunk_end_time)\n            pred['row_id'].append(row_id)\n            bird_no = 0\n            for bird in class_labels:         \n                pred[bird].append(outputs[i,bird_no])\n                bird_no += 1\n        gc.collect()","metadata":{"papermill":{"duration":0.011209,"end_time":"2025-03-12T14:13:02.593243","exception":false,"start_time":"2025-03-12T14:13:02.582034","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.154261Z","iopub.execute_input":"2025-05-19T08:05:06.154592Z","iopub.status.idle":"2025-05-19T08:05:06.162535Z","shell.execute_reply.started":"2025-05-19T08:05:06.154564Z","shell.execute_reply":"2025-05-19T08:05:06.161315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = {'row_id': []}\nfor species_code in class_labels:\n    pred[species_code] = []\n    \nstart = time.time()\nwith concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor:\n    _ = list(executor.map(prediction, file_list))\nend_t = time.time()\n\nif debug == True:\n    print(700*(end_t - start)/60/debug_num)","metadata":{"papermill":{"duration":6.823541,"end_time":"2025-03-12T14:13:09.419521","exception":false,"start_time":"2025-03-12T14:13:02.59598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.163821Z","iopub.execute_input":"2025-05-19T08:05:06.16425Z","iopub.status.idle":"2025-05-19T08:05:06.18475Z","shell.execute_reply.started":"2025-05-19T08:05:06.164205Z","shell.execute_reply":"2025-05-19T08:05:06.183544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = pd.DataFrame(pred, columns = ['row_id'] + class_labels) \ndisplay(results.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.185713Z","iopub.execute_input":"2025-05-19T08:05:06.186062Z","iopub.status.idle":"2025-05-19T08:05:06.217557Z","shell.execute_reply.started":"2025-05-19T08:05:06.186022Z","shell.execute_reply":"2025-05-19T08:05:06.216272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results.to_csv(\"submission.csv\", index=False)    \n\nsub = pd.read_csv('submission.csv')\ncols = sub.columns[1:]\ngroups = sub['row_id'].str.rsplit('_', n=1).str[0]\ngroups = groups.values\nfor group in np.unique(groups):\n    sub_group = sub[group == groups]\n    predictions = sub_group[cols].values\n    new_predictions = predictions.copy()\n    for i in range(1, predictions.shape[0]-1):\n        new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n    new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n    new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n    sub_group[cols] = new_predictions\n    sub[group == groups] = sub_group\nsub.to_csv(\"submission.csv\", index=False)\n\n\nif debug:\n    display(results)","metadata":{"papermill":{"duration":0.097214,"end_time":"2025-03-12T14:13:09.519812","exception":false,"start_time":"2025-03-12T14:13:09.422598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.218426Z","iopub.execute_input":"2025-05-19T08:05:06.218819Z","iopub.status.idle":"2025-05-19T08:05:06.274236Z","shell.execute_reply.started":"2025-05-19T08:05:06.218779Z","shell.execute_reply":"2025-05-19T08:05:06.272572Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Finaly Blending》》》\n</b></h1> ","metadata":{}},{"cell_type":"code","source":"# ------------------------------------------- #\n# [IMPORTANT]\n# * Blending Weight\n# ------------------------------------------- #\nsub_w=[0.6,0.4]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.275771Z","iopub.execute_input":"2025-05-19T08:05:06.276374Z","iopub.status.idle":"2025-05-19T08:05:06.28183Z","shell.execute_reply.started":"2025-05-19T08:05:06.276329Z","shell.execute_reply":"2025-05-19T08:05:06.280227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_TARGETs = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\nlist_targets_0 = [f'{TARGET} 0' for TARGET in list_TARGETs]\nlist_targets_1 = [f'{TARGET} 1' for TARGET in list_TARGETs]\n\ndf0 = pd.read_csv(\"/kaggle/working/submission.csv\")\ndf1 = pd.read_csv(\"/kaggle/working/submission1.csv\")\n\ndf0 = df0.rename(columns={TARGET : f'{TARGET} 0' for TARGET in list_TARGETs})\ndf1 = df1.rename(columns={TARGET : f'{TARGET} 1' for TARGET in list_TARGETs})\n\ndfs = pd.merge(df0,df1,on=['row_id'])\n\nfor i in range(len(list_TARGETs)):\n    dfs[list_TARGETs[i]] = dfs[list_targets_0[i]]*sub_w[0] + sub_w[1]*dfs[list_targets_1[i]]\n             \nfor col0,col1 in zip(list_targets_0, list_targets_1):\n    del dfs[col0]\n    del dfs[col1]\n    \n    \ndfs.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T08:05:06.283063Z","iopub.execute_input":"2025-05-19T08:05:06.283482Z","iopub.status.idle":"2025-05-19T08:05:08.285608Z","shell.execute_reply.started":"2025-05-19T08:05:06.283453Z","shell.execute_reply":"2025-05-19T08:05:08.284034Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}