{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11053663,"sourceType":"datasetVersion","datasetId":6886569},{"sourceId":12109390,"sourceType":"datasetVersion","datasetId":7624096},{"sourceId":12111412,"sourceType":"datasetVersion","datasetId":7625450}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport timm\nimport cv2\nfrom tqdm import tqdm\nfrom collections import defaultdict","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-09T20:46:11.441809Z","iopub.execute_input":"2025-06-09T20:46:11.442092Z","iopub.status.idle":"2025-06-09T20:46:30.600121Z","shell.execute_reply.started":"2025-06-09T20:46:11.442068Z","shell.execute_reply":"2025-06-09T20:46:30.59916Z"}},"outputs":[],"execution_count":1},{"cell_type":"markdown","source":"# get pseudo label data","metadata":{}},{"cell_type":"code","source":"class CFG:\n    test_soundscapes = \"/kaggle/input/birdclef-2025/train_soundscapes\"\n    taxonomy_csv     = \"/kaggle/input/birdclef-2025/taxonomy.csv\"\n    model_dir        = \"/kaggle/input/efficientnetv2-in21k-focal\"\n    SR = 32000\n    WINDOW_SIZE = 5\n    N_FFT = 1034\n    HOP_LENGTH = 64\n    N_MELS = 136\n    FMIN = 20\n    FMAX = 16000\n    TARGET_SHAPE = (256, 256)\n    model_name = 'tf_efficientnetv2_s.in21k_ft_in1k'\n    use_all_folds = False\n    in_channels = 1\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    pseudo_label_threshold = 0.93 # for fallback in case dynamic threshold fails\n\ncfg = CFG()\n\ntaxonomy_df = pd.read_csv(cfg.taxonomy_csv)\nspecies_ids = taxonomy_df['primary_label'].tolist()\nnum_classes = len(species_ids)\nprint(f\"Loaded taxonomy with {num_classes} species classes.\")\nprint(f\"Using device: {cfg.device}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df_original = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\nlabel_counts = train_df_original['primary_label'].value_counts().to_dict()\n\nmin_thresh = 0.55\nmax_thresh = 0.99\n\nmax_count = max(label_counts.values())\nmin_count = min(label_counts.values())\n\nper_class_thresholds = {}\nfor label, count in label_counts.items():\n    commonness = (count - min_count) / (max_count - min_count + 1e-8)  # 0 = rare, 1 = common\n    per_class_thresholds[label] = min_thresh + (commonness ** 0.5) * (max_thresh - min_thresh)  # Apply square root\n\n# Sanity check\nprint(\"Dynamic threshold example:\")\nfor label in sorted(label_counts, key=label_counts.get)[:5]:  # 5 least frequent\n    print(f\"{label} (rare): {per_class_thresholds[label]:.3f}\")\nfor label in sorted(label_counts, key=label_counts.get, reverse=True)[:5]:  # 5 most frequent\n    print(f\"{label} (common): {per_class_thresholds[label]:.3f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFModel(nn.Module):\n    def __init__(self, cfg, num_classes):\n        super().__init__()\n        self.backbone = timm.create_model(\n            cfg.model_name,\n            pretrained=True,\n            in_chans=cfg.in_channels,\n            drop_rate=0.0,\n            drop_path_rate=0.0,\n        )\n        \n        if 'efficientnet' in cfg.model_name:\n            backbone_out = self.backbone.classifier.in_features\n            self.backbone.classifier = nn.Identity()\n        elif 'resnet' in cfg.model_name:\n            backbone_out = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        else:\n            backbone_out = self.backbone.get_classifier().in_features\n            self.backbone.reset_classifier(0, '') \n        \n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        self.classifier = nn.Linear(backbone_out, num_classes)\n    \n    def forward(self, x):\n        features = self.backbone(x)\n        if isinstance(features, dict):\n            features = features.get('features', features.get('out', features))\n        if features.dim() == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        logits = self.classifier(features)\n        return logits\n\nmodel = BirdCLEFModel(cfg, num_classes=num_classes)\nmodel.to(cfg.device)\nmodel.eval() ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio_to_melspec(audio_data, cfg):\n    if np.isnan(audio_data).any():\n        mean_val = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_val)\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data, sr=cfg.SR, n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH, n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN, fmax=cfg.FMAX, power=2.0\n    )\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    return mel_spec_norm\n\ndef process_audio_segment(audio_segment, cfg):\n    segment_length = cfg.SR * cfg.WINDOW_SIZE\n    if len(audio_segment) < segment_length:\n        audio_segment = np.pad(audio_segment, (0, segment_length - len(audio_segment)), mode='constant')\n    mel = audio_to_melspec(audio_segment, cfg)\n    if mel.shape != cfg.TARGET_SHAPE:\n        mel = cv2.resize(mel, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n    return mel.astype(np.float32)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_preds = defaultdict(list)\npseudo_mels = {}\ntotal_segments = 0\nTOP_N = 3\n\naudio_files = sorted(Path(cfg.test_soundscapes).glob(\"*.ogg\"))\nprint(f\"Processing {len(audio_files)} files...\")\n\nwith torch.no_grad():\n    for audio_path in tqdm(audio_files, desc=\"Audio files\", dynamic_ncols=True):\n        audio_name = audio_path.name\n        y, sr = librosa.load(audio_path, sr=cfg.SR)\n        if sr != cfg.SR:\n            y = librosa.resample(y, orig_sr=sr, target_sr=cfg.SR)\n\n        seg_samples = cfg.SR * cfg.WINDOW_SIZE  # 5 seconds\n        total_segments += 1  # only one segment per file\n\n        if len(y) < seg_samples:\n            n_repeat = int(np.ceil(seg_samples / len(y)))\n            y = np.tile(y, n_repeat)\n        \n        start = max(0, len(y) // 2 - seg_samples // 2)\n        end = start + seg_samples\n        segment = y[start:end]\n\n        if len(segment) < seg_samples:\n            segment = np.pad(segment, (0, seg_samples - len(segment)), mode='constant')\n\n        mel = process_audio_segment(segment, cfg)\n        tensor = torch.tensor(mel).unsqueeze(0).unsqueeze(0).to(cfg.device)\n\n        probs = torch.sigmoid(model(tensor)).cpu().numpy().squeeze()\n        top_n_indices = np.argsort(probs)[-TOP_N:][::-1]\n\n        for idx in top_n_indices:\n            prob = float(probs[idx])\n            label = species_ids[idx]\n            threshold = per_class_thresholds.get(label, cfg.pseudo_label_threshold)\n\n            if prob >= threshold:\n                segment_name = f\"{audio_name.replace('.ogg', '')}_center5s.ogg\"\n                class_preds[label].append((prob, segment_name, mel))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Keep only top-K per class\nresults = []\nresults_with_conf = []  # for confidence analysis\nMAX_PER_CLASS = 400  # configurable cap\n\nfor label, entries in class_preds.items():\n    top_entries = sorted(entries, key=lambda x: -x[0])[:MAX_PER_CLASS]\n    for prob, segment_name, mel in top_entries:\n        results.append([segment_name, label, prob])\n        results_with_conf.append({\n            \"filename\": segment_name,\n            \"primary_label\": label,\n            \"confidence\": prob,\n            \"threshold\": per_class_thresholds.get(label, cfg.pseudo_label_threshold)\n        })\n        pseudo_mels[segment_name.replace('.ogg', '')] = mel\n\nprint(f\"Done. Collected {len(results)} pseudo-labels from {total_segments} segments.\")\n\n# DEBUG: show that our keys line up with the filenames in results\nprint(\"First 10 filenames from results (with .ogg):\")\nprint([row[0] for row in results[:10]])\nprint(\"First 10 keys in pseudo_mels dict:\")\nprint(list(pseudo_mels.keys())[:10])\n\nprint(\"\\nCheck each of the first 10:\")\nfor fn, lbl, prob in results[:10]:\n    k = fn.replace(\".ogg\",\"\")\n    print(f\"{fn}  → key='{k}'  in dict? {k in pseudo_mels}\")\n\nnp.save(\"pseudo_mels.npy\", pseudo_mels)\n\n# Save confidence info separately for analysis\nconf_df = pd.DataFrame(results_with_conf)\nconf_df.to_csv(\"pseudo_label_confidences.csv\", index=False)\nprint(f\"Saved {len(conf_df)} entries with confidence scores to 'pseudo_label_confidences.csv'\")\n\n# Save train.csv-compatible format\npseudo_train = pd.DataFrame({\n    \"filename\": [row[0] for row in results],\n    \"primary_label\": [row[1] for row in results],\n    \"secondary_labels\": [[] for _ in results],\n    \"latitude\": [None] * len(results),\n    \"longitude\": [None] * len(results),\n    \"author\": [\"pseudo\"] * len(results),\n    \"rating\": [0] * len(results),\n    \"collection\": [\"pseudo\"] * len(results)\n})\n\npseudo_train.to_csv(\"pseudo_train.csv\", index=False)\nprint(f\"Saved {len(pseudo_train)} pseudo-labeled samples to 'pseudo_train.csv'\")\npseudo_train.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_df = pd.read_csv(\"/kaggle/input/pseudo-label/pseudo_label_confidences.csv\")\nlabel_counts = conf_df['primary_label'].value_counts()\n\nplt.figure(figsize=(20, 5))\nplt.bar(range(len(label_counts)), label_counts.values)\nplt.title(\"Number of Samples per Species\")\nplt.xlabel(\"Species Index\")\nplt.ylabel(\"Sample Count\")\nplt.tight_layout()\nplt.show()\nlen(label_counts)\n\nplt.figure(figsize=(8, 5))\nsns.histplot(conf_df['confidence'], bins=30, kde=True)\nplt.title(\"Distribution of Confidence\")\nplt.xlabel(\"Confidence\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"import os\nimport logging\nimport random\nimport gc\nimport time\nimport cv2\nimport math\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport librosa\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.auto import tqdm\n\nimport timm\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    \n    seed = 42\n    debug = False  \n    apex = False\n    print_freq = 100\n    num_workers = 2\n\n    early_stopping_patience = 3  # stop if no improvement for 3 epochs\n    early_stopping_delta = 0.0001  # minimum improvement to count as progress\n    \n    OUTPUT_DIR = '/kaggle/working/'\n\n    train_datadir = '/kaggle/input/birdclef-2025/train_audio'\n    train_csv = '/kaggle/input/birdclef-2025/train.csv'\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n\n    spectrogram_npy = '/kaggle/input/birdclef25-mel-spectrograms/birdclef2025_melspec_5sec_256_256.npy'\n \n    model_name = 'efficientnet_b0'  \n    pretrained = True\n    in_channels = 1\n\n    LOAD_DATA = True  \n    FS = 32000\n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)\n    \n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    \n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    epochs = 10  \n    batch_size = 32  \n    criterion = 'BCEWithLogitsLoss'\n\n    n_fold = 5\n    selected_folds = [0, 1, 2, 3, 4]   \n\n    optimizer = 'AdamW'\n    lr = 5e-4 \n    weight_decay = 1e-5\n  \n    scheduler = 'CosineAnnealingLR'\n    min_lr = 1e-6\n    T_max = epochs\n\n    aug_prob = 0.5  \n    mixup_alpha = 0.5  \n    \n    def update_debug_settings(self):\n        if self.debug:\n            self.epochs = 2\n            self.selected_folds = [0]\n\ncfg = CFG()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_seed(seed=42):\n    \"\"\"\n    Set seed for reproducibility\n    \"\"\"\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(cfg.seed)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFDatasetFromNPY(Dataset):\n    def __init__(self, df, cfg, spectrograms=None, mode=\"train\"):\n        self.df = df\n        self.cfg = cfg\n        self.mode = mode\n\n        self.spectrograms = spectrograms\n        \n        taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = taxonomy_df['primary_label'].tolist()\n        self.num_classes = len(self.species_ids)\n        self.label_to_idx = {label: idx for idx, label in enumerate(self.species_ids)}\n\n        if 'filepath' not in self.df.columns:\n            self.df['filepath'] = self.cfg.train_datadir + '/' + self.df.filename\n        \n        if 'samplename' not in self.df.columns:\n            self.df['samplename'] = self.df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n\n        sample_names = set(self.df['samplename'])\n        if self.spectrograms:\n            found_samples = sum(1 for name in sample_names if name in self.spectrograms)\n            print(f\"Found {found_samples} matching spectrograms for {mode} dataset out of {len(self.df)} samples\")\n        \n        if cfg.debug:\n            self.df = self.df.sample(min(1000, len(self.df)), random_state=cfg.seed).reset_index(drop=True)\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        samplename = row['samplename']\n        spec = None\n\n        if self.spectrograms and samplename in self.spectrograms:\n            spec = self.spectrograms[samplename]\n        elif not self.cfg.LOAD_DATA:\n            spec = process_audio_file(row['filepath'], self.cfg)\n\n        if spec is None:\n            spec = np.zeros(self.cfg.TARGET_SHAPE, dtype=np.float32)\n            if self.mode == \"train\":  # Only print warning during training\n                print(f\"Warning: Spectrogram for {samplename} not found and could not be generated\")\n\n        spec = torch.tensor(spec, dtype=torch.float32).unsqueeze(0)  # Add channel dimension\n\n        if self.mode == \"train\" and random.random() < self.cfg.aug_prob:\n            spec = self.apply_spec_augmentations(spec)\n        \n        target = self.encode_label(row['primary_label'], smoothing=0.05)\n        \n        if 'secondary_labels' in row and row['secondary_labels'] not in [[''], None, np.nan]:\n            if isinstance(row['secondary_labels'], str):\n                secondary_labels = eval(row['secondary_labels'])\n            else:\n                secondary_labels = row['secondary_labels']\n            \n            for label in secondary_labels:\n                if label in self.label_to_idx:\n                    target[self.label_to_idx[label]] = 1.0\n        \n        return {\n            'melspec': spec, \n            'target': torch.tensor(target, dtype=torch.float32),\n            'filename': row['filename']\n        }\n    \n    def apply_spec_augmentations(self, spec):\n        \"\"\"Apply augmentations to spectrogram\"\"\"\n    \n        # Time masking (horizontal stripes)\n        if random.random() < 0.5:\n            num_masks = random.randint(1, 3)\n            for _ in range(num_masks):\n                width = random.randint(5, 20)\n                start = random.randint(0, spec.shape[2] - width)\n                spec[0, :, start:start+width] = 0\n        \n        # Frequency masking (vertical stripes)\n        if random.random() < 0.5:\n            num_masks = random.randint(1, 3)\n            for _ in range(num_masks):\n                height = random.randint(5, 20)\n                start = random.randint(0, spec.shape[1] - height)\n                spec[0, start:start+height, :] = 0\n        \n        # Random brightness/contrast\n        if random.random() < 0.5:\n            gain = random.uniform(0.8, 1.2)\n            bias = random.uniform(-0.1, 0.1)\n            spec = spec * gain + bias\n            spec = torch.clamp(spec, 0, 1) \n            \n        return spec\n    \n    def encode_label(self, label, smoothing=0.05):\n        \"\"\"Encode label to smoothed one-hot vector\"\"\"\n        target = np.full(self.num_classes, smoothing)  # smooth negatives\n        if label in self.label_to_idx:\n            target[self.label_to_idx[label]] = 1.0 - smoothing  # smooth positive\n        return target","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collate_fn(batch):\n    \"\"\"Custom collate function to handle different sized spectrograms\"\"\"\n    batch = [item for item in batch if item is not None]\n    if len(batch) == 0:\n        return {}\n        \n    result = {key: [] for key in batch[0].keys()}\n    \n    for item in batch:\n        for key, value in item.items():\n            result[key].append(value)\n    \n    for key in result:\n        if key == 'target' and isinstance(result[key][0], torch.Tensor):\n            result[key] = torch.stack(result[key])\n        elif key == 'melspec' and isinstance(result[key][0], torch.Tensor):\n            shapes = [t.shape for t in result[key]]\n            if len(set(str(s) for s in shapes)) == 1:\n                result[key] = torch.stack(result[key])\n    \n    return result","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFModel(nn.Module):\n    def __init__(self, cfg):\n        super().__init__()\n        self.cfg = cfg\n        \n        taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n        cfg.num_classes = len(taxonomy_df)\n        \n        self.backbone = timm.create_model(\n            cfg.model_name,\n            pretrained=cfg.pretrained,\n            in_chans=cfg.in_channels,\n            drop_rate=0.2,\n            drop_path_rate=0.2\n        )\n        \n        if 'efficientnet' in cfg.model_name:\n            backbone_out = self.backbone.classifier.in_features\n            self.backbone.classifier = nn.Identity()\n        elif 'resnet' in cfg.model_name:\n            backbone_out = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        else:\n            backbone_out = self.backbone.get_classifier().in_features\n            self.backbone.reset_classifier(0, '')\n        \n        self.pooling = nn.AdaptiveAvgPool2d(1)\n            \n        self.feat_dim = backbone_out\n        \n        self.classifier = nn.Linear(backbone_out, cfg.num_classes)\n        \n        self.mixup_enabled = hasattr(cfg, 'mixup_alpha') and cfg.mixup_alpha > 0\n        if self.mixup_enabled:\n            self.mixup_alpha = cfg.mixup_alpha\n            \n    def forward(self, x, targets=None):\n    \n        if self.training and self.mixup_enabled and targets is not None:\n            mixed_x, targets_a, targets_b, lam = self.mixup_data(x, targets)\n            x = mixed_x\n        else:\n            targets_a, targets_b, lam = None, None, None\n        \n        features = self.backbone(x)\n        \n        if isinstance(features, dict):\n            features = features['features']\n            \n        if len(features.shape) == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        \n        logits = self.classifier(features)\n        \n        if self.training and self.mixup_enabled and targets is not None:\n            loss = self.mixup_criterion(F.binary_cross_entropy_with_logits, \n                                       logits, targets_a, targets_b, lam)\n            return logits, loss\n            \n        return logits\n    \n    def mixup_data(self, x, targets):\n        \"\"\"Applies mixup to the data batch\"\"\"\n        batch_size = x.size(0)\n\n        lam = np.random.beta(self.mixup_alpha, self.mixup_alpha)\n\n        indices = torch.randperm(batch_size).to(x.device)\n\n        mixed_x = lam * x + (1 - lam) * x[indices]\n        \n        return mixed_x, targets, targets[indices], lam\n    \n    def mixup_criterion(self, criterion, pred, y_a, y_b, lam):\n        \"\"\"Applies mixup to the loss function\"\"\"\n        return lam * criterion(pred, y_a) + (1 - lam) * criterion(pred, y_b)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_optimizer(model, cfg):\n  \n    if cfg.optimizer == 'Adam':\n        optimizer = optim.Adam(\n            model.parameters(),\n            lr=cfg.lr,\n            weight_decay=cfg.weight_decay\n        )\n    elif cfg.optimizer == 'AdamW':\n        optimizer = optim.AdamW(\n            model.parameters(),\n            lr=cfg.lr,\n            weight_decay=cfg.weight_decay\n        )\n    elif cfg.optimizer == 'SGD':\n        optimizer = optim.SGD(\n            model.parameters(),\n            lr=cfg.lr,\n            momentum=0.9,\n            weight_decay=cfg.weight_decay\n        )\n    else:\n        raise NotImplementedError(f\"Optimizer {cfg.optimizer} not implemented\")\n        \n    return optimizer\n\ndef get_scheduler(optimizer, cfg):\n   \n    if cfg.scheduler == 'CosineAnnealingLR':\n        scheduler = lr_scheduler.CosineAnnealingLR(\n            optimizer,\n            T_max=cfg.T_max,\n            eta_min=cfg.min_lr\n        )\n    elif cfg.scheduler == 'ReduceLROnPlateau':\n        scheduler = lr_scheduler.ReduceLROnPlateau(\n            optimizer,\n            mode='min',\n            factor=0.5,\n            patience=2,\n            min_lr=cfg.min_lr,\n            verbose=True\n        )\n    elif cfg.scheduler == 'StepLR':\n        scheduler = lr_scheduler.StepLR(\n            optimizer,\n            step_size=cfg.epochs // 3,\n            gamma=0.5\n        )\n    elif cfg.scheduler == 'OneCycleLR':\n        scheduler = None  \n    else:\n        scheduler = None\n        \n    return scheduler\n\ndef get_criterion(cfg):\n \n    if cfg.criterion == 'BCEWithLogitsLoss':\n        criterion = nn.BCEWithLogitsLoss()\n    else:\n        raise NotImplementedError(f\"Criterion {cfg.criterion} not implemented\")\n        \n    return criterion","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, criterion, device, scheduler=None):\n    \n    model.train()\n    losses = []\n    all_targets = []\n    all_outputs = []\n    \n    pbar = tqdm(enumerate(loader), total=len(loader), desc=\"Training\")\n    \n    for step, batch in pbar:\n    \n        if isinstance(batch['melspec'], list):\n            batch_outputs = []\n            batch_losses = []\n            \n            for i in range(len(batch['melspec'])):\n                inputs = batch['melspec'][i].unsqueeze(0).to(device)\n                target = batch['target'][i].unsqueeze(0).to(device)\n                \n                optimizer.zero_grad()\n                output = model(inputs)\n                loss = criterion(output, target)\n                loss.backward()\n                \n                batch_outputs.append(output.detach().cpu())\n                batch_losses.append(loss.item())\n            \n            optimizer.step()\n            outputs = torch.cat(batch_outputs, dim=0).numpy()\n            loss = np.mean(batch_losses)\n            targets = batch['target'].numpy()\n            \n        else:\n            inputs = batch['melspec'].to(device)\n            targets = batch['target'].to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs)\n            \n            if isinstance(outputs, tuple):\n                outputs, loss = outputs  \n            else:\n                loss = criterion(outputs, targets)\n                \n            loss.backward()\n            optimizer.step()\n            \n            outputs = outputs.detach().cpu().numpy()\n            targets = targets.detach().cpu().numpy()\n        \n        if scheduler is not None and isinstance(scheduler, lr_scheduler.OneCycleLR):\n            scheduler.step()\n            \n        all_outputs.append(outputs)\n        all_targets.append(targets)\n        losses.append(loss if isinstance(loss, float) else loss.item())\n        \n        pbar.set_postfix({\n            'train_loss': np.mean(losses[-10:]) if losses else 0,\n            'lr': optimizer.param_groups[0]['lr']\n        })\n    \n    all_outputs = np.concatenate(all_outputs)\n    all_targets = np.concatenate(all_targets)\n    auc = calculate_auc(all_targets, all_outputs)\n    avg_loss = np.mean(losses)\n    \n    return avg_loss, auc\n\ndef validate(model, loader, criterion, device):\n   \n    model.eval()\n    losses = []\n    all_targets = []\n    all_outputs = []\n    \n    with torch.no_grad():\n        for batch in tqdm(loader, desc=\"Validation\"):\n            if isinstance(batch['melspec'], list):\n                batch_outputs = []\n                batch_losses = []\n                \n                for i in range(len(batch['melspec'])):\n                    inputs = batch['melspec'][i].unsqueeze(0).to(device)\n                    target = batch['target'][i].unsqueeze(0).to(device)\n                    \n                    output = model(inputs)\n                    loss = criterion(output, target)\n                    \n                    batch_outputs.append(output.detach().cpu())\n                    batch_losses.append(loss.item())\n                \n                outputs = torch.cat(batch_outputs, dim=0).numpy()\n                loss = np.mean(batch_losses)\n                targets = batch['target'].numpy()\n                \n            else:\n                inputs = batch['melspec'].to(device)\n                targets = batch['target'].to(device)\n                \n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n                \n                outputs = outputs.detach().cpu().numpy()\n                targets = targets.detach().cpu().numpy()\n            \n            all_outputs.append(outputs)\n            all_targets.append(targets)\n            losses.append(loss if isinstance(loss, float) else loss.item())\n    \n    all_outputs = np.concatenate(all_outputs)\n    all_targets = np.concatenate(all_targets)\n    \n    auc = calculate_auc(all_targets, all_outputs)\n    avg_loss = np.mean(losses)\n    \n    return avg_loss, auc\n\ndef calculate_auc(targets, outputs):\n    targets = np.array(targets)\n    outputs = np.array(outputs)\n    probs = torch.sigmoid(torch.tensor(outputs)).numpy()\n\n    # Round smoothed labels to binary for AUC computation\n    binary_targets = (targets >= 0.5).astype(int)\n\n    aucs = []\n    for i in range(binary_targets.shape[1]):\n        if np.sum(binary_targets[:, i]) == 0:\n            continue\n        try:\n            auc = roc_auc_score(binary_targets[:, i], probs[:, i])\n            aucs.append(auc)\n        except ValueError:\n            continue\n\n    return np.mean(aucs) if aucs else 0.5","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_training(df, cfg, spectrograms=None):\n    \"\"\"Training function that can either use pre-computed spectrograms or generate them on-the-fly\"\"\"\n\n    taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n    species_ids = taxonomy_df['primary_label'].tolist()\n    cfg.num_classes = len(species_ids)\n    \n    if cfg.debug:\n        cfg.update_debug_settings()\n\n    if cfg.LOAD_DATA:\n        if spectrograms is not None:\n            print(f\"Using {len(spectrograms)} pre-computed mel spectrograms passed into the function.\")\n        else:\n            print(\"Loading pre-computed mel spectrograms from NPY file...\")\n            try:\n                spectrograms = np.load(cfg.spectrogram_npy, allow_pickle=True).item()\n                print(f\"Loaded {len(spectrograms)} pre-computed mel spectrograms\")\n            except Exception as e:\n                print(f\"Error loading pre-computed spectrograms: {e}\")\n                print(\"Will generate spectrograms on-the-fly instead.\")\n                cfg.LOAD_DATA = False\n\n    \n    if not cfg.LOAD_DATA:\n        print(\"Will generate spectrograms on-the-fly during training.\")\n        if 'filepath' not in df.columns:\n            df['filepath'] = cfg.train_datadir + '/' + df.filename\n        if 'samplename' not in df.columns:\n            df['samplename'] = df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n        \n    # Mark which rows are pseudo-labeled (for filtering validation set only)\n    df[\"is_pseudo\"] = df[\"author\"].eq(\"pseudo\")\n    \n    skf = StratifiedKFold(n_splits=cfg.n_fold, shuffle=True, random_state=cfg.seed)\n    \n    best_scores = []\n    \n    for fold, (train_idx, val_idx) in enumerate(skf.split(df, df['primary_label'])):\n        if fold not in cfg.selected_folds:\n            continue\n    \n        print(f'\\n{\"=\"*30} Fold {fold} {\"=\"*30}')\n        \n        train_df = df.iloc[train_idx].reset_index(drop=True)\n        val_df = df.iloc[val_idx]\n    \n        # Remove pseudo-labels from validation set\n        val_df = val_df[val_df[\"is_pseudo\"] == False].reset_index(drop=True)\n    \n        # Drop the helper column before passing to the model\n        train_df = train_df.drop(columns=[\"is_pseudo\"])\n        val_df = val_df.drop(columns=[\"is_pseudo\"])\n    \n        print(f'Training set: {len(train_df)} samples')\n        print(f'Validation set: {len(val_df)} samples')\n\n        \n        train_dataset = BirdCLEFDatasetFromNPY(train_df, cfg, spectrograms=spectrograms, mode='train')\n        val_dataset = BirdCLEFDatasetFromNPY(val_df, cfg, spectrograms=spectrograms, mode='valid')\n        \n        train_loader = DataLoader(\n            train_dataset, \n            batch_size=cfg.batch_size, \n            shuffle=True, \n            num_workers=cfg.num_workers,\n            pin_memory=True,\n            collate_fn=collate_fn,\n            drop_last=True\n        )\n        \n        val_loader = DataLoader(\n            val_dataset, \n            batch_size=cfg.batch_size, \n            shuffle=False, \n            num_workers=cfg.num_workers,\n            pin_memory=True,\n            collate_fn=collate_fn\n        )\n        \n        model = BirdCLEFModel(cfg).to(cfg.device)\n        optimizer = get_optimizer(model, cfg)\n        criterion = get_criterion(cfg)\n        \n        if cfg.scheduler == 'OneCycleLR':\n            scheduler = lr_scheduler.OneCycleLR(\n                optimizer,\n                max_lr=cfg.lr,\n                steps_per_epoch=len(train_loader),\n                epochs=cfg.epochs,\n                pct_start=0.1\n            )\n        else:\n            scheduler = get_scheduler(optimizer, cfg)\n        \n        best_auc = 0\n        best_epoch = 0\n        \n        best_auc = 0\n        best_epoch = 0\n        patience_counter = 0  # for early stopping\n\n        for epoch in range(cfg.epochs):\n            print(f\"\\nEpoch {epoch+1}/{cfg.epochs}\")\n            \n            train_loss, train_auc = train_one_epoch(\n                model, \n                train_loader, \n                optimizer, \n                criterion, \n                cfg.device,\n                scheduler if isinstance(scheduler, lr_scheduler.OneCycleLR) else None\n            )\n            \n            val_loss, val_auc = validate(model, val_loader, criterion, cfg.device)\n\n            if scheduler is not None and not isinstance(scheduler, lr_scheduler.OneCycleLR):\n                if isinstance(scheduler, lr_scheduler.ReduceLROnPlateau):\n                    scheduler.step(val_loss)\n                else:\n                    scheduler.step()\n\n            print(f\"Train Loss: {train_loss:.4f}, Train AUC: {train_auc:.4f}\")\n            print(f\"Val Loss: {val_loss:.4f}, Val AUC: {val_auc:.4f}\")\n            \n            # Check if improvement is enough\n            if val_auc > best_auc + cfg.early_stopping_delta:\n                best_auc = val_auc\n                best_epoch = epoch + 1\n                patience_counter = 0  # reset counter\n                print(f\"New best AUC: {best_auc:.4f} at epoch {best_epoch}\")\n\n                torch.save({\n                    'model_state_dict': model.state_dict(),\n                    'optimizer_state_dict': optimizer.state_dict(),\n                    'scheduler_state_dict': scheduler.state_dict() if scheduler else None,\n                    'epoch': epoch,\n                    'val_auc': val_auc,\n                    'train_auc': train_auc,\n                    'cfg': cfg\n                }, f\"model_fold{fold}.pth\")\n            else:\n                patience_counter += 1\n                print(f\"No improvement. Patience: {patience_counter}/{cfg.early_stopping_patience}\")\n                if patience_counter >= cfg.early_stopping_patience:\n                    print(\"Early stopping triggered.\")\n                    break\n\n        \n        best_scores.append(best_auc)\n        print(f\"\\nBest AUC for fold {fold}: {best_auc:.4f} at epoch {best_epoch}\")\n        \n        # Clear memory\n        del model, optimizer, scheduler, train_loader, val_loader\n        torch.cuda.empty_cache()\n        gc.collect()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"Cross-Validation Results:\")\n    for fold, score in enumerate(best_scores):\n        print(f\"Fold {cfg.selected_folds[fold]}: {score:.4f}\")\n    print(f\"Mean AUC: {np.mean(best_scores):.4f}\")\n    print(\"=\"*60)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    import time\n    from pathlib import Path\n\n    print(\"\\nLoading training data...\")\n    train_df = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\n    print(f\">Loaded original train.csv with {len(train_df)} rows\")\n\n    # add pseudo-labeled data\n    try:\n        pseudo_df = pd.read_csv(\"/kaggle/input/pseudo-label/pseudo_train.csv\")\n        print(f\"Pseudo CSV sample:\\n{pseudo_df.head(3)}\")\n        train_df = pd.concat([train_df, pseudo_df], ignore_index=True)\n        print(f\"Loaded pseudo_train.csv and added {len(pseudo_df)} pseudo samples.\")\n    except Exception as e:\n        print(f\"Could not load pseudo_train.csv: {e}\")\n\n    print(f\"Total combined train_df rows before filtering: {len(train_df)}\")\n\n    # load mel spectrogram dicts\n    try:\n        original_mels = np.load(\n            \"/kaggle/input/birdclef25-mel-spectrograms/birdclef2025_melspec_5sec_256_256.npy\",\n            allow_pickle=True\n        ).item()\n        pseudo_mels = np.load(\n            \"/kaggle/input/pseudo-label/pseudo_mels.npy\",\n            allow_pickle=True\n        ).item()\n        print(f\"Loaded {len(original_mels)} original and {len(pseudo_mels)} pseudo spectrograms.\")\n    except Exception as e:\n        print(f\"Error loading spectrograms: {e}\")\n        original_mels = {}\n        pseudo_mels = {}\n\n    combined_mels = {**original_mels, **pseudo_mels}\n    cfg.LOAD_DATA = True\n    spectrograms = combined_mels\n    print(f\"Total spectrograms available: {len(spectrograms)}\")\n\n    # DEBUG: show examples from train_df and mel keys\n    print(\"\\nDEBUG PREVIEW\")\n    print(\"First 10 train_df filenames:\")\n    print(train_df['filename'].tolist()[:10])\n\n    stems = [Path(f).stem for f in train_df['filename'].tolist()[:10]]\n    print(\"Derived samplenames from those filenames:\")\n    print(stems)\n\n    print(\"First 10 keys in combined_mels:\")\n    print(list(combined_mels.keys())[:10])\n\n    # Build the same 'samplename' field your dataset class expects\n    train_df['samplename'] = train_df.filename.map(lambda x: x.replace('.ogg', '').replace('/', '-'))\n\n    # Keep only rows for which you have a precomputed mel\n    mask = train_df['samplename'].isin(combined_mels.keys())\n    train_df = train_df[mask].reset_index(drop=True)\n    print(f\"\\nFiltered dataset: {len(train_df)} samples with matching mel spectrograms.\")\n\n    # load taxonomy and start training\n    taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n\n    print(\"\\nStarting training...\")\n    run_training(train_df, cfg, spectrograms=spectrograms)\n    print(\"\\nTraining complete!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}