{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11053663,"sourceType":"datasetVersion","datasetId":6886569},{"sourceId":11918002,"sourceType":"datasetVersion","datasetId":7492364}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **BirdCLEF 2025 Training Notebook**\n\nThis is a baseline training pipeline for BirdCLEF 2025 using EfficientNetB0 with PyTorch and Timm(for pretrained EffNet). You can check inference and preprocessing notebooks in the following links: \n\n- [EfficientNet B0 Pytorch [Inference] | BirdCLEF'25](https://www.kaggle.com/code/kadircandrisolu/efficientnet-b0-pytorch-inference-birdclef-25)\n\n  \n- [Transforming Audio-to-Mel Spec. | BirdCLEF'25](https://www.kaggle.com/code/kadircandrisolu/transforming-audio-to-mel-spec-birdclef-25)  \n\nNote that by default this notebook is in Debug Mode, so it will only train the model with 2 epochs, but the [weight](https://www.kaggle.com/datasets/kadircandrisolu/birdclef25-effnetb0-starter-weight) I used in the inference notebook was obtained after 10 epochs of training.\n\n**Features**\n* Implement with Pytorch and Timm\n* Flexible audio processing with both pre-computed and on-the-fly mel spectrograms\n* Stratified 5-fold cross-validation with ensemble capability\n* Mixup training for improved generalization\n* Spectrogram augmentations (time/frequency masking, brightness adjustment)\n* AdamW optimizer with Cosine Annealing LR scheduling\n* Debug mode for quick experimentation with smaller datasets\n\n**Pre-computed Spectrograms**\nFor faster training, you can use pre-computed mel spectrograms from [this dataset](https://www.kaggle.com/datasets/kadircandrisolu/birdclef25-mel-spectrograms) by setting `LOAD_DATA = True`","metadata":{}},{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport logging\nimport random\nimport gc\nimport time\nimport cv2\nimport math\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport librosa\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport torchaudio\nfrom torchvision.ops import sigmoid_focal_loss\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.auto import tqdm\n\nimport timm\nimport lightgbm as lgb\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)\n\nclass FocalLoss(nn.Module):\n    def __init__(self, alpha=0.25, gamma=2.0, reduction='mean'):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n        self.reduction = reduction\n\n    def forward(self, inputs, targets):\n        return sigmoid_focal_loss(\n            inputs, targets,\n            alpha=self.alpha,\n            gamma=self.gamma,\n            reduction=self.reduction\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:05.815697Z","iopub.execute_input":"2025-06-05T08:01:05.816192Z","iopub.status.idle":"2025-06-05T08:01:05.827248Z","shell.execute_reply.started":"2025-06-05T08:01:05.816154Z","shell.execute_reply":"2025-06-05T08:01:05.825804Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    # Basic\n    seed = 42\n    debug = True  \n    apex = False\n    print_freq = 100\n    num_workers = 2\n\n    # Paths\n    OUTPUT_DIR = '/kaggle/working/'\n    train_datadir = '/kaggle/input/birdclef-2025/train_audio'\n    train_csv = '/kaggle/input/birdclef-2025/train.csv'\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    spectrogram_npy = '/kaggle/input/birdclef25-mel-spectrograms/birdclef2025_melspec_5sec_256_256.npy'\n    fabio_csv_path = '/kaggle/input/fabio-csv/fabio.csv'\n\n    # model\n    model_name = 'efficientnet_b0'  \n    pretrained = True\n    in_channels = 1\n    num_classes = None  # Will be set dynamically\n\n    # data\n    FS = 32000\n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)\n    \n    # Audio\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    \n    # Training\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    epochs = 10\n    batch_size = 32\n    criterion = 'FocalBCE'\n\n    # Augmentation\n    aug_prob = 1.0\n    mixup_prob = 0.6     # 1? 0.8? 0.6?\n    mixup_alpha = 0.4\n\n    def update_debug_settings(self):\n        if self.debug:\n            self.epochs = 2\n\ncfg = CFG()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:05.829258Z","iopub.execute_input":"2025-06-05T08:01:05.829672Z","iopub.status.idle":"2025-06-05T08:01:05.857854Z","shell.execute_reply.started":"2025-06-05T08:01:05.829618Z","shell.execute_reply":"2025-06-05T08:01:05.856546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"def set_seed(seed=42):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(cfg.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:05.859932Z","iopub.execute_input":"2025-06-05T08:01:05.860539Z","iopub.status.idle":"2025-06-05T08:01:05.892264Z","shell.execute_reply.started":"2025-06-05T08:01:05.860494Z","shell.execute_reply":"2025-06-05T08:01:05.890859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFDatasetFromNPY(Dataset):\n    def __init__(self, df, cfg, spectrograms=None, negative_spectrograms=None, mode=\"train\"):\n        self.df = df.copy()\n        self.cfg = cfg\n        self.mode = mode\n        self.spectrograms = spectrograms\n        self.negative_spectrograms = negative_spectrograms\n\n        # Load Fabio intervals\n        if os.path.exists(cfg.fabio_csv_path):\n            fabio_df = pd.read_csv(cfg.fabio_csv_path)\n            self.fabio_intervals = {row['filename']: (row['start'], row['stop']) for _, row in fabio_df.iterrows()}\n        else:\n            self.fabio_intervals = {}\n        \n        # Load taxonomy\n        taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = taxonomy_df['primary_label'].tolist()\n        self.num_classes = len(self.species_ids)\n        self.label_to_idx = {label: idx for idx, label in enumerate(self.species_ids)}\n        \n        # Update config\n        cfg.num_classes = self.num_classes\n\n        # File paths\n        if 'filepath' not in self.df.columns:\n            self.df['filepath'] = self.cfg.train_datadir + '/' + self.df['filename']\n\n        # Sample name: key of spectrogram dictionary\n        if 'samplename' not in self.df.columns:\n            self.df['samplename'] = self.df['filename'].map(\n                lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0]\n            )\n\n        # Add negative samples to dataframe if provided\n        if self.negative_spectrograms is not None and mode == \"train\":\n            negative_df = pd.DataFrame({\n                'filename': list(self.negative_spectrograms.keys()),\n                'primary_label': ['nocall'] * len(self.negative_spectrograms),\n                'samplename': list(self.negative_spectrograms.keys()),\n                'filepath': [''] * len(self.negative_spectrograms)  # Empty filepath for negative samples\n            })\n            self.df = pd.concat([self.df, negative_df], ignore_index=True)\n\n        # 클래스별 샘플 수 계산 (초기화 마지막 부분에 추가)\n        if mode == \"train\":\n            self.class_counts = self.df['primary_label'].value_counts().to_dict()\n            self.rare_threshold = 20\n            self.target_samples = 50\n            print(f\"Classes with < {self.rare_threshold} samples: {sum(1 for count in self.class_counts.values() if count < self.rare_threshold)}\")\n\n        # Debug mode\n        if cfg.debug:\n            self.df = self.df.sample(min(1000, len(self.df)), random_state=cfg.seed).reset_index(drop=True)\n            \n        # Check spectrograms availability\n        if self.spectrograms:\n            sample_names = set(self.df['samplename'])\n            found_samples = sum(1 for name in sample_names if name in self.spectrograms)\n            print(f\"Found {found_samples} matching positive spectrograms for {mode} dataset\")\n        \n        if self.negative_spectrograms:\n            neg_sample_names = set(self.negative_spectrograms.keys())\n            found_neg_samples = len(neg_sample_names)\n            print(f\"Found {found_neg_samples} negative spectrograms for {mode} dataset\")\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        samplename = row['samplename']\n        filename = row['filename']\n        primary_label = row['primary_label']\n        \n        # Get spectrogram\n        spec = self._get_spectrogram(samplename)\n        \n        if spec is None:\n            spec = np.zeros(self.cfg.TARGET_SHAPE, dtype=np.float32)\n            if self.mode == \"train\":\n                print(f\"Warning: No spectrogram found for {samplename}. Using zero spectrogram.\")\n\n        spec = torch.tensor(spec, dtype=torch.float32).unsqueeze(0)  # Add channel dimension\n\n        # Apply augmentations (primary_label 전달)\n        if self.mode == \"train\" and random.random() < self.cfg.aug_prob:\n            spec = self.apply_spec_augmentations(spec, primary_label)\n        \n        # Encode labels\n        target = self.encode_label(primary_label)\n        \n        # Handle secondary labels\n        if 'secondary_labels' in row and pd.notna(row['secondary_labels']) and row['secondary_labels'] != '':\n            secondary_labels = self._parse_secondary_labels(row['secondary_labels'])\n            for label in secondary_labels:\n                if label in self.label_to_idx:\n                    target[self.label_to_idx[label]] = 1.0\n\n        # Apply mixup\n        if self.mode == \"train\" and getattr(self.cfg, 'mixup_prob', 0) > 0 and random.random() < self.cfg.mixup_prob:\n            spec, target = self._apply_mixup(spec, target, idx)\n        \n        return {\n            'melspec': spec, \n            'target': torch.tensor(target, dtype=torch.float32),\n            'filename': filename\n        }\n    \n    def _get_spectrogram(self, samplename):\n        \"\"\"Get spectrogram from cache\"\"\"\n        if self.spectrograms and samplename in self.spectrograms:\n            return self.spectrograms[samplename]\n        elif self.negative_spectrograms and samplename in self.negative_spectrograms:\n            return self.negative_spectrograms[samplename]\n        else:\n            return None\n    \n    def _parse_secondary_labels(self, secondary_labels):\n        \"\"\"Parse secondary labels from string or list\"\"\"\n        if isinstance(secondary_labels, str):\n            try:\n                return eval(secondary_labels)\n            except:\n                return []\n        elif isinstance(secondary_labels, list):\n            return secondary_labels\n        return []\n\n    def _smart_mixup_pairing(self, idx, target):\n        \"\"\"동일하거나 유사한 클래스끼리 우선 페어링\"\"\"\n        current_classes = np.where(target > 0)[0]\n        \n        # Positive 샘플만 후보로 선택\n        positive_candidates = []\n        same_class_candidates = []\n        \n        for i, row in self.df.iterrows():\n            if i != idx and row['primary_label'] != 'nocall':\n                positive_candidates.append(i)\n                \n                candidate_target = self.encode_label(row['primary_label'])\n                candidate_classes = np.where(candidate_target > 0)[0]\n                \n                # 겹치는 클래스가 있으면 같은 클래스 후보에 추가\n                if len(np.intersect1d(current_classes, candidate_classes)) > 0:\n                    same_class_candidates.append(i)\n        \n        # 우선순위: 같은 클래스 > positive 샘플 > 전체 랜덤\n        if same_class_candidates:\n            return random.choice(same_class_candidates)\n        elif positive_candidates:\n            return random.choice(positive_candidates)\n        else:\n            return random.randint(0, len(self.df) - 1)\n       \n    def _apply_mixup(self, spec, target, idx):\n        row = self.df.iloc[idx]\n        \n        # Dismiss negative \n        if row['primary_label'] == 'nocall':\n            return spec, target\n        \n        # pairing\n        mix_idx = self._smart_mixup_pairing(idx, target)\n        row2 = self.df.iloc[mix_idx]\n        \n        # Load spectrogram\n        spec2 = self._get_spectrogram(row2['samplename'])\n        if spec2 is None:\n            return spec, target\n        \n        spec2 = torch.tensor(spec2, dtype=torch.float32).unsqueeze(0)\n        \n        # Target encoding\n        target2 = self.encode_label(row2['primary_label'])\n        \n        # Deal with Secondary labels\n        if 'secondary_labels' in row2 and pd.notna(row2['secondary_labels']) and row2['secondary_labels'] != '':\n            secondary_labels2 = self._parse_secondary_labels(row2['secondary_labels'])\n            for label in secondary_labels2:\n                if label in self.label_to_idx:\n                    target2[self.label_to_idx[label]] = 1.0\n        \n        # Set alpha\n        alpha = random.uniform(0.2, 0.8)\n        lam = np.random.beta(alpha, alpha)\n        \n        # Mixup\n        mixed_spec = lam * spec + (1 - lam) * spec2\n        mixed_target = lam * target + (1 - lam) * target2\n        \n        return mixed_spec, mixed_target\n    \n    def apply_spec_augmentations(self, spec, primary_label=None):\n        \"\"\"클래스별 적응적 augmentation\"\"\"\n        \n        # 클래스별 샘플 수 확인\n        if primary_label and hasattr(self, 'class_counts') and primary_label in self.class_counts:\n            sample_count = self.class_counts[primary_label]\n            is_rare_class = sample_count < self.rare_threshold\n        else:\n            is_rare_class = False\n        \n        # Rare class: 강한 augmentation (80% 확률, 더 많은 기법)\n        if is_rare_class:\n            aug_prob = 0.8  # 높은 확률\n            max_techniques = 4  # 더 많은 기법 적용\n        # Normal class: 일반 augmentation (50% 확률)\n        else:\n            aug_prob = 0.5\n            max_techniques = 3\n        \n        applied_count = 0\n        \n        # Time masking\n        if random.random() < aug_prob and applied_count < max_techniques:\n            if is_rare_class:\n                # Rare class: 더 강한 마스킹\n                num_masks = random.randint(1, 4)\n                for _ in range(num_masks):\n                    width = random.randint(3, 25)  # 더 넓은 범위\n                    start = random.randint(0, max(1, spec.shape[2] - width))\n                    spec[0, :, start:start+width] = 0\n            else:\n                # Normal class: 기존 방식\n                num_masks = random.randint(1, 3)\n                for _ in range(num_masks):\n                    width = random.randint(5, 20)\n                    start = random.randint(0, max(1, spec.shape[2] - width))\n                    spec[0, :, start:start+width] = 0\n            applied_count += 1\n        \n        # Frequency masking\n        if random.random() < aug_prob and applied_count < max_techniques:\n            if is_rare_class:\n                # Rare class: 더 강한 마스킹\n                num_masks = random.randint(1, 4)\n                for _ in range(num_masks):\n                    height = random.randint(3, 25)\n                    start = random.randint(0, max(1, spec.shape[1] - height))\n                    spec[0, start:start+height, :] = 0\n            else:\n                # Normal class: 기존 방식\n                num_masks = random.randint(1, 3)\n                for _ in range(num_masks):\n                    height = random.randint(5, 20)\n                    start = random.randint(0, max(1, spec.shape[1] - height))\n                    spec[0, start:start+height, :] = 0\n            applied_count += 1\n        \n        # Random brightness/contrast\n        if random.random() < aug_prob and applied_count < max_techniques:\n            if is_rare_class:\n                # Rare class: 더 강한 변형\n                gain = random.uniform(0.7, 1.3)\n                bias = random.uniform(-0.15, 0.15)\n            else:\n                # Normal class: 기존 방식\n                gain = random.uniform(0.8, 1.2)\n                bias = random.uniform(-0.1, 0.1)\n            \n            spec = spec * gain + bias\n            spec = torch.clamp(spec, 0, 1)\n            applied_count += 1\n\n        # Gaussian noise\n        if random.random() < aug_prob and applied_count < max_techniques:\n            if is_rare_class:\n                # Rare class: 더 강한 노이즈\n                noise_level = random.uniform(0.03, 0.08)\n            else:\n                # Normal class: 기존 방식\n                noise_level = 0.05\n            \n            noise = torch.randn_like(spec) * noise_level\n            spec = spec + noise\n            spec = torch.clamp(spec, 0, 1)\n            applied_count += 1\n\n        # Random erasing\n        if random.random() < aug_prob and applied_count < max_techniques:\n            if is_rare_class:\n                # Rare class: 더 많은 erasing\n                num_erases = random.randint(1, 3)\n                for _ in range(num_erases):\n                    erase_height = random.randint(3, 25)\n                    erase_width = random.randint(3, 25)\n                    max_x = spec.shape[2] - erase_width\n                    max_y = spec.shape[1] - erase_height\n                    if max_x > 0 and max_y > 0:\n                        x = random.randint(0, max_x)\n                        y = random.randint(0, max_y)\n                        spec[0, y:y+erase_height, x:x+erase_width] = 0\n            else:\n                # Normal class: 기존 방식\n                erase_height = random.randint(5, 20)\n                erase_width = random.randint(5, 20)\n                max_x = spec.shape[2] - erase_width\n                max_y = spec.shape[1] - erase_height\n                if max_x > 0 and max_y > 0:\n                    x = random.randint(0, max_x)\n                    y = random.randint(0, max_y)\n                    spec[0, y:y+erase_height, x:x+erase_width] = 0\n            applied_count += 1\n            \n        return spec\n    \n    def encode_label(self, label):\n        \"\"\"Encode label to one-hot vector, handle 'nocall' as all zeros\"\"\"\n        target = np.zeros(self.num_classes)\n        if label == 'nocall' or label == '':\n            return target  # All zeros for negative samples\n        if label in self.label_to_idx:\n            target[self.label_to_idx[label]] = 1.0\n        return target\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:06.009645Z","iopub.execute_input":"2025-06-05T08:01:06.010094Z","iopub.status.idle":"2025-06-05T08:01:06.051824Z","shell.execute_reply.started":"2025-06-05T08:01:06.010057Z","shell.execute_reply":"2025-06-05T08:01:06.050312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collate_fn(batch):\n    \"\"\"Custom collate function to handle different sized spectrograms\"\"\"\n    batch = [item for item in batch if item is not None]\n    if len(batch) == 0:\n        return {}\n        \n    result = {key: [] for key in batch[0].keys()}\n    \n    for item in batch:\n        for key, value in item.items():\n            result[key].append(value)\n    \n    for key in result:\n        if key in ['target', 'melspec'] and isinstance(result[key][0], torch.Tensor):\n            try:\n                result[key] = torch.stack(result[key])\n            except RuntimeError as e:\n                print(f\"Error stacking {key}: {e}\")\n                continue\n    \n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:06.053788Z","iopub.execute_input":"2025-06-05T08:01:06.054252Z","iopub.status.idle":"2025-06-05T08:01:06.076457Z","shell.execute_reply.started":"2025-06-05T08:01:06.054212Z","shell.execute_reply":"2025-06-05T08:01:06.075450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ConvNeXtFeatureExtractor(nn.Module):\n    def __init__(self, cfg):\n        super().__init__()\n        self.backbone = timm.create_model(\n            'convnextv2_nano.fcmae',\n            pretrained=True,\n            in_chans=cfg.in_channels,\n            num_classes=0  # Remove classification layer\n        )\n        self.pooling = nn.AdaptiveAvgPool2d(1)\n\n    def forward(self, x):\n        features = self.backbone(x)  # (B, C, H, W)\n        if len(features.shape) == 4:\n            pooled = self.pooling(features)  # (B, C, 1, 1)\n            return pooled.view(pooled.size(0), -1)  # (B, C)\n        return features\n\ndef extract_features(model, dataloader, device):\n    \"\"\"Extract features using the model\"\"\"\n    model.eval()\n    features = []\n    labels = []\n    \n    with torch.no_grad():\n        for batch in tqdm(dataloader, desc=\"Extracting features\"):\n            if 'melspec' not in batch or 'target' not in batch:\n                continue\n                \n            x = batch['melspec'].to(device)\n            feats = model(x)  # (B, C)\n            features.append(feats.cpu().numpy())\n            labels.append(batch['target'].cpu().numpy())\n            \n    if features:\n        features = np.concatenate(features, axis=0)\n        labels = np.concatenate(labels, axis=0)\n        return features, labels\n    else:\n        return np.array([]), np.array([])\n\ndef train_lightgbm(X, y):\n    \"\"\"Train LightGBM models for each class with validation split\"\"\"\n    models = []\n    print(f\"Training LightGBM for {y.shape[1]} classes...\")\n    \n    for i in tqdm(range(y.shape[1]), desc=\"Training LightGBM\"):\n        # Train/validation split\n        X_train, X_val, y_train, y_val = train_test_split(\n            X, y[:, i], \n            test_size=0.2, \n            random_state=42,\n            stratify=y[:, i] if len(np.unique(y[:, i])) > 1 else None\n        )\n        \n        lgb_train = lgb.Dataset(X_train, label=y_train)\n        lgb_val = lgb.Dataset(X_val, label=y_val)\n        \n        params = {\n            'objective': 'binary',\n            'metric': 'auc',\n            'boosting_type': 'gbdt',\n            'num_leaves': 31,\n            'learning_rate': 0.05,\n            'feature_fraction': 0.9,\n            'bagging_fraction': 0.8,\n            'bagging_freq': 5,\n            'verbose': -1\n        }\n        \n        model = lgb.train(\n            params,\n            lgb_train,\n            num_boost_round=100,\n            valid_sets=[lgb_val],\n            valid_names=['valid'],\n            callbacks=[\n                lgb.early_stopping(10),\n                lgb.log_evaluation(0)\n            ]\n        )\n        models.append(model)\n    \n    return models\n\ndef load_spectrograms(cfg):\n    \"\"\"Load spectrograms from numpy file\"\"\"\n    if os.path.exists(cfg.spectrogram_npy):\n        print(\"Loading spectrograms from numpy file...\")\n        all_spectrograms = np.load(cfg.spectrogram_npy, allow_pickle=True).item()\n        \n        # Separate positive and negative spectrograms\n        positive_spectrograms = {k: v for k, v in all_spectrograms.items() if not k.startswith('negative-')}\n        negative_spectrograms = {k: v for k, v in all_spectrograms.items() if k.startswith('negative-')}\n        \n        print(f\"Loaded {len(positive_spectrograms)} positive spectrograms\")\n        print(f\"Loaded {len(negative_spectrograms)} negative spectrograms\")\n        \n        return positive_spectrograms, negative_spectrograms\n    else:\n        print(\"Spectrogram file not found\")\n        return None, None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:06.078722Z","iopub.execute_input":"2025-06-05T08:01:06.079184Z","iopub.status.idle":"2025-06-05T08:01:06.105626Z","shell.execute_reply.started":"2025-06-05T08:01:06.079139Z","shell.execute_reply":"2025-06-05T08:01:06.104335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    \"\"\"Main training function\"\"\"\n    print(\"Starting BirdCLEF 2025 training pipeline...\")\n    \n    # Update debug settings\n    cfg.update_debug_settings()\n    \n    # Load data\n    print(\"Loading training data...\")\n    train_df = pd.read_csv(cfg.train_csv)\n    print(f\"Loaded {len(train_df)} training samples\")\n    \n    # Load spectrograms\n    positive_spectrograms, negative_spectrograms = load_spectrograms(cfg)\n    \n    # Create feature extractor\n    print(\"Initializing ConvNeXt feature extractor...\")\n    feature_model = ConvNeXtFeatureExtractor(cfg).to(cfg.device)\n    \n    # Create dataset and dataloader\n    print(\"Creating dataset...\")\n    train_dataset = BirdCLEFDatasetFromNPY(\n        train_df, \n        cfg, \n        spectrograms=positive_spectrograms, \n        negative_spectrograms=negative_spectrograms,\n        mode='train'\n    )\n    train_loader = DataLoader(\n        train_dataset, \n        batch_size=cfg.batch_size, \n        shuffle=False,\n        num_workers=cfg.num_workers,\n        pin_memory=True,\n        collate_fn=collate_fn\n    )\n    \n    # Extract features\n    print(\"Extracting features...\")\n    X_train, y_train = extract_features(feature_model, train_loader, cfg.device)\n    \n    if len(X_train) == 0:\n        print(\"No features extracted. Check your data and model.\")\n        return\n    \n    print(f\"Extracted features shape: {X_train.shape}\")\n    print(f\"Labels shape: {y_train.shape}\")\n    \n    # Check positive/negative ratio\n    positive_samples = np.sum(np.any(y_train == 1, axis=1))\n    negative_samples = len(y_train) - positive_samples\n    print(f\"Positive samples: {positive_samples}\")\n    print(f\"Negative samples: {negative_samples}\")\n    print(f\"Negative ratio: {negative_samples/len(y_train)*100:.1f}%\")\n    \n    # Train LightGBM\n    print(\"Training LightGBM models...\")\n    lgbm_models = train_lightgbm(X_train, y_train)\n    \n    print(f\"Training completed! Trained {len(lgbm_models)} LightGBM models.\")\n    \n    return lgbm_models\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T08:01:06.107267Z","iopub.execute_input":"2025-06-05T08:01:06.107739Z"}},"outputs":[],"execution_count":null}]}