{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":238662100,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **BirdCLEF 2025 Inference Notebook**\n\nThis notebook performs inference using pre-trained EfficientNetB0 models (one for each of the 5 folds) trained in the accompanying training notebook.\nIt processes the test soundscapes, generates predictions for each 5-second segment, ensembles the results from the 5 models, and creates the final `submission.csv` file.\n\n**Assumptions:**\n*   The trained model weights (`model_fold0.pth`, `model_fold1.pth`, ..., `model_fold4.pth`) are available in a Kaggle dataset mounted at `/kaggle/input/your-model-weights-dataset/`. **Remember to replace this path with the actual path to your weights.**\n*   The competition data is mounted at `/kaggle/input/birdclef-2025/`.","metadata":{}},{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport math\nimport random\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport cv2\n\nimport torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\n\nimport timm\nfrom tqdm.auto import tqdm\n\nwarnings.filterwarnings(\"ignore\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    debug = False # Set to True for testing with fewer files\n    num_workers = 2\n    \n    # Paths\n    # !!! IMPORTANT: Update this path to where your trained model weights are located !!!\n    model_weights_dir = '/kaggle/input/efficientnet-b0-pytorch-train-birdclef-25' # Example path, adjust as needed\n    \n    test_datadir = '/kaggle/input/birdclef-2025/test_soundscapes/'\n    train_csv = '/kaggle/input/birdclef-2025/train.csv' # Needed for label mapping if not in taxonomy\n    submission_csv_path = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    output_dir = '/kaggle/working/'\n    \n    # Model parameters (should match training)\n    model_name = 'efficientnet_b0'\n    pretrained = False # We load weights, not pre-trained from timm\n    in_channels = 3\n    num_classes = 206 # Will be updated based on taxonomy.csv\n    \n    # Audio parameters (should match training)\n    FS = 32000\n    TARGET_DURATION_SEC = 5.0\n    TARGET_SAMPLES = int(TARGET_DURATION_SEC * FS)\n    TARGET_SHAPE = (256, 256)\n    \n    N_FFT = 2048\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 20\n    FMAX = 16000\n    \n    # Inference parameters\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    batch_size = 16 # Adjust based on GPU memory\n    num_folds = 5\n\ncfg = CFG()\n\n# Read taxonomy to get all species labels\ntaxonomy_df = pd.read_csv(cfg.taxonomy_csv)\ncfg.species_ids = sorted(taxonomy_df['primary_label'].unique().tolist())\ncfg.num_classes = len(cfg.species_ids)\nprint(f\"Number of classes: {cfg.num_classes}\")\nprint(f\"Device: {cfg.device}\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"def set_seed(seed=42):\n    # \"\"Set seed for reproducibility\"\"\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    # No need for deterministic settings in inference, prioritize speed\n    # torch.backends.cudnn.deterministic = True \n    # torch.backends.cudnn.benchmark = False\n\nset_seed(cfg.seed)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Audio Pre-processing","metadata":{}},{"cell_type":"code","source":"def audio2melspec(audio_data, cfg):\n    # \"\"Convert audio data to mel spectrogram (identical to training)\"\"\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    # Normalize to [0, 1]\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    \n    # Resize if necessary (should match target shape)\n    if mel_spec_norm.shape != cfg.TARGET_SHAPE:\n       mel_spec_norm = cv2.resize(mel_spec_norm, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n        \n    return mel_spec_norm.astype(np.float32)\n\ndef process_soundscape(audio_path, cfg):\n    # \"\"Loads a soundscape, splits it into 5s chunks, and converts each to a mel spectrogram.\"\"\n    specs = []\n    try:\n        # Load the full soundscape\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS, mono=True)\n        \n        total_duration = librosa.get_duration(y=audio_data, sr=cfg.FS)\n        num_segments = int(math.ceil(total_duration / cfg.TARGET_DURATION_SEC))\n        \n        for i in range(num_segments):\n            start_sample = i * cfg.TARGET_SAMPLES\n            end_sample = start_sample + cfg.TARGET_SAMPLES\n            \n            # Extract segment\n            segment = audio_data[start_sample:end_sample]\n            \n            # Pad if segment is shorter than target duration (last segment)\n            if len(segment) < cfg.TARGET_SAMPLES:\n                segment = np.pad(segment, (0, cfg.TARGET_SAMPLES - len(segment)), mode='constant')\n            \n            # Convert to mel spectrogram\n            mel_spec = audio2melspec(segment, cfg)\n            specs.append(mel_spec)\n            \n        if not specs: # Handle potential errors or empty audio\n             print(f\"Warning: No segments processed for {audio_path}\")\n             # Return a list of zero spectrograms matching the expected output structure\n             # Assuming 1 minute = 12 segments\n             num_expected_segments = 12 \n             return [np.zeros(cfg.TARGET_SHAPE, dtype=np.float32) for _ in range(num_expected_segments)]\n            \n        return specs\n        \n    except Exception as e:\n        print(f\"Error processing soundscape {audio_path}: {e}\")\n        # Return list of zero spectrograms if error occurs\n        num_expected_segments = 12 # Assuming 1 minute test files\n        return [np.zeros(cfg.TARGET_SHAPE, dtype=np.float32) for _ in range(num_expected_segments)]","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset for Inference","metadata":{}},{"cell_type":"code","source":"class BirdCLEFInferenceDataset(Dataset):\n    def __init__(self, soundscape_paths, cfg):\n        self.soundscape_paths = soundscape_paths\n        self.cfg = cfg\n\n    def __len__(self):\n        return len(self.soundscape_paths)\n\n    def __getitem__(self, idx):\n        audio_path = self.soundscape_paths[idx]\n        soundscape_id = Path(audio_path).stem\n        \n        # Process the entire soundscape into a list of spectrograms\n        list_of_specs = process_soundscape(audio_path, self.cfg)\n        \n        # Stack spectrograms into a single tensor for batching\n        # Shape: (num_segments, channels, height, width)\n        specs_tensor = torch.tensor(np.array(list_of_specs), dtype=torch.float32).unsqueeze(1)\n        \n        return {\n            'soundscape_id': soundscape_id,\n            'specs': specs_tensor # Tensor of shape [12, 1, H, W]\n        }","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Definition (Identical to Training)","metadata":{}},{"cell_type":"code","source":"class FocalLossBCE(torch.nn.Module):\n    def __init__(\n            self,\n            alpha: float = 0.25,\n            gamma: float = 2,\n            reduction: str = \"mean\",\n            bce_weight: float = 0.6,\n            focal_weight: float = 1.4,\n    ):\n        super().__init__()\n        self.alpha = alpha\n        self.gamma = gamma\n        self.reduction = reduction\n        self.bce = torch.nn.BCEWithLogitsLoss(reduction=reduction)\n        self.bce_weight = bce_weight\n        self.focal_weight = focal_weight\n\n    def forward(self, logits, targets):\n        focal_loss = torchvision.ops.sigmoid_focal_loss(\n            inputs=logits,\n            targets=targets,\n            alpha=self.alpha,\n            gamma=self.gamma,\n            reduction=self.reduction,\n        )\n        bce_loss = self.bce(logits, targets)\n        return self.bce_weight * bce_loss + self.focal_weight * focal_loss\n\ndef get_criterion(cfg):\n    return FocalLossBCE()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEFModel(nn.Module):\n    def __init__(self, cfg):\n        super().__init__()\n        self.cfg = cfg\n        \n        taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n        cfg.num_classes = len(taxonomy_df)\n        \n        self.backbone = timm.create_model(\n            cfg.model_name,\n            pretrained=cfg.pretrained,\n            in_chans=cfg.in_channels,\n            drop_rate=0.2,\n            drop_path_rate=0.2\n        )\n        \n        if 'efficientnet' in cfg.model_name:\n            backbone_out = self.backbone.classifier.in_features\n            self.backbone.classifier = nn.Identity()\n        elif 'resnet' in cfg.model_name:\n            backbone_out = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        else:\n            backbone_out = self.backbone.get_classifier().in_features\n            self.backbone.reset_classifier(0, '')\n        \n        self.pooling = nn.AdaptiveAvgPool2d(1)\n            \n        self.feat_dim = backbone_out\n        \n        self.classifier = nn.Linear(backbone_out, cfg.num_classes)\n        \n        self.mixup_enabled = hasattr(cfg, 'mixup_alpha') and cfg.mixup_alpha > 0\n        if self.mixup_enabled:\n            self.mixup_alpha = cfg.mixup_alpha\n            \n    def forward(self, x, targets=None):\n    \n        if self.training and self.mixup_enabled and targets is not None:\n            mixed_x, targets_a, targets_b, lam = self.mixup_data(x, targets)\n            x = mixed_x\n        else:\n            targets_a, targets_b, lam = None, None, None\n        \n        features = self.backbone(x)\n        \n        if isinstance(features, dict):\n            features = features['features']\n            \n        if len(features.shape) == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        \n        logits = self.classifier(features)\n        \n        if self.training and self.mixup_enabled and targets is not None:\n            loss = self.mixup_criterion(F.binary_cross_entropy_with_logits, \n                                       logits, targets_a, targets_b, lam)\n            return logits, loss\n            \n        return logits\n    \n    def mixup_data(self, x, targets):\n        \"\"\"Applies mixup to the data batch\"\"\"\n        batch_size = x.size(0)\n\n        lam = np.random.beta(self.mixup_alpha, self.mixup_alpha)\n\n        indices = torch.randperm(batch_size).to(x.device)\n\n        mixed_x = lam * x + (1 - lam) * x[indices]\n        \n        return mixed_x, targets, targets[indices], lam\n    \n    def mixup_criterion(self, criterion, pred, y_a, y_b, lam):\n        \"\"\"Applies mixup to the loss function\"\"\"\n        return lam * criterion(pred, y_a) + (1 - lam) * criterion(pred, y_b)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Models","metadata":{}},{"cell_type":"code","source":"loaded_models = [] \nprint(f\"Loading {cfg.num_folds} models from {cfg.model_weights_dir}\")\nfor fold in range(cfg.num_folds):\n    model_path = os.path.join(cfg.model_weights_dir, f'model_fold{fold}.pth')\n    if not os.path.exists(model_path):\n        print(f\"ERROR: Model weight file not found at {model_path}\")\n        print(\"Please ensure the 'model_weights_dir' in CFG points to the correct dataset/directory.\")\n        raise FileNotFoundError(f\"Model weight not found: {model_path}\")\n\n    # Create the model instance using the correct 'models' module\n    model = BirdCLEFModel(cfg)\n    try:\n        # Load the state dict\n        checkpoint = torch.load(model_path, map_location=torch.device(cfg.device))\n\n        if 'model_state_dict' in checkpoint:\n            state_dict = checkpoint['model_state_dict']\n        else:\n            state_dict = checkpoint\n\n        # Optional: Adjust keys if needed\n        # state_dict = {k.replace('module.', ''): v for k, v in state_dict.items()}\n\n        model.load_state_dict(state_dict)\n        model.to(cfg.device)\n        model.eval()\n        loaded_models.append(model) # <<<--- Appending to the renamed list\n        print(f\"Loaded model from {model_path}\")\n    except Exception as e:\n        print(f\"Error loading model from {model_path}: {e}\")\n        raise e\n\n# Check the renamed list\nif len(loaded_models) != cfg.num_folds:\n    print(f\"Warning: Expected {cfg.num_folds} models, but only loaded {len(loaded_models)}. Check paths and files.\")\n    if len(loaded_models) == 0:\n         raise RuntimeError(\"No models were loaded successfully. Cannot proceed with inference.\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Inference Loop","metadata":{}},{"cell_type":"code","source":"def run_inference(models, cfg):\n    all_predictions = []\n    \n    # Find test soundscape files\n    test_files = []\n    if os.path.exists(cfg.test_datadir):\n        test_files = [os.path.join(cfg.test_datadir, f) for f in os.listdir(cfg.test_datadir) if f.endswith('.ogg')]\n        print(f\"Found {len(test_files)} test soundscapes.\")\n    else:\n        print(f\"Test directory {cfg.test_datadir} not found.\")\n\n    if cfg.debug:\n        test_files = test_files[:3] # Limit files for debugging\n        print(f\"Debug mode: Processing only {len(test_files)} files.\")\n        \n    # === Modified Condition ===\n    # If no test files were found (directory non-existent or empty), generate sample submission\n    if not test_files:\n         print(\"Test directory empty or non-existent. Generating sample submission format.\")\n         try:\n             sample_df = pd.read_csv(cfg.submission_csv_path)\n         except FileNotFoundError:\n              print(f\"Error: Sample submission file not found at {cfg.submission_csv_path}\")\n              # Create a fallback structure if sample submission is missing\n              return pd.DataFrame(columns=['row_id'] + cfg.species_ids)\n         \n         # Fill with a default value (e.g., 1/num_classes) or zeros\n         default_prob = 1 / cfg.num_classes \n         for species in cfg.species_ids:\n             if species in sample_df.columns:\n                 sample_df[species] = default_prob\n             else:\n                 # Add missing species column if it's in taxonomy but not sample submission\n                 print(f\"Info: Species {species} from taxonomy added to submission columns.\")\n                 sample_df[species] = default_prob\n                 \n         # Ensure all required columns exist and are in order\n         final_cols = ['row_id'] + cfg.species_ids\n         # Select existing columns first, then add any missing ones\n         sample_df = sample_df[[col for col in final_cols if col in sample_df.columns]]\n         for col in final_cols:\n             if col not in sample_df.columns:\n                 sample_df[col] = default_prob # Should not happen often due to logic above\n                 \n         return sample_df[final_cols] # Return with columns in correct order\n\n    # --- Proceed with inference if test files exist ---\n    \n    # Create dataset and dataloader\n    inference_dataset = BirdCLEFInferenceDataset(test_files, cfg)\n    # Batch size for dataloader is 1, as we process one full soundscape at a time\n    inference_loader = DataLoader(inference_dataset, batch_size=1, shuffle=False, num_workers=cfg.num_workers)\n\n    with torch.no_grad():\n        for batch in tqdm(inference_loader, desc=\"Inference\"):\n            soundscape_id = batch['soundscape_id'][0] # Dataloader returns list for batch_size=1\n            specs_batch = batch['specs'][0].to(cfg.device) # Shape [12, 1, H, W]\n            \n            num_segments = specs_batch.shape[0]\n            fold_predictions = []\n            \n            # Get predictions from each fold's model\n            for model in models:\n                # Process segments in mini-batches if necessary (GPU memory)\n                segment_preds = []\n                for i in range(0, num_segments, cfg.batch_size):\n                    mini_batch = specs_batch[i:i+cfg.batch_size]\n                    logits = model(mini_batch)\n                    probs = torch.sigmoid(logits) # Convert logits to probabilities\n                    segment_preds.append(probs.cpu().numpy())\n                \n                # Concatenate predictions for the soundscape from this fold\n                fold_soundscape_preds = np.concatenate(segment_preds, axis=0) # Shape [12, num_classes]\n                fold_predictions.append(fold_soundscape_preds)\n            \n            # Ensemble predictions: Average probabilities across folds\n            # Shape: (num_folds, num_segments, num_classes) -> (num_segments, num_classes)\n            ensembled_preds = np.mean(np.stack(fold_predictions, axis=0), axis=0)\n            \n            # Store predictions for this soundscape\n            for i in range(num_segments):\n                end_time = (i + 1) * int(cfg.TARGET_DURATION_SEC)\n                row_id = f\"{soundscape_id}_{end_time}\"\n                \n                prediction_dict = {'row_id': row_id}\n                # Fill probabilities for each species\n                for j, species_id in enumerate(cfg.species_ids):\n                    prediction_dict[species_id] = ensembled_preds[i, j]\n                    \n                all_predictions.append(prediction_dict)\n            \n            # Clean up memory\n            del specs_batch, fold_predictions, ensembled_preds\n            if cfg.device == 'cuda':\n                torch.cuda.empty_cache()\n            gc.collect()\n            \n    # Create submission DataFrame from collected predictions\n    if not all_predictions: # Should not happen if test_files was not empty\n        print('Warning: No predictions were generated even though test files were found.')\n        # Return empty df with correct columns as fallback\n        return pd.DataFrame(columns=['row_id'] + cfg.species_ids) \n        \n    submission_df = pd.DataFrame(all_predictions)\n    \n    # Ensure all required columns are present and in the correct order\n    final_cols = ['row_id'] + cfg.species_ids\n    # Add any missing species columns (e.g., if a species was in taxonomy but somehow missed in prediction dict)\n    for col in final_cols:\n        if col not in submission_df.columns:\n            print(f\"Warning: Column {col} missing in submission DataFrame, adding with zeros.\")\n            submission_df[col] = 0.0 # Or default_prob\n            \n    # Select and order columns - this should now work\n    submission_df = submission_df[final_cols] \n    \n    return submission_df","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Generate Submission File","metadata":{}},{"cell_type":"code","source":"start_time = time.time()\n\n# Ensure the renamed list 'loaded_models' is populated before calling run_inference\n# Check if 'loaded_models' exists and is not empty\nif 'loaded_models' not in locals() or not loaded_models:\n    print(\"Error: Models were not loaded into 'loaded_models' list. Cannot run inference.\")\n    # As a fallback for Kaggle commit, create an empty submission\n    # Consider raising an error if debugging locally: raise RuntimeError(\"Models not loaded.\")\n    submission_df = pd.DataFrame(columns=['row_id'] + cfg.species_ids)\nelse:\n    # Pass the correct list to the inference function\n    submission_df = run_inference(loaded_models, cfg) # <<<--- Passing the renamed list\n\n# Save the submission file\noutput_path = os.path.join(cfg.output_dir, 'submission.csv')\nsubmission_df.to_csv(output_path, index=False)\n\nend_time = time.time()\nprint(f\"Submission file created at: {output_path}\")\nprint(f\"Total inference time: {end_time - start_time:.2f} seconds\")\n\n# Display the first few rows of the submission\nprint(\"Submission DataFrame head:\")\nprint(submission_df.head())","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}