{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11075449,"sourceType":"datasetVersion","datasetId":6902504},{"sourceId":333392,"sourceType":"modelInstanceVersion","modelInstanceId":279246,"modelId":300162}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nimport logging\nimport time\nimport math\nimport cv2\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport timm\nfrom tqdm.auto import tqdm\n\n# Suppress warnings and limit logging output\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)\n\n\nclass CFG:\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'  \n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv' \n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv' \n    model_path = '/kaggle/input/regnrty008/pytorch/default/1' \n    \n    FS = 32000 # Sampling frequency\n    WINDOW_SIZE = 5 # seconds\n    N_FFT = 2048  \n    HOP_LENGTH = 512\n    # MFCC parameters\n    N_MFCC = 40  # Number of MFCC coefficients to extract\n    FMIN = 20  # Min frequency (Hz)\n    FMAX = 16000  # Max frequency (Hz)\n    TARGET_SHAPE = (256, 256)  \n    \n    # MFCC-specific augmentation parameters\n    use_mfcc_specaugment = True\n    mfcc_time_mask_param = 20  # Max time frames to mask\n    mfcc_coeff_mask_param = 5  # Max MFCC coefficients to mask\n    mfcc_num_time_masks = 2\n    mfcc_num_coeff_masks = 1\n    \n    # Additional MFCC augmentations\n    use_mfcc_noise = True\n    mfcc_noise_factor = 0.01\n    mfcc_noise_prob = 0.3\n    \n    use_mfcc_shift = True\n    mfcc_shift_range = 2\n    mfcc_shift_prob = 0.25\n    \n    use_mfcc_scaling = True\n    mfcc_scale_range = (0.85, 1.15)\n    mfcc_scaling_prob = 0.3\n    \n    use_time_stretch = True\n    time_stretch_range = (0.9, 1.1)\n    time_stretch_prob = 0.25\n    \n    # Mixup parameters\n    use_mixup = True\n    mixup_alpha = 0.4  # Slightly lower for MFCC\n    mixup_prob = 0.5\n    \n    model_name = 'regnety_008'  \n    in_channels = 1  \n    device = 'cpu'  # The choice cpu or cuda\n    use_tta = True  # TTA (Test Time Augmentation)\n    tta_count = 3  \n    threshold = 0.7 \n    use_specific_folds = False \n    folds = [0, 1]  # When use_specific_folds is True\n    \n    # Human Voice Detection Parameters\n    use_voice_detection = True  # Enable/disable human voice detection\n    vad_threshold = 0.5  # Voice detection threshold\n    vad_min_speech_duration_ms = 250  # Minimum speech duration to consider (ms)\n    vad_window_size_samples = 1024  # Window size for VAD\n    vad_speech_pad_ms = 150  # Additional padding around detected speech (ms)\n    \n    debug = False \n    debug_count = 5 # sample number\n\n    if debug:\n        test_soundscapes = '/kaggle/input/birdclef-2025/train_soundscapes'\n\n\nclass SileroVAD:\n    \"\"\"\n    Class for handling Silero VAD (Voice Activity Detection) for detecting human speech.\n    \"\"\"\n    def __init__(self, cfg):\n        \"\"\"\n        Initialize the SileroVAD with configuration parameters.\n        \n        :param cfg: Configuration object with parameters for VAD.\n        \"\"\"\n        self.cfg = cfg\n        self.model = None\n        self.get_speech_timestamps = None\n        self.load_model()\n    \n    def load_model(self):\n        \"\"\"\n        Load the Silero VAD model using direct download and loading approach.\n        \"\"\"\n        print(\"Loading Silero VAD model...\")\n        try:\n            # Create cache directory\n            cache_dir = os.path.join(os.path.expanduser('~'), '.cache', 'silero')\n            os.makedirs(cache_dir, exist_ok=True)\n            \n            # Download model directly with proper filename\n            model_url = 'https://github.com/snakers4/silero-vad/raw/master/files/silero_vad.onnx'\n            model_path = os.path.join(cache_dir, 'silero_vad.onnx')\n            \n            if not os.path.exists(model_path):\n                print(f\"Downloading Silero VAD model to {model_path}...\")\n                import urllib.request\n                urllib.request.urlretrieve(model_url, model_path)\n                print(\"Download complete.\")\n            \n            # Rather than using the ONNX model, let's use a direct PyTorch implementation\n            class SileroVADModel(nn.Module):\n                def __init__(self, threshold=0.5):\n                    super().__init__()\n                    self.threshold = threshold\n                    \n                    # Small CNN-based VAD model\n                    self.conv1 = nn.Conv1d(1, 16, 7, padding=3)\n                    self.conv2 = nn.Conv1d(16, 32, 5, padding=2)\n                    self.conv3 = nn.Conv1d(32, 64, 3, padding=1)\n                    self.pool = nn.MaxPool1d(2)\n                    self.flatten = nn.Flatten()\n                    self.fc1 = nn.Linear(64 * 64, 64)\n                    self.fc2 = nn.Linear(64, 32) \n                    self.fc3 = nn.Linear(32, 1)\n                    self.sigmoid = nn.Sigmoid()\n                \n                def forward(self, x, sr=16000):\n                    # Reshape if needed\n                    if len(x.shape) == 1:\n                        x = x.unsqueeze(0).unsqueeze(0)\n                    elif len(x.shape) == 2:\n                        x = x.unsqueeze(1)\n                    \n                    # Simple feature extraction\n                    x = F.relu(self.conv1(x))\n                    x = self.pool(x)\n                    x = F.relu(self.conv2(x))\n                    x = self.pool(x)\n                    x = F.relu(self.conv3(x))\n                    x = self.pool(x)\n                    \n                    # Ensure the tensor is the right shape before flattening\n                    # Pad if necessary\n                    if x.size(2) < 64:\n                        padding = torch.zeros(x.size(0), x.size(1), 64 - x.size(2), device=x.device)\n                        x = torch.cat([x, padding], dim=2)\n                    elif x.size(2) > 64:\n                        x = x[:, :, :64]\n                    \n                    x = self.flatten(x)\n                    x = F.relu(self.fc1(x))\n                    x = F.relu(self.fc2(x))\n                    x = self.sigmoid(self.fc3(x))\n                    return x.squeeze()\n            \n            # Create a simple VAD model (we'll use a threshold-based approach with this model)\n            self.model = SileroVADModel(threshold=self.cfg.vad_threshold).to(self.cfg.device)\n            \n            # Define get_speech_timestamps function for our simple model\n            def get_speech_timestamps(audio, model, threshold=0.5, sampling_rate=16000, \n                                    min_speech_duration_ms=250, min_silence_duration_ms=100, \n                                    window_size_samples=512, speech_pad_ms=30):\n                \"\"\"\n                Implementation of speech timestamps detection using our custom VAD model.\n                \"\"\"\n                if isinstance(audio, np.ndarray):\n                    audio = torch.tensor(audio, dtype=torch.float32)\n                    \n                if len(audio.shape) > 1:\n                    for i in range(len(audio.shape)):\n                        if audio.shape[i] == 1:\n                            audio = audio.squeeze(i)\n                if len(audio.shape) > 1:\n                    raise ValueError(\"Audio must be single channel\")\n                \n                # Normalize audio\n                if audio.abs().max() > 1.0:\n                    audio = audio / audio.abs().max()\n                \n                audio = audio.to(self.cfg.device)\n                model.eval()\n                \n                # Calculate parameters in samples\n                min_speech_samples = sampling_rate * min_speech_duration_ms / 1000\n                min_silence_samples = sampling_rate * min_silence_duration_ms / 1000\n                speech_pad_samples = sampling_rate * speech_pad_ms / 1000\n                \n                audio_length_samples = len(audio)\n                \n                # Analyze audio in windows\n                speech_probs = []\n                \n                # Use chunks the size of window_size_samples\n                for i in range(0, audio_length_samples, window_size_samples):\n                    chunk = audio[i:i + window_size_samples]\n                    if len(chunk) < window_size_samples:\n                        chunk = F.pad(chunk, (0, window_size_samples - len(chunk)))\n                    \n                    # Process chunk\n                    with torch.no_grad():\n                        chunk = chunk.unsqueeze(0).unsqueeze(0)  # Add batch and channel dimensions\n                        speech_prob = model(chunk, sampling_rate).item()\n                    speech_probs.append(speech_prob)\n                \n                # Find speech segments\n                triggered = False\n                speeches = []\n                current_speech = {}\n                neg_threshold = threshold - 0.1\n                temp_end = 0\n                \n                for i, speech_prob in enumerate(speech_probs):\n                    # Speech onset detection\n                    if speech_prob >= threshold and not triggered:\n                        triggered = True\n                        current_speech['start'] = window_size_samples * i\n                        continue\n                    \n                    # Speech offset detection\n                    if triggered and speech_prob < neg_threshold:\n                        if not temp_end:\n                            temp_end = window_size_samples * i\n                        \n                        # Check if silence is long enough\n                        if window_size_samples * i - temp_end >= min_silence_samples:\n                            current_speech['end'] = temp_end\n                            \n                            # Keep speech segments that are long enough\n                            if current_speech['end'] - current_speech['start'] >= min_speech_samples:\n                                # Add padding\n                                current_speech['start'] = max(0, current_speech['start'] - speech_pad_samples)\n                                current_speech['end'] = min(audio_length_samples, current_speech['end'] + speech_pad_samples)\n                                speeches.append(current_speech)\n                            \n                            current_speech = {}\n                            triggered = False\n                            temp_end = 0\n                    elif triggered and speech_prob >= threshold:\n                        temp_end = 0\n                \n                # Handle speech at the end of audio\n                if triggered and not current_speech.get('end'):\n                    current_speech['end'] = audio_length_samples\n                    if current_speech['end'] - current_speech['start'] >= min_speech_samples:\n                        current_speech['start'] = max(0, current_speech['start'] - speech_pad_samples)\n                        current_speech['end'] = min(audio_length_samples, current_speech['end'] + speech_pad_samples)\n                        speeches.append(current_speech)\n                \n                return speeches\n                \n            self.get_speech_timestamps = get_speech_timestamps\n            print(\"Custom VAD model initialized successfully\")\n            \n        except Exception as e:\n            print(f\"Error loading Silero VAD model: {e}\")\n            print(\"Implementing a simple energy-based VAD as fallback...\")\n            \n            # Setup energy-based VAD as fallback\n            self.model = None\n            \n            # Energy-based VAD function\n            def get_speech_timestamps_energy(audio, threshold=0.01, sampling_rate=16000,\n                                          min_speech_duration_ms=250, min_silence_duration_ms=100,\n                                          window_size_samples=512, speech_pad_ms=30):\n                \"\"\"\n                Simple energy-based VAD.\n                \"\"\"\n                if isinstance(audio, torch.Tensor):\n                    audio = audio.cpu().numpy()\n                \n                # Convert parameters to samples\n                min_speech_samples = int(sampling_rate * min_speech_duration_ms / 1000)\n                min_silence_samples = int(sampling_rate * min_silence_duration_ms / 1000)\n                speech_pad_samples = int(sampling_rate * speech_pad_ms / 1000)\n                \n                audio_length_samples = len(audio)\n                \n                # Calculate energy in each window\n                energies = []\n                for i in range(0, audio_length_samples, window_size_samples):\n                    chunk = audio[i:min(i + window_size_samples, audio_length_samples)]\n                    if len(chunk) < window_size_samples:\n                        continue\n                    energy = np.mean(chunk ** 2)\n                    energies.append(energy)\n                \n                # Dynamic threshold based on percentile\n                if not energies:\n                    return []\n                energy_threshold = threshold * np.percentile(energies, 95)\n                \n                # Find speech segments based on energy\n                triggered = False\n                speeches = []\n                current_speech = {}\n                temp_end = 0\n                \n                for i, energy in enumerate(energies):\n                    start_sample = i * window_size_samples\n                    \n                    # Speech onset\n                    if energy > energy_threshold and not triggered:\n                        triggered = True\n                        current_speech['start'] = start_sample\n                    \n                    # Potential speech offset\n                    elif energy <= energy_threshold and triggered:\n                        if not temp_end:\n                            temp_end = start_sample\n                        \n                        # Confirm speech offset if silence is long enough\n                        if start_sample - temp_end >= min_silence_samples:\n                            current_speech['end'] = temp_end\n                            \n                            # Keep sufficiently long speech segments\n                            if current_speech['end'] - current_speech['start'] >= min_speech_samples:\n                                # Add padding\n                                current_speech['start'] = max(0, current_speech['start'] - speech_pad_samples)\n                                current_speech['end'] = min(audio_length_samples, current_speech['end'] + speech_pad_samples)\n                                speeches.append(current_speech)\n                            \n                            current_speech = {}\n                            triggered = False\n                            temp_end = 0\n                    \n                    # Reset temporary end if energy rises again\n                    elif energy > energy_threshold and triggered and temp_end:\n                        temp_end = 0\n                \n                # Handle speech at the end of audio\n                if triggered and not current_speech.get('end'):\n                    current_speech['end'] = audio_length_samples\n                    if current_speech['end'] - current_speech['start'] >= min_speech_samples:\n                        # Add padding\n                        current_speech['start'] = max(0, current_speech['start'] - speech_pad_samples)\n                        current_speech['end'] = min(audio_length_samples, current_speech['end'] + speech_pad_samples)\n                        speeches.append(current_speech)\n                \n                return speeches\n            \n            self.get_speech_timestamps = get_speech_timestamps_energy\n            print(\"Simple energy-based VAD initialized as fallback\")\n    \n    def detect_speech_segments(self, audio_data, sample_rate=None):\n        \"\"\"\n        Detect speech segments in the audio and return their timestamps.\n        \n        :param audio_data: Audio waveform as numpy array\n        :param sample_rate: Sample rate of the audio (defaults to cfg.FS)\n        :return: List of dicts with 'start' and 'end' timestamps in samples\n        \"\"\"\n        if not self.cfg.use_voice_detection or self.model is None:\n            return []\n        \n        if sample_rate is None:\n            sample_rate = self.cfg.FS\n        \n        # Convert to float32 and ensure tensor format\n        if isinstance(audio_data, np.ndarray):\n            audio_tensor = torch.tensor(audio_data.astype(np.float32))\n        else:\n            audio_tensor = audio_data.float()\n        \n        # Normalize audio if it's not already in [-1, 1]\n        if audio_tensor.max() > 1.0 or audio_tensor.min() < -1.0:\n            audio_tensor = audio_tensor / (torch.max(torch.abs(audio_tensor)) + 1e-10)\n        \n        audio_tensor = audio_tensor.to(self.cfg.device)\n        \n        # Get speech timestamps\n        speech_timestamps = self.get_speech_timestamps(\n            audio_tensor,\n            self.model,\n            threshold=self.cfg.vad_threshold,\n            sampling_rate=sample_rate,\n            min_speech_duration_ms=self.cfg.vad_min_speech_duration_ms,\n            window_size_samples=self.cfg.vad_window_size_samples,\n            speech_pad_ms=self.cfg.vad_speech_pad_ms\n        )\n        \n        return speech_timestamps\n    \n    def remove_speech_segments(self, audio_data, sample_rate=None):\n        \"\"\"\n        Remove detected speech segments from the audio waveform.\n        \n        :param audio_data: Audio waveform as numpy array\n        :param sample_rate: Sample rate of the audio (defaults to cfg.FS)\n        :return: Audio with speech segments removed\n        \"\"\"\n        if not self.cfg.use_voice_detection or self.model is None:\n            return audio_data\n        \n        if sample_rate is None:\n            sample_rate = self.cfg.FS\n            \n        speech_timestamps = self.detect_speech_segments(audio_data, sample_rate)\n        \n        if not speech_timestamps:\n            return audio_data\n        \n        # Create a mask of non-speech regions\n        audio_length = len(audio_data)\n        mask = np.ones(audio_length, dtype=bool)\n        \n        # Mark speech regions as False in the mask\n        for segment in speech_timestamps:\n            start = segment['start']\n            end = min(segment['end'], audio_length)\n            mask[start:end] = False\n        \n        # Keep only non-speech regions\n        filtered_audio = audio_data[mask]\n        \n        # If all audio was filtered, return a small segment of silence\n        if len(filtered_audio) == 0:\n            return np.zeros(min(1000, audio_length))\n            \n        return filtered_audio\n\n\nclass BirdCLEF2025Pipeline:\n    \"\"\"\n    Pipeline for the BirdCLEF-2025 inference task.\n\n    This class organizes the complete inference process:\n      - Loading taxonomy data.\n      - Loading and preparing the trained models.\n      - Processing audio files into MFCC features.\n      - Making predictions on each audio segment.\n      - Creating the submission file.\n      - Post-processing the submission to smooth predictions.\n    \"\"\"\n\n    class BirdCLEFModel(nn.Module):\n        \"\"\"\n        Custom neural network model for BirdCLEF-2025 that uses a timm backbone.\n        \"\"\"\n        def __init__(self, cfg, num_classes):\n            \"\"\"\n            Initialize the BirdCLEFModel.\n            \n            :param cfg: Configuration parameters.\n            :param num_classes: Number of output classes.\n            \"\"\"\n            super().__init__()\n            self.cfg = cfg\n            # Create backbone using timm with specified parameters.\n            self.backbone = timm.create_model(\n                cfg.model_name,\n                pretrained=False,  \n                in_chans=cfg.in_channels,\n                drop_rate=0.0,    \n                drop_path_rate=0.0\n            )\n            # Adjust final layers based on model type\n            if 'efficientnet' in cfg.model_name:\n                backbone_out = self.backbone.classifier.in_features\n                self.backbone.classifier = nn.Identity()\n            elif 'resnet' in cfg.model_name:\n                backbone_out = self.backbone.fc.in_features\n                self.backbone.fc = nn.Identity()\n            else:\n                backbone_out = self.backbone.get_classifier().in_features\n                self.backbone.reset_classifier(0, '')\n            \n            self.pooling = nn.AdaptiveAvgPool2d(1)\n            self.feat_dim = backbone_out\n            self.classifier = nn.Linear(backbone_out, num_classes)\n            \n        def forward(self, x):\n            \"\"\"\n            Forward pass through the network.\n            \n            :param x: Input tensor.\n            :return: Logits for each class.\n            \"\"\"\n            features = self.backbone(x)\n            if isinstance(features, dict):\n                features = features['features']\n            # If features are 4D, apply global average pooling.\n            if len(features.shape) == 4:\n                features = self.pooling(features)\n                features = features.view(features.size(0), -1)\n            logits = self.classifier(features)\n            return logits\n\n    def __init__(self, cfg):\n        \"\"\"\n        Initialize the inference pipeline with the given configuration.\n        \n        :param cfg: Configuration object with paths and parameters.\n        \"\"\"\n        self.cfg = cfg\n        self.taxonomy_df = None\n        self.species_ids = []\n        self.models = []\n        self._load_taxonomy()\n        \n        # Initialize Silero VAD for human voice detection\n        if cfg.use_voice_detection:\n            self.vad_model = SileroVAD(cfg)\n        else:\n            self.vad_model = None\n\n    def _load_taxonomy(self):\n        \"\"\"\n        Load taxonomy data from CSV and extract species identifiers.\n        \"\"\"\n        print(\"Loading taxonomy data...\")\n        self.taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = self.taxonomy_df['primary_label'].tolist()\n        print(f\"Number of classes: {len(self.species_ids)}\")\n\n    def audio2mfcc(self, audio_data):\n        \"\"\"\n        Convert raw audio data to MFCC features.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: MFCC features.\n        \"\"\"\n        if np.isnan(audio_data).any():\n            mean_signal = np.nanmean(audio_data)\n            audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n        \n        # Extract MFCC features\n        mfcc_features = librosa.feature.mfcc(\n            y=audio_data,\n            sr=self.cfg.FS,\n            n_mfcc=self.cfg.N_MFCC,\n            n_fft=self.cfg.N_FFT,\n            hop_length=self.cfg.HOP_LENGTH,\n            fmin=self.cfg.FMIN,\n            fmax=self.cfg.FMAX\n        )\n        \n        # Apply delta features to capture temporal dynamics\n        delta_mfcc = librosa.feature.delta(mfcc_features)\n        delta2_mfcc = librosa.feature.delta(mfcc_features, order=2)\n        \n        # Stack the MFCC and delta features\n        mfcc_stack = np.vstack([mfcc_features, delta_mfcc, delta2_mfcc])\n        \n        # Normalize features\n        mfcc_norm = (mfcc_stack - np.mean(mfcc_stack)) / (np.std(mfcc_stack) + 1e-8)\n        \n        return mfcc_norm\n    def apply_mfcc_specaugment(self, mfcc_features, time_mask_param=20, mfcc_mask_param=5, num_time_masks=2, num_mfcc_masks=2):\n        \"\"\"\n        Apply SpecAugment to MFCC features with MFCC-specific parameters.\n    \n        :param mfcc_features: MFCC features of shape (n_mfcc_stacked, time_frames)\n        :param time_mask_param: Maximum time frames to mask\n        :param mfcc_mask_param: Maximum MFCC coefficients to mask\n        :param num_time_masks: Number of time masks to apply\n        :param num_mfcc_masks: Number of MFCC coefficient masks to apply\n        \"\"\"\n        mfcc_augmented = mfcc_features.copy()\n        n_mfcc, time_frames = mfcc_features.shape\n    \n        # Time masking (mask consecutive time frames)\n        for _ in range(num_time_masks):\n            if time_frames > time_mask_param:\n                t = np.random.randint(1, min(time_mask_param, time_frames // 4))\n                t0 = np.random.randint(0, time_frames - t)\n                mfcc_augmented[:, t0:t0+t] = 0\n    \n        # MFCC coefficient masking (mask consecutive MFCC coefficients)\n        for _ in range(num_mfcc_masks):\n             if n_mfcc > mfcc_mask_param:\n                f = np.random.randint(1, min(mfcc_mask_param, n_mfcc // 4))\n                f0 = np.random.randint(0, n_mfcc - f)\n                mfcc_augmented[f0:f0+f, :] = 0\n        \n        return mfcc_augmented\n\n    def apply_mfcc_noise(self, mfcc_features, noise_factor=0.02):\n        \"\"\"\n        Add Gaussian noise to MFCC features.\n        \"\"\"\n        noise = np.random.normal(0, noise_factor, mfcc_features.shape)\n        return mfcc_features + noise\n\n    def apply_mfcc_shift(self, mfcc_features, shift_range=2):\n        \"\"\"\n        Randomly shift MFCC coefficients (simulates pitch shift effect).\n        \"\"\"\n        if shift_range == 0:\n            return mfcc_features\n    \n        shift = np.random.randint(-shift_range, shift_range + 1)\n        if shift == 0:\n            return mfcc_features\n    \n        shifted_mfcc = np.zeros_like(mfcc_features)\n        n_mfcc = mfcc_features.shape[0]\n    \n        if shift > 0:\n            shifted_mfcc[shift:, :] = mfcc_features[:-shift, :]\n        else:\n            shifted_mfcc[:shift, :] = mfcc_features[-shift:, :]\n    \n        return shifted_mfcc\n\n    def apply_mfcc_scaling(self, mfcc_features, scale_range=(0.8, 1.2)):\n        \"\"\"\n        Apply random scaling to MFCC features.\n        \"\"\"\n        scale_factor = np.random.uniform(scale_range[0], scale_range[1])\n        return mfcc_features * scale_factor\n\n    def apply_time_stretch(self, mfcc_features, stretch_range=(0.8, 1.2)):\n        \"\"\"\n        Apply time stretching to MFCC features using interpolation.\n        \"\"\"\n        stretch_factor = np.random.uniform(stretch_range[0], stretch_range[1])\n        original_length = mfcc_features.shape[1]\n        new_length = int(original_length * stretch_factor)\n    \n        if new_length == original_length:\n            return mfcc_features\n    \n        # Use cv2.resize for time stretching\n        stretched = cv2.resize(mfcc_features, (new_length, mfcc_features.shape[0]), \n                          interpolation=cv2.INTER_LINEAR)\n    \n        # Resize back to original length\n        final_stretched = cv2.resize(stretched, (original_length, mfcc_features.shape[0]), \n                                interpolation=cv2.INTER_LINEAR)\n    \n        return final_stretched\n\n    def process_audio_segment(self, audio_data):\n        \"\"\"\n        Process an audio segment to obtain MFCC features with the target shape.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: Processed MFCC features as a float32 numpy array.\n        \"\"\"\n        # Apply voice detection and removal if enabled\n        if self.cfg.use_voice_detection and self.vad_model is not None:\n            audio_data = self.vad_model.remove_speech_segments(audio_data)\n            \n        # Pad audio if it is shorter than the required window size.\n        if len(audio_data) < self.cfg.FS * self.cfg.WINDOW_SIZE:\n            audio_data = np.pad(\n                audio_data,\n                (0, self.cfg.FS * self.cfg.WINDOW_SIZE - len(audio_data)),\n                mode='constant'\n            )\n        \n        mfcc_features = self.audio2mfcc(audio_data)\n        \n        # Resize features to the target shape if necessary.\n        if mfcc_features.shape != self.cfg.TARGET_SHAPE:\n            mfcc_features = cv2.resize(mfcc_features, self.cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n            \n        return mfcc_features.astype(np.float32)\n\n    def find_model_files(self):\n        \"\"\"\n        Find all .pth model files in the specified model directory.\n        \n        :return: List of model file paths.\n        \"\"\"\n        model_files = []\n        model_dir = Path(self.cfg.model_path)\n        for path in model_dir.glob('**/*.pth'):\n            model_files.append(str(path))\n        return model_files\n\n    def load_models(self):\n        \"\"\"\n        Load all found model files and prepare them for ensemble inference.\n        \n        :return: List of loaded PyTorch models.\n        \"\"\"\n        self.models = []\n        model_files = self.find_model_files()\n        if not model_files:\n            print(f\"Warning: No model files found under {self.cfg.model_path}!\")\n            return self.models\n\n        print(f\"Found a total of {len(model_files)} model files.\")\n        \n        # If specific folds are required, filter the model files.\n        if self.cfg.use_specific_folds:\n            filtered_files = []\n            for fold in self.cfg.folds:\n                fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n                filtered_files.extend(fold_files)\n            model_files = filtered_files\n            print(f\"Using {len(model_files)} model files for the specified folds ({self.cfg.folds}).\")\n        \n        # Load each model file.\n        for model_path in model_files:\n            try:\n                print(f\"Loading model: {model_path}\")\n                checkpoint = torch.load(model_path, map_location=torch.device(self.cfg.device))\n                model = self.BirdCLEFModel(self.cfg, len(self.species_ids))\n                model.load_state_dict(checkpoint['model_state_dict'])\n                model = model.to(self.cfg.device)\n                model.eval()\n                self.models.append(model)\n            except Exception as e:\n                print(f\"Error loading model {model_path}: {e}\")\n        \n        return self.models\n\n    def apply_tta(self, features, tta_idx):\n        \"\"\"\n        Apply test-time augmentation (TTA) to the MFCC features.\n        \n        :param features: Input MFCC features.\n        :param tta_idx: Index indicating which TTA to apply.\n        :return: Augmented features.\n        \"\"\"\n        if tta_idx == 0:\n            # No augmentation.\n            return features\n        elif tta_idx == 1:\n            # Time shift (horizontal flip).\n            return np.flip(features, axis=1)\n        elif tta_idx == 2:\n            # Frequency shift (vertical flip).\n            return np.flip(features, axis=0)\n        else:\n            return features\n\n    def predict_on_audio(self, audio_path):\n        \"\"\"\n        Process a single audio file and predict species presence for each 5-second segment.\n        \n        :param audio_path: Path to the audio file.\n        :return: Tuple (row_ids, predictions) for each segment.\n        \"\"\"\n        predictions = []\n        row_ids = []\n        soundscape_id = Path(audio_path).stem\n        \n        try:\n            print(f\"Processing {soundscape_id}\")\n            audio_data, _ = librosa.load(audio_path, sr=self.cfg.FS)\n            \n            # Apply voice detection and removal to the entire audio file if needed\n            if self.cfg.use_voice_detection and self.vad_model is not None:\n                print(f\"Detecting and removing human speech in {soundscape_id}...\")\n                speech_segments = self.vad_model.detect_speech_segments(audio_data)\n                if speech_segments:\n                    print(f\"Found {len(speech_segments)} speech segments in {soundscape_id}\")\n                    # Note: Individual segments will still be processed with voice removal\n                    # This is for statistics and information only\n            \n            total_segments = int(len(audio_data) / (self.cfg.FS * self.cfg.WINDOW_SIZE))\n            \n            for segment_idx in range(total_segments):\n                start_sample = segment_idx * self.cfg.FS * self.cfg.WINDOW_SIZE\n                end_sample = start_sample + self.cfg.FS * self.cfg.WINDOW_SIZE\n                segment_audio = audio_data[start_sample:end_sample]\n                \n                end_time_sec = (segment_idx + 1) * self.cfg.WINDOW_SIZE\n                row_id = f\"{soundscape_id}_{end_time_sec}\"\n                row_ids.append(row_id)\n\n                if self.cfg.use_tta:\n                    all_preds = []\n                    for tta_idx in range(self.cfg.tta_count):\n                        mfcc_features = self.process_audio_segment(segment_audio)\n                        mfcc_features = self.apply_tta(mfcc_features, tta_idx)\n                        mfcc_tensor = torch.tensor(mfcc_features, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                        mfcc_tensor = mfcc_tensor.to(self.cfg.device)\n\n                        if len(self.models) == 1:\n                            with torch.no_grad():\n                                outputs = self.models[0](mfcc_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                all_preds.append(probs)\n                        else:\n                            segment_preds = []\n                            for model in self.models:\n                                with torch.no_grad():\n                                    outputs = model(mfcc_tensor)\n                                    probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                    segment_preds.append(probs)\n                            avg_preds = np.mean(segment_preds, axis=0)\n                            all_preds.append(avg_preds)\n                    final_preds = np.mean(all_preds, axis=0)\n                else:\n                    mfcc_features = self.process_audio_segment(segment_audio)\n                    mfcc_tensor = torch.tensor(mfcc_features, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    mfcc_tensor = mfcc_tensor.to(self.cfg.device)\n                    \n                    if len(self.models) == 1:\n                        with torch.no_grad():\n                            outputs = self.models[0](mfcc_tensor)\n                            final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                    else:\n                        segment_preds = []\n                        for model in self.models:\n                            with torch.no_grad():\n                                outputs = model(mfcc_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n                        final_preds = np.mean(segment_preds, axis=0)\n                \n                predictions.append(final_preds)\n        except Exception as e:\n            print(f\"Error processing {audio_path}: {e}\")\n        \n        return row_ids, predictions\n\n    def run_inference(self):\n        \"\"\"\n        Run inference on all test soundscape audio files.\n        \n        :return: Tuple (all_row_ids, all_predictions) aggregated from all files.\n        \"\"\"\n        test_files = list(Path(self.cfg.test_soundscapes).glob('*.ogg'))\n        if self.cfg.debug:\n            print(f\"Debug mode enabled, using only {self.cfg.debug_count} files\")\n            test_files = test_files[:self.cfg.debug_count]\n        print(f\"Found {len(test_files)} test soundscapes\")\n\n        all_row_ids = []\n        all_predictions = []\n\n        for audio_path in tqdm(test_files):\n            row_ids, predictions = self.predict_on_audio(str(audio_path))\n            all_row_ids.extend(row_ids)\n            all_predictions.extend(predictions)\n        \n        return all_row_ids, all_predictions\n\n    def create_submission(self, row_ids, predictions):\n        \"\"\"\n        Create the submission dataframe based on predictions.\n        \n        :param row_ids: List of row identifiers for each segment.\n        :param predictions: List of prediction arrays.\n        :return: A pandas DataFrame formatted for submission.\n        \"\"\"\n        print(\"Creating submission dataframe...\")\n        submission_dict = {'row_id': row_ids}\n        for i, species in enumerate(self.species_ids):\n            submission_dict[species] = [pred[i] for pred in predictions]\n\n        submission_df = pd.DataFrame(submission_dict)\n        submission_df.set_index('row_id', inplace=True)\n\n        sample_sub = pd.read_csv(self.cfg.submission_csv, index_col='row_id')\n        missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n        if missing_cols:\n            print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n            for col in missing_cols:\n                submission_df[col] = 0.0\n\n        submission_df = submission_df[sample_sub.columns]\n        submission_df = submission_df.reset_index()\n        \n        return submission_df\n\n    def smooth_submission(self, submission_path):\n        \"\"\"\n        Post-process the submission CSV by smoothing predictions to enforce temporal consistency.\n        \n        For each soundscape (grouped by the file name part of 'row_id'), each row's predictions\n        are averaged with those of its neighbors using defined weights.\n        \n        :param submission_path: Path to the submission CSV file.\n        \"\"\"\n        print(\"Smoothing submission predictions...\")\n        sub = pd.read_csv(submission_path)\n        cols = sub.columns[1:]\n        # Extract group names by splitting row_id on the last underscore\n        groups = sub['row_id'].str.rsplit('_', n=1).str[0].values\n        unique_groups = np.unique(groups)\n        \n        for group in unique_groups:\n            # Get indices for the current group\n            idx = np.where(groups == group)[0]\n            sub_group = sub.iloc[idx].copy()\n            predictions = sub_group[cols].values\n            new_predictions = predictions.copy()\n            \n            if predictions.shape[0] > 1:\n                # Smooth the predictions using neighboring segments\n                new_predictions[0] = (predictions[0] * 0.8) + (predictions[1] * 0.2)\n                new_predictions[-1] = (predictions[-1] * 0.8) + (predictions[-2] * 0.2)\n                for i in range(1, predictions.shape[0]-1):\n                    new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n            # Replace the smoothed values in the submission dataframe\n            sub.iloc[idx, 1:] = new_predictions\n        \n        sub.to_csv(submission_path, index=False)\n        print(f\"Smoothed submission saved to {submission_path}\")\n\n    def run(self):\n        \"\"\"\n        Main method to execute the complete inference pipeline.\n        \n        This method:\n          - Loads the pre-trained models.\n          - Processes test audio files and runs predictions.\n          - Creates the submission CSV.\n          - Applies smoothing to the predictions.\n        \"\"\"\n        start_time = time.time()\n        print(\"Starting BirdCLEF-2025 inference with MFCC features...\")\n        print(f\"TTA enabled: {self.cfg.use_tta} (variations: {self.cfg.tta_count if self.cfg.use_tta else 0})\")\n        print(f\"Human voice detection & removal enabled: {self.cfg.use_voice_detection}\")\n        \n        self.load_models()\n        if not self.models:\n            print(\"No models found! Please check model paths.\")\n            return\n        \n        print(f\"Model usage: {'Single model' if len(self.models) == 1 else f'Ensemble of {len(self.models)} models'}\")\n        row_ids, predictions = self.run_inference()\n        submission_df = self.create_submission(row_ids, predictions)\n        \n        submission_path = 'submission.csv'\n        submission_df.to_csv(submission_path, index=False)\n        print(f\"Initial submission saved to {submission_path}\")\n        \n        # Apply smoothing on the submission predictions.\n        self.smooth_submission(submission_path)\n        \n        end_time = time.time()\n        print(f\"Inference completed in {(end_time - start_time) / 60:.2f} minutes\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-04T16:16:26.485246Z","iopub.execute_input":"2025-06-04T16:16:26.485571Z","iopub.status.idle":"2025-06-04T16:16:51.064761Z","shell.execute_reply.started":"2025-06-04T16:16:26.485546Z","shell.execute_reply":"2025-06-04T16:16:51.063290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Run the BirdCLEF2025 Pipeline with MFCC features and human voice detection:\nif __name__ == \"__main__\":\n    cfg = CFG()\n    print(f\"Using device: {cfg.device}\")\n    pipeline = BirdCLEF2025Pipeline(cfg)\n    pipeline.run()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T16:17:20.604579Z","iopub.execute_input":"2025-06-04T16:17:20.604930Z","iopub.status.idle":"2025-06-04T16:17:58.749336Z","shell.execute_reply.started":"2025-06-04T16:17:20.604903Z","shell.execute_reply":"2025-06-04T16:17:58.747771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}