{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11513634,"sourceType":"datasetVersion","datasetId":7220082}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport logging\nimport random\nimport gc\nimport time\nimport cv2\nimport math\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport librosa\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.auto import tqdm\n\nimport timm\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-28T05:36:42.156361Z","iopub.execute_input":"2025-04-28T05:36:42.156664Z","iopub.status.idle":"2025-04-28T05:36:42.162670Z","shell.execute_reply.started":"2025-04-28T05:36:42.156640Z","shell.execute_reply":"2025-04-28T05:36:42.161966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass CFG:\n    \"\"\"\n    Configuration class holding all paths and parameters required for the inference pipeline.\n    \"\"\"\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    model_path = '/kaggle/input/pth-birdclsefnet'\n    \n    # Audio parameters\n    FS = 32000  \n    WINDOW_SIZE = 5  \n    \n    # Mel spectrogram parameters\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    TARGET_SHAPE = (256, 256)\n    \n    model_name = 'efficientnet_b0'\n    in_channels = 1\n    device = 'cpu'  \n    \n    # Inference parameters\n    batch_size = 16\n    use_tta = False  \n    tta_count = 3   \n    threshold = 0.7\n    \n    use_specific_folds = False  # If False, use all found models\n    folds = [0, 1]  # Used only if use_specific_folds is True\n    \n    debug = False\n    debug_count = 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T05:36:42.163746Z","iopub.execute_input":"2025-04-28T05:36:42.164028Z","iopub.status.idle":"2025-04-28T05:36:42.178297Z","shell.execute_reply.started":"2025-04-28T05:36:42.164003Z","shell.execute_reply":"2025-04-28T05:36:42.177530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class BirdCLEF2025Pipeline:\n    \"\"\"\n    Pipeline for the BirdCLEF-2025 inference task.\n\n    This class organizes the complete inference process:\n      - Loading taxonomy data.\n      - Loading and preparing the trained models.\n      - Processing audio files into mel spectrograms.\n      - Making predictions on each audio segment.\n      - Creating the submission file.\n      - Post-processing the submission to smooth predictions.\n    \"\"\"\n\n    class BirdCLEFModel(nn.Module):\n        \"\"\"\n        Custom neural network model for BirdCLEF-2025 that uses a timm backbone.\n        \"\"\"\n        def __init__(self, cfg, num_classes):\n            \"\"\"\n            Initialize the BirdCLEFModel.\n            \n            :param cfg: Configuration parameters.\n            :param num_classes: Number of output classes.\n            \"\"\"\n            super().__init__()\n            self.cfg = cfg\n            # Create backbone using timm with specified parameters.\n            self.backbone = timm.create_model(\n                cfg.model_name,\n                pretrained=False,  \n                in_chans=cfg.in_channels,\n                drop_rate=0.0,    \n                drop_path_rate=0.0\n            )\n            # Adjust final layers based on model type\n            if 'efficientnet' in cfg.model_name:\n                backbone_out = self.backbone.classifier.in_features\n                self.backbone.classifier = nn.Identity()\n            elif 'resnet' in cfg.model_name:\n                backbone_out = self.backbone.fc.in_features\n                self.backbone.fc = nn.Identity()\n            else:\n                backbone_out = self.backbone.get_classifier().in_features\n                self.backbone.reset_classifier(0, '')\n            \n            self.pooling = nn.AdaptiveAvgPool2d(1)\n            self.feat_dim = backbone_out\n            self.classifier = nn.Linear(backbone_out, num_classes)\n            \n        def forward(self, x):\n            \"\"\"\n            Forward pass through the network.\n            \n            :param x: Input tensor.\n            :return: Logits for each class.\n            \"\"\"\n            features = self.backbone(x)\n            if isinstance(features, dict):\n                features = features['features']\n            # If features are 4D, apply global average pooling.\n            if len(features.shape) == 4:\n                features = self.pooling(features)\n                features = features.view(features.size(0), -1)\n            logits = self.classifier(features)\n            return logits\n\n    def __init__(self, cfg):\n        \"\"\"\n        Initialize the inference pipeline with the given configuration.\n        \n        :param cfg: Configuration object with paths and parameters.\n        \"\"\"\n        self.cfg = cfg\n        self.taxonomy_df = None\n        self.species_ids = []\n        self.models = []\n        self._load_taxonomy()\n\n    def _load_taxonomy(self):\n        \"\"\"\n        Load taxonomy data from CSV and extract species identifiers.\n        \"\"\"\n        print(\"Loading taxonomy data...\")\n        self.taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = self.taxonomy_df['primary_label'].tolist()\n        print(f\"Number of classes: {len(self.species_ids)}\")\n\n    def audio2melspec(self, audio_data):\n        \"\"\"\n        Convert raw audio data to a normalized mel spectrogram.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: Normalized mel spectrogram.\n        \"\"\"\n        if np.isnan(audio_data).any():\n            mean_signal = np.nanmean(audio_data)\n            audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n        \n        mel_spec = librosa.feature.melspectrogram(\n            y=audio_data,\n            sr=self.cfg.FS,\n            n_fft=self.cfg.N_FFT,\n            hop_length=self.cfg.HOP_LENGTH,\n            n_mels=self.cfg.N_MELS,\n            fmin=self.cfg.FMIN,\n            fmax=self.cfg.FMAX,\n            power=2.0\n        )\n        mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n        mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n        return mel_spec_norm\n\n    def process_audio_segment(self, audio_data):\n        \"\"\"\n        Process an audio segment to obtain a mel spectrogram with the target shape.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: Processed mel spectrogram as a float32 numpy array.\n        \"\"\"\n        ##\n        #add\n        ##\n        # Pad audio if it is shorter than the required window size.\n        if len(audio_data) < self.cfg.FS * self.cfg.WINDOW_SIZE:\n            audio_data = np.pad(\n                audio_data,\n                (0, self.cfg.FS * self.cfg.WINDOW_SIZE - len(audio_data)),\n                mode='constant'\n            )\n        \n        mel_spec = self.audio2melspec(audio_data)\n        \n        # Resize spectrogram to the target shape if necessary.\n        if mel_spec.shape != self.cfg.TARGET_SHAPE:\n            mel_spec = cv2.resize(mel_spec, self.cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n            \n        return mel_spec.astype(np.float32)\n\n    def find_model_files(self):\n        \"\"\"\n        Find all .pth model files in the specified model directory.\n        \n        :return: List of model file paths.\n        \"\"\"\n        model_files = []\n        model_dir = Path(self.cfg.model_path)\n        for path in model_dir.glob('**/*.pth'):\n            model_files.append(str(path))\n        return model_files\n\n    def load_models(self):\n        \"\"\"\n        Load all found model files and prepare them for ensemble inference.\n        \n        :return: List of loaded PyTorch models.\n        \"\"\"\n        self.models = []\n        model_files = self.find_model_files()\n        if not model_files:\n            print(f\"Warning: No model files found under {self.cfg.model_path}!\")\n            return self.models\n\n        print(f\"Found a total of {len(model_files)} model files.\")\n        \n        # If specific folds are required, filter the model files.\n        if self.cfg.use_specific_folds:\n            filtered_files = []\n            for fold in self.cfg.folds:\n                fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n                filtered_files.extend(fold_files)\n            model_files = filtered_files\n            print(f\"Using {len(model_files)} model files for the specified folds ({self.cfg.folds}).\")\n        \n        # Load each model file.\n        for model_path in model_files:\n            try:\n                print(f\"Loading model: {model_path}\")\n                checkpoint = torch.load(model_path, map_location=torch.device(self.cfg.device))\n                model = self.BirdCLEFModel(self.cfg, len(self.species_ids))\n                model.load_state_dict(checkpoint['model_state_dict'])\n                model = model.to(self.cfg.device)\n                model.eval()\n                self.models.append(model)\n            except Exception as e:\n                print(f\"Error loading model {model_path}: {e}\")\n        \n        return self.models\n\n    def apply_tta(self, spec, tta_idx):\n        \"\"\"\n        Apply test-time augmentation (TTA) to the spectrogram.\n        \n        :param spec: Input mel spectrogram.\n        :param tta_idx: Index indicating which TTA to apply.\n        :return: Augmented spectrogram.\n        \"\"\"\n        if tta_idx == 0:\n            # No augmentation.\n            return spec\n        elif tta_idx == 1:\n            # Time shift (horizontal flip).\n            return np.flip(spec, axis=1)\n        elif tta_idx == 2:\n            # Frequency shift (vertical flip).\n            return np.flip(spec, axis=0)\n        else:\n            return spec\n\n    def predict_on_spectrogram(self, audio_path):\n        \"\"\"\n        Process a single audio file and predict species presence for each 5-second segment.\n        \n        :param audio_path: Path to the audio file.\n        :return: Tuple (row_ids, predictions) for each segment.\n        \"\"\"\n        predictions = []\n        row_ids = []\n        soundscape_id = Path(audio_path).stem\n        \n        try:\n            print(f\"Processing {soundscape_id}\")\n            audio_data, _ = librosa.load(audio_path, sr=self.cfg.FS)\n            total_segments = int(len(audio_data) / (self.cfg.FS * self.cfg.WINDOW_SIZE))\n            \n            for segment_idx in range(total_segments):\n                start_sample = segment_idx * self.cfg.FS * self.cfg.WINDOW_SIZE\n                end_sample = start_sample + self.cfg.FS * self.cfg.WINDOW_SIZE\n                segment_audio = audio_data[start_sample:end_sample]\n                \n                end_time_sec = (segment_idx + 1) * self.cfg.WINDOW_SIZE\n                row_id = f\"{soundscape_id}_{end_time_sec}\"\n                row_ids.append(row_id)\n\n                if self.cfg.use_tta:\n                    all_preds = []\n                    for tta_idx in range(self.cfg.tta_count):\n                        mel_spec = self.process_audio_segment(segment_audio)\n                        mel_spec = self.apply_tta(mel_spec, tta_idx)\n                        mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                        mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n\n                        if len(self.models) == 1:\n                            with torch.no_grad():\n                                outputs = self.models[0](mel_spec_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                all_preds.append(probs)\n                        else:\n                            segment_preds = []\n                            for model in self.models:\n                                with torch.no_grad():\n                                    outputs = model(mel_spec_tensor)\n                                    probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                    segment_preds.append(probs)\n                            avg_preds = np.mean(segment_preds, axis=0)\n                            all_preds.append(avg_preds)\n                    final_preds = np.mean(all_preds, axis=0)\n                else:\n                    mel_spec = self.process_audio_segment(segment_audio)\n                    mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n                    \n                    if len(self.models) == 1:\n                        with torch.no_grad():\n                            outputs = self.models[0](mel_spec_tensor)\n                            final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                    else:\n                        segment_preds = []\n                        for model in self.models:\n                            with torch.no_grad():\n                                outputs = model(mel_spec_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n                        final_preds = np.mean(segment_preds, axis=0)\n                \n                predictions.append(final_preds)\n        except Exception as e:\n            print(f\"Error processing {audio_path}: {e}\")\n        \n        return row_ids, predictions\n\n    def run_inference(self):\n        \"\"\"\n        Run inference on all test soundscape audio files.\n        \n        :return: Tuple (all_row_ids, all_predictions) aggregated from all files.\n        \"\"\"\n        test_files = list(Path(self.cfg.test_soundscapes).glob('*.ogg'))\n        if self.cfg.debug:\n            print(f\"Debug mode enabled, using only {self.cfg.debug_count} files\")\n            test_files = test_files[:self.cfg.debug_count]\n        print(f\"Found {len(test_files)} test soundscapes\")\n\n        all_row_ids = []\n        all_predictions = []\n\n        for audio_path in tqdm(test_files):\n            row_ids, predictions = self.predict_on_spectrogram(str(audio_path))\n            all_row_ids.extend(row_ids)\n            all_predictions.extend(predictions)\n        \n        return all_row_ids, all_predictions\n\n    def create_submission(self, row_ids, predictions):\n        \"\"\"\n        Create the submission dataframe based on predictions.\n        \n        :param row_ids: List of row identifiers for each segment.\n        :param predictions: List of prediction arrays.\n        :return: A pandas DataFrame formatted for submission.\n        \"\"\"\n        print(\"Creating submission dataframe...\")\n        submission_dict = {'row_id': row_ids}\n        for i, species in enumerate(self.species_ids):\n            submission_dict[species] = [pred[i] for pred in predictions]\n\n        submission_df = pd.DataFrame(submission_dict)\n        submission_df.set_index('row_id', inplace=True)\n\n        sample_sub = pd.read_csv(self.cfg.submission_csv, index_col='row_id')\n        missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n        if missing_cols:\n            print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n            for col in missing_cols:\n                submission_df[col] = 0.0\n\n        submission_df = submission_df[sample_sub.columns]\n        submission_df = submission_df.reset_index()\n        \n        return submission_df\n\n    def smooth_submission(self, submission_path):\n        \"\"\"\n        Post-process the submission CSV by smoothing predictions to enforce temporal consistency.\n        \n        For each soundscape (grouped by the file name part of 'row_id'), each row's predictions\n        are averaged with those of its neighbors using defined weights.\n        \n        :param submission_path: Path to the submission CSV file.\n        \"\"\"\n        print(\"Smoothing submission predictions...\")\n        sub = pd.read_csv(submission_path)\n        cols = sub.columns[1:]\n        # Extract group names by splitting row_id on the last underscore\n        groups = sub['row_id'].str.rsplit('_', n=1).str[0].values\n        unique_groups = np.unique(groups)\n        \n        for group in unique_groups:\n            # Get indices for the current group\n            idx = np.where(groups == group)[0]\n            sub_group = sub.iloc[idx].copy()\n            predictions = sub_group[cols].values\n            new_predictions = predictions.copy()\n            \n            if predictions.shape[0] > 1:\n                # Smooth the predictions using neighboring segments\n                new_predictions[0] = (predictions[0] * 0.8) + (predictions[1] * 0.2)\n                new_predictions[-1] = (predictions[-1] * 0.8) + (predictions[-2] * 0.2)\n                for i in range(1, predictions.shape[0]-1):\n                    new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n            # Replace the smoothed values in the submission dataframe\n            sub.iloc[idx, 1:] = new_predictions\n        \n        sub.to_csv(submission_path, index=False)\n        print(f\"Smoothed submission saved to {submission_path}\")\n\n    def run(self):\n        \"\"\"\n        Main method to execute the complete inference pipeline.\n        \n        This method:\n          - Loads the pre-trained models.\n          - Processes test audio files and runs predictions.\n          - Creates the submission CSV.\n          - Applies smoothing to the predictions.\n        \"\"\"\n        start_time = time.time()\n        print(\"Starting BirdCLEF-2025 inference...\")\n        print(f\"TTA enabled: {self.cfg.use_tta} (variations: {self.cfg.tta_count if self.cfg.use_tta else 0})\")\n        \n        self.load_models()\n        if not self.models:\n            print(\"No models found! Please check model paths.\")\n            return\n        \n        print(f\"Model usage: {'Single model' if len(self.models) == 1 else f'Ensemble of {len(self.models)} models'}\")\n        row_ids, predictions = self.run_inference()\n        submission_df = self.create_submission(row_ids, predictions)\n        \n        submission_path = 'submission.csv'\n        submission_df.to_csv(submission_path, index=False)\n        print(f\"Initial submission saved to {submission_path}\")\n        \n        # Apply smoothing on the submission predictions.\n        self.smooth_submission(submission_path)\n        \n        end_time = time.time()\n        print(f\"Inference completed in {(end_time - start_time) / 60:.2f} minutes\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T05:36:42.344087Z","iopub.execute_input":"2025-04-28T05:36:42.344670Z","iopub.status.idle":"2025-04-28T05:36:42.379034Z","shell.execute_reply.started":"2025-04-28T05:36:42.344642Z","shell.execute_reply":"2025-04-28T05:36:42.378379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the BirdCLEF2025 Pipeline:\nif __name__ == \"__main__\":\n    cfg = CFG()\n    print(f\"Using device: {cfg.device}\")\n    pipeline = BirdCLEF2025Pipeline(cfg)\n    pipeline.run()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-28T05:36:42.380291Z","iopub.execute_input":"2025-04-28T05:36:42.380570Z","iopub.status.idle":"2025-04-28T05:36:42.629353Z","shell.execute_reply.started":"2025-04-28T05:36:42.380551Z","shell.execute_reply":"2025-04-28T05:36:42.628729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}