{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":349245,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":291621,"modelId":312281}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# \n# Thanks for the following notebooks:\n# https://www.kaggle.com/code/kumarandatascientist/lb-0-804-efficientnet-b0-pytorch-pipeline\n# https://www.kaggle.com/code/kadircandrisolu/transforming-audio-to-mel-spec-birdclef-25\n# https://www.kaggle.com/code/kadircandrisolu/efficientnet-b0-pytorch-train-birdclef-25\n# \n# to toggle comments in rstudio select text then 'ctrl' + 'shift' + 'c'\n# to toggle comments in spyder select text then 'ctrl' + '1'\n# \n# To add or replace secondary_labels:\n# \n# 1. Download models from infer notebook and put in models directory of working dir\n# \n# 2. Modify infer notebook to predict on train.csv like in audio-to-mel notebook.\n#    code below\n# \n# 3. Modify  secondary_labels in train.csv to add to or replace labels above\n#    a percentage in submission. code below\n# \n# 4, Run audio-to-mel notebook. Need to add np.save('bird25mel.npy', all_bird_data)\n#    to save output.\n# \n# 5. Run train notebook with train.csv and bird25mel.npy as inputs\n# \n# N_FFT = 1024\n# HOP_LENGTH = 64\n# N_MELS = 136\n# FMIN = 20\n# FMAX = 16000\n# TARGET_SHAPE = (256, 256)\n# \n# The working dir is set in code in RStudio.\n# The working dir is set from top menu in spyder.\n# \n# \n# spyder python notebook to predict from train.csv\n# #################\n# ###\n# \n# import os\n# import gc\n# import warnings\n# import logging\n# import time\n# import math\n# import cv2\n# from pathlib import Path\n# \n# import numpy as np\n# import pandas as pd\n# import librosa\n# import torch\n# import torch.nn as nn\n# import torch.nn.functional as F\n# import timm\n# from tqdm.auto import tqdm\n# \n# # Suppress warnings and limit logging output\n# warnings.filterwarnings(\"ignore\")\n# logging.basicConfig(level=logging.ERROR)\n# \n# train_csv = pd.read_csv('train.csv')\n# class CFG:\n#     \"\"\"\n#     Configuration class holding all paths and parameters required for the inference pipeline.\n#     \"\"\"\n#     test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n#     submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n#     taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n#     model_path = '/kaggle/input/birdclef-2025-efficientnet-b0'\n# \n#     test_soundscapes = 'c:/bird25/test_soundscapes'\n#     submission_csv = 'c:/bird25/sample_submission.csv'\n#     taxonomy_csv = 'c:/bird25/taxonomy.csv'\n#     model_path = 'c:/bird25/models'\n# \n# \n# \n#     # Audio parameters\n#     FS = 32000\n#     WINDOW_SIZE = 5\n# \n#     # Mel spectrogram parameters\n#     N_FFT = 1034\n#     HOP_LENGTH = 64\n#     N_MELS = 136\n#     FMIN = 20\n#     FMAX = 16000\n#     TARGET_SHAPE = (256, 256)\n# \n#     model_name = 'efficientnet_b0'\n#     in_channels = 1\n#     device = 'cuda' if torch.cuda.is_available() else 'cpu'\n#     # Inference parameters\n#     batch_size = 16\n#     use_tta = False\n#     tta_count = 3\n#     threshold = 0.7\n# \n#     use_specific_folds = False  # If False, use all found models\n#     folds = [0, 1]  # Used only if use_specific_folds is True\n# \n#     debug = False\n#     debug_count = 3\n# \n# \n# class BirdCLEF2025Pipeline:\n#     \"\"\"\n#     Pipeline for the BirdCLEF-2025 inference task.\n# \n#     This class organizes the complete inference process:\n#       - Loading taxonomy data.\n#       - Loading and preparing the trained models.\n#       - Processing audio files into mel spectrograms.\n#       - Making predictions on each audio segment.\n#       - Creating the submission file.\n#       - Post-processing the submission to smooth predictions.\n#     \"\"\"\n# \n#     class BirdCLEFModel(nn.Module):\n#         \"\"\"\n#         Custom neural network model for BirdCLEF-2025 that uses a timm backbone.\n#         \"\"\"\n#         def __init__(self, cfg, num_classes):\n#             \"\"\"\n#             Initialize the BirdCLEFModel.\n# \n#             :param cfg: Configuration parameters.\n#             :param num_classes: Number of output classes.\n#             \"\"\"\n#             super().__init__()\n#             self.cfg = cfg\n#             # Create backbone using timm with specified parameters.\n#             self.backbone = timm.create_model(\n#                 cfg.model_name,\n#                 pretrained=False,\n#                 in_chans=cfg.in_channels,\n#                 drop_rate=0.0,\n#                 drop_path_rate=0.0\n#             )\n#             # Adjust final layers based on model type\n#             if 'efficientnet' in cfg.model_name:\n#                 backbone_out = self.backbone.classifier.in_features\n#                 self.backbone.classifier = nn.Identity()\n#             elif 'resnet' in cfg.model_name:\n#                 backbone_out = self.backbone.fc.in_features\n#                 self.backbone.fc = nn.Identity()\n#             else:\n#                 backbone_out = self.backbone.get_classifier().in_features\n#                 self.backbone.reset_classifier(0, '')\n# \n#             self.pooling = nn.AdaptiveAvgPool2d(1)\n#             self.feat_dim = backbone_out\n#             self.classifier = nn.Linear(backbone_out, num_classes)\n# \n#         def forward(self, x):\n#             \"\"\"\n#             Forward pass through the network.\n# \n#             :param x: Input tensor.\n#             :return: Logits for each class.\n#             \"\"\"\n#             features = self.backbone(x)\n#             if isinstance(features, dict):\n#                 features = features['features']\n#             # If features are 4D, apply global average pooling.\n#             if len(features.shape) == 4:\n#                 features = self.pooling(features)\n#                 features = features.view(features.size(0), -1)\n#             logits = self.classifier(features)\n#             return logits\n# \n#     def __init__(self, cfg):\n#         \"\"\"\n#         Initialize the inference pipeline with the given configuration.\n# \n#         :param cfg: Configuration object with paths and parameters.\n#         \"\"\"\n#         self.cfg = cfg\n#         self.taxonomy_df = None\n#         self.species_ids = []\n#         self.models = []\n#         self._load_taxonomy()\n# \n#     def _load_taxonomy(self):\n#         \"\"\"\n#         Load taxonomy data from CSV and extract species identifiers.\n#         \"\"\"\n#         print(\"Loading taxonomy data...\")\n#         self.taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n#         self.species_ids = self.taxonomy_df['primary_label'].tolist()\n#         print(f\"Number of classes: {len(self.species_ids)}\")\n# \n#     def audio2melspec(self, audio_data):\n#         \"\"\"\n#         Convert raw audio data to a normalized mel spectrogram.\n# \n#         :param audio_data: 1D numpy array of audio samples.\n#         :return: Normalized mel spectrogram.\n#         \"\"\"\n#         if np.isnan(audio_data).any():\n#             mean_signal = np.nanmean(audio_data)\n#             audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n# \n#         mel_spec = librosa.feature.melspectrogram(\n#             y=audio_data,\n#             sr=self.cfg.FS,\n#             n_fft=self.cfg.N_FFT,\n#             hop_length=self.cfg.HOP_LENGTH,\n#             n_mels=self.cfg.N_MELS,\n#             fmin=self.cfg.FMIN,\n#             fmax=self.cfg.FMAX,\n#             power=2.0\n#         )\n#         mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n#         mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n#         return mel_spec_norm\n# \n#     def process_audio_segment(self, audio_data):\n#         \"\"\"\n#         Process an audio segment to obtain a mel spectrogram with the target shape.\n# \n#         :param audio_data: 1D numpy array of audio samples.\n#         :return: Processed mel spectrogram as a float32 numpy array.\n#         \"\"\"\n#         # Pad audio if it is shorter than the required window size.\n#         if len(audio_data) < self.cfg.FS * self.cfg.WINDOW_SIZE:\n#             audio_data = np.pad(\n#                 audio_data,\n#                 (0, self.cfg.FS * self.cfg.WINDOW_SIZE - len(audio_data)),\n#                 mode='constant'\n#             )\n# \n#         mel_spec = self.audio2melspec(audio_data)\n# \n#         # Resize spectrogram to the target shape if necessary.\n#         if mel_spec.shape != self.cfg.TARGET_SHAPE:\n#             mel_spec = cv2.resize(mel_spec, self.cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n# \n#         return mel_spec.astype(np.float32)\n# \n#     def find_model_files(self):\n#         \"\"\"\n#         Find all .pth model files in the specified model directory.\n# \n#         :return: List of model file paths.\n#         \"\"\"\n#         model_files = []\n#         model_dir = Path(self.cfg.model_path)\n#         for path in model_dir.glob('**/*.pth'):\n#             model_files.append(str(path))\n#         return model_files\n# \n#     def load_models(self):\n#         \"\"\"\n#         Load all found model files and prepare them for ensemble inference.\n# \n#         :return: List of loaded PyTorch models.\n#         \"\"\"\n#         self.models = []\n#         model_files = self.find_model_files()\n#         if not model_files:\n#             print(f\"Warning: No model files found under {self.cfg.model_path}!\")\n#             return self.models\n# \n#         print(f\"Found a total of {len(model_files)} model files.\")\n# \n#         # If specific folds are required, filter the model files.\n#         if self.cfg.use_specific_folds:\n#             filtered_files = []\n#             for fold in self.cfg.folds:\n#                 fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n#                 filtered_files.extend(fold_files)\n#             model_files = filtered_files\n#             print(f\"Using {len(model_files)} model files for the specified folds ({self.cfg.folds}).\")\n# \n#         # Load each model file.\n#         for model_path in model_files:\n#             try:\n#                 print(f\"Loading model: {model_path}\")\n#                 checkpoint = torch.load(model_path,weights_only = False, map_location=torch.device(self.cfg.device))\n#                 model = self.BirdCLEFModel(self.cfg, len(self.species_ids))\n#                 model.load_state_dict(checkpoint['model_state_dict'])\n#                 model = model.to(self.cfg.device)\n#                 model.eval()\n#                 self.models.append(model)\n#             except Exception as e:\n#                 print(f\"Error loading model {model_path}: {e}\")\n# \n#         return self.models\n# \n#     def apply_tta(self, spec, tta_idx):\n#         \"\"\"\n#         Apply test-time augmentation (TTA) to the spectrogram.\n# \n#         :param spec: Input mel spectrogram.\n#         :param tta_idx: Index indicating which TTA to apply.\n#         :return: Augmented spectrogram.\n#         \"\"\"\n#         if tta_idx == 0:\n#             # No augmentation.\n#             return spec\n#         elif tta_idx == 1:\n#             # Time shift (horizontal flip).\n#             return np.flip(spec, axis=1)\n#         elif tta_idx == 2:\n#             # Frequency shift (vertical flip).\n#             return np.flip(spec, axis=0)\n#         else:\n#             return spec\n# \n#     def predict_on_spectrogram(self, audio_path):\n#         \"\"\"\n#         Process a single audio file and predict species presence for each 5-second segment.\n# \n#         :param audio_path: Path to the audio file.\n#         :return: Tuple (row_ids, predictions) for each segment.\n#         \"\"\"\n#         predictions = []\n#         row_ids = []\n#         soundscape_id = Path(audio_path).stem\n# \n#         try:\n#             print(f\"Processing {soundscape_id}\")\n#             audio_data, _ = librosa.load(audio_path, sr=self.cfg.FS)\n#             total_segments = 1\n# \n#             for segment_idx in range(total_segments):\n#                 target_samples = int(5*32000)\n#                 if len(audio_data) < target_samples:\n#                     n_copy = math.ceil(target_samples / len(audio_data))\n#                     if n_copy > 1:\n#                        audio_data = np.concatenate([audio_data] * n_copy)\n#                 start_idx = max(0, int(len(audio_data) / 2 - target_samples / 2))\n#                 end_idx = min(len(audio_data), start_idx + target_samples)\n#                 center_audio = audio_data[start_idx:end_idx]\n#                 if len(center_audio) < target_samples:\n#                    center_audio = np.pad(center_audio,\n#                                  (0, target_samples - len(center_audio)),\n#                                  mode='constant')\n#                 segment_audio = center_audio\n# \n#                 end_time_sec = 5\n#                 row_id = f\"{soundscape_id}_{end_time_sec}\"\n#                 row_ids.append(row_id)\n# \n#                 if self.cfg.use_tta:\n#                     all_preds = []\n#                     for tta_idx in range(self.cfg.tta_count):\n#                         mel_spec = self.process_audio_segment(segment_audio)\n#                         mel_spec = self.apply_tta(mel_spec, tta_idx)\n#                         mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n#                         mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n# \n#                         if len(self.models) == 1:\n#                             with torch.no_grad():\n#                                 outputs = self.models[0](mel_spec_tensor)\n#                                 probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n#                                 all_preds.append(probs)\n#                         else:\n#                             segment_preds = []\n#                             for model in self.models:\n#                                 with torch.no_grad():\n#                                     outputs = model(mel_spec_tensor)\n#                                     probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n#                                     segment_preds.append(probs)\n#                             avg_preds = np.mean(segment_preds, axis=0)\n#                             all_preds.append(avg_preds)\n#                     final_preds = np.mean(all_preds, axis=0)\n#                 else:\n#                     mel_spec = self.process_audio_segment(segment_audio)\n#                     mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n#                     mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n# \n#                     if len(self.models) == 1:\n#                         with torch.no_grad():\n#                             outputs = self.models[0](mel_spec_tensor)\n#                             final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n#                     else:\n#                         segment_preds = []\n#                         for model in self.models:\n#                             with torch.no_grad():\n#                                 outputs = model(mel_spec_tensor)\n#                                 probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n#                                 segment_preds.append(probs)\n#                         final_preds = np.mean(segment_preds, axis=0)\n# \n#                 predictions.append(final_preds)\n#         except Exception as e:\n#             print(f\"Error processing {audio_path}: {e}\")\n# \n#         return row_ids, predictions\n# \n#     def run_inference(self):\n#         \"\"\"\n#         Run inference on all test soundscape audio files.\n# \n#         :return: Tuple (all_row_ids, all_predictions) aggregated from all files.\n#         \"\"\"\n# \n#         all_row_ids = []\n#         all_predictions = []\n#         train_csv['fnames']='train_audio/' + train_csv[ 'filename']\n#         for n in range(len(train_csv)):\n#             row_ids, predictions = self.predict_on_spectrogram(str(train_csv['fnames'][n]))\n#             all_row_ids.extend(row_ids)\n#             all_predictions.extend(predictions)\n#             print(n)\n#         return all_row_ids, all_predictions\n# \n#     def create_submission(self, row_ids, predictions):\n#         \"\"\"\n#         Create the submission dataframe based on predictions.\n# \n#         :param row_ids: List of row identifiers for each segment.\n#         :param predictions: List of prediction arrays.\n#         :return: A pandas DataFrame formatted for submission.\n#         \"\"\"\n#         print(\"Creating submission dataframe...\")\n#         submission_dict = {'row_id': row_ids}\n#         for i, species in enumerate(self.species_ids):\n#             submission_dict[species] = [pred[i] for pred in predictions]\n# \n#         submission_df = pd.DataFrame(submission_dict)\n#         submission_df.set_index('row_id', inplace=True)\n# \n#         sample_sub = pd.read_csv(self.cfg.submission_csv, index_col='row_id')\n#         missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n#         if missing_cols:\n#             print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n#             for col in missing_cols:\n#                 submission_df[col] = 0.0\n# \n#         submission_df = submission_df[sample_sub.columns]\n#         submission_df = submission_df.reset_index()\n# \n#         return submission_df\n# \n#     def smooth_submission(self, submission_path):\n#         \"\"\"\n#         Post-process the submission CSV by smoothing predictions to enforce temporal consistency.\n# \n#         For each soundscape (grouped by the file name part of 'row_id'), each row's predictions\n#         are averaged with those of its neighbors using defined weights.\n# \n#         :param submission_path: Path to the submission CSV file.\n#         \"\"\"\n#         print(\"Smoothing submission predictions...\")\n#         sub = pd.read_csv(submission_path)\n#         cols = sub.columns[1:]\n#         # Extract group names by splitting row_id on the last underscore\n#         groups = sub['row_id'].str.rsplit('_', n=1).str[0].values\n#         unique_groups = np.unique(groups)\n# \n#         for group in unique_groups:\n#             # Get indices for the current group\n#             idx = np.where(groups == group)[0]\n#             sub_group = sub.iloc[idx].copy()\n#             predictions = sub_group[cols].values\n#             new_predictions = predictions.copy()\n# \n#             if predictions.shape[0] > 1:\n#                 # Smooth the predictions using neighboring segments\n#                 new_predictions[0] = (predictions[0] * 0.8) + (predictions[1] * 0.2)\n#                 new_predictions[-1] = (predictions[-1] * 0.8) + (predictions[-2] * 0.2)\n#                 for i in range(1, predictions.shape[0]-1):\n#                     new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n#             # Replace the smoothed values in the submission dataframe\n#             sub.iloc[idx, 1:] = new_predictions\n# \n#         sub.to_csv(submission_path, index=False)\n#         print(f\"Smoothed submission saved to {submission_path}\")\n# \n#     def run(self):\n#         \"\"\"\n#         Main method to execute the complete inference pipeline.\n# \n#         This method:\n#           - Loads the pre-trained models.\n#           - Processes test audio files and runs predictions.\n#           - Creates the submission CSV.\n#           - Applies smoothing to the predictions.\n#         \"\"\"\n#         start_time = time.time()\n#         print(\"Starting BirdCLEF-2025 inference...\")\n#         print(f\"TTA enabled: {self.cfg.use_tta} (variations: {self.cfg.tta_count if self.cfg.use_tta else 0})\")\n# \n#         self.load_models()\n#         if not self.models:\n#             print(\"No models found! Please check model paths.\")\n#             return\n# \n#         print(f\"Model usage: {'Single model' if len(self.models) == 1 else f'Ensemble of {len(self.models)} models'}\")\n#         row_ids, predictions = self.run_inference()\n#         submission_df = self.create_submission(row_ids, predictions)\n# \n#         submission_path = 'submission.csv'\n#         submission_df.to_csv(submission_path, index=False)\n#         print(f\"Initial submission saved to {submission_path}\")\n# \n#         # Apply smoothing on the submission predictions.\n#        # self.smooth_submission(submission_path)\n# \n#         end_time = time.time()\n#         print(f\"Inference completed in {(end_time - start_time) / 60:.2f} minutes\")\n# \n# \n# \n# # In[2]:\n# \n# \n# # Run the BirdCLEF2025 Pipeline:\n# if __name__ == \"__main__\":\n#     cfg = CFG()\n#     print(f\"Using device: {cfg.device}\")\n#     pipeline = BirdCLEF2025Pipeline(cfg)\n#     pipeline.run()\n# \n# ###\n# ################\n# \n# \n# rstudio code to Modify  secondary_labels in train.csv\n# ##############\n# ###\n#     library(tidyverse)\n# \n#     # replace or add to secondary_labels\n#     REPLACE = FALSE\n# \n#     # probability to select from submission\n#     PROB = 0.7\n# \n#     setwd(\"C:/bird25\")\n# \n#     # save original train.csv on first run\n#     if(!file.exists(\"train_org.csv\")){\n#       file.copy(from=\"train.csv\", to=\"train_org.csv\")\n#     }\n# \n#     tt<-  read.csv(\"train_org.csv\")\n#     ss<- read.csv(\"submission.csv\",check.names = FALSE)\n#     ss$row_id<- str_sub(ss$row_id, 1, str_length(ss$row_id)-2)\n#     names(ss)[1]<- \"id\"\n#     tt$id<- ss$id\n#     ss$primary_label <- tt$primary_label\n#     ss<-ss %>% relocate(primary_label, .after=id)\n#     pl <- sort(unique(ss$primary_label))\n# \n#     # relocates are just for easier view\n#     tt<-tt %>% relocate(secondary_labels, .after=primary_label)\n#     if(REPLACE==TRUE){\n#       tt$secc<-\"['']\"\n#     }else{\n#       tt$secc<-tt$secondary_labels\n#     }\n#     tt<-tt %>% relocate(secc, .after=secondary_labels)\n# \n#     kn<-0\n#     for(n in 1:length(pl)){\n#       # n=111\n#       s2<-ss[,c(1:2,n+2)]\n#       names(s2)[3]<- \"v\"\n#       s2<- arrange(s2, desc(v))\n#       s2<- s2[s2$v>= PROB,]\n#       s2<- s2[s2$primary_label!=pl[n],]\n# \n#       if(nrow(s2)>0){\n#         for(j in 1:nrow(s2)){\n#           # j=1\n#           g3<-tt[tt$id==s2$id[j] & tt$primary_label==s2$primary_label[j],]\n#           z<- g3$secc\n#           print(z)\n#           if(str_length(z)<5){\n#             g3$secc<- paste0(\"[\",\"'\",pl[n],\"'\",\"]\")\n#           }else{\n#             z<-str_sub(z,2,-2)\n#             g3$secc<- paste0(\"[\",z,\",\",\"'\",pl[n],\"'\",\"]\")\n#           }\n#           tt[tt$id==s2$id[j] & tt$primary_label==s2$primary_label[j],]<-g3\n#           print(g3$secc)\n#           kn=kn+1\n#         }\n#       }\n# \n#     }\n#     print(kn)\n#     tt$secondary_labels<- tt$secc\n#     tt$secc<-NULL\n#     tt$id<-NULL\n#     write.csv(tt, \"train.csv\", row.names = FALSE)\n# \n# ###\n# #############\n# \n# \n\n\nimport os\nimport gc\nimport warnings\nimport logging\nimport time\nimport math\nimport cv2\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport timm\nfrom tqdm.auto import tqdm\n\n# Suppress warnings and limit logging output\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)\n\n\nclass CFG:\n    \"\"\"\n    Configuration class holding all paths and parameters required for the inference pipeline.\n    \"\"\"\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    model_path = '/kaggle/input/modelmon2/pytorch/default/1'\n    \n    # Audio parameters\n    FS = 32000  \n    WINDOW_SIZE = 5  \n    \n    # Mel spectrogram parameters\n    N_FFT = 1034\n    HOP_LENGTH = 64\n    N_MELS = 136\n    FMIN = 20\n    FMAX = 16000\n    TARGET_SHAPE = (256, 256)\n    \n    model_name = 'efficientnet_b0'\n    in_channels = 1\n    device = 'cpu'  \n    \n    # Inference parameters\n    batch_size = 16\n    use_tta = False  \n    tta_count = 3   \n    threshold = 0.7\n    \n    use_specific_folds = False  # If False, use all found models\n    folds = [0, 1]  # Used only if use_specific_folds is True\n    \n    debug = False\n    debug_count = 3\n\n\nclass BirdCLEF2025Pipeline:\n    \"\"\"\n    Pipeline for the BirdCLEF-2025 inference task.\n\n    This class organizes the complete inference process:\n      - Loading taxonomy data.\n      - Loading and preparing the trained models.\n      - Processing audio files into mel spectrograms.\n      - Making predictions on each audio segment.\n      - Creating the submission file.\n      - Post-processing the submission to smooth predictions.\n    \"\"\"\n\n    class BirdCLEFModel(nn.Module):\n        \"\"\"\n        Custom neural network model for BirdCLEF-2025 that uses a timm backbone.\n        \"\"\"\n        def __init__(self, cfg, num_classes):\n            \"\"\"\n            Initialize the BirdCLEFModel.\n            \n            :param cfg: Configuration parameters.\n            :param num_classes: Number of output classes.\n            \"\"\"\n            super().__init__()\n            self.cfg = cfg\n            # Create backbone using timm with specified parameters.\n            self.backbone = timm.create_model(\n                cfg.model_name,\n                pretrained=False,  \n                in_chans=cfg.in_channels,\n                drop_rate=0.0,    \n                drop_path_rate=0.0\n            )\n            # Adjust final layers based on model type\n            if 'efficientnet' in cfg.model_name:\n                backbone_out = self.backbone.classifier.in_features\n                self.backbone.classifier = nn.Identity()\n            elif 'resnet' in cfg.model_name:\n                backbone_out = self.backbone.fc.in_features\n                self.backbone.fc = nn.Identity()\n            else:\n                backbone_out = self.backbone.get_classifier().in_features\n                self.backbone.reset_classifier(0, '')\n            \n            self.pooling = nn.AdaptiveAvgPool2d(1)\n            self.feat_dim = backbone_out\n            self.classifier = nn.Linear(backbone_out, num_classes)\n            \n        def forward(self, x):\n            \"\"\"\n            Forward pass through the network.\n            \n            :param x: Input tensor.\n            :return: Logits for each class.\n            \"\"\"\n            features = self.backbone(x)\n            if isinstance(features, dict):\n                features = features['features']\n            # If features are 4D, apply global average pooling.\n            if len(features.shape) == 4:\n                features = self.pooling(features)\n                features = features.view(features.size(0), -1)\n            logits = self.classifier(features)\n            return logits\n\n    def __init__(self, cfg):\n        \"\"\"\n        Initialize the inference pipeline with the given configuration.\n        \n        :param cfg: Configuration object with paths and parameters.\n        \"\"\"\n        self.cfg = cfg\n        self.taxonomy_df = None\n        self.species_ids = []\n        self.models = []\n        self._load_taxonomy()\n\n    def _load_taxonomy(self):\n        \"\"\"\n        Load taxonomy data from CSV and extract species identifiers.\n        \"\"\"\n        print(\"Loading taxonomy data...\")\n        self.taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = self.taxonomy_df['primary_label'].tolist()\n        print(f\"Number of classes: {len(self.species_ids)}\")\n\n    def audio2melspec(self, audio_data):\n        \"\"\"\n        Convert raw audio data to a normalized mel spectrogram.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: Normalized mel spectrogram.\n        \"\"\"\n        if np.isnan(audio_data).any():\n            mean_signal = np.nanmean(audio_data)\n            audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n        \n        mel_spec = librosa.feature.melspectrogram(\n            y=audio_data,\n            sr=self.cfg.FS,\n            n_fft=self.cfg.N_FFT,\n            hop_length=self.cfg.HOP_LENGTH,\n            n_mels=self.cfg.N_MELS,\n            fmin=self.cfg.FMIN,\n            fmax=self.cfg.FMAX,\n            power=2.0\n        )\n        mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n        mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n        return mel_spec_norm\n\n    def process_audio_segment(self, audio_data):\n        \"\"\"\n        Process an audio segment to obtain a mel spectrogram with the target shape.\n        \n        :param audio_data: 1D numpy array of audio samples.\n        :return: Processed mel spectrogram as a float32 numpy array.\n        \"\"\"\n        # Pad audio if it is shorter than the required window size.\n        if len(audio_data) < self.cfg.FS * self.cfg.WINDOW_SIZE:\n            audio_data = np.pad(\n                audio_data,\n                (0, self.cfg.FS * self.cfg.WINDOW_SIZE - len(audio_data)),\n                mode='constant'\n            )\n        \n        mel_spec = self.audio2melspec(audio_data)\n        \n        # Resize spectrogram to the target shape if necessary.\n        if mel_spec.shape != self.cfg.TARGET_SHAPE:\n            mel_spec = cv2.resize(mel_spec, self.cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n            \n        return mel_spec.astype(np.float32)\n\n    def find_model_files(self):\n        \"\"\"\n        Find all .pth model files in the specified model directory.\n        \n        :return: List of model file paths.\n        \"\"\"\n        model_files = []\n        model_dir = Path(self.cfg.model_path)\n        for path in model_dir.glob('**/*.pth'):\n            model_files.append(str(path))\n        return model_files\n\n    def load_models(self):\n        \"\"\"\n        Load all found model files and prepare them for ensemble inference.\n        \n        :return: List of loaded PyTorch models.\n        \"\"\"\n        self.models = []\n        model_files = self.find_model_files()\n        if not model_files:\n            print(f\"Warning: No model files found under {self.cfg.model_path}!\")\n            return self.models\n\n        print(f\"Found a total of {len(model_files)} model files.\")\n        \n        # If specific folds are required, filter the model files.\n        if self.cfg.use_specific_folds:\n            filtered_files = []\n            for fold in self.cfg.folds:\n                fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n                filtered_files.extend(fold_files)\n            model_files = filtered_files\n            print(f\"Using {len(model_files)} model files for the specified folds ({self.cfg.folds}).\")\n        \n        # Load each model file.\n        for model_path in model_files:\n            try:\n                print(f\"Loading model: {model_path}\")\n                checkpoint = torch.load(model_path, map_location=torch.device(self.cfg.device))\n                model = self.BirdCLEFModel(self.cfg, len(self.species_ids))\n                model.load_state_dict(checkpoint['model_state_dict'])\n                model = model.to(self.cfg.device)\n                model.eval()\n                self.models.append(model)\n            except Exception as e:\n                print(f\"Error loading model {model_path}: {e}\")\n        \n        return self.models\n\n    def apply_tta(self, spec, tta_idx):\n        \"\"\"\n        Apply test-time augmentation (TTA) to the spectrogram.\n        \n        :param spec: Input mel spectrogram.\n        :param tta_idx: Index indicating which TTA to apply.\n        :return: Augmented spectrogram.\n        \"\"\"\n        if tta_idx == 0:\n            # No augmentation.\n            return spec\n        elif tta_idx == 1:\n            # Time shift (horizontal flip).\n            return np.flip(spec, axis=1)\n        elif tta_idx == 2:\n            # Frequency shift (vertical flip).\n            return np.flip(spec, axis=0)\n        else:\n            return spec\n\n    def predict_on_spectrogram(self, audio_path):\n        \"\"\"\n        Process a single audio file and predict species presence for each 5-second segment.\n        \n        :param audio_path: Path to the audio file.\n        :return: Tuple (row_ids, predictions) for each segment.\n        \"\"\"\n        predictions = []\n        row_ids = []\n        soundscape_id = Path(audio_path).stem\n        \n        try:\n            print(f\"Processing {soundscape_id}\")\n            audio_data, _ = librosa.load(audio_path, sr=self.cfg.FS)\n            total_segments = int(len(audio_data) / (self.cfg.FS * self.cfg.WINDOW_SIZE))\n            \n            for segment_idx in range(total_segments):\n                start_sample = segment_idx * self.cfg.FS * self.cfg.WINDOW_SIZE\n                end_sample = start_sample + self.cfg.FS * self.cfg.WINDOW_SIZE\n                segment_audio = audio_data[start_sample:end_sample]\n                \n                end_time_sec = (segment_idx + 1) * self.cfg.WINDOW_SIZE\n                row_id = f\"{soundscape_id}_{end_time_sec}\"\n                row_ids.append(row_id)\n\n                if self.cfg.use_tta:\n                    all_preds = []\n                    for tta_idx in range(self.cfg.tta_count):\n                        mel_spec = self.process_audio_segment(segment_audio)\n                        mel_spec = self.apply_tta(mel_spec, tta_idx)\n                        mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                        mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n\n                        if len(self.models) == 1:\n                            with torch.no_grad():\n                                outputs = self.models[0](mel_spec_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                all_preds.append(probs)\n                        else:\n                            segment_preds = []\n                            for model in self.models:\n                                with torch.no_grad():\n                                    outputs = model(mel_spec_tensor)\n                                    probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                    segment_preds.append(probs)\n                            avg_preds = np.mean(segment_preds, axis=0)\n                            all_preds.append(avg_preds)\n                    final_preds = np.mean(all_preds, axis=0)\n                else:\n                    mel_spec = self.process_audio_segment(segment_audio)\n                    mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    mel_spec_tensor = mel_spec_tensor.to(self.cfg.device)\n                    \n                    if len(self.models) == 1:\n                        with torch.no_grad():\n                            outputs = self.models[0](mel_spec_tensor)\n                            final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                    else:\n                        segment_preds = []\n                        for model in self.models:\n                            with torch.no_grad():\n                                outputs = model(mel_spec_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n                        final_preds = np.mean(segment_preds, axis=0)\n                \n                predictions.append(final_preds)\n        except Exception as e:\n            print(f\"Error processing {audio_path}: {e}\")\n        \n        return row_ids, predictions\n\n    def run_inference(self):\n        \"\"\"\n        Run inference on all test soundscape audio files.\n        \n        :return: Tuple (all_row_ids, all_predictions) aggregated from all files.\n        \"\"\"\n        test_files = list(Path(self.cfg.test_soundscapes).glob('*.ogg'))\n        if self.cfg.debug:\n            print(f\"Debug mode enabled, using only {self.cfg.debug_count} files\")\n            test_files = test_files[:self.cfg.debug_count]\n        print(f\"Found {len(test_files)} test soundscapes\")\n\n        all_row_ids = []\n        all_predictions = []\n\n        for audio_path in tqdm(test_files):\n            row_ids, predictions = self.predict_on_spectrogram(str(audio_path))\n            all_row_ids.extend(row_ids)\n            all_predictions.extend(predictions)\n        \n        return all_row_ids, all_predictions\n\n    def create_submission(self, row_ids, predictions):\n        \"\"\"\n        Create the submission dataframe based on predictions.\n        \n        :param row_ids: List of row identifiers for each segment.\n        :param predictions: List of prediction arrays.\n        :return: A pandas DataFrame formatted for submission.\n        \"\"\"\n        print(\"Creating submission dataframe...\")\n        submission_dict = {'row_id': row_ids}\n        for i, species in enumerate(self.species_ids):\n            submission_dict[species] = [pred[i] for pred in predictions]\n\n        submission_df = pd.DataFrame(submission_dict)\n        submission_df.set_index('row_id', inplace=True)\n\n        sample_sub = pd.read_csv(self.cfg.submission_csv, index_col='row_id')\n        missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n        if missing_cols:\n            print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n            for col in missing_cols:\n                submission_df[col] = 0.0\n\n        submission_df = submission_df[sample_sub.columns]\n        submission_df = submission_df.reset_index()\n        \n        return submission_df\n\n    def smooth_submission(self, submission_path):\n        \"\"\"\n        Post-process the submission CSV by smoothing predictions to enforce temporal consistency.\n        \n        For each soundscape (grouped by the file name part of 'row_id'), each row's predictions\n        are averaged with those of its neighbors using defined weights.\n        \n        :param submission_path: Path to the submission CSV file.\n        \"\"\"\n        print(\"Smoothing submission predictions...\")\n        sub = pd.read_csv(submission_path)\n        cols = sub.columns[1:]\n        # Extract group names by splitting row_id on the last underscore\n        groups = sub['row_id'].str.rsplit('_', n=1).str[0].values\n        unique_groups = np.unique(groups)\n        \n        for group in unique_groups:\n            # Get indices for the current group\n            idx = np.where(groups == group)[0]\n            sub_group = sub.iloc[idx].copy()\n            predictions = sub_group[cols].values\n            new_predictions = predictions.copy()\n            \n            if predictions.shape[0] > 1:\n                # Smooth the predictions using neighboring segments\n                new_predictions[0] = (predictions[0] * 0.8) + (predictions[1] * 0.2)\n                new_predictions[-1] = (predictions[-1] * 0.8) + (predictions[-2] * 0.2)\n                for i in range(1, predictions.shape[0]-1):\n                    new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n            # Replace the smoothed values in the submission dataframe\n            sub.iloc[idx, 1:] = new_predictions\n        \n        sub.to_csv(submission_path, index=False)\n        print(f\"Smoothed submission saved to {submission_path}\")\n\n    def run(self):\n        \"\"\"\n        Main method to execute the complete inference pipeline.\n        \n        This method:\n          - Loads the pre-trained models.\n          - Processes test audio files and runs predictions.\n          - Creates the submission CSV.\n          - Applies smoothing to the predictions.\n        \"\"\"\n        start_time = time.time()\n        print(\"Starting BirdCLEF-2025 inference...\")\n        print(f\"TTA enabled: {self.cfg.use_tta} (variations: {self.cfg.tta_count if self.cfg.use_tta else 0})\")\n        \n        self.load_models()\n        if not self.models:\n            print(\"No models found! Please check model paths.\")\n            return\n        \n        print(f\"Model usage: {'Single model' if len(self.models) == 1 else f'Ensemble of {len(self.models)} models'}\")\n        row_ids, predictions = self.run_inference()\n        submission_df = self.create_submission(row_ids, predictions)\n        \n        submission_path = 'submission.csv'\n        submission_df.to_csv(submission_path, index=False)\n        print(f\"Initial submission saved to {submission_path}\")\n        \n        # Apply smoothing on the submission predictions.\n        self.smooth_submission(submission_path)\n        \n        end_time = time.time()\n        print(f\"Inference completed in {(end_time - start_time) / 60:.2f} minutes\")\n\n# Run the BirdCLEF2025 Pipeline:\nif __name__ == \"__main__\":\n    cfg = CFG()\n    print(f\"Using device: {cfg.device}\")\n    pipeline = BirdCLEF2025Pipeline(cfg)\n    pipeline.run()\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}