{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":12054436,"sourceType":"datasetVersion","datasetId":7586535},{"sourceId":12064760,"sourceType":"datasetVersion","datasetId":7593951},{"sourceId":3729,"sourceType":"modelInstanceVersion","modelInstanceId":2656,"modelId":312}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is the inference notebook for the training notebook : https://www.kaggle.com/code/hosseinbahraminekoo/birdclef2025-train","metadata":{}},{"cell_type":"markdown","source":"* Import all necessary libraries for data processing, audio analysis, model building, evaluation, visualization, and system utilities.\n* Includes packages for parallel processing, plotting, deep learning with PyTorch and timm, and handling warnings.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa\nimport glob\nimport torch\nimport torch.nn as nn\nimport os\nimport random\nfrom joblib import Parallel, delayed\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nfrom ast import literal_eval\nimport timm\nimport pandas.api.types\n\nimport sklearn.metrics\nfrom tqdm import tqdm\nimport gc\nfrom warnings import filterwarnings\nfilterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:01:11.546990Z","iopub.execute_input":"2025-06-05T07:01:11.547299Z","iopub.status.idle":"2025-06-05T07:01:38.292919Z","shell.execute_reply.started":"2025-06-05T07:01:11.547266Z","shell.execute_reply":"2025-06-05T07:01:38.291509Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Configuration class defining dataset paths, audio processing parameters, model input shape, and seed for reproducibility.\n* Also sets the number of classes and checks if test soundscape files are available to determine submission mode.","metadata":{}},{"cell_type":"code","source":"class Config:\n    train_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    seed = 42\n    train_csv = \"/kaggle/input/birdclef-2025/train.csv\"\n    sample_submission_csv = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\n    train_soundscapes = \"/kaggle/input/birdclef-2025/train_soundscapes\"\n    test_soundscapes = \"/kaggle/input/birdclef-2025/test_soundscapes\"\n    # test_soundscapes = \"/kaggle/input/birdclef-2025/test_audio\"\n    sr = int(32e3)\n\n    num_classes = 206\n    n_fft = 2048\n    hop_length = 500\n\n    n_mels = 256\n    fmin = 50\n    fmax = 16000\n    power = 2\n    image_shape = (128, 640, 1)\n    submission_mode = len(glob.glob(\"/kaggle/input/birdclef-2025/test_soundscapes/*.ogg\")) > 0 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:01:38.295532Z","iopub.execute_input":"2025-06-05T07:01:38.296239Z","iopub.status.idle":"2025-06-05T07:01:38.308701Z","shell.execute_reply.started":"2025-06-05T07:01:38.296202Z","shell.execute_reply":"2025-06-05T07:01:38.306803Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Set random seed across Python, NumPy, and PyTorch (CPU and GPU) for reproducibility.\n* Configure PyTorch backend for deterministic behavior to ensure consistent results across runs.","metadata":{}},{"cell_type":"code","source":"def set_seed(seed: int = 42):\n    random.seed(seed)\n    np.random.seed(seed)\n    # reproducible weight initialization\n    torch.manual_seed(seed)\n\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n        \n    torch.backends.cudnn.determinstic = True\n    torch.backends.cudnn.benchmark = False\n\n    print(f\"[info] set seed: {seed}\")\n\nset_seed()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:01:38.310935Z","iopub.execute_input":"2025-06-05T07:01:38.311369Z","iopub.status.idle":"2025-06-05T07:01:38.369965Z","shell.execute_reply.started":"2025-06-05T07:01:38.311338Z","shell.execute_reply":"2025-06-05T07:01:38.367873Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Select soundscape audio files based on submission mode:\n- If in submission mode, load all test soundscape .ogg files\n- Otherwise, load a small subset (5 files) from training soundscapes for debugging or development","metadata":{}},{"cell_type":"code","source":"if Config.submission_mode:\n    sound_dir = glob.glob(Config.test_soundscapes + \"/*.ogg\")\nelse:\n    sound_dir = glob.glob(Config.train_soundscapes + \"/*.ogg\")[:5]\n\nsound_dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:01:38.371068Z","iopub.execute_input":"2025-06-05T07:01:38.371734Z","iopub.status.idle":"2025-06-05T07:01:38.460985Z","shell.execute_reply.started":"2025-06-05T07:01:38.371685Z","shell.execute_reply":"2025-06-05T07:01:38.459541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Process audio files by:\n- Loading audio and scaling amplitude\n- Splitting audio into 5-second chunks\n- Computing Mel spectrograms with specified audio parameters\n- Converting spectrogram to decibel scale and normalizing features to [0,1]\n- Truncating or padding spectrogram width to fixed size (640 frames)\n- Mapping each chunk to a unique row ID for later use\n- Parallelize processing of all soundscape files to speed up feature extraction\n- Combine all chunk mappings into a global dictionary and print total processed chunks","metadata":{}},{"cell_type":"code","source":"%%time\ndef process(audio_path):\n    filename = audio_path.split(\"/\")[-1].split(\".\")[0]\n    data, _ = librosa.load(audio_path, sr = Config.sr)\n\n    data = data * 1024\n\n    # Dividing the data into 5s chunks\n    chunk_duration = 5\n    min_len = chunk_duration * Config.sr\n\n    local_mapper = {}\n    for i in range(0, len(data), min_len):\n        # Making row ids\n        t = i // Config.sr\n        row_id = f\"{filename}_{t + chunk_duration}\"\n\n        chunk_5s = data[i: i + min_len]\n        chunk_10s = np.tile(chunk_5s, 2)\n\n        chunk_10s = chunk_10s.reshape(-1, len(chunk_10s))\n\n        # Converting to mel spectrogram\n        mel_sp = librosa.feature.melspectrogram(\n            y = chunk_10s,\n            sr = Config.sr,\n            fmin = Config.fmin,\n            fmax = Config.fmax,\n            power = Config.power,\n            n_mels = Config.n_mels,\n            n_fft = Config.n_fft,\n            hop_length = Config.hop_length\n        )\n\n        mel_sp = librosa.power_to_db(mel_sp, ref = 1)\n\n        # Normalizing the features\n        eps = 1e-12\n        mel_sp = (mel_sp - mel_sp.min())/(mel_sp.max() - mel_sp.min() + eps)\n\n        mel_sp = mel_sp[:, :, :640]\n        local_mapper[row_id] = mel_sp\n    return local_mapper    \n\n# Loading audio files\nall_mappers = Parallel(\n    n_jobs = -1,\n    backend = \"loky\"\n)(delayed(process)(filepath) for filepath in sound_dir)\n\n# Creating complete mapping\nglobal_mapper = {}\nfor mapper in all_mappers: global_mapper.update(mapper)\n\nprint(f\"[INFO] loaded all audio files, total_items: {len(global_mapper)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:01:38.462100Z","iopub.execute_input":"2025-06-05T07:01:38.462494Z","iopub.status.idle":"2025-06-05T07:02:17.919736Z","shell.execute_reply.started":"2025-06-05T07:01:38.462425Z","shell.execute_reply":"2025-06-05T07:02:17.917727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Display the first key and its corresponding Mel spectrogram array from the processed audio chunks to verify data format","metadata":{}},{"cell_type":"code","source":"for key, val in global_mapper.items():\n    print(key)\n    print(val)\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:02:17.921504Z","iopub.execute_input":"2025-06-05T07:02:17.921891Z","iopub.status.idle":"2025-06-05T07:02:17.930634Z","shell.execute_reply.started":"2025-06-05T07:02:17.921864Z","shell.execute_reply":"2025-06-05T07:02:17.928725Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Define and load EfficientNet-B0 models from specified checkpoint paths for inference:\n- Use a custom Model class wrapping timm’s EfficientNet with single-channel input\n- Load each checkpoint’s state dict onto CPU device for Kaggle compatibility\n- Set models to evaluation mode and store in a list for ensemble prediction","metadata":{}},{"cell_type":"code","source":"%%time\n\nmodel_paths = [\n    '/kaggle/input/effnetb0-mixup-epoch20-fold-3/fold_1_epoch_6_best_effnetB0_val_auc_0.9798_val_loss_0.011425114528982332.pth',\n    '/kaggle/input/effnetb0-mixup-epoch20-fold-3/fold_2_epoch_1_best_effnetB0_val_auc_0.9930_val_loss_0.006461535613466642.pth',\n    #'/kaggle/input/effnetb0-mixup-epoch20-fold-3/fold_0_epoch_19_best_effnetB0_val_auc_0.9527_val_loss_0.015609626224437954.pth'\n]\n\ndevice = torch.device(\"cpu\")  # Explicit for Kaggle\n\nclass Model(nn.Module):\n    def __init__(self, model_name: str, num_classes: int):\n        super().__init__()\n        self.base_model = timm.create_model(\n            model_name=model_name,\n            num_classes=num_classes,\n            pretrained=False,\n            in_chans=1\n        )\n\n    def forward(self, x):\n        return self.base_model(x)\n\ndef load_models(model_paths, model_name, num_classes):\n    models = []\n    for path in model_paths:\n        model = Model(model_name, num_classes)\n        state = torch.load(path, map_location=device)\n        model.load_state_dict(state)\n        model.to(device)\n        model.eval()\n        models.append(model)\n    return models\n\nmodels_pool = load_models(\n    model_paths,\n    model_name=\"tf_efficientnet_b0\",\n    num_classes=Config.num_classes\n)\n\nprint(f\"[INFO] Loaded {len(models_pool)} models.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:02:17.933645Z","iopub.execute_input":"2025-06-05T07:02:17.934023Z","iopub.status.idle":"2025-06-05T07:02:18.438361Z","shell.execute_reply.started":"2025-06-05T07:02:17.933980Z","shell.execute_reply":"2025-06-05T07:02:18.436840Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Generate mini test-time augmentations by returning the original Mel spectrogram tensor\n* along with a randomly time-shifted version to improve model robustness during inference.","metadata":{}},{"cell_type":"code","source":"def mini_tta_batch(mels_t):\n    \"\"\"Return original and time-shifted versions\"\"\"\n    shifted = torch.roll(mels_t, shifts=random.randint(-20, 20), dims=2)\n    return [mels_t, shifted]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:02:18.439834Z","iopub.execute_input":"2025-06-05T07:02:18.440230Z","iopub.status.idle":"2025-06-05T07:02:18.448036Z","shell.execute_reply.started":"2025-06-05T07:02:18.440202Z","shell.execute_reply":"2025-06-05T07:02:18.446530Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Define a PyTorch Dataset class for test data that returns row IDs and corresponding Mel spectrogram tensors.\n* Create a DataLoader for batch processing test samples without shuffling for consistent order.\n* Load best thresholds for binary classification per class from external file.\n* Select the first model from the ensemble to avoid timeout issues and set all models to evaluation mode.\n* Perform inference with mini test-time augmentation (original + random time-shifted spectrogram) using no_grad for efficiency.\n* Average predictions over TTA versions and apply per-class thresholds to generate binary predictions.\n* Store predictions keyed by row ID and verify total predictions match total samples.\n* Prepare the submission DataFrame with row IDs and predicted labels, then save it as a CSV file for submission.","metadata":{}},{"cell_type":"code","source":"class TestDataset(torch.utils.data.Dataset):\n    def __init__(self, mapper):\n        self.mapper = mapper\n        self.ids = list(self.mapper.keys())\n\n    def __len__(self): return len(self.mapper)\n\n    # def __getitem__(self, idx): return self.ids[idx], self.mapper[self.ids[idx]]\n\n    def __getitem__(self, idx):\n        row_id = self.ids[idx]\n        mel_np = self.mapper[row_id]           # NumPy array\n        mel_tensor = torch.from_numpy(mel_np)  # Convert once, fast\n        return row_id, mel_tensor\n\ntest_loader = torch.utils.data.DataLoader(\n    test_ds := TestDataset(global_mapper),\n    batch_size=16,\n    num_workers=0,      # Safer for CPU-only\n    shuffle=False,\n    drop_last=False\n)\n\npred_mapper = {}\nbest_thresholds = np.load(\"/kaggle/input/effnetb0-birdclef2025-fold2-epoch1-best-threshold/best_thresholds.npy\")\n\n# Use only 1 model to avoid timeout\nmodel = models_pool[0]\nmodel.eval()\n\n# Set all models to eval mode just to be safe\nfor model in models_pool:\n    model.eval()\n# Use torch.no_grad() globally\nwith torch.no_grad():\n    for row_ids, mels_t in test_loader:\n        mels_t = mels_t.to(device).float()  # Ensure float32 type\n\n        # Mini TTA\n        versions = mini_tta_batch(mels_t)\n    \n        # Collect predictions\n        batch_preds = []\n        for v in versions:\n            outputs = model(v)\n            probs = torch.sigmoid(outputs).cpu().numpy()\n            batch_preds.append(probs)\n\n        # Average and apply threshold\n        avg_preds = np.mean(batch_preds, axis=0)\n        for i in range(len(row_ids)):\n            pred = (avg_preds[i] >= best_thresholds).astype(float)\n            pred_mapper[row_ids[i]] = pred\n        \n# Sanity check\nprint(f\"[INFO] Total samples: {len(global_mapper)} — Predictions generated: {len(pred_mapper)}\")\n\n# Create submission file\nsample_sub = pd.read_csv(Config.sample_submission_csv)\ncolumns = sample_sub.columns\n\npred_values = list(pred_mapper.values())\nrow_ids = list(pred_mapper.keys())\n\nsub_df = pd.DataFrame(data=pred_values, columns=columns[1:])\nsub_df.insert(0, 'row_id', row_ids)\n\nsub_df.to_csv('submission.csv', index=False)\nprint(f\"[INFO] Submission with TTA saved with shape: {sub_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T07:02:18.449787Z","iopub.execute_input":"2025-06-05T07:02:18.450160Z","iopub.status.idle":"2025-06-05T07:02:51.838491Z","shell.execute_reply.started":"2025-06-05T07:02:18.450133Z","shell.execute_reply":"2025-06-05T07:02:51.837058Z"}},"outputs":[],"execution_count":null}]}