{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":239797157,"sourceType":"kernelVersion"},{"sourceId":237327950,"sourceType":"kernelVersion"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"IS_SUBMISSION = False\nFOLD = 0 # -1 means take all folds and ensemble them\n#NUM_TEST_FILES = 700","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:41.951877Z","iopub.execute_input":"2025-11-25T20:30:41.952384Z","iopub.status.idle":"2025-11-25T20:30:41.958729Z","shell.execute_reply.started":"2025-11-25T20:30:41.952360Z","shell.execute_reply":"2025-11-25T20:30:41.958056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torchaudio\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nimport timm\nimport os\nfrom glob import glob\nimport json\nfrom types import SimpleNamespace\nimport torchvision\nimport itertools\n\n# Set seed\nnp.random.seed(42)\ntorch.manual_seed(42);","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:45.301084Z","iopub.execute_input":"2025-11-25T20:30:45.301871Z","iopub.status.idle":"2025-11-25T20:30:50.179086Z","shell.execute_reply.started":"2025-11-25T20:30:45.301840Z","shell.execute_reply":"2025-11-25T20:30:50.178292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nDEVICE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.180605Z","iopub.execute_input":"2025-11-25T20:30:50.181088Z","iopub.status.idle":"2025-11-25T20:30:50.217127Z","shell.execute_reply.started":"2025-11-25T20:30:50.181064Z","shell.execute_reply":"2025-11-25T20:30:50.216317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IDX_TO_LABEL = sorted(pd.read_csv('/kaggle/input/birdclef-2025/train.csv').primary_label.unique())\nNUM_LABELS = len(IDX_TO_LABEL)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.218111Z","iopub.execute_input":"2025-11-25T20:30:50.218370Z","iopub.status.idle":"2025-11-25T20:30:50.346390Z","shell.execute_reply.started":"2025-11-25T20:30:50.218352Z","shell.execute_reply":"2025-11-25T20:30:50.345640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"global_model_dir = glob('/kaggle/input/bc25-train-*')[0]\nglobal_model_dir","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.348239Z","iopub.execute_input":"2025-11-25T20:30:50.348528Z","iopub.status.idle":"2025-11-25T20:30:50.354644Z","shell.execute_reply.started":"2025-11-25T20:30:50.348506Z","shell.execute_reply":"2025-11-25T20:30:50.353858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(f'{global_model_dir}/CFG.json', 'r') as file:\n    CFG = json.load(file)\nCFG = SimpleNamespace(**CFG)\nCFG","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.355508Z","iopub.execute_input":"2025-11-25T20:30:50.355792Z","iopub.status.idle":"2025-11-25T20:30:50.370987Z","shell.execute_reply.started":"2025-11-25T20:30:50.355761Z","shell.execute_reply":"2025-11-25T20:30:50.370230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Model1(torch.nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.backbone = timm.create_model(\n            'efficientnet_b0',\n            pretrained=False,\n            in_chans=1,\n            drop_rate=0.2,\n            drop_path_rate=0.2\n        )\n        backbone_out = self.backbone.classifier.in_features\n        self.backbone.classifier = torch.nn.Identity()\n        self.pooling = torch.nn.AdaptiveAvgPool2d(1)\n        self.feat_dim = backbone_out\n        self.classifier = torch.nn.Linear(backbone_out, NUM_LABELS)\n\n    def forward(self, x):\n        features = self.backbone(x)\n        return self.classifier(features)\n\n\ndef make_model():\n    if CFG.MODEL_VERSION == 0:\n        return timm.create_model(\n            'tf_efficientnet_b0',\n            in_chans=1,\n            num_classes=NUM_LABELS,\n            pretrained=False,\n        )\n    elif CFG.MODEL_VERSION == 1:\n        return Model1()\n    elif CFG.MODEL_VERSION == 2:\n        return timm.create_model(\n            'tf_efficientnet_b0',\n            in_chans=1,\n            num_classes=NUM_LABELS,\n            pretrained=False,\n        )\n    elif CFG.MODEL_VERSION == 3:\n        return timm.create_model(\n            'efficientnet_b0',\n            in_chans=1,\n            num_classes=NUM_LABELS,\n            pretrained=False,\n        )\n    elif CFG.MODEL_VERSION == 6:\n        return timm.create_model(\n            'efficientnet_b0',\n            pretrained=False,\n            in_chans=CFG.IN_CHANNELS,\n            drop_rate=0.2,\n            drop_path_rate=0.2,\n            num_classes=NUM_LABELS,\n        )\n    elif CFG.MODEL_VERSION == 7:\n        return timm.create_model(\n            'efficientnet_b2',\n            pretrained=False,\n            in_chans=CFG.IN_CHANNELS,\n            drop_rate=0.2,\n            drop_path_rate=0.2,\n            num_classes=NUM_LABELS,\n        )\n\n\ndef load_model(fold):\n    weight_paths = glob(f'/kaggle/input/bc25-train-*/model_state_dict_fold_{fold}_epoch_*.pt')\n    model_dir = os.path.dirname(weight_paths[0])\n    \n    if hasattr(CFG, 'FULL_DATA') and CFG.FULL_DATA:\n        best_epoch = max(\n            int(e.split('_')[-1].split('.')[0])\n            for e in glob(f'{model_dir}/model_state_dict_fold_{fold}_epoch_*.pt')\n        )\n    else:\n        val_metrics = torch.load(f'{model_dir}/val_metrics_fold_{fold}.pt', weights_only=True)\n        best_epoch = np.argmin([m['loss'] for m in val_metrics])\n\n    weights_path = f'{model_dir}/model_state_dict_fold_{fold}_epoch_{best_epoch}.pt'\n    print(weights_path)\n    weights = torch.load(weights_path, weights_only=True, map_location=DEVICE)\n    model = make_model().to(DEVICE)\n    model.load_state_dict(weights)\n    model.eval()\n    return model\n\n\nif FOLD == -1:\n    models = Parallel(n_jobs=-1)(\n        delayed(load_model)(fold) for fold in tqdm(range(5), desc=\"Loading models\")\n    )\nelse:\n    models = [load_model(FOLD)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.372344Z","iopub.execute_input":"2025-11-25T20:30:50.373095Z","iopub.status.idle":"2025-11-25T20:30:50.687884Z","shell.execute_reply.started":"2025-11-25T20:30:50.373075Z","shell.execute_reply":"2025-11-25T20:30:50.687229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"to_spec = torch.nn.Sequential(\n    torchaudio.transforms.MelSpectrogram(\n        sample_rate=32000,\n        n_mels=CFG.NUM_MELS,\n        n_fft=CFG.N_FFT,\n        hop_length=CFG.HOP_LENGTH,\n        center=False,\n        power=2,\n    ),\n    torchaudio.transforms.AmplitudeToDB(\n        stype=\"power\",\n        top_db=80.0,\n    )\n).to(DEVICE)\n\nresize_spec = (\n    torchvision.transforms.Resize(CFG.RESIZE_TARGET) \n    if hasattr(CFG, 'RESIZE_TARGET') and CFG.RESIZE_TARGET \n    else None\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.688801Z","iopub.execute_input":"2025-11-25T20:30:50.689036Z","iopub.status.idle":"2025-11-25T20:30:50.710997Z","shell.execute_reply.started":"2025-11-25T20:30:50.689017Z","shell.execute_reply":"2025-11-25T20:30:50.710169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\ntest_soundscape_path = (\n    '/kaggle/input/birdclef-2025/test_soundscapes/' \n    if IS_SUBMISSION else \n    '/kaggle/input/birdclef-2025/train_soundscapes'\n)\ntest_soundscapes = [\n    os.path.join(test_soundscape_path, afile) \n    for afile in sorted(os.listdir(test_soundscape_path)) \n    if afile.endswith('.ogg')\n]\ntest_soundscapes = (\n    test_soundscapes\n    if IS_SUBMISSION else\n    test_soundscapes#[:NUM_TEST_FILES]\n)\n\"\"\";","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.711838Z","iopub.execute_input":"2025-11-25T20:30:50.712061Z","iopub.status.idle":"2025-11-25T20:30:50.716069Z","shell.execute_reply.started":"2025-11-25T20:30:50.712044Z","shell.execute_reply":"2025-11-25T20:30:50.715252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/bc25-eda/train_metadata_joined.parquet')\ntest_soundscapes = [f'/kaggle/input/birdclef-2025/train_audio/{p}' for p in list(df.filename)]\ntest_soundscapes = test_soundscapes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.716918Z","iopub.execute_input":"2025-11-25T20:30:50.717262Z","iopub.status.idle":"2025-11-25T20:30:50.811549Z","shell.execute_reply.started":"2025-11-25T20:30:50.717236Z","shell.execute_reply":"2025-11-25T20:30:50.810906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Quantizer:\n    def __init__(self, num_bits):\n        self.range = 2**num_bits\n        self.max = 2**(num_bits - 1) - 1\n        self.min = -2**(num_bits - 1)\n        if num_bits <= 8:\n            self.dtype = torch.int8\n        elif num_bits <= 16:\n            self.dtype = torch.int16\n        elif num_bits <= 32:\n            self.dtype = torch.int32\n\n    def quantize(self, tensor):\n        min_val = tensor.min()\n        max_val = tensor.max()\n        if min_val == max_val: # Edge case: all values are the same\n            return torch.full_like(tensor, 0, dtype=self.dtype), min_val, max_val\n        scale = self.range / (max_val - min_val)\n        quantized_tensor = torch.round((tensor - min_val) * scale + self.min).clamp(self.min, self.max).to(self.dtype)\n        return quantized_tensor, min_val, max_val\n\n    def dequantize(self, quantized_tensor, min_val, max_val):\n        if min_val == max_val:\n            return torch.full_like(quantized_tensor, min_val, dtype=torch.float32)\n        scale = (max_val - min_val) / self.range\n        return (quantized_tensor.to(torch.float32) - self.min) * scale + min_val\n\n\nq = Quantizer(16)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.813247Z","iopub.execute_input":"2025-11-25T20:30:50.813712Z","iopub.status.idle":"2025-11-25T20:30:50.820385Z","shell.execute_reply.started":"2025-11-25T20:30:50.813685Z","shell.execute_reply":"2025-11-25T20:30:50.819504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\n\ndef circular_pad(x, n):\n    s = x.shape[-1]\n    n_extra = math.ceil(n / s) + 1\n    y = torch.concat([x]*n_extra, axis=-1)\n    return y[:s+n]\n\ncircular_pad(torch.arange(10), 5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.821096Z","iopub.execute_input":"2025-11-25T20:30:50.821304Z","iopub.status.idle":"2025-11-25T20:30:50.839234Z","shell.execute_reply.started":"2025-11-25T20:30:50.821287Z","shell.execute_reply":"2025-11-25T20:30:50.838504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NUM_FRAMES = 32000*5\n\ndef get_spec_chunks(soundscape):\n    audio = torchaudio.load(soundscape, num_frames=NUM_FRAMES)[0][0]\n    max_range = audio.shape[0] - NUM_FRAMES\n    if max_range < 0:\n        audio = circular_pad(audio, -max_range)\n    audio = audio.to(DEVICE).reshape(-1, 1, NUM_FRAMES)\n    spec = to_spec(audio)\n    return q.quantize(spec)\n\nall_chunks = Parallel(n_jobs=-1)(\n    delayed(get_spec_chunks)(f) for f in tqdm(test_soundscapes, desc=\"Loading files\")\n)\n#all_chunks = list(itertools.chain.from_iterable(all_chunks)) # flatten\nlen(all_chunks), all_chunks[0][0].shape if len(all_chunks) > 0 else 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:30:50.840043Z","iopub.execute_input":"2025-11-25T20:30:50.840282Z","iopub.status.idle":"2025-11-25T20:32:57.100266Z","shell.execute_reply.started":"2025-11-25T20:30:50.840265Z","shell.execute_reply":"2025-11-25T20:32:57.099537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FILE_IN_BATCH = 16","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:32:57.101797Z","iopub.execute_input":"2025-11-25T20:32:57.102048Z","iopub.status.idle":"2025-11-25T20:32:57.105606Z","shell.execute_reply.started":"2025-11-25T20:32:57.102030Z","shell.execute_reply":"2025-11-25T20:32:57.104837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare(chunk):\n    spec = q.dequantize(*chunk)\n    if CFG.IN_CHANNELS > 1:\n        spec = spec.expand(-1, CFG.IN_CHANNELS, -1, -1)\n    else:\n        spec = spec[None]\n    return spec[0]\n\nprobs = []\nwith torch.inference_mode():\n    for i in tqdm(range(0, len(all_chunks), FILE_IN_BATCH), desc=\"Running inference\"):\n        batch = torch.stack([prepare(c) for c in all_chunks[i:i+FILE_IN_BATCH]])\n        logits = torch.stack([model(batch) for model in models])\n        #logits = stacked_model(batch)\n        probs.append(torch.nn.functional.softmax(logits, dim=-1).mean(0))\n        \n    if len(probs) > 0:\n        #probs = smoothen(torch.concat(probs)).to('cpu')\n        probs = torch.concat(probs).to('cpu')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:32:57.106411Z","iopub.execute_input":"2025-11-25T20:32:57.106761Z","iopub.status.idle":"2025-11-25T20:33:12.622387Z","shell.execute_reply.started":"2025-11-25T20:32:57.106735Z","shell.execute_reply":"2025-11-25T20:33:12.621747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"torch.save(probs, \"probs.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:33:12.623375Z","iopub.execute_input":"2025-11-25T20:33:12.623738Z","iopub.status.idle":"2025-11-25T20:33:12.634000Z","shell.execute_reply.started":"2025-11-25T20:33:12.623710Z","shell.execute_reply":"2025-11-25T20:33:12.633202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"row_ids = []\nfor soundscape in test_soundscapes:\n    for i in range(12):\n        row_ids.append(os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}')\n\npredictions = pd.DataFrame(probs, columns=IDX_TO_LABEL)\npredictions['row_id'] = df.filename\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T20:33:12.634844Z","iopub.execute_input":"2025-11-25T20:33:12.635036Z","iopub.status.idle":"2025-11-25T20:33:14.287682Z","shell.execute_reply.started":"2025-11-25T20:33:12.635021Z","shell.execute_reply":"2025-11-25T20:33:14.286920Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"code","source":"LABEL_TO_IDX = dict((v, k) for k, v in enumerate(IDX_TO_LABEL))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"N = probs.shape[0]\n\ntop = probs.max(1)\ntop_predicted_idx = top[1]\ntop_predicted_label = [IDX_TO_LABEL[i] for i in top_predicted_idx]\ntop_predicted_prob = top[0]\n\nprimary_label = list(df.primary_label)[:N]\nprimary_label_prob = [probs[i, LABEL_TO_IDX[label]].item() for i, label in enumerate(primary_label)]\n\ngt_idx = torch.tensor([LABEL_TO_IDX[l] for l in primary_label])\nsorted_idx = probs.argsort(dim=1, descending=True)\nprimary_label_rank = (sorted_idx == gt_idx.unsqueeze(1)).nonzero()[:,1]\n\naccuracy = [float(top_predicted_label[i] == primary_label[i]) for i in range(N)]\nsplit = ['val' if f else 'train' for f in df.fold[:N] == FOLD]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metrics = pd.DataFrame(dict(\n    filename=df.filename[:N],\n    top_predicted_label=top_predicted_label,\n    top_predicted_prob=top_predicted_prob,\n    primary_label=primary_label,\n    primary_label_prob=primary_label_prob,\n    primary_label_rank=primary_label_rank,\n    accuracy=accuracy,\n    split=split\n))\nmetrics.to_csv('metrics.csv', index=False)\nmetrics","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metrics.groupby('split').accuracy.describe()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}