{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11014516,"sourceType":"datasetVersion","datasetId":5202665},{"sourceId":11752775,"sourceType":"datasetVersion","datasetId":7281527}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q /kaggle/input/openvino-package/openvino-2025.0.0-17942-cp310-cp310-manylinux2014_x86_64.whl --no-index --find-links /kaggle/input/openvino-package","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-04T11:01:24.413409Z","iopub.execute_input":"2025-05-04T11:01:24.413748Z","iopub.status.idle":"2025-05-04T11:01:32.442546Z","shell.execute_reply.started":"2025-05-04T11:01:24.413711Z","shell.execute_reply":"2025-05-04T11:01:32.441225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchaudio\nimport torch.nn as nn\nimport torchvision.transforms.v2 as v2\nimport torchvision.transforms.functional as F\nimport random\nimport numpy as np\nfrom torch.nn import functional as F_nn\nimport openvino as ov\nimport numpy as np\nimport torch\nimport timm\nimport torchaudio\nimport torch.nn as nn\nfrom collections import OrderedDict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:08:38.205584Z","iopub.execute_input":"2025-04-30T17:08:38.205922Z","iopub.status.idle":"2025-04-30T17:08:52.027926Z","shell.execute_reply.started":"2025-04-30T17:08:38.205893Z","shell.execute_reply":"2025-04-30T17:08:52.026567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioToSpec:\n    def __init__(self, \n                 melSpecParams,\n                 top_db,\n                 image_size,\n                 ):\n        self.image_size = image_size\n        self.melSpec_transform = torchaudio.transforms.MelSpectrogram(**melSpecParams)\n        self.db_transform = torchaudio.transforms.AmplitudeToDB(stype='power', top_db=top_db)\n\n        if image_size is not None:\n            self.val_transforms = v2.Compose([\n                v2.Resize(size=self.image_size),\n            ])\n        else:\n            self.val_transforms = None\n\n    def __call__(self, audio): #Take Audio Tensor as input (1 x Audio)\n        # Generate spectrogram\n        spec = self.db_transform(self.melSpec_transform(audio)) #1 x Mel_H x Mel_W\n        spec = self.normalize_melspec(spec) #1 x Mel_H x Mel_W\n        spec = spec.expand(3, -1, -1)\n        if self.val_transforms is not None:\n            spec = self.val_transforms(spec)\n        return spec\n    \n    def normalize_melspec(self, X, eps=1e-6):\n        return (X + 80) / 80\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:08:55.277819Z","iopub.execute_input":"2025-04-30T17:08:55.27819Z","iopub.status.idle":"2025-04-30T17:08:55.285868Z","shell.execute_reply.started":"2025-04-30T17:08:55.278163Z","shell.execute_reply":"2025-04-30T17:08:55.284495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport timm\n\ndef init_layer(layer):\n    \"\"\"Initialize a Linear or Convolutional layer.\"\"\"\n    nn.init.xavier_uniform_(layer.weight)\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\ndef init_bn(bn):\n    \"\"\"Initialize a Batch Normalization layer.\"\"\"\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.)\n\n\nclass AttBlockV2(nn.Module):\n    \"\"\"Attention block for SED tasks.\"\"\"\n    def __init__(self, in_features, out_features, activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True\n        )\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True\n        )\n        self.activation = activation\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        # x: (batch_size, channels, time)\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n\n        if self.activation == \"sigmoid\":\n            eps = 1e-6\n            x = torch.clamp(x, eps, 1 - eps)\n        return x\n\n    def nonlinear_transform(self, x):\n        if self.activation == \"linear\":\n            return x\n        elif self.activation == \"sigmoid\":\n            return torch.sigmoid(x)\n\ndef interpolate(x, ratio):\n    \"\"\"Interpolate data in time domain.\"\"\"\n    (batch_size, time_steps, classes_num) = x.shape\n    upsampled = x[:, :, None, :].repeat(1, 1, ratio, 1)\n    upsampled = upsampled.reshape(batch_size, time_steps * ratio, classes_num)\n    return upsampled\n\ndef pad_framewise_output(framewise_output, frames_num):\n    \"\"\"Pad framewise_output to the same length as input frames.\"\"\"\n    output = torch.nn.functional.interpolate(\n        framewise_output.unsqueeze(1),\n        size=(frames_num, framewise_output.size(2)),\n        align_corners=True,\n        mode=\"bilinear\",\n    ).squeeze(1)\n    return output\n\nclass SEDModel(nn.Module):\n    def __init__(self, \n                 model_name, \n                 num_classes, \n                 n_mels,\n                 in_chans=1, \n                 pretrained=True, \n                 drop_path_rate=0.2, \n                 drop_rate=0.5,\n                 freq_wise = False\n                 ):\n        super().__init__()\n        \n        self.freq_wise = freq_wise\n        self.num_classes = num_classes\n        self.in_chans = in_chans\n        \n        # Batch normalization layer for mel spectrogram input\n        self.bn0 = nn.BatchNorm2d(n_mels)\n        \n        # Create backbone model using timm\n        base_model = timm.create_model(\n            model_name,\n            pretrained=pretrained,\n            in_chans=in_chans,\n            drop_path_rate=drop_path_rate,\n            drop_rate=drop_rate,\n        )\n        \n        # Extract all layers except the classification head\n        layers = list(base_model.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n        \n        # Determine the feature dimension based on model architecture\n        if \"efficientnet\" in model_name:\n            in_features = base_model.classifier.in_features\n        elif \"eca\" in model_name:\n            in_features = base_model.head.fc.in_features\n        elif \"res\" in model_name:\n            in_features = base_model.fc.in_features\n        else:\n            # Default fallback for other architectures\n            in_features = base_model.num_features\n            \n        # Add fully connected layer\n        self.fc1 = nn.Linear(in_features, in_features, bias=True)\n        \n        # Add attention block for SED\n        self.att_block = AttBlockV2(in_features, num_classes, activation=\"sigmoid\")\n\n        \n        \n        # Initialize weights\n        self.init_weight()\n    \n    def init_weight(self):\n        init_layer(self.fc1)\n        init_bn(self.bn0)\n    \n    def forward(self, x):\n        \"\"\"\n        Args:\n            x: Input tensor of shape (batch_size, channels, time, freq)\n                or (batch_size, channels, freq, time) depending on your data\n        \n        Returns:\n            clipwise_output: Predicted probabilities for each class (batch_size, num_classes)\n        \"\"\"\n        # Expected input shape: (batch_size, channels, time, freq)\n        # Transpose to (batch_size, channels, freq, time) if needed\n\n\n        if not self.freq_wise:\n            x = x.permute(0, 1, 3, 2)\n                \n        frames_num = x.shape[2]\n        \n        # Apply batch normalization\n        x = x.transpose(1, 3).contiguous()\n        x = self.bn0(x)\n        x = x.transpose(1, 3)\n        \n        # Re-transpose for the CNN\n        x = x.transpose(2, 3).contiguous()\n        # Now shape: (batch_size, channels, freq, time)\n        \n        # Pass through encoder\n        x = self.encoder(x)\n        \n        # Average pooling over frequency dimension\n        x = torch.mean(x, dim=2)\n        \n        # Apply channel smoothing\n        x1 = torch.nn.functional.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = torch.nn.functional.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n        \n        x = torch.nn.functional.dropout(x, p=0.5, training=self.training)\n\n        # Apply FC layer\n        x = x.transpose(1, 2).contiguous()\n        x = torch.nn.functional.relu_(self.fc1(x))\n        x = x.transpose(1, 2).contiguous()\n        \n        x = torch.nn.functional.dropout(x, p=0.5, training=self.training)\n\n        # # Get clipwise output through attention mechanism\n        clipwise_output = self.att_block(x)\n        \n        return clipwise_output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:08:56.670091Z","iopub.execute_input":"2025-04-30T17:08:56.670463Z","iopub.status.idle":"2025-04-30T17:08:56.690838Z","shell.execute_reply.started":"2025-04-30T17:08:56.670435Z","shell.execute_reply":"2025-04-30T17:08:56.689746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TimmCNN(torch.nn.Module):\n    def __init__(self, backbone, pretrained, num_classes):\n        super().__init__()\n\n        self.backbone = timm.create_model(\n            backbone,\n            pretrained=pretrained,\n            in_chans=3,\n            num_classes=num_classes,\n        )\n\n    def forward(self, x):\n        x = self.backbone(x).sigmoid()\n        return x\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:00.15899Z","iopub.execute_input":"2025-04-30T17:09:00.15938Z","iopub.status.idle":"2025-04-30T17:09:00.164973Z","shell.execute_reply.started":"2025-04-30T17:09:00.159349Z","shell.execute_reply":"2025-04-30T17:09:00.163817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio2spec_params = {\n    \"384\" : {'sample_rate': 32000, 'n_mels': 384, 'f_min': 0, 'f_max': 16000, 'n_fft': 3072, 'normalized': True, 'hop_length': 420},\n    \"448\" : {'sample_rate': 32000, 'n_mels': 448, 'f_min': 50, 'f_max': 16000, 'n_fft': 4096, 'normalized': True, 'hop_length': 334},\n}\n\nimage_size_params = {\n    \"384\" : (384, 384),\n    \"448\" : (448, 448),\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:02.519753Z","iopub.execute_input":"2025-04-30T17:09:02.520095Z","iopub.status.idle":"2025-04-30T17:09:02.525481Z","shell.execute_reply.started":"2025-04-30T17:09:02.520068Z","shell.execute_reply":"2025-04-30T17:09:02.524352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio2spec = {\n    \"384\" : AudioToSpec(melSpecParams=audio2spec_params[\"384\"], top_db=80, image_size=image_size_params[\"384\"]),\n    \"448\" : AudioToSpec(melSpecParams=audio2spec_params[\"448\"], top_db=80, image_size=image_size_params[\"448\"]),\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:03.059703Z","iopub.execute_input":"2025-04-30T17:09:03.060059Z","iopub.status.idle":"2025-04-30T17:09:03.178396Z","shell.execute_reply.started":"2025-04-30T17:09:03.060033Z","shell.execute_reply":"2025-04-30T17:09:03.177136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class VINOEngine:\n    def __init__(self, modelname, model_path, audio2spec, is_pytorch=True, is_sed = False):\n        core = ov.Core()\n        \n        if is_pytorch:\n            # Load PyTorch model\n            if is_sed:\n                pt_model = SEDModel(modelname, \n                            206, \n                            384,\n                            in_chans=3, \n                            pretrained=False, \n                            drop_rate=0.5,\n                            drop_path_rate=0.2,\n                            freq_wise=False\n                            )\n            else:\n                pt_model = TimmCNN(backbone=modelname, \n                            pretrained=False, \n                            num_classes=206)\n            \n            state_dict = torch.load(model_path, map_location='cpu')\n\n            # Remove 'module.' prefix if it exists\n            new_state_dict = OrderedDict()\n            for k, v in state_dict.items():\n                new_key = k.replace('module.', '') if k.startswith('module.') else k\n                new_state_dict[new_key] = v\n            \n            # Load into your model\n            pt_model.load_state_dict(new_state_dict)\n            pt_model.eval()\n            data = torch.randn(1, 32000*5)\n            data = audio2spec(data).unsqueeze(0)\n            print (data.shape)\n            model = ov.convert_model(pt_model, example_input=data)\n        else:\n            # Use existing ONNX path\n            model = core.read_model(model_path)\n        \n        self.compiled_model = core.compile_model(model=model, device_name='AUTO')\n        self.output_layer = self.compiled_model.output(0)\n        \n    def __call__(self, data):\n        result_infer = self.compiled_model(data)[self.output_layer]\n        return result_infer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:04.522336Z","iopub.execute_input":"2025-04-30T17:09:04.52271Z","iopub.status.idle":"2025-04-30T17:09:04.531242Z","shell.execute_reply.started":"2025-04-30T17:09:04.52268Z","shell.execute_reply":"2025-04-30T17:09:04.530083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"STAGE_1_BASE_PATH = \"/kaggle/input/seg-stage1-birdclef25\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:04.838745Z","iopub.execute_input":"2025-04-30T17:09:04.839185Z","iopub.status.idle":"2025-04-30T17:09:04.84494Z","shell.execute_reply.started":"2025-04-30T17:09:04.839153Z","shell.execute_reply":"2025-04-30T17:09:04.843398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vino_models = {\n    'v2_b0_cnn_384_381_3072_0': [VINOEngine(\"tf_efficientnetv2_b0.in1k\", f'{STAGE_1_BASE_PATH}/v2_b0_cnn_384_381_3072_0_conf_final_model.pt', \n                           audio2spec[\"384\"], is_pytorch=True)],\n    'b0_cnn_384_381_3072_0': [VINOEngine(\"tf_efficientnet_b0_ns\", f'{STAGE_1_BASE_PATH}/b0_cnn_384_381_3072_0_conf_final_model.pt', \n                           audio2spec[\"384\"], is_pytorch=True)],\n    'v2_b1_cnn_384_381_3072_0': [VINOEngine(\"tf_efficientnetv2_b1.in1k\", f'{STAGE_1_BASE_PATH}/v2_b1_cnn_384_381_3072_0_conf_final_model.pt', \n                           audio2spec[\"384\"], is_pytorch=True)],\n    'l0_cnn_384_381_3072_0': [VINOEngine(\"eca_nfnet_l0\", f'{STAGE_1_BASE_PATH}/l0_cnn_384_381_3072_0_conf_final_model.pt', \n                           audio2spec[\"384\"], is_pytorch=True)],\n    # 'b0_cnn_448_3': [VINOEngine(f'{STAGE_0_BASE_PATH}/b0_cnn_448_3_model_soup_final.pt', \n    #                        audio2spec[\"448\"], is_pytorch=True)],\n    # 'b0_cnn_448_4': [VINOEngine(f'{STAGE_0_BASE_PATH}/b0_cnn_448_4_model_soup_final.pt', \n    #                        audio2spec[\"448\"], is_pytorch=True)],\n    # 'b0_sed_384_3': [VINOEngine(f'{STAGE_0_BASE_PATH}/b0_sed_384_3_final.pt', \n    #                        audio2spec[\"384\"], is_pytorch=True, is_sed=True)],\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:09:26.403648Z","iopub.execute_input":"2025-04-30T17:09:26.404388Z","iopub.status.idle":"2025-04-30T17:09:37.715298Z","shell.execute_reply.started":"2025-04-30T17:09:26.404339Z","shell.execute_reply":"2025-04-30T17:09:37.712855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_single_model(models, data):\n    spec_384 = audio2spec[\"384\"](data).unsqueeze(0)\n    # spec_448 = audio2spec[\"448\"](data).unsqueeze(0)\n\n    b0_cnn_384_381_3072_0 = models['b0_cnn_384_381_3072_0'][0](spec_384)[0]\n    v2_b0_cnn_384_381_3072_0 = models['v2_b0_cnn_384_381_3072_0'][0](spec_384)[0]\n    v2_b1_cnn_384_381_3072_0 = models['v2_b1_cnn_384_381_3072_0'][0](spec_384)[0]\n    l0_cnn_384_381_3072_0 = models['l0_cnn_384_381_3072_0'][0](spec_384)[0]\n\n\n    \n    # out_1 = models['b0_cnn_384_4'][0](spec_384)[0]\n    \n    # out_2 = models['b0_cnn_448_3'][0](spec_448)[0]\n    # out_3 = models['b0_cnn_448_4'][0](spec_448)[0]\n\n    \n    # out_4 = models['b0_sed_384_3'][0](spec_384)[0]\n\n    \n    return {\n        \"b0_cnn_384_381_3072_0\" : b0_cnn_384_381_3072_0,\n        \"reg008_cnn_384_381_3072_0\" : v2_b0_cnn_384_381_3072_0,\n        \"v2_b1_cnn_384_381_3072_0\" : v2_b1_cnn_384_381_3072_0,\n        \"l0_cnn_384_381_3072_0\" : l0_cnn_384_381_3072_0\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:10:22.006145Z","iopub.execute_input":"2025-04-30T17:10:22.006909Z","iopub.status.idle":"2025-04-30T17:10:22.014665Z","shell.execute_reply.started":"2025-04-30T17:10:22.006859Z","shell.execute_reply":"2025-04-30T17:10:22.013535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import soundfile as sf\nimport os\nimport librosa\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:10:23.745669Z","iopub.execute_input":"2025-04-30T17:10:23.746035Z","iopub.status.idle":"2025-04-30T17:10:24.390784Z","shell.execute_reply.started":"2025-04-30T17:10:23.746007Z","shell.execute_reply":"2025-04-30T17:10:24.389456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs(\"dummy\", exist_ok=True)\n# Parameters\nduration = 60  # 1 minute\nsampling_rate = 32000  # 32kHz\nfrequency = 440.0  # Frequency of the tone (A4 note)\namplitude = 0.5  # Amplitude of the signal\n\n# Time array\nt = np.linspace(0, duration, int(sampling_rate * duration), endpoint=False)\n\n# Generate a sine wave\naudio_data = amplitude * np.sin(2 * np.pi * frequency * t)\n\n# Save as OGG file\noutput_file = \"dummy/soundscape_8358733.ogg\"\nsf.write(output_file, audio_data, sampling_rate, format='OGG', subtype='VORBIS')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:10:28.749043Z","iopub.execute_input":"2025-04-30T17:10:28.749453Z","iopub.status.idle":"2025-04-30T17:10:29.405751Z","shell.execute_reply.started":"2025-04-30T17:10:28.749423Z","shell.execute_reply":"2025-04-30T17:10:29.404504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def smooth_array_general(predictions):\n    \n    max_map = np.max(predictions, axis=0)\n    less_mask = max_map <= 0.1\n    predictions[:, less_mask] = (predictions[:, less_mask]) ** 2\n\n    less_mask = max_map >= 0.8\n    predictions[:, less_mask] = (predictions[:, less_mask]) ** 2\n    \n    new_predictions = predictions.copy()\n    for i in range(1, predictions.shape[0]-1):\n        new_predictions[i] = (predictions[i-1] * 0.15) + (predictions[i] * 0.7) + (predictions[i+1] * 0.15)\n    new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n    new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n\n    return new_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:10:29.407876Z","iopub.execute_input":"2025-04-30T17:10:29.408356Z","iopub.status.idle":"2025-04-30T17:10:29.417244Z","shell.execute_reply.started":"2025-04-30T17:10:29.408311Z","shell.execute_reply":"2025-04-30T17:10:29.415557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(42)\n\nclass_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\nif len(test_soundscapes) == 0:\n    test_soundscape_path = 'dummy/'\n    test_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n\nsub = pd.read_csv(\"/kaggle/input/birdclef-2025/sample_submission.csv\")\ntarget_columns = sub.columns.tolist()[1:]\nnum_classes = len(target_columns)\nbird2id = {b: i for i, b in enumerate(target_columns)}\n\npredictions = pd.DataFrame(columns=['row_id'] + class_labels)\nfor soundscape in test_soundscapes:\n\n    # Load audio\n    sig, rate = sf.read(soundscape)\n\n    # Split into 5-second chunks\n    chunks = []\n    for i in range(0, len(sig), rate*5):\n        chunk = sig[i:i+rate*5]\n        chunks.append(chunk)\n\n    dict_predictions = {\n        \"b0_cnn_384_381_3072_0\" : [],\n        \"reg008_cnn_384_381_3072_0\" : [],\n        \"v2_b1_cnn_384_381_3072_0\" : [],\n        \"l0_cnn_384_381_3072_0\" : []\n        \n    }\n    for i, chunk in enumerate(chunks):\n        \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n\n        chunk = chunk.astype(np.float32)\n        max_v = np.max(np.abs(chunk))\n        if max_v > 1:\n            chunk = chunk / max_v\n\n        wave = torch.from_numpy(chunk).float()\n        scores = run_single_model(vino_models, wave.unsqueeze(0))\n        dict_predictions[\"b0_cnn_384_381_3072_0\"].append(scores[\"b0_cnn_384_381_3072_0\"])\n        dict_predictions[\"reg008_cnn_384_381_3072_0\"].append(scores[\"reg008_cnn_384_381_3072_0\"])\n        dict_predictions[\"v2_b1_cnn_384_381_3072_0\"].append(scores[\"v2_b1_cnn_384_381_3072_0\"])\n        dict_predictions[\"l0_cnn_384_381_3072_0\"].append(scores[\"l0_cnn_384_381_3072_0\"])\n\n    dict_predictions[\"b0_cnn_384_381_3072_0\"] = np.stack(dict_predictions[\"b0_cnn_384_381_3072_0\"])\n    dict_predictions[\"reg008_cnn_384_381_3072_0\"] = np.stack(dict_predictions[\"reg008_cnn_384_381_3072_0\"])\n    dict_predictions[\"v2_b1_cnn_384_381_3072_0\"] = np.stack(dict_predictions[\"v2_b1_cnn_384_381_3072_0\"])\n    dict_predictions[\"l0_cnn_384_381_3072_0\"] = np.stack(dict_predictions[\"l0_cnn_384_381_3072_0\"])\n\n    \n    dict_predictions[\"b0_cnn_384_381_3072_0\"] = smooth_array_general(dict_predictions[\"b0_cnn_384_381_3072_0\"])\n    dict_predictions[\"reg008_cnn_384_381_3072_0\"] = smooth_array_general(dict_predictions[\"reg008_cnn_384_381_3072_0\"])\n    dict_predictions[\"v2_b1_cnn_384_381_3072_0\"] = smooth_array_general(dict_predictions[\"v2_b1_cnn_384_381_3072_0\"])\n    dict_predictions[\"l0_cnn_384_381_3072_0\"] = smooth_array_general(dict_predictions[\"l0_cnn_384_381_3072_0\"])\n\n    dict_predictions = (dict_predictions[\"b0_cnn_384_381_3072_0\"] + dict_predictions[\"reg008_cnn_384_381_3072_0\"]  + \n                        dict_predictions[\"v2_b1_cnn_384_381_3072_0\"]   + dict_predictions[\"l0_cnn_384_381_3072_0\"]) / 4.0\n    \n    \n    for i, chunk in enumerate(chunks):\n        \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n        scores = dict_predictions[i]\n        \n        scores = [scores[bird2id[each]] for each in class_labels]\n        new_row = pd.DataFrame([[row_id] + list(scores)], columns=['row_id'] + class_labels)\n        predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n        \n# Save prediction as csv\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T17:11:03.222994Z","iopub.execute_input":"2025-04-30T17:11:03.223402Z","iopub.status.idle":"2025-04-30T17:11:05.399348Z","shell.execute_reply.started":"2025-04-30T17:11:03.223369Z","shell.execute_reply":"2025-04-30T17:11:05.39773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}