{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":10607811,"sourceType":"datasetVersion","datasetId":6566575},{"sourceId":11995011,"sourceType":"datasetVersion","datasetId":7544996},{"sourceId":12011642,"sourceType":"datasetVersion","datasetId":7556732},{"sourceId":12012809,"sourceType":"datasetVersion","datasetId":7557406},{"sourceId":12017197,"sourceType":"datasetVersion","datasetId":7560506},{"sourceId":12021546,"sourceType":"datasetVersion","datasetId":7563295},{"sourceId":12025683,"sourceType":"datasetVersion","datasetId":7566098},{"sourceId":12029772,"sourceType":"datasetVersion","datasetId":7568943},{"sourceId":12082942,"sourceType":"datasetVersion","datasetId":7576844}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import glob\nimport os\nimport random\nimport sys\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport timm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.utils.data as torchdata\nfrom torchaudio.transforms import AmplitudeToDB, MelSpectrogram\nfrom tqdm.auto import tqdm\nimport glob\nimport concurrent.futures\nimport shutil\nimport albumentations as A\nimport torchaudio\n\nfrom collections import defaultdict\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-05T14:59:05.148377Z","iopub.execute_input":"2025-06-05T14:59:05.148751Z","iopub.status.idle":"2025-06-05T14:59:58.035978Z","shell.execute_reply.started":"2025-06-05T14:59:05.148704Z","shell.execute_reply":"2025-06-05T14:59:58.033809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install /kaggle/input/onnxruntime/humanfriendly-10.0-py2.py3-none-any.whl\n!pip install /kaggle/input/onnxruntime/coloredlogs-15.0.1-py2.py3-none-any.whl\n!pip install /kaggle/input/onnxruntime/onnxruntime-1.20.1-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T14:59:58.037227Z","iopub.execute_input":"2025-06-05T14:59:58.038543Z","iopub.status.idle":"2025-06-05T15:00:17.340266Z","shell.execute_reply.started":"2025-06-05T14:59:58.038468Z","shell.execute_reply":"2025-06-05T15:00:17.338379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/birdclef-2025/sample_submission.csv\")\ntarget_columns_ = sub.columns.tolist()\ntarget_columns = sub.columns.tolist()[1:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:17.341603Z","iopub.execute_input":"2025-06-05T15:00:17.342027Z","iopub.status.idle":"2025-06-05T15:00:17.378243Z","shell.execute_reply.started":"2025-06-05T15:00:17.341968Z","shell.execute_reply":"2025-06-05T15:00:17.376163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TOTAL_SECONDS_CHUNKS = 12\ntest_path = \"/kaggle/input/birdclef-2025/test_soundscapes/\"\nfiles = glob.glob(f'{test_path}*')\n\nseconds = [i for i in range(5, (TOTAL_SECONDS_CHUNKS*5) + 5, 5)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:27.035562Z","iopub.execute_input":"2025-06-05T15:06:27.036464Z","iopub.status.idle":"2025-06-05T15:06:27.044484Z","shell.execute_reply.started":"2025-06-05T15:06:27.036406Z","shell.execute_reply":"2025-06-05T15:06:27.042536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"files = glob.glob(f'{test_path}*')\nif len(files) == 1:\n    shutil.copy('/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_074000.ogg', '/kaggle/working/H02_20230420_074000.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_112000.ogg', '/kaggle/working/H02_20230420_112000.ogg'),\n    shutil.copy('/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_074000.ogg', '/kaggle/working/H02_20230420_074000t.ogg')\n    shutil.copy('/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_112000.ogg', '/kaggle/working/H02_20230420_112000t.ogg')\n    test_path = \"/kaggle/working/\"\n    \nprint(test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:27.30283Z","iopub.execute_input":"2025-06-05T15:06:27.303321Z","iopub.status.idle":"2025-06-05T15:06:27.31795Z","shell.execute_reply.started":"2025-06-05T15:06:27.303285Z","shell.execute_reply":"2025-06-05T15:06:27.316168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:27.692555Z","iopub.execute_input":"2025-06-05T15:06:27.692962Z","iopub.status.idle":"2025-06-05T15:06:27.700835Z","shell.execute_reply.started":"2025-06-05T15:06:27.692903Z","shell.execute_reply":"2025-06-05T15:06:27.699263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mel_spec_params_128 = {\n        \"sample_rate\": 32000,\n        \"n_mels\": 128,\n        \"f_min\": 0,\n        \"f_max\": 16000,\n        \"n_fft\": 2048,\n        \"hop_length\": 512,\n        \"normalized\": True,\n        \"center\": True,\n        \"pad_mode\" : \"constant\",\n        \"norm\" : \"slaney\",\n        \"mel_scale\" : \"htk\",\n    }\n\nmel_spec_params_96 = {\n        \"sample_rate\": 32000,\n        \"n_mels\": 96,\n        \"f_min\": 0,\n        \"f_max\": 16000,\n        \"n_fft\": 2048,\n        \"hop_length\": 512,\n        \"normalized\": True,\n        \"center\": True,\n        \"pad_mode\" : \"constant\",\n        \"norm\" : \"slaney\",\n        \"mel_scale\" : \"htk\",\n    }\ntop_db = 80","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:28.427647Z","iopub.execute_input":"2025-06-05T15:06:28.428057Z","iopub.status.idle":"2025-06-05T15:06:28.434403Z","shell.execute_reply.started":"2025-06-05T15:06:28.428024Z","shell.execute_reply":"2025-06-05T15:06:28.433074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize_melspec(X, eps=1e-6):\n    mean = X.mean((1, 2), keepdim=True)\n    std = X.std((1, 2), keepdim=True)\n    Xstd = (X - mean) / (std + eps)\n\n    norm_min, norm_max = (\n        Xstd.min(-1)[0].min(-1)[0],\n        Xstd.max(-1)[0].max(-1)[0],\n    )\n    fix_ind = (norm_max - norm_min) > eps * torch.ones_like(\n        (norm_max - norm_min)\n    )\n    V = torch.zeros_like(Xstd)\n    if fix_ind.sum():\n        V_fix = Xstd[fix_ind]\n        norm_max_fix = norm_max[fix_ind, None, None]\n        norm_min_fix = norm_min[fix_ind, None, None]\n        V_fix = torch.max(\n            torch.min(V_fix, norm_max_fix),\n            norm_min_fix,\n        )\n        V_fix = (V_fix - norm_min_fix) / (norm_max_fix - norm_min_fix)\n        V[fix_ind] = V_fix\n    return V","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:30.244741Z","iopub.execute_input":"2025-06-05T15:06:30.245231Z","iopub.status.idle":"2025-06-05T15:06:30.252961Z","shell.execute_reply.started":"2025-06-05T15:06:30.245196Z","shell.execute_reply":"2025-06-05T15:06:30.251739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_val = A.Compose([\n    A.Resize(224, 224),\n    A.Normalize()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:32.004855Z","iopub.execute_input":"2025-06-05T15:06:32.005282Z","iopub.status.idle":"2025-06-05T15:06:32.011278Z","shell.execute_reply.started":"2025-06-05T15:06:32.005248Z","shell.execute_reply":"2025-06-05T15:06:32.009966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TestDataset(torchdata.Dataset):\n    def __init__(self, \n                 df: pd.DataFrame, \n                 clip: np.ndarray,\n                ):\n        \n        self.df = df\n        self.clip = clip\n        self.mel_transform_96 = torchaudio.transforms.MelSpectrogram(**mel_spec_params_96)\n        self.mel_transform_128 = torchaudio.transforms.MelSpectrogram(**mel_spec_params_128)\n        self.db_transform = torchaudio.transforms.AmplitudeToDB(stype='power', top_db=top_db)\n        self.transform = transforms_val\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx: int):\n\n        sample = self.df.loc[idx, :]\n        row_id = sample.row_id\n\n        end_seconds = int(sample.seconds)\n        start_seconds = int(end_seconds - 5)\n        \n        wave = self.clip[:, 32000 * start_seconds : 32000 * end_seconds]\n        \n        mel_spectrogram = normalize_melspec(self.db_transform(self.mel_transform_96(wave)))\n        mel_spectrogram = mel_spectrogram * 255\n        mel_spectrogram = mel_spectrogram.expand(3, -1, -1).permute(1, 2, 0).numpy()\n        \n        res = self.transform(image=mel_spectrogram)\n        spec = res['image'].astype(np.float32)\n        spec = spec.transpose(2, 0, 1)\n\n        spec_96 = spec.copy()\n\n        mel_spectrogram = normalize_melspec(self.db_transform(self.mel_transform_128(wave)))\n        mel_spectrogram = mel_spectrogram * 255\n        mel_spectrogram = mel_spectrogram.expand(3, -1, -1).permute(1, 2, 0).numpy()\n        \n        res = self.transform(image=mel_spectrogram)\n        spec = res['image'].astype(np.float32)\n        spec = spec.transpose(2, 0, 1)\n        \n        return {\n            \"row_id\": row_id,\n            \"wave_96\": spec_96,\n            \"wave_128\": spec,\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:33.690243Z","iopub.execute_input":"2025-06-05T15:06:33.690575Z","iopub.status.idle":"2025-06-05T15:06:33.700793Z","shell.execute_reply.started":"2025-06-05T15:06:33.690549Z","shell.execute_reply":"2025-06-05T15:06:33.699429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom typing import Union\n\ndef apply_power_to_low_ranked_cols(\n    p: np.ndarray,\n    top_k: int = 30,\n    exponent: Union[int, float] = 2,\n    inplace: bool = True\n) -> np.ndarray:\n    \"\"\"\n    Rank columns by their column‑wise maximum and raise every column whose\n    rank falls below `top_k` to a given power.\n\n    Parameters\n    ----------\n    p : np.ndarray\n        A 2‑D array of shape **(n_chunks, n_classes)**.\n\n        - **n_chunks** is the number of fixed‑length time chunks obtained\n          after slicing the input audio (or other sequential data).  \n          *Example:* In the BirdCLEF `test_soundscapes` set, each file is\n          60 s long. If you extract non‑overlapping 5 s windows,  \n          `n_chunks = 60 s / 5 s = 12`.\n        - **n_classes** is the number of classes being predicted.\n        - Each element `p[i, j]` is the score or probability of class *j*\n          in chunk *i*.\n\n    top_k : int, default=30\n        The highest‑ranked columns (by their maximum value) that remain\n        unchanged.\n\n    exponent : int or float, default=2\n        The power applied to the selected low‑ranked columns  \n        (e.g. `2` squares, `0.5` takes the square root, `3` cubes).\n\n    inplace : bool, default=True\n        If `True`, modify `p` in place.  \n        If `False`, operate on a copy and leave the original array intact.\n\n    Returns\n    -------\n    np.ndarray\n        The transformed array. It is the same object as `p` when\n        `inplace=True`; otherwise, it is a new array.\n\n    \"\"\"\n    if not inplace:\n        p = p.copy()\n\n    # Identify columns whose max value ranks below `top_k`\n    tail_cols = np.argsort(-p.max(axis=0))[top_k:]\n\n    # Apply the power transformation to those columns\n    p[:, tail_cols] = p[:, tail_cols] ** exponent\n    return p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:35.989019Z","iopub.execute_input":"2025-06-05T15:06:35.989384Z","iopub.status.idle":"2025-06-05T15:06:35.996144Z","shell.execute_reply.started":"2025-06-05T15:06:35.989355Z","shell.execute_reply":"2025-06-05T15:06:35.995031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sigmoid(x):\n    \"\"\"Calculate sigmoid function.\"\"\"\n    return 1 / (1 + np.exp(-x))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:06:41.939435Z","iopub.execute_input":"2025-06-05T15:06:41.939798Z","iopub.status.idle":"2025-06-05T15:06:41.947332Z","shell.execute_reply.started":"2025-06-05T15:06:41.939761Z","shell.execute_reply":"2025-06-05T15:06:41.945959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prediction_for_clip(audio_path):\n    prediction_dict = {}\n\n    wav, org_sr = torchaudio.load(audio_path, normalize=True)\n    clip = torchaudio.functional.resample(wav, orig_freq=org_sr, new_freq=32000)\n\n    name_ = audio_path.split(\".ogg\")[0].split(\"/\")[-1]\n    row_ids = [name_+f\"_{second}\" for second in seconds]\n\n    test_df = pd.DataFrame({\n        \"row_id\": row_ids,\n        \"seconds\": seconds,\n    })\n\n    dataset = TestDataset(\n        df=test_df, \n        clip=clip,\n    )\n\n    loader = torchdata.DataLoader(\n        dataset,\n        batch_size=4, \n        num_workers=2,\n        drop_last=False,\n        shuffle=False,\n        pin_memory=True\n    )\n\n    global_outputs = defaultdict(list)  # 用于全局收集每个模型的输出\n    all_row_ids = []  # 保留所有 row_id 的顺序\n\n    for inputs in loader:\n        row_ids = inputs['row_id']\n        inputs.pop('row_id')\n        all_row_ids.extend(row_ids)\n\n        outputs = defaultdict(list)\n\n        for k in models.keys():\n            for model in models[k]:\n                if \"SED_96\" in k:\n                    _, segmentwise_output, clipwise_output, _, _, _ = model.run(\n                        None, {\"input\": inputs[\"wave_96\"].cpu().numpy()})\n                    logit_temp = (clipwise_output[:, -206:] + np.max(segmentwise_output, 1)[:, -206:]) / 2\n                elif \"SED_128\" in k:\n                    _, segmentwise_output, clipwise_output, _, _, _ = model.run(\n                        None, {\"input\": inputs[\"wave_128\"].cpu().numpy()})\n                    logit_temp = (clipwise_output[:, -206:] + np.max(segmentwise_output, 1)[:, -206:]) / 2\n                elif \"CNN_96\" in k:\n                    logit = model.run(None, {\"input\": inputs[\"wave_96\"].cpu().numpy()})[0]\n                    logit_temp = sigmoid(logit)[:, -206:]\n                elif \"CNN_128\" in k:\n                    logit = model.run(None, {\"input\": inputs[\"wave_128\"].cpu().numpy()})[0]\n                    logit_temp = sigmoid(logit)[:, -206:]\n                else:\n                    raise NotImplementError\n\n                outputs[k].append(logit_temp)\n\n        # 按 model group 聚合并添加到 global_outputs\n        for k, array_list in outputs.items():\n            stacked = np.stack(array_list)  # (n_models_in_group, batch_size, n_classes)\n            global_outputs[k].append(stacked)\n\n    # 所有 batch 处理完后进行 Geo Mean\n    for k in global_outputs:\n        global_outputs[k] = np.concatenate(global_outputs[k], axis=1)  # (n_models_in_group, total_chunks, n_classes)\n        # print(global_outputs[k].shape)\n    # print(stacked.shape)\n\n    per_key_gmeans = {}\n    for k, stacked in global_outputs.items():\n        # print(k, stacked.shape, stacked[0].shape, stacked[0, :, :].shape)\n        if \"SED\" in k:\n            for i in range(stacked.shape[0]):\n                stacked[i, :, :] = apply_power_to_low_ranked_cols(stacked[i, :, :], top_k=30, exponent=2)\n\n        # weights = np.array([0.25, 0.3, 0.25, 0.1, 0.1], dtype=np.float32)\n        weights = np.array([1,1,1], dtype=np.float32)\n\n        weights = weights / weights.sum()  # 归一化以确保总和为 1\n\n        log_stack = np.log(stacked)  # shape: (4, 12, 206)\n        weighted_log_mean = np.tensordot(weights, log_stack, axes=(0, 0))  # shape: (12, 206)\n        per_key_gmeans[k] = np.exp(weighted_log_mean)\n\n\n        # log_mean = np.mean(np.log(stacked), axis=0)  # (total_chunks, n_classes)\n        # per_key_gmeans[k] = np.exp(log_mean)\n        # if \"SED\" in k:\n        #     per_key_gmeans[k] = apply_power_to_low_ranked_cols(per_key_gmeans[k], top_k=30,exponent=2)\n        # print(k, per_key_gmeans[k].shape)\n\n    stacked_gmeans = np.stack(list(per_key_gmeans.values()))  # (n_model_groups, total_chunks, n_classes)\n    overall_log_mean = np.mean(np.log(stacked_gmeans + 1e-9), axis=0)\n    overall_gmean = np.exp(overall_log_mean)  # (total_chunks, n_classes)\n    # print(overall_gmean.shape)\n\n    for row_id_idx, row_id in enumerate(all_row_ids):\n        prediction_dict[str(row_id)] = {}\n        for label in range(len(target_columns)):\n            prediction_dict[row_id][target_columns[label]] = overall_gmean[row_id_idx, label]\n\n    return prediction_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:21:03.818449Z","iopub.execute_input":"2025-06-05T15:21:03.818975Z","iopub.status.idle":"2025-06-05T15:21:03.839464Z","shell.execute_reply.started":"2025-06-05T15:21:03.818904Z","shell.execute_reply":"2025-06-05T15:21:03.837219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import onnxruntime\noptions = onnxruntime.SessionOptions()\n\n# Enable graph optimization\noptions.graph_optimization_level = onnxruntime.GraphOptimizationLevel.ORT_ENABLE_ALL\noptions.intra_op_num_threads = 1  # Threads used to parallelize operations within one node\noptions.inter_op_num_threads = 1  # Threads used to parallelize execution between nodes\n\nmodels = defaultdict(list)\nmodels[\"SED_96\"] = [\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnet_b0_ns_mel96.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnetv2_b3_mel96.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnetv2_s.in21k_ft_in1k_mel96.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_mnasnet_100_mel96.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_spnasnet_100_mel96.onnx\", options),\n]\nmodels[\"SED_128\"] = [\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnet_b0_ns_mel128.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnetv2_b3_mel128.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_tf_efficientnetv2_s.in21k_ft_in1k_mel128.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_mnasnet_100_mel128.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/SED_spnasnet_100_mel128.onnx\", options),\n]\nmodels[\"CNN_96\"] = [\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnet_b0_ns_mel96.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnetv2_b3_mel96.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnetv2_s.in21k_ft_in1k_mel96.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_mnasnet_100_mel96.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_spnasnet_100_mel96.onnx\", options),\n]\nmodels[\"CNN_128\"] = [\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnet_b0_ns_mel128.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnetv2_b3_mel128.onnx\", options),\n    onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_tf_efficientnetv2_s.in21k_ft_in1k_mel128.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_mnasnet_100_mel128.onnx\", options),\n    # onnxruntime.InferenceSession(\"/kaggle/input/birdfinal-onnx/CNN_spnasnet_100_mel128.onnx\", options),\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:21:06.502864Z","iopub.execute_input":"2025-06-05T15:21:06.50335Z","iopub.status.idle":"2025-06-05T15:21:12.278291Z","shell.execute_reply.started":"2025-06-05T15:21:06.503317Z","shell.execute_reply":"2025-06-05T15:21:12.276714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_audios = list(glob.glob(f'{test_path}*.ogg'))\n\n# if len(all_audios) < 100:\n#   all_audios = list(glob.glob('/kaggle/input/birdclef-2025/train_soundscapes/*.ogg'))[:700]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:21:12.28023Z","iopub.execute_input":"2025-06-05T15:21:12.280631Z","iopub.status.idle":"2025-06-05T15:21:12.28706Z","shell.execute_reply.started":"2025-06-05T15:21:12.280592Z","shell.execute_reply":"2025-06-05T15:21:12.285377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nwith concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor:\n    dicts = list(executor.map(prediction_for_clip, all_audios))\n\nprediction_dicts = {}\nfor d in dicts:\n    prediction_dicts.update(d)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:21:12.28903Z","iopub.execute_input":"2025-06-05T15:21:12.289384Z","iopub.status.idle":"2025-06-05T15:21:43.639696Z","shell.execute_reply.started":"2025-06-05T15:21:12.28935Z","shell.execute_reply":"2025-06-05T15:21:43.63719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame.from_dict(prediction_dicts, \"index\").rename_axis(\"row_id\").reset_index()\n\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:07:21.887365Z","iopub.execute_input":"2025-06-05T15:07:21.887883Z","iopub.status.idle":"2025-06-05T15:07:21.961785Z","shell.execute_reply.started":"2025-06-05T15:07:21.887828Z","shell.execute_reply":"2025-06-05T15:07:21.960548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.max().iloc[:30]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.199909Z","iopub.execute_input":"2025-06-05T15:00:45.200378Z","iopub.status.idle":"2025-06-05T15:00:45.214115Z","shell.execute_reply.started":"2025-06-05T15:00:45.200343Z","shell.execute_reply":"2025-06-05T15:00:45.212271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.215542Z","iopub.execute_input":"2025-06-05T15:00:45.216003Z","iopub.status.idle":"2025-06-05T15:00:45.250736Z","shell.execute_reply.started":"2025-06-05T15:00:45.215969Z","shell.execute_reply":"2025-06-05T15:00:45.249592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # https://www.kaggle.com/competitions/birdclef-2024/discussion/511527\n# def smooth_array_general(array, w=[0.1, 0.2, 0.4, 0.2, 0.1]):\n#     smoothed_array = np.zeros_like(array)\n#     timesteps = array.shape[0]\n#     radius = len(w) // 2\n\n#     for t in range(timesteps):\n#         for i, weight in enumerate(w):\n#             index = t - radius + i\n#             if index < 0: \n#                 smoothed_array[t] += array[0] * weight\n#             elif index >= timesteps: \n#                 smoothed_array[t] += array[-1] * weight\n#             else:\n#                 smoothed_array[t] += array[index] * weight\n#     for c in range(array.shape[1]):\n#         smoothed_array[:, c] = smoothed_array[:, c] * 0.8 + smoothed_array[:, c].mean(keepdims=True) * 0.2\n#     return smoothed_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.251857Z","iopub.execute_input":"2025-06-05T15:00:45.252281Z","iopub.status.idle":"2025-06-05T15:00:45.257203Z","shell.execute_reply.started":"2025-06-05T15:00:45.252247Z","shell.execute_reply":"2025-06-05T15:00:45.256073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sub_array = submission.iloc[:, 1:].values\n\n# timestep = 12\n# num_classes = 206\n# num_files = sub_array.shape[0] // timestep\n# reshaped_array = sub_array.reshape(num_files, timestep, num_classes)\n\n# smoothed_results = []\n# for i in range(num_files):\n#     smoothed = smooth_array_general(reshaped_array[i])  # shape: (12, 206)\n#     smoothed_results.append(smoothed)\n\n# final_array = np.vstack(smoothed_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.258558Z","iopub.execute_input":"2025-06-05T15:00:45.258923Z","iopub.status.idle":"2025-06-05T15:00:45.277848Z","shell.execute_reply.started":"2025-06-05T15:00:45.25889Z","shell.execute_reply":"2025-06-05T15:00:45.276159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission['48124']= 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.279409Z","iopub.execute_input":"2025-06-05T15:00:45.280149Z","iopub.status.idle":"2025-06-05T15:00:45.306313Z","shell.execute_reply.started":"2025-06-05T15:00:45.280093Z","shell.execute_reply":"2025-06-05T15:00:45.30425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission['48124'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.307769Z","iopub.execute_input":"2025-06-05T15:00:45.308155Z","iopub.status.idle":"2025-06-05T15:00:45.334187Z","shell.execute_reply.started":"2025-06-05T15:00:45.308122Z","shell.execute_reply":"2025-06-05T15:00:45.33303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.iloc[:, 1:] = final_array\n# submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.335393Z","iopub.execute_input":"2025-06-05T15:00:45.335735Z","iopub.status.idle":"2025-06-05T15:00:45.354243Z","shell.execute_reply.started":"2025-06-05T15:00:45.335706Z","shell.execute_reply":"2025-06-05T15:00:45.352633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm *.ogg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.355417Z","iopub.execute_input":"2025-06-05T15:00:45.355814Z","iopub.status.idle":"2025-06-05T15:00:45.566618Z","shell.execute_reply.started":"2025-06-05T15:00:45.355771Z","shell.execute_reply":"2025-06-05T15:00:45.565133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sub_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.568414Z","iopub.execute_input":"2025-06-05T15:00:45.568874Z","iopub.status.idle":"2025-06-05T15:00:45.573785Z","shell.execute_reply.started":"2025-06-05T15:00:45.568824Z","shell.execute_reply":"2025-06-05T15:00:45.572609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.574918Z","iopub.execute_input":"2025-06-05T15:00:45.575786Z","iopub.status.idle":"2025-06-05T15:00:45.594501Z","shell.execute_reply.started":"2025-06-05T15:00:45.575747Z","shell.execute_reply":"2025-06-05T15:00:45.592727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.iloc[:, 1:] = final_array","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.595736Z","iopub.execute_input":"2025-06-05T15:00:45.596162Z","iopub.status.idle":"2025-06-05T15:00:45.614296Z","shell.execute_reply.started":"2025-06-05T15:00:45.596116Z","shell.execute_reply":"2025-06-05T15:00:45.613055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.loc[:, notuse_sample_labels] = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.615565Z","iopub.execute_input":"2025-06-05T15:00:45.615956Z","iopub.status.idle":"2025-06-05T15:00:45.636481Z","shell.execute_reply.started":"2025-06-05T15:00:45.615892Z","shell.execute_reply":"2025-06-05T15:00:45.634969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.to_csv(\"submission.csv\", index=False)\n# print (\"Done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.637728Z","iopub.execute_input":"2025-06-05T15:00:45.638089Z","iopub.status.idle":"2025-06-05T15:00:45.658218Z","shell.execute_reply.started":"2025-06-05T15:00:45.638053Z","shell.execute_reply":"2025-06-05T15:00:45.656621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T15:00:45.659438Z","iopub.execute_input":"2025-06-05T15:00:45.659904Z","iopub.status.idle":"2025-06-05T15:00:45.686422Z","shell.execute_reply.started":"2025-06-05T15:00:45.659859Z","shell.execute_reply":"2025-06-05T15:00:45.684955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}