{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11458343,"sourceType":"datasetVersion","datasetId":6930139},{"sourceId":12046911,"sourceType":"datasetVersion","datasetId":6930285},{"sourceId":12051518,"sourceType":"datasetVersion","datasetId":7243711},{"sourceId":12091361,"sourceType":"datasetVersion","datasetId":7611682},{"sourceId":12133948,"sourceType":"datasetVersion","datasetId":7641340},{"sourceId":243313432,"sourceType":"kernelVersion"},{"sourceId":244266989,"sourceType":"kernelVersion"},{"sourceId":245028034,"sourceType":"kernelVersion"},{"sourceId":243447401,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":64.08478,"end_time":"2025-04-16T20:13:48.159484","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-04-16T20:12:44.074704","version":"2.5.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/bird-clef-2025-addones/openvino-2025.0.0-17942-cp311-cp311-manylinux2014_x86_64.whl --no-deps\n# !pip install /kaggle/input/bird-clef-2024-addones/onnxruntime-1.17.3-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl --no-deps\n!pip list | grep openvino\n!pip list | grep numpy","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:23.422865Z","iopub.execute_input":"2025-12-28T09:11:23.423168Z","iopub.status.idle":"2025-12-28T09:11:29.549769Z","shell.execute_reply.started":"2025-12-28T09:11:23.423147Z","shell.execute_reply":"2025-12-28T09:11:29.548551Z"},"papermill":{"duration":4.08287,"end_time":"2025-04-16T20:12:51.724666","exception":false,"start_time":"2025-04-16T20:12:47.641796","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/bird-clef-2025-code/main_folder/main_folder/')","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.551924Z","iopub.execute_input":"2025-12-28T09:11:29.552259Z","iopub.status.idle":"2025-12-28T09:11:29.558111Z","shell.execute_reply.started":"2025-12-28T09:11:29.552231Z","shell.execute_reply":"2025-12-28T09:11:29.557151Z"},"papermill":{"duration":0.013862,"end_time":"2025-04-16T20:12:56.409308","exception":false,"start_time":"2025-04-16T20:12:56.395446","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa\nimport seaborn as sns\nimport os\nimport json\nimport IPython.display as ipd\nimport soundfile as sf\nimport torch\nimport h5py\n# import onnxruntime as ort\nimport openvino as ov\n\nfrom glob import glob\nfrom tqdm import tqdm\nfrom matplotlib import pyplot as plt\nfrom itertools import chain\nfrom os.path import join as pjoin\nfrom copy import deepcopy\n\n\nfrom code_base.datasets import WaveDataset, WaveAllFileDataset\nfrom code_base.inefernce import BirdsInference\nfrom code_base.utils import load_json, compose_submission_dataframe\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.559284Z","iopub.execute_input":"2025-12-28T09:11:29.559722Z","iopub.status.idle":"2025-12-28T09:11:29.598697Z","shell.execute_reply.started":"2025-12-28T09:11:29.559698Z","shell.execute_reply":"2025-12-28T09:11:29.595573Z"},"papermill":{"duration":7.508944,"end_time":"2025-04-16T20:13:03.924934","exception":false,"start_time":"2025-04-16T20:12:56.41599","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.006128,"end_time":"2025-04-16T20:13:03.942799","exception":false,"start_time":"2025-04-16T20:13:03.936671","status":"completed"},"tags":[]}},{"cell_type":"code","source":"POSTFIX = \"\"\nDATASET_NAME = \"bird-clef-2025-models-v2\"\nEXP_NAME = \"tf_efficientnetv2_s_in21k_Exp_noamp_64bs_5sec_BasicAug_EqualBalancing_AdamW1e4_CosBatchLR1e6_Epoch50_FocalBCELoss_LSF1005_FromPrebs1_PseudoF2PT05MT01P04I2_AddRareBirdsNoLeak\" + POSTFIX\nTRAIN_PERIOD = 5\nprint(\"Possible checkpoints:\\n\\n{}\".format(\"\\n\".join(set([\n    os.path.basename(el) for el in glob(f\"/kaggle/input/{DATASET_NAME}/{EXP_NAME}/{EXP_NAME}/**\", recursive=True) if \"train\" not in os.path.basename(el)\n]))))","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.599335Z","iopub.status.idle":"2025-12-28T09:11:29.599658Z","shell.execute_reply.started":"2025-12-28T09:11:29.599515Z","shell.execute_reply":"2025-12-28T09:11:29.599528Z"},"papermill":{"duration":0.027164,"end_time":"2025-04-16T20:13:03.976496","exception":false,"start_time":"2025-04-16T20:13:03.949332","status":"completed"},"tags":[],"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IS_TEST = len(glob(\"/kaggle/input/birdclef-2025/test_soundscapes/*.ogg\")) > 1\nIS_TEST","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.601058Z","iopub.status.idle":"2025-12-28T09:11:29.601889Z","shell.execute_reply.started":"2025-12-28T09:11:29.601698Z","shell.execute_reply":"2025-12-28T09:11:29.60172Z"},"papermill":{"duration":0.020578,"end_time":"2025-04-16T20:13:04.009088","exception":false,"start_time":"2025-04-16T20:13:03.98851","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CONFIG = {\n    # Inference Class\n    \"use_sigmoid\": False,\n    \"use_compiled_fp16\": False,\n    # Data config\n    \"test_data_root\": \"/kaggle/input/birdclef-2025/test_soundscapes/*.ogg\" if IS_TEST else \"/kaggle/input/birdclef-2025/train_soundscapes/*.ogg\",\n    \"label_map_data_path\":'/kaggle/input/bird-clef-2025-models-v2/bird2int_2025.json',\n    \"lookback\":None,\n    \"lookahead\":None,\n    \"segment_len\":5,\n    \"step\": None,\n    \"late_normalize\": True,\n    \"load_normalize\": False,\n    # Model config\n    \"exp_name\":EXP_NAME,\n    # Post Processing\n    \"global_target_smoothing_coef\": None\n}\n\nif CONFIG.get(\"use_sed_mode\", False):\n    assert CONFIG[\"step\"] is not None\nelse:\n    assert CONFIG[\"step\"] is None\n","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.602784Z","iopub.status.idle":"2025-12-28T09:11:29.603053Z","shell.execute_reply.started":"2025-12-28T09:11:29.602919Z","shell.execute_reply":"2025-12-28T09:11:29.602931Z"},"papermill":{"duration":0.015912,"end_time":"2025-04-16T20:13:04.031851","exception":false,"start_time":"2025-04-16T20:13:04.015939","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data","metadata":{"papermill":{"duration":0.006282,"end_time":"2025-04-16T20:13:04.045411","exception":false,"start_time":"2025-04-16T20:13:04.039129","status":"completed"},"tags":[]}},{"cell_type":"code","source":"bird2id = load_json(CONFIG[\"label_map_data_path\"])","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.604168Z","iopub.status.idle":"2025-12-28T09:11:29.604552Z","shell.execute_reply.started":"2025-12-28T09:11:29.604404Z","shell.execute_reply":"2025-12-28T09:11:29.60442Z"},"papermill":{"duration":0.022074,"end_time":"2025-04-16T20:13:04.075989","exception":false,"start_time":"2025-04-16T20:13:04.053915","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_au_pathes = glob(CONFIG[\"test_data_root\"])\nif not IS_TEST:\n    test_au_pathes = test_au_pathes[:10]   ###### 700 for performance test\n\ntest_df = pd.DataFrame({\n    \"filename\": test_au_pathes,\n    #\"duration_s\": [librosa.get_duration(filename=el) for el in test_au_pathes]\n    \"duration_s\": [60.0 for el in test_au_pathes]  # Fernando: prior is slow and all files have 60.0s\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.60585Z","iopub.status.idle":"2025-12-28T09:11:29.606158Z","shell.execute_reply.started":"2025-12-28T09:11:29.60602Z","shell.execute_reply":"2025-12-28T09:11:29.606033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds_config_test = {\n   \"root\": \"\",\n   \"label_str2int_mapping_path\": CONFIG[\"label_map_data_path\"],\n   \"n_cores\": 64,\n   \"use_audio_cache\": True,\n   \"test_mode\": True,\n   \"segment_len\": CONFIG[\"segment_len\"],\n   \"lookback\":CONFIG[\"lookback\"],\n   \"lookahead\":CONFIG[\"lookahead\"],\n    \"sample_id\": None,\n    \"late_normalize\": CONFIG[\"late_normalize\"],\n    \"load_normalize\": CONFIG.get(\"load_normalize\", True),\n    \"step\": CONFIG[\"step\"],\n    \"validate_sr\": 32_000\n}\nloader_config = {\n    \"batch_size\": 24,\n    \"drop_last\": False,\n    \"shuffle\": False,\n    \"num_workers\": 0,\n}","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.606937Z","iopub.status.idle":"2025-12-28T09:11:29.607398Z","shell.execute_reply.started":"2025-12-28T09:11:29.607153Z","shell.execute_reply":"2025-12-28T09:11:29.607172Z"},"papermill":{"duration":0.015432,"end_time":"2025-04-16T20:13:09.164632","exception":false,"start_time":"2025-04-16T20:13:09.1492","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds_test = WaveAllFileDataset(df=test_df, **ds_config_test)\nloader_test = torch.utils.data.DataLoader(\n    ds_test,\n    **loader_config,\n)","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.60902Z","iopub.status.idle":"2025-12-28T09:11:29.609482Z","shell.execute_reply.started":"2025-12-28T09:11:29.60924Z","shell.execute_reply":"2025-12-28T09:11:29.609257Z"},"papermill":{"duration":0.02197,"end_time":"2025-04-16T20:13:09.193678","exception":false,"start_time":"2025-04-16T20:13:09.171708","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Vino Model","metadata":{"papermill":{"duration":0.006641,"end_time":"2025-04-16T20:13:09.207448","exception":false,"start_time":"2025-04-16T20:13:09.200807","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def compile_model(exp_name, postfix=POSTFIX):\n    core = ov.Core()\n    model = core.read_model(model=f\"/kaggle/input/{DATASET_NAME}/{exp_name}/{exp_name}/onnx_ensem_5first_folds{postfix}_openvino_fp16/model_simpl.xml\")\n    return core.compile_model(model=model, device_name=\"CPU\")\n\ncompiled_models = [compile_model(e_n) for e_n in [\n    \"tf_efficientnetv2_s_in21k_Exp_noamp_64bs_5sec_BasicAug_EqualBalancing_AdamW1e4_CosBatchLR1e6_Epoch50_FocalBCELoss_LSF1005_FromPrebs1_PseudoF2PT05MT01P04I2_AddRareBirdsNoLeak\",\n    \"eca_nfnet_l0_Exp_noamp_64bs_5sec_BasicAug_SqrtBalancing_Radamlr1e3_CosBatchLR1e6_Epoch50_FocalBCELoss_LSF1005_FromXCV2Best_PseudoF2PT05MT01P04I3_MinorOverSampleV1\"\n]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.611202Z","iopub.status.idle":"2025-12-28T09:11:29.611544Z","shell.execute_reply.started":"2025-12-28T09:11:29.611378Z","shell.execute_reply":"2025-12-28T09:11:29.611394Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Vino Inference Class\n","metadata":{"papermill":{"duration":0.006732,"end_time":"2025-04-16T20:13:10.9402","exception":false,"start_time":"2025-04-16T20:13:10.933468","status":"completed"},"tags":[]}},{"cell_type":"code","source":"inference_class = BirdsInference(\n    device=\"cpu\",\n    verbose_tqdm=True,\n    use_sigmoid=CONFIG[\"use_sigmoid\"],\n    use_compiled_fp16=CONFIG[\"use_compiled_fp16\"],\n)","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.613094Z","iopub.status.idle":"2025-12-28T09:11:29.613427Z","shell.execute_reply.started":"2025-12-28T09:11:29.613229Z","shell.execute_reply":"2025-12-28T09:11:29.613243Z"},"papermill":{"duration":0.014366,"end_time":"2025-04-16T20:13:10.961674","exception":false,"start_time":"2025-04-16T20:13:10.947308","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Vino prediction","metadata":{"papermill":{"duration":0.007229,"end_time":"2025-04-16T20:13:10.976049","exception":false,"start_time":"2025-04-16T20:13:10.96882","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_preds, test_dfidx, test_end = inference_class.predict_test_loader(\n#     nn_models=model,\n    nn_models=compiled_models,\n    data_loader=loader_test,\n#     is_onnx_model=True\n    is_openvino_model=True\n)\ntest_pred_df = compose_submission_dataframe(\n    probs=test_preds,\n    dfidxs=test_dfidx,\n    end_seconds=test_end,\n    filenames=loader_test.dataset.df[loader_test.dataset.name_col].copy(),\n    bird2id=bird2id,\n)\nsample_submission = pd.read_csv(\"/kaggle/input/birdclef-2025/sample_submission.csv\")\nif test_pred_df.shape[1] > sample_submission.shape[1]:\n    print(\"Shrinking columns\")\n    test_pred_df = test_pred_df[sample_submission.columns]","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.61444Z","iopub.status.idle":"2025-12-28T09:11:29.614719Z","shell.execute_reply.started":"2025-12-28T09:11:29.614595Z","shell.execute_reply":"2025-12-28T09:11:29.614606Z"},"papermill":{"duration":33.815759,"end_time":"2025-04-16T20:13:44.798833","exception":false,"start_time":"2025-04-16T20:13:10.983074","status":"completed"},"tags":[],"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ONNX prediction","metadata":{}},{"cell_type":"code","source":"!pip install -q onnxruntime --no-index --find-links /kaggle/input/b5-onnxruntime-v1-22","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.615602Z","iopub.status.idle":"2025-12-28T09:11:29.615874Z","shell.execute_reply.started":"2025-12-28T09:11:29.61574Z","shell.execute_reply":"2025-12-28T09:11:29.615755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 1: # libs onnx\n    import gc\n    import time\n    import glob\n    import math\n    import concurrent.futures\n    from tqdm.autonotebook import tqdm\n    \n    import torchaudio\n    import torchaudio.transforms as T\n    \n    import onnx\n    import onnxruntime as ort\n\n    # ----------------------------------------------------------------------------------------------------\n    # 0. constants\n    # ----------------------------------------------------------------------------------------------------\n\n    BIRDS = [\n        '1139490', '1192948', '1194042', '126247', '1346504', '134933', '135045', '1462711', '1462737', '1564122', '21038', '21116',\n        '21211', '22333', '22973', '22976', '24272', '24292', '24322', '41663', '41778', '41970', '42007', '42087', '42113', '46010',\n        '47067', '476537', '476538', '48124', '50186', '517119', '523060', '528041', '52884', '548639', '555086', '555142', '566513',\n        '64862', '65336', '65344', '65349', '65373', '65419', '65448', '65547', '65962', '66016', '66531', '66578', '66893', '67082',\n        '67252', '714022', '715170', '787625', '81930', '868458', '963335', \n        'amakin1', 'amekes', 'ampkin1', 'anhing', 'babwar', 'bafibi1', 'banana', 'baymac', 'bbwduc', 'bicwre1', 'bkcdon', 'bkmtou1', \n        'blbgra1', 'blbwre1', 'blcant4', 'blchaw1', 'blcjay1', 'blctit1', 'blhpar1', 'blkvul', 'bobfly1', 'bobher1', 'brtpar1', 'bubcur1',\n        'bubwre1', 'bucmot3', 'bugtan', 'butsal1', 'cargra1', 'cattyr', 'chbant1', 'chfmac1', 'cinbec1', 'cocher1', 'cocwoo1', 'colara1',\n        'colcha1', 'compau', 'compot1', 'cotfly1', 'crbtan1', 'crcwoo1', 'crebob1', 'cregua1', 'creoro1', 'eardov1', 'fotfly', 'gohman1',\n        'grasal4', 'grbhaw1', 'greani1', 'greegr', 'greibi1', 'grekis', 'grepot1', 'gretin1', 'grnkin', 'grysee1', 'gybmar', 'gycwor1', \n        'labter1', 'laufal1', 'leagre', 'linwoo1', 'littin1', 'mastit1', 'neocor', 'norscr1', 'olipic1', 'orcpar', 'palhor2', 'paltan1',\n        'pavpig2', 'piepuf1', 'pirfly1', 'piwtyr1', 'plbwoo1', 'plctan1', 'plukit1', 'purgal2', 'ragmac1', 'rebbla1', 'recwoo1', 'rinkin1',\n        'roahaw', 'rosspo1', 'royfly1', 'rtlhum', 'rubsee1', 'rufmot1', 'rugdov', 'rumfly1', 'ruther1', 'rutjac1', 'rutpuf1', 'saffin',\n        'sahpar1', 'savhaw1', 'secfly1', 'shghum1', 'shtfly1', 'smbani', 'snoegr', 'sobtyr1', 'socfly1', 'solsan', 'soulap1', 'spbwoo1',\n        'speowl1', 'spepar1', 'srwswa1', 'stbwoo2', 'strcuc1', 'strfly1', 'strher', 'strowl1', 'tbsfin1', 'thbeup1', 'thlsch3', 'trokin',\n        'tropar', 'trsowl', 'turvul', 'verfly', 'watjac1', 'wbwwre1', 'whbant1', 'whbman1', 'whfant1', 'whmtyr1', 'whtdov', 'whttro1',\n        'whwswa1', 'woosto', 'y00678', 'yebela1', 'yebfly1', 'yebsee1', 'yecspi2', 'yectyr1', 'yehbla2', 'yehcar1', 'yelori1', 'yeofly1',\n        'yercac1', 'ywcpar', \n    ]\n    \n    # ----------------------------------------------------------------------------------------------------\n    # 1. miscelaneous functions\n    # ----------------------------------------------------------------------------------------------------\n    \n    class dotdict(dict):\n        def __getattr__(self, name):\n            return self.get(name, None)\n\n        def __setattr__(self, name, val):\n            self[name] = val\n\n        def __set__(self, name, val):\n            self.__setattr__(name, val)\n            \n        def to_dict(self):\n            return eval(str(self))\n\n        # the following are required by save_obj\n        def __getstate__(self):\n            return self.__dict__\n\n        def __setstate__(self, d):\n            self.__dict__.update(d)\n\n    \n    def clean_config(models_config, models=None):\n        out, out_dict = [], dict()\n        for row in models_config:\n            if len(row) == 5:\n                name, folds, file, folder, config = row\n                folder = f'{ROOT}/{folder}'\n            elif len(row) == 4:\n                name, folds, folder, config = row\n                file = name\n                folder = f'{ROOT}/{folder}'\n            else:\n                name, folds, config = row\n                file = name\n                folder = f'{ROOT}/b5-models'\n                \n            if models is not None and not name in models:\n                continue\n            config = dotdict(config)\n            if config.aug:\n                config.aug = dotdict(config.aug)\n            out.append((name, folds, file, folder, config))\n            out_dict[name] = config\n        return out, out_dict\n\n\n    def sigmoid(x):\n        return np.where(x >= 0, 1 / (1 + np.exp(-x)), np.exp(x) / (1 + np.exp(x)))\n\n    # ----------------------------------------------------------------------------------------------------\n    # 2. melspec functions\n    # ----------------------------------------------------------------------------------------------------\n\n    len_audio_1m = 1 * 60 * 32_000\n   \n    hop_length_256_5 = 32_000 * 5 // (256 - 1)\n    \n    melspec_128_256_5 = T.MelSpectrogram(\n        n_fft=2048, hop_length=hop_length_256_5, f_min=50, f_max=16_000, sample_rate=32_000,\n        n_mels=128, norm='slaney', mel_scale='slaney', pad_mode='constant')\n\n    db_transform = T.AmplitudeToDB(stype='power', top_db=80)\n\n    def transform_to_spec(dim=(128, 256), norm=False, duration=5, snipet=True, spec=None, bits=8):\n        \"\"\" converts an audio array to spectrograms. \"\"\"\n        melspec_fn = melspec_128_256_5\n        db_transform_ = db_transform\n        eps = 1e-6\n        \n        def _process(audio):\n            spec = melspec_fn(audio)\n            spec = db_transform(spec)\n            min_ = torch.amin(spec)\n            max_ = torch.amax(spec)\n            spec = (spec - min_) / (max_ - min_)\n            if bits == 8 or bits is None:\n                spec = (spec * 255).to(torch.uint8) / 255\n            elif bits == 16:\n                spec = (spec * 65535).to(torch.uint32) / 65535\n            return spec\n\n        return _process\n\n\n    # ----------------------------------------------------------------------------------------------------\n    # 3. submit functions\n    # ----------------------------------------------------------------------------------------------------\n\n    def split_spec_configs(ws, models_config_dict):\n        \"\"\" split models based on the melspec parameters \"\"\"\n        spec_configs = dict()\n        for name in ws.keys():\n            cfg = models_config_dict[name.split('_')[0]]\n            spec = (cfg.spec, cfg.img_dim, cfg.img_duration, cfg.spec_norm, cfg.snipet)\n            if spec not in spec_configs:\n                spec_configs[spec] = [name]\n            else:\n                spec_configs[spec].append(name)\n        return spec_configs\n\n\n    def prepare_data(files, num_workers, verbose=1, norm=False, dim=(128,256), duration=5,\n                     snipet=False, spec=None, bits=8):\n        \"\"\" generate specs for submission in parallel \"\"\"\n        transform_to_spec_ = transform_to_spec(\n            dim=dim, norm=norm, duration=duration, snipet=snipet, spec=spec, bits=bits)\n        len_duration = duration * 32_000\n        \n        def _process_chunk(chunk):\n            out = []\n            for file in tqdm(chunk):\n                audio, sr = torchaudio.load(file)\n                audio = audio[:, :len_audio_1m].view(-1, len_duration)\n                out.append(transform_to_spec_(audio))\n            out = torch.cat(out, axis=0)\n            out = out.unsqueeze(1)\n            return out\n\n        t1 = time.time()\n        chunk_size = math.ceil(len(files) / num_workers)\n        chunks = [files[i : i+chunk_size] for i in range(0, len(files), chunk_size)]\n\n        with concurrent.futures.ThreadPoolExecutor(max_workers=num_workers) as executor:\n            out = list(executor.map(_process_chunk, chunks))\n        out = torch.cat(out, axis=0)\n        duration = time.time() - t1\n        if verbose:\n            print(f'specs: {out.shape}')\n            print(f'time: {duration:,.0f} s for {len(files):,.0f} files  -  all: {duration * 700 / len(files) / 60:,.0f} m')\n        return out\n\n\n    def load_models_onnx(models_config):\n        models = dict()\n        for name, folds, filename, folder, config in models_config:\n            for fold in folds:\n                filename = f'{name}_f{fold}'\n                onnx_model = onnx.load(f'{folder}/{name}/{filename}.onnx')\n                onnx_session = ort.InferenceSession(onnx_model.SerializeToString())\n                models[filename] = onnx_session\n        print(f'models: {list(models.keys())}')\n        return models\n\n\n    def prepare_predict():\n        \"\"\" create models \"\"\"\n        models = load_models_onnx(models_config)\n        predict_fn = predict_fn_onnx\n        return models, predict_fn\n        \n\n    def predict_fn_onnx(specs):\n        \"\"\" run prediction for specs using onnx models. \"\"\"\n        stop = False\n        batch_size = ARGS.batch_size\n        out = []\n        specs = specs.numpy()\n        n = math.ceil(specs.shape[0] / batch_size)\n        \n        for model, bits in zip(models, all_bits):\n            out_model = []\n            for i in tqdm(range(n)):\n                x = specs[i*batch_size : (i+1)*batch_size]\n                x = ((x * 255).astype(np.uint8) / 255).astype(np.float32)\n                y = model.run(['y'], {'x': x})[0]\n                out_model.append(y)\n            out_model = np.concatenate(out_model, axis=0)\n            out.append(out_model)\n        out = np.stack(out, axis=0)\n        return out\n\n\n    def ensemble_oof(coef, oof, names):\n        w, idxs = [], []\n        if isinstance(coef, list):\n            coef = {x:1 for x in coef}\n        for name, factor in coef.items():\n            w.append(factor)\n            idxs.append(names.index(name))\n        w = np.array(w).reshape((-1,1,1))\n        e = (oof[idxs] * w).sum(axis=0) / w.sum(axis=0)\n        return e\n\n    print('done')","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.616798Z","iopub.status.idle":"2025-12-28T09:11:29.617109Z","shell.execute_reply.started":"2025-12-28T09:11:29.616961Z","shell.execute_reply":"2025-12-28T09:11:29.616976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models_config = [\n('ebs.426', [0,1,2,3,4], 'b5-models-onnx', {'accumulation_steps': 1, 'add_position': False, 'adv_eps': 0.01, 'adv_lr': 0.005, 'adv_th': 0.3, 'aug': {'CoarseDropout': (0.375, 0.375, 1, 0.7), 'Flip': 0.5, 'audio': False, 'mixup': 1, 'volume': (0.3333333333333333, 3)}, 'awp': False, 'background_noise_cache_size': 1000, 'background_noise_max_usage': 6, 'background_noise_prob': 0.5, 'background_noise_reference': True, 'betas': (0.9, 0.999), 'check_val_every_n_epoch': 1, 'curation_mode': 'speech', 'dataset': 'h5-6', 'dataset_val': 'val-128-256-m1-4', 'drop_path_rate': None, 'duration': {'hours': 11, 'minutes': 55}, 'encoder': 'tf_efficientnetv2_s', 'end_lr': 1e-06, 'epochs': 50, 'eps': 1e-08, 'extension': '.ogg', 'files_edges': '/kaggle/input/b5-cache/files_edges speech 5', 'filter': (False, 0), 'fold_col': 'author', 'gem_p': 1.8, 'head_dropout': 0.0, 'img_dim': (128, 256), 'img_duration': 5, 'in_chans': 1, 'label_smoothing': 0.0, 'log_every_n_steps': 50, 'loss_schedule': {0: 'focal_volodymyr'}, 'lr': [(1e-05, 0.001), (3, 0.00025)], 'mask_non_labels': True, 'max_grad_norm': 10, 'mode': 'max', 'model_name': 'SpecNetImg', 'monitor': 'val_auc', 'montage': (0, 3), 'montage_cache_size': 10, 'n_folds': 5, 'n_preload_species': 15, 'norm_audio': False, 'num_workers': 6, 'optimizer': 'AdamW', 'out_indices': 2, 'precision_schedule': {0: 'float32'}, 'prefetch_factor': 2, 'previous_dataset': 'add-h5-6', 'pseudo_label_zero_th': 0.1, 'sample_unlabeled_prob': 0.0, 'saved_optimizer': None, 'saved_scheduler': None, 'schedule_type': 'multi_lr', 'seed': 10, 'spec': None, 'spec_norm': False, 'start_lr': None, 'step_scheduler_after': 'step', 'swa': None, 'train_batch_size': 64, 'train_mode': 'random', 'train_primary_th': 0.5, 'train_size': 28000, 'train_unlabeled_dataset': 'h5-unlabeled-2', 'unlabeled_primary_th': 0.5, 'unlabeled_weight': 1, 'use_mask': False, 'val_batch_size': 64, 'warmup_share': 0.02, 'weight_decay': 1e-06, 'ws_power': 0.5, 'name': 'ebs.1', 'folds': [4], 'unlabeled_pseudo_preds': ['b5-data-pseudo-pred-v3/unlabeled pseudo pred v2'], 'montage_bird_prob': False, 'pretrained_encoder': '/kaggle/input/b5-pretrained-weights/tf_efficientnetv2_s_in21k_Pretrainversion1.pth'}),\n]\n\nROOT = '/kaggle/input'\nmodels_config, models_config_dict = clean_config(models_config)\n\nebs_426 = {'ebs.426_f0': 1/5, 'ebs.426_f1': 1/5, 'ebs.426_f2': 1/5, 'ebs.426_f3': 1/5, 'ebs.426_f4': 1/5}\n\nclass ARGS:\n    engine = 'onnx'  # opt onnx vino\n    lot_size = 90\n    num_workers = 3\n    batch_size = 4\n    ws = ebs_426\n    ensemble_logit = False\n    norm_pred = False\n        \nif ARGS.ws is not None:\n    ARGS.ws = {f'{x}_f0'if '_' not in x else x : w for x, w in ARGS.ws.items()}\n    ws_names = set(x.split('_')[0] for x in ARGS.ws.keys())\n    models_config = [x for x in models_config if x[0] in ws_names]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.618844Z","iopub.status.idle":"2025-12-28T09:11:29.619342Z","shell.execute_reply.started":"2025-12-28T09:11:29.619181Z","shell.execute_reply":"2025-12-28T09:11:29.619199Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntrain_path = '/kaggle/input/birdclef-2025/train_soundscapes'\ntest_files_ = glob.glob(f'{test_path}*')\nfolder = test_path if len(test_files_) > 1 else train_path\nrepeated_files = ['_'.join(x.split('_')[:-1]) for x in test_pred_df.row_id]\nfiles = []\nfor x in repeated_files:\n    if len(files) == 0 or x != files[-1]:\n        files.append(x)\nfiles = [f'{folder}/{x}.ogg' for x in files]\n\nprint(f'files: {len(files):,.0f}')","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.620923Z","iopub.status.idle":"2025-12-28T09:11:29.621374Z","shell.execute_reply.started":"2025-12-28T09:11:29.621136Z","shell.execute_reply":"2025-12-28T09:11:29.621153Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 1:  # predict\n    file_lots = [files[i : i+ARGS.lot_size] for i in range(0, len(files), ARGS.lot_size)]\n    models_dict, predict_fn = prepare_predict()\n    predict_fn = predict_fn_onnx\n    spec_configs = split_spec_configs(ARGS.ws, models_config_dict)\n    print(spec_configs)\n    names = []\n    oof = []\n    for (spec, img_dim, img_duration, spec_norm, snipet), spec_model_names in spec_configs.items():\n        print('-'*20, 'spec', spec, img_dim, img_duration, spec_norm, snipet)\n        models = [models_dict[f'{name}'] for name in spec_model_names]\n        all_bits = [models_config_dict[name.split('_')[0]].bits for name in spec_model_names]\n        names += spec_model_names\n        preds = []\n        for lot in file_lots:\n            specs = prepare_data(lot, 4, verbose=0, norm=spec_norm, dim=img_dim, duration=img_duration,\n                                 snipet=snipet, spec=spec, bits=32)\n            chunk_size = math.ceil(len(specs) / ARGS.num_workers)\n            chunks = [specs[i : i+chunk_size] for i in range(0, len(specs), chunk_size)]\n            with concurrent.futures.ThreadPoolExecutor(max_workers=ARGS.num_workers) as executor:\n                chunk_preds = list(executor.map(predict_fn, chunks))\n            chunk_preds = np.concatenate(chunk_preds, axis=1)\n            print('chunk_preds', chunk_preds.shape)\n            preds.append(chunk_preds)\n        oof_spec = np.concatenate(preds, axis=1)\n        oof.append(oof_spec)\n\n    oof = np.concatenate(oof, axis=0)\n    print(oof.shape)\n    gc.collect()\n\n    # ensemble\n    oof = sigmoid(oof)\n    onnx_probs = ensemble_oof(ARGS.ws, oof, names)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.622657Z","iopub.status.idle":"2025-12-28T09:11:29.623034Z","shell.execute_reply.started":"2025-12-28T09:11:29.622843Z","shell.execute_reply":"2025-12-28T09:11:29.622858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# merge solutions\nvino_probs = test_pred_df[BIRDS].values.astype(np.float32)\nN_, F_ = vino_probs.shape\nonly_probs = vino_probs * 2/3 + onnx_probs * 1/3\ntest_pred_df[BIRDS] = only_probs.reshape((N_, F_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-28T09:11:29.625531Z","iopub.status.idle":"2025-12-28T09:11:29.625825Z","shell.execute_reply.started":"2025-12-28T09:11:29.625688Z","shell.execute_reply":"2025-12-28T09:11:29.6257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Postprocessing","metadata":{}},{"cell_type":"code","source":"def postprocessing(input_df, top=6):\n    print(\"Top\", top)\n    only_probs = input_df.iloc[:, 1:].values\n    N, F = only_probs.shape\n    only_probs = only_probs.reshape((N//12, 12, F))\n    mean_ = np.mean(np.sort(only_probs, axis=1)[:, -top:], axis=1, keepdims=True)\n    only_probs *= mean_\n    input_df.iloc[:, 1:] = only_probs.reshape((N, F))\n    return input_df","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.627152Z","iopub.status.idle":"2025-12-28T09:11:29.62763Z","shell.execute_reply.started":"2025-12-28T09:11:29.627422Z","shell.execute_reply":"2025-12-28T09:11:29.62744Z"},"papermill":{"duration":0.018099,"end_time":"2025-04-16T20:13:44.871809","exception":false,"start_time":"2025-04-16T20:13:44.85371","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred_df = postprocessing(test_pred_df, top=1)","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.629648Z","iopub.status.idle":"2025-12-28T09:11:29.630088Z","shell.execute_reply.started":"2025-12-28T09:11:29.62986Z","shell.execute_reply":"2025-12-28T09:11:29.629878Z"},"papermill":{"duration":0.016011,"end_time":"2025-04-16T20:13:44.897044","exception":false,"start_time":"2025-04-16T20:13:44.881033","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Prediction","metadata":{"papermill":{"duration":0.009274,"end_time":"2025-04-16T20:13:45.312816","exception":false,"start_time":"2025-04-16T20:13:45.303542","status":"completed"},"tags":[],"_kg_hide-input":true}},{"cell_type":"code","source":"# sample_submission = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nassert set(sample_submission.columns) == set(test_pred_df.columns)\ntest_pred_df = test_pred_df[sample_submission.columns]\ntest_pred_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2025-12-28T09:11:29.630882Z","iopub.status.idle":"2025-12-28T09:11:29.631213Z","shell.execute_reply.started":"2025-12-28T09:11:29.631075Z","shell.execute_reply":"2025-12-28T09:11:29.631087Z"},"papermill":{"duration":0.066031,"end_time":"2025-04-16T20:13:45.388411","exception":false,"start_time":"2025-04-16T20:13:45.32238","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}