{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"vscode":{"interpreter":{"hash":"f7241b2af102f7e024509099765066b36197b195077f7bfac6e5bc041ba17c8c"}},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"datasetVersion","sourceId":8680930,"datasetId":4783443,"databundleVersionId":8832631},{"sourceType":"datasetVersion","sourceId":8650354,"datasetId":4784404,"databundleVersionId":8800521},{"sourceType":"datasetVersion","sourceId":8665785,"datasetId":4966492,"databundleVersionId":8816811},{"sourceType":"datasetVersion","sourceId":8460492,"datasetId":4988031,"databundleVersionId":8599449},{"sourceType":"datasetVersion","sourceId":8108072,"datasetId":4789213,"databundleVersionId":8226301},{"sourceType":"kernelVersion","sourceId":172595154}],"dockerImageVersionId":30684,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport glob\nimport json\nimport torch\nimport joblib\nimport shutil\nimport librosa\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom scipy.special import expit\n\nwarnings.simplefilter(action=\"ignore\", category=UserWarning)\ntorch.set_num_threads(os.cpu_count())","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-13T12:31:25.082615Z","iopub.execute_input":"2024-06-13T12:31:25.082972Z","iopub.status.idle":"2024-06-13T12:31:30.658786Z","shell.execute_reply.started":"2024-06-13T12:31:25.082945Z","shell.execute_reply":"2024-06-13T12:31:30.657818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append('/kaggle/input/birdclef-2024-code/src')\n\nfrom util.logger import Config\nfrom util.metrics import macro_auc\nfrom util.torch import load_model_weights\n\nfrom data.preparation import prepare_data, prepare_folds\nfrom params import CLASSES\nfrom model_zoo.models import define_model\n\nfrom inference.predict import load_sample, infer_sample","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-13T12:31:30.664180Z","iopub.execute_input":"2024-06-13T12:31:30.667054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qqq /kaggle/input/onnxruntime/humanfriendly-10.0-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n!pip install -qqq /kaggle/input/onnxruntime/coloredlogs-15.0.1-py2.py3-none-any.whl --no-index --find-links /kaggle/input/onnxruntime\n!pip install -qqq /kaggle/input/onnxruntime/onnxruntime-1.17.3-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl --no-index --find-links /kaggle/input/onnxruntime","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Params","metadata":{}},{"cell_type":"code","source":"EVAL = False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if EVAL:\n    DATA_PATH = \"/kaggle/input/birdclef-2024/train_audio/\"\nelse:\n    DATA_PATH = \"/kaggle/input/birdclef-2024/test_soundscapes/\"\n    LIM = None\n    \n    if len(os.listdir(DATA_PATH)) < 5:\n        DATA_PATH = \"/kaggle/input/birdclef-2024/unlabeled_soundscapes/\"\n        LIM = 100\n#         LIM = 1100\n        LIM = 3","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 48\nUSE_FP16 = True\nNUM_WORKERS = 4\n\nDEVICE = \"cpu\" \n\nDURATION = 5\nSR = 32000\n\nUSE_PP = True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FOLD = 0 if EVAL else \"fullfit_0\"\n\nEXP_FOLDERS = [\n#     (\"/kaggle/input/birdclef-2024-weights-4/2024-05-16_12/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),  # effvit-b0 8xPLnoaug CPMPMel head no augs       <-- 0.71\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-21_4/\", [f\"fullfit_{i}\" for i in range(5)], \"onnx\"),   # mnasnet PL2 no augs extra data start sampling  \n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-22_9/\", [f\"fullfit_{i}\" for i in range(5)], \"onnx\"),   # mnasnet PL3\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-23_6/\", [f\"fullfit_{i}\" for i in range(5)], \"onnx\"),   # mnasnet PL2.5\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-23_8/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # mnasnet PL2.5 more epochs   <-- 0.71\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-24_0/\", [f\"fullfit_{i}\" for i in range(5)], \"onnx\"),   # effvit-b0 PL2.5 more epochs\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-31_1/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet start_sampling xc 60eps avg_pp\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-05-31_2/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 avg_pp\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-02_1/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 more epochs less reg old comp\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-03_0/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet more epochs nodrop old comp lower lr\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-03_3/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 more epochs old comp mixup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-03_4/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet more epochs old comp mixup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-04_2/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 more epochs old comp mixup no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-04_3/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet more epochs old comp mixup no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-05_2/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 more epochs old comp mixup no rating=1 dedup more_pl up_low_freq\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-05_3/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet more epochs old comp mixup no rating=1 dedup more_pl up_low_freq\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-05_4/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2 efficientvit_b0 more epochs old comp more mixup no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-06_0/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL2.5 mnasnet more epochs old comp more mixup no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-06_10/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),  # PL4 efficientvit_b0 more epochs old comp mixup no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-07_0/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL4.5 mnasnet more epochs old comp more no rating=1 dedup\n    (\"/kaggle/input/birdclef-2024-weights-1/2024-06-07_10/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),  # PL5 efficientvit_b0 more epochs old comp mixup no rating=1 dedup\n    (\"/kaggle/input/birdclef-2024-weights-1/2024-06-07_11/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),   # PL5.5 mnasnet more epochs old comp more no rating=1 dedup\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-09_10/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),  # AVES efficientvit_b0\n#     (\"/kaggle/input/birdclef-2024-weights-1/2024-06-09_11/\", [f\"fullfit_{i}\" for i in range(3)], \"onnx\"),  # AVES mnasnet\n\n]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data","metadata":{}},{"cell_type":"code","source":"if EVAL:\n    df = pd.DataFrame({\"path\": glob.glob(DATA_PATH + \"*/*\")})\n    df[\"id\"] = df[\"path\"].apply(lambda x: x.split(\"/\")[-1][:-4])\n\n    folds = pd.read_csv('/kaggle/input/birdclef-2024-weights-1/folds_4.csv')\n    folds['id'] = folds['filename'].apply(lambda x: x.split('/')[-1][:-4])\n    df = df.merge(folds)\n    df = df[df['fold'] == 0].reset_index(drop=True)\n\n    df[\"primary_label\"] = df[\"path\"].apply(lambda x:  x.split('/')[-2])\nelse:\n    df = pd.DataFrame({\"path\": glob.glob(DATA_PATH + \"*\")})\n    df[\"id\"] = df[\"path\"].apply(lambda x: x.split(\"/\")[-1].rsplit('.', 1)[0])\n\n    if LIM:\n        df = df.head(LIM * 2)\n        df[\"duration\"] = df[\"path\"].apply(lambda x: librosa.get_duration(path=x))\n        df = df[df[\"duration\"] == 240].reset_index(drop=True)\n    \n        df = df.head(LIM)\n        \ndisplay(df.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Models","metadata":{}},{"cell_type":"code","source":"models = []\nfor e in EXP_FOLDERS:\n    exp_folder, folds, runtime = e\n    config = Config(json.load(open(exp_folder + \"config.json\", \"r\")))\n    \n    for fold in folds:\n        weights = exp_folder + f\"{config.name}_{fold}.pt\"\n\n        model = define_model(\n            config.name,\n            config.melspec_config,\n            head=config.head,\n            aug_config=config.aug_config,\n            num_classes=config.num_classes,\n            n_channels=config.n_channels,\n            drop_rate=config.drop_rate,\n            drop_path_rate=config.drop_path_rate,\n            norm=config.norm if hasattr(config, \"norm\") else \"min_max\",\n            top_db=config.top_db if hasattr(config, \"top_db\") else None,\n            verbose=True,\n            pretrained=False\n        )\n        model = model.to(DEVICE).eval()\n\n        model = load_model_weights(model, weights, verbose=config.local_rank == 0)\n        models.append((model, runtime))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Export","metadata":{}},{"cell_type":"code","source":"sessions = [None for _ in range(len(models))]\n\nif any([runtime != \"torch\" for _, runtime in models]):\n    sessions = []\n    import onnx\n    import onnxruntime as ort\n#     from onnxconverter_common import float16\n\n    input_names = ['x']\n    output_names = ['output']\n\n    input_tensor = torch.randn(\n        1 if EVAL else BATCH_SIZE,\n        config.n_channels,\n        config.melspec_config['n_mels'],\n        313 if config.melspec_config['hop_length'] == 512 else 224\n    )\n\n    for i, (model, runtime) in tqdm(enumerate(models), total=len(models)):\n        name = f\"model_{i}.onnx\"\n        \n        torch.onnx.export(\n            model.encoder.cpu(),\n            input_tensor,\n            name,\n            verbose=False,\n            input_names=input_names,\n            output_names=output_names,\n        )\n        onnx_model = onnx.load(name)\n        # onnx_model = float16.convert_float_to_float16(onnx_model)\n        # onnx.save(onnx_model, f\"model_{i}.onnx\")\n        onnx.checker.check_model(onnx_model)\n        ort_session = ort.InferenceSession(f\"model_{i}.onnx\")\n\n        if runtime == \"onnx\":\n            sessions.append(ort_session)\n            print(f'\\n-> Converted model {name} to onnx\\n')\n\n        elif runtime == \"openvino\":\n            try:\n                import openvino.runtime as ov\n            except ImportError:\n                print('\\nInstalling openvino ...\\n')\n                !python -m pip install -qqq --no-index --find-links=/kaggle/input/openvino -r /kaggle/input/openvino/requirements.txt\n                print('\\nDone !\\n')\n                import openvino.runtime as ov\n                \n            ov_name = f\"model_{i}.xml\"\n            \n            !ovc $name --output_model $ov_name #--compress_to_fp16=False\n            \n            core = ov.Core()\n            openvino_model = core.read_model(model=name[:-4] + \"xml\")\n            compiled_model = core.compile_model(openvino_model, device_name=\"CPU\")\n            infer_request = compiled_model.create_infer_request()\n            sessions.append(infer_request)\n\n            print(f'\\n-> Converted model {ov_name} to openvino\\n')\n        else:\n            sessions.append(None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference\n- 5x vit-b0 : 4'10 onnx\n- 5x mnasnet: 4'00 onnx - 3'20 openvino\n- 5x vit-b0 + 5x mnasnet : 8'","metadata":{}},{"cell_type":"code","source":"try:\n    batches = np.array_split(np.arange(len(df)), len(df) / 100)\nexcept:\n    batches = [np.arange(len(df))]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ninference_rows = []\n\nfor i, batch in enumerate(batches):\n    print(f\"-> Batch {i + 1}/{len(batches)}\")\n    df_batch = df.iloc[batch].reset_index(drop=True)\n\n    waves = joblib.Parallel(n_jobs=os.cpu_count())(\n        joblib.delayed(load_sample)(\n            path,\n            evaluate=EVAL,\n            sr=SR,\n            duration=DURATION,\n            normalize=config.wav_norm if hasattr(config, \"wav_norm\") else \"librosa\"\n        )\n        for path in tqdm(df_batch[\"path\"].values)\n    )\n    all_preds = [\n        infer_sample(\n            wave,\n            models,\n            sessions,\n            device=DEVICE,\n            use_fp16=USE_FP16,\n        )\n        for wave in tqdm(waves)\n    ]\n\n    del waves\n    gc.collect()\n\n    for idx in range(len(df_batch)):\n        y_pred = all_preds[idx]  # n_models x 48 x 182\n        \n        if USE_PP:\n            preds = y_pred.mean(0)  # 48 x 182\n            max_preds = preds.max(0, keepdims=True)  # 1 x 182\n            max_preds = max_preds + (preds.mean() - max_preds.mean())\n            preds = preds + max_preds\n            preds = expit(preds)\n            \n            # Sliding window smoothing\n            preds_smoothed = preds.copy()\n            for i in range(preds.shape[1]):\n                p = np.pad(preds[:, i], (2, 2), mode=\"edge\")\n                preds_smoothed[:, i] = np.convolve(p, np.array([0.1, 0.2, 0.4, 0.2, 0.1]), mode=\"valid\")\n            preds = preds_smoothed\n\n        else:\n            preds = expit(y_pred).mean(0)\n\n        for t, pred in enumerate(preds):\n            predictions = dict([(l, p) for l, p in zip(CLASSES, pred)])\n            inference_rows.append(\n                {\"row_id\": f\"{df_batch.id[idx]}_{(t + 1) * 5}\"} | predictions\n            )\n\n    del all_preds\n    gc.collect()\n\nsub = pd.DataFrame(inference_rows)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)\ndisplay(sub.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Done ! ","metadata":{}}]}