{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!rm -rf /tmp/birdclef2022\n!cp -r \"/kaggle/input/birdclef22-clone-source-code-repository/birdclef2022\" /tmp/birdclef2022","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-24T08:49:00.101635Z","iopub.execute_input":"2022-05-24T08:49:00.101889Z","iopub.status.idle":"2022-05-24T08:49:02.909011Z","shell.execute_reply.started":"2022-05-24T08:49:00.101815Z","shell.execute_reply":"2022-05-24T08:49:02.908083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! tree -L 2 /tmp/birdclef2022","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:02.910869Z","iopub.execute_input":"2022-05-24T08:49:02.911133Z","iopub.status.idle":"2022-05-24T08:49:03.621394Z","shell.execute_reply.started":"2022-05-24T08:49:02.911105Z","shell.execute_reply":"2022-05-24T08:49:03.620249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%bash\ncat << 'EOF' > /tmp/run.bash\nPIP_DEP_PATH='/kaggle/input/birdclef22-create-build-environment/pip_deps'\necho ${PIP_DEP_PATH}\npip install ${PIP_DEP_PATH}/* -f ./ --no-index --no-deps --find-links=\"${PIP_DEP_PATH}\"\n\nEOF\nchmod +x /tmp/run.bash","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:03.623174Z","iopub.execute_input":"2022-05-24T08:49:03.623450Z","iopub.status.idle":"2022-05-24T08:49:03.655397Z","shell.execute_reply.started":"2022-05-24T08:49:03.623412Z","shell.execute_reply":"2022-05-24T08:49:03.654536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!/tmp/run.bash","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:49:03.657985Z","iopub.execute_input":"2022-05-24T08:49:03.658484Z","iopub.status.idle":"2022-05-24T08:49:17.920124Z","shell.execute_reply.started":"2022-05-24T08:49:03.658444Z","shell.execute_reply":"2022-05-24T08:49:17.919274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import timm\n\nprint(f\"timm version: {timm.__version__}\")","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:17.922557Z","iopub.execute_input":"2022-05-24T08:49:17.922843Z","iopub.status.idle":"2022-05-24T08:49:24.046040Z","shell.execute_reply.started":"2022-05-24T08:49:17.922806Z","shell.execute_reply":"2022-05-24T08:49:24.045311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /tmp/birdclef2022/binary_classifier","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:24.047385Z","iopub.execute_input":"2022-05-24T08:49:24.047632Z","iopub.status.idle":"2022-05-24T08:49:24.054053Z","shell.execute_reply.started":"2022-05-24T08:49:24.047599Z","shell.execute_reply":"2022-05-24T08:49:24.053207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%bash\n\n(cat << 'EOF'\nexport DATA_DIR=\"/kaggle/input/birdclef-2022\"\nEOF\n) > .env","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:24.055779Z","iopub.execute_input":"2022-05-24T08:49:24.056271Z","iopub.status.idle":"2022-05-24T08:49:24.080130Z","shell.execute_reply.started":"2022-05-24T08:49:24.056234Z","shell.execute_reply":"2022-05-24T08:49:24.079326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import importlib\nimport json\nimport os\nfrom itertools import product\nfrom os.path import basename\n\nimport glob\nimport numpy as np\nimport multiprocessing as mp\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sns\nimport torch\n\nfrom copy import copy\nfrom torch.utils.data import DataLoader\nfrom tqdm.notebook import tqdm\n\nfrom train_util import set_module_path, auto_set_config_param\nfrom torch.optim.swa_utils import AveragedModel\n\nplt.style.use(\"ggplot\")\n\n%load_ext dotenv\n%load_ext lab_black\n%load_ext autoreload\n%dotenv\n%autoreload 2","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:24.081612Z","iopub.execute_input":"2022-05-24T08:49:24.081875Z","iopub.status.idle":"2022-05-24T08:49:25.025178Z","shell.execute_reply.started":"2022-05-24T08:49:24.081842Z","shell.execute_reply":"2022-05-24T08:49:25.024443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"DATA_DIR\"]","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.026464Z","iopub.execute_input":"2022-05-24T08:49:25.026691Z","iopub.status.idle":"2022-05-24T08:49:25.087919Z","shell.execute_reply.started":"2022-05-24T08:49:25.026660Z","shell.execute_reply":"2022-05-24T08:49:25.087274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_module_path()","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.091309Z","iopub.execute_input":"2022-05-24T08:49:25.091572Z","iopub.status.idle":"2022-05-24T08:49:25.151578Z","shell.execute_reply.started":"2022-05-24T08:49:25.091535Z","shell.execute_reply":"2022-05-24T08:49:25.150778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_test_df(test_soundscapes):\n    file_ids = [basename(path)[:-4] for path in test_soundscapes]\n    test_df = pd.DataFrame(\n        {\n            \"file_id\": file_ids,\n            \"filename\": [f\"{fid}.ogg\" for fid in file_ids],\n        }\n    )\n    return test_df","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.152779Z","iopub.execute_input":"2022-05-24T08:49:25.153353Z","iopub.status.idle":"2022-05-24T08:49:25.219720Z","shell.execute_reply.started":"2022-05-24T08:49:25.153310Z","shell.execute_reply":"2022-05-24T08:49:25.218906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_soundscapes = glob.glob(\n    \"/kaggle/input/birdclef-2022/test_soundscapes/soundscape_*.ogg\"\n)\n\n\ntest_df = create_test_df(test_soundscapes)\ntest_df = test_df","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.221016Z","iopub.execute_input":"2022-05-24T08:49:25.221763Z","iopub.status.idle":"2022-05-24T08:49:25.289587Z","shell.execute_reply.started":"2022-05-24T08:49:25.221724Z","shell.execute_reply":"2022-05-24T08:49:25.288876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.290946Z","iopub.execute_input":"2022-05-24T08:49:25.291223Z","iopub.status.idle":"2022-05-24T08:49:25.362361Z","shell.execute_reply.started":"2022-05-24T08:49:25.291188Z","shell.execute_reply":"2022-05-24T08:49:25.361675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMP_FOLDER = \"/kaggle/input/birdclef-2022\"\nTEST_AUDIO_ROOT = f\"{COMP_FOLDER}/test_soundscapes\"\n\n\nsample_submission = pd.read_csv(f\"{COMP_FOLDER}/sample_submission.csv\")\nN_CORES = mp.cpu_count()\nPUBLIC_RUN = False\n\nRAM_CHECK = False\nMIXED_PRECISION = False\nDEVICE = \"cuda\"","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.363685Z","iopub.execute_input":"2022-05-24T08:49:25.363929Z","iopub.status.idle":"2022-05-24T08:49:25.434295Z","shell.execute_reply.started":"2022-05-24T08:49:25.363895Z","shell.execute_reply":"2022-05-24T08:49:25.433647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_fns = [item for item in os.listdir(TEST_AUDIO_ROOT) if item.endswith(\".ogg\")]","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.435375Z","iopub.execute_input":"2022-05-24T08:49:25.435685Z","iopub.status.idle":"2022-05-24T08:49:25.494779Z","shell.execute_reply.started":"2022-05-24T08:49:25.435647Z","shell.execute_reply":"2022-05-24T08:49:25.494010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv(\"/kaggle/working/test_metadata.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.496940Z","iopub.execute_input":"2022-05-24T08:49:25.497614Z","iopub.status.idle":"2022-05-24T08:49:25.559277Z","shell.execute_reply.started":"2022-05-24T08:49:25.497571Z","shell.execute_reply":"2022-05-24T08:49:25.558220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"rating\"] = 5\ntest_df[\"target\"] = 0\ntest_df[\"secondary_labels\"] = \"[]\"\ntest_df[\"pseudo_labels\"] = \" \".join([\"0\"] * 152)\ntest_df[\"length\"] = 32_000 * 60\ntest_df[\"fold\"] = -1","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:25.560397Z","iopub.execute_input":"2022-05-24T08:49:25.560864Z","iopub.status.idle":"2022-05-24T08:49:25.631761Z","shell.execute_reply.started":"2022-05-24T08:49:25.560816Z","shell.execute_reply":"2022-05-24T08:49:25.630967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_passt_1_v3\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\ncfg.infer = True\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfilepaths = [\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-passt_1_v3-fold-1-seed970058:v3/checkpoint_swa_model_seed970058.pth\",\n]\nfor filepath in filepaths:\n    state_dicts.append(filepath)\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_1 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_1 += [preds_]\n\npreds_1 = np.array(preds_1)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:49:25.633382Z","iopub.execute_input":"2022-05-24T08:49:25.633649Z","iopub.status.idle":"2022-05-24T08:49:46.036849Z","shell.execute_reply.started":"2022-05-24T08:49:25.633615Z","shell.execute_reply":"2022-05-24T08:49:46.036046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_panns_2_v5\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    # sd = {k.replace(\"module.\", \"\"): v for k, v in sd.items()}\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfor filepath in glob.iglob(\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-panns_2_v5-fold-1-seed781952:v3/checkpoint_swa_model_seed781952.pth\"\n):\n    state_dicts.append(filepath)\n    backbones.append(\"resnet34\")\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_2 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_2 += [preds_]\n\npreds_2 = np.array(preds_2)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:49:46.038466Z","iopub.execute_input":"2022-05-24T08:49:46.038728Z","iopub.status.idle":"2022-05-24T08:49:48.164485Z","shell.execute_reply.started":"2022-05-24T08:49:46.038691Z","shell.execute_reply":"2022-05-24T08:49:48.163490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_panns_2_v6\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    # sd = {k.replace(\"module.\", \"\"): v for k, v in sd.items()}\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfor filepath in glob.iglob(\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-panns_2_v6-fold-1-seed58594:v3/checkpoint_swa_model_seed58594.pth\"\n):\n    state_dicts.append(filepath)\n    backbones.append(\"resnet34\")\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_3 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_3 += [preds_]\n\npreds_3 = np.array(preds_3)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:48.166570Z","iopub.execute_input":"2022-05-24T08:49:48.166891Z","iopub.status.idle":"2022-05-24T08:49:49.952957Z","shell.execute_reply.started":"2022-05-24T08:49:48.166851Z","shell.execute_reply":"2022-05-24T08:49:49.952120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_panns_2_v7\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    # sd = {k.replace(\"module.\", \"\"): v for k, v in sd.items()}\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfor filepath in glob.iglob(\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-panns_2_v7-fold-1-seed285669:v3/checkpoint_swa_model_seed285669.pth\"\n):\n    state_dicts.append(filepath)\n    backbones.append(\"eca_nfnet_l0\")\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_4 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_4 += [preds_]\n\npreds_4 = np.array(preds_4)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:49.954639Z","iopub.execute_input":"2022-05-24T08:49:49.954941Z","iopub.status.idle":"2022-05-24T08:49:51.941697Z","shell.execute_reply.started":"2022-05-24T08:49:49.954906Z","shell.execute_reply":"2022-05-24T08:49:51.940878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_panns_2_v8\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    # sd = {k.replace(\"module.\", \"\"): v for k, v in sd.items()}\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfor filepath in glob.iglob(\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-panns_2_v8-fold-1-seed174663:v3/checkpoint_swa_model_seed174663.pth\"\n):\n    state_dicts.append(filepath)\n    backbones.append(\"tf_efficientnet_b0_ns\")\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_5 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_5 += [preds_]\n\npreds_5 = np.array(preds_5)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:49:51.943406Z","iopub.execute_input":"2022-05-24T08:49:51.943661Z","iopub.status.idle":"2022-05-24T08:49:52.935043Z","shell.execute_reply.started":"2022-05-24T08:49:51.943627Z","shell.execute_reply":"2022-05-24T08:49:52.934209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_passt_1_v5\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\ncfg.infer = True\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfilepaths = [\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-passt_1_v5-fold-1-seed167728:v3/checkpoint_swa_model_seed167728.pth\",\n]\nfor filepath in filepaths:\n    state_dicts.append(filepath)\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_6 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_6 += [preds_]\n\npreds_6 = np.array(preds_6)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:49:52.936850Z","iopub.execute_input":"2022-05-24T08:49:52.937202Z","iopub.status.idle":"2022-05-24T08:49:57.673025Z","shell.execute_reply.started":"2022-05-24T08:49:52.937147Z","shell.execute_reply":"2022-05-24T08:49:57.672223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_passt_1_v6\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\ncfg.infer = True\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfilepaths = [\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-passt_1_v6-fold-1-seed280349:v3/checkpoint_swa_model_seed280349.pth\",\n]\nfor filepath in filepaths:\n    state_dicts.append(filepath)\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_7 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_7 += [preds_]\n\npreds_7 = np.array(preds_7)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:49:57.674614Z","iopub.execute_input":"2022-05-24T08:49:57.674849Z","iopub.status.idle":"2022-05-24T08:50:02.309058Z","shell.execute_reply.started":"2022-05-24T08:49:57.674823Z","shell.execute_reply":"2022-05-24T08:50:02.308174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = importlib.import_module(\"default_config\")\nimportlib.reload(cfg)\ncfg = importlib.import_module(\"cfg_passt_1_v7\")\nimportlib.reload(cfg)\ncfg = copy(cfg.cfg)\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_bins)\n\ncfg.test_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\ncfg.pretrained_weights = None\ncfg.infer = True\n\nauto_set_config_param(cfg)\n\nds = importlib.import_module(cfg.dataset)\nimportlib.reload(ds)\n\nCustomDataset = ds.CustomDataset\nbatch_to_device = ds.batch_to_device\n\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"test\")\ntest_dl = DataLoader(\n    test_ds, shuffle=False, batch_size=cfg.batch_size, num_workers=N_CORES\n)\n\nmodel = importlib.import_module(cfg.model)\nimportlib.reload(model)\nNet = model.Net\n\n\ndef get_state_dict(sd_fp):\n    state_dict = torch.load(sd_fp, map_location=\"cpu\")\n    sd = state_dict[\"model\"]\n    return sd\n\n\nstate_dicts = []\nbackbones = []\nfilepaths = [\n    \"/kaggle/input/birdclef2022-model-checkpoints/swa-model-passt_1_v7-fold-1-seed572879:v3/checkpoint_swa_model_seed572879.pth\",\n]\nfor filepath in filepaths:\n    state_dicts.append(filepath)\n\nnets = []\n\nfor i, state_dict in enumerate(state_dicts):\n    net = Net(cfg).eval().cuda()\n    swa_model = AveragedModel(net, device=cfg.device)\n    swa_model.update_parameters(net)\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    swa_model.load_state_dict(sd, strict=True)\n    nets += [swa_model]\n\n# %%checkerror\nfrom scipy.stats.mstats import gmean\n\nwith torch.no_grad():\n    preds_8 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)[\"logits\"]\n                preds_ += [out.cpu().numpy()]\n\n        preds_8 += [preds_]\n\npreds_8 = np.array(preds_8)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-24T08:50:02.310791Z","iopub.execute_input":"2022-05-24T08:50:02.311051Z","iopub.status.idle":"2022-05-24T08:50:07.630377Z","shell.execute_reply.started":"2022-05-24T08:50:02.311017Z","shell.execute_reply":"2022-05-24T08:50:07.629581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_1.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.631711Z","iopub.execute_input":"2022-05-24T08:50:07.631979Z","iopub.status.idle":"2022-05-24T08:50:07.706005Z","shell.execute_reply.started":"2022-05-24T08:50:07.631943Z","shell.execute_reply":"2022-05-24T08:50:07.705344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_2.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.707077Z","iopub.execute_input":"2022-05-24T08:50:07.707837Z","iopub.status.idle":"2022-05-24T08:50:07.776586Z","shell.execute_reply.started":"2022-05-24T08:50:07.707799Z","shell.execute_reply":"2022-05-24T08:50:07.775692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_3.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.781339Z","iopub.execute_input":"2022-05-24T08:50:07.781541Z","iopub.status.idle":"2022-05-24T08:50:07.848288Z","shell.execute_reply.started":"2022-05-24T08:50:07.781515Z","shell.execute_reply":"2022-05-24T08:50:07.847531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_4.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.849691Z","iopub.execute_input":"2022-05-24T08:50:07.849971Z","iopub.status.idle":"2022-05-24T08:50:07.917766Z","shell.execute_reply.started":"2022-05-24T08:50:07.849932Z","shell.execute_reply":"2022-05-24T08:50:07.916805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_5.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.919455Z","iopub.execute_input":"2022-05-24T08:50:07.919733Z","iopub.status.idle":"2022-05-24T08:50:07.988108Z","shell.execute_reply.started":"2022-05-24T08:50:07.919695Z","shell.execute_reply":"2022-05-24T08:50:07.987082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_6.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:07.989471Z","iopub.execute_input":"2022-05-24T08:50:07.989816Z","iopub.status.idle":"2022-05-24T08:50:08.056967Z","shell.execute_reply.started":"2022-05-24T08:50:07.989779Z","shell.execute_reply":"2022-05-24T08:50:08.056271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_7.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:08.058122Z","iopub.execute_input":"2022-05-24T08:50:08.058528Z","iopub.status.idle":"2022-05-24T08:50:08.124825Z","shell.execute_reply.started":"2022-05-24T08:50:08.058491Z","shell.execute_reply":"2022-05-24T08:50:08.124039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_8.shape  # (bs, n_models, parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:08.125999Z","iopub.execute_input":"2022-05-24T08:50:08.126842Z","iopub.status.idle":"2022-05-24T08:50:08.195257Z","shell.execute_reply.started":"2022-05-24T08:50:08.126774Z","shell.execute_reply":"2022-05-24T08:50:08.194359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make Submission","metadata":{}},{"cell_type":"code","source":"def create_submit_df(test_df, cfg):\n    file_ids = test_df.file_id.tolist()\n    scored_birds = cfg.birds[:21]\n    parts = list(range(5, 65, 5))\n    row_ids = [\n        f\"{f}_{s}_{p}\"\n        for f, p, s in product(file_ids, parts, scored_birds)  # (bs, parts, n_classes)\n    ]\n    submit_df = pd.DataFrame({\"row_id\": row_ids})\n    submit_df[\"target\"] = None\n    return submit_df","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:50:08.196614Z","iopub.execute_input":"2022-05-24T08:50:08.197211Z","iopub.status.idle":"2022-05-24T08:50:08.272079Z","shell.execute_reply.started":"2022-05-24T08:50:08.197150Z","shell.execute_reply":"2022-05-24T08:50:08.271122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_scored = 21\npreds = np.concatenate(\n    [preds_1, preds_2, preds_3, preds_4, preds_5, preds_6, preds_7, preds_8], axis=1\n)\nprint(preds.shape)\n\nbn, n_models, part, n_classes = preds.shape\ngem_p = 3\npreds = (preds**gem_p).mean(axis=1) ** (1 / gem_p)  # (batch, parts, n_classes)\n# preds = preds.max(axis=1)  # (batch, parts, n_classes)\nprint(preds.shape)\n\npreds = preds[..., :n_scored]\npreds = preds.reshape(bn * part, n_scored)\npreds.shape  # (batch * parts, n_classes)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:06.379617Z","iopub.execute_input":"2022-05-24T08:53:06.379881Z","iopub.status.idle":"2022-05-24T08:53:06.464143Z","shell.execute_reply.started":"2022-05-24T08:53:06.379850Z","shell.execute_reply":"2022-05-24T08:53:06.463385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groups = [preds[..., :5], preds[..., 5:10], preds[..., 10:15], preds[..., 15:]]\npos_ratios = [0.100, 0.550, 0.350, 0.334]\n\nthresholds = np.concatenate(\n    [np.quantile(g, 1 - pr, axis=0) for g, pr in zip(groups, pos_ratios)],\n    axis=-1,\n)\nthresholds = thresholds.reshape(1, -1)\nthresholds, thresholds.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:07.031498Z","iopub.execute_input":"2022-05-24T08:53:07.032376Z","iopub.status.idle":"2022-05-24T08:53:07.115545Z","shell.execute_reply.started":"2022-05-24T08:53:07.032329Z","shell.execute_reply":"2022-05-24T08:53:07.114829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submits = preds > thresholds  # (batch * parts, n_classes)\nsubmits = submits.reshape(-1)  # (batch * parts * n_classes)\nsubmits.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:07.701214Z","iopub.execute_input":"2022-05-24T08:53:07.701680Z","iopub.status.idle":"2022-05-24T08:53:07.772970Z","shell.execute_reply.started":"2022-05-24T08:53:07.701642Z","shell.execute_reply":"2022-05-24T08:53:07.772265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df = create_submit_df(test_df, cfg)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:08.080935Z","iopub.execute_input":"2022-05-24T08:53:08.081218Z","iopub.status.idle":"2022-05-24T08:53:08.149087Z","shell.execute_reply.started":"2022-05-24T08:53:08.081184Z","shell.execute_reply":"2022-05-24T08:53:08.148215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\nsubmit_df[\"target\"] = submits","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:08.293589Z","iopub.execute_input":"2022-05-24T08:53:08.293852Z","iopub.status.idle":"2022-05-24T08:53:08.363802Z","shell.execute_reply.started":"2022-05-24T08:53:08.293821Z","shell.execute_reply":"2022-05-24T08:53:08.362976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:08.507275Z","iopub.execute_input":"2022-05-24T08:53:08.507790Z","iopub.status.idle":"2022-05-24T08:53:08.581855Z","shell.execute_reply.started":"2022-05-24T08:53:08.507753Z","shell.execute_reply":"2022-05-24T08:53:08.581061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df.target.sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:08.719732Z","iopub.execute_input":"2022-05-24T08:53:08.719998Z","iopub.status.idle":"2022-05-24T08:53:08.789889Z","shell.execute_reply.started":"2022-05-24T08:53:08.719968Z","shell.execute_reply":"2022-05-24T08:53:08.788969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:08.918665Z","iopub.execute_input":"2022-05-24T08:53:08.918934Z","iopub.status.idle":"2022-05-24T08:53:08.986780Z","shell.execute_reply.started":"2022-05-24T08:53:08.918890Z","shell.execute_reply":"2022-05-24T08:53:08.985955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize","metadata":{}},{"cell_type":"code","source":"import warnings\nfrom types import SimpleNamespace\n\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nwarnings.filterwarnings(\"ignore\")\nplt.style.use(\"ggplot\")\n\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:09.330371Z","iopub.execute_input":"2022-05-24T08:53:09.331109Z","iopub.status.idle":"2022-05-24T08:53:09.404356Z","shell.execute_reply.started":"2022-05-24T08:53:09.331071Z","shell.execute_reply":"2022-05-24T08:53:09.403227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize(rel_path, vis_df):\n    path = f\"{TEST_AUDIO_ROOT}/{rel_path}\"\n    assert os.path.isfile(path), path\n    display(ipd.Audio(path))\n\n    # show mel spec\n    fig, (ax1, ax2) = plt.subplots(\n        ncols=1, nrows=2, figsize=(18, 8), gridspec_kw={\"height_ratios\": [1, 3]}\n    )\n    audio, _ = librosa.core.load(path, sr=cfg.sample_rate, mono=True)\n    melspec = librosa.feature.melspectrogram(\n        audio,\n        sr=cfg.sample_rate,\n        n_fft=cfg.window_size,\n        hop_length=cfg.hop_length,\n        n_mels=cfg.mel_bins,\n        power=1.0,\n        fmin=cfg.fmin,\n        fmax=cfg.fmax,\n    )\n    spec = librosa.pcen(\n        melspec * (2**31),\n        time_constant=0.06,\n        eps=1e-6,\n        gain=0.8,\n        power=0.25,\n        bias=10,\n        sr=cfg.sample_rate,\n        hop_length=cfg.hop_length,\n    )\n    colormesh = librosa.display.specshow(\n        spec,\n        hop_length=cfg.hop_length,\n        sr=cfg.sample_rate,\n        fmin=cfg.fmin,\n        fmax=cfg.fmax,\n        x_axis=\"time\",\n        y_axis=\"mel\",\n        ax=ax1,\n    )\n    ax1.set_title(\n        f\"[{rel_path}]\",\n        fontsize=15,\n    )\n\n    sns.heatmap(vis_df, cmap=\"viridis\", ax=ax2, cbar=False, vmin=0, vmax=1)\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-24T08:53:09.541951Z","iopub.execute_input":"2022-05-24T08:53:09.542229Z","iopub.status.idle":"2022-05-24T08:53:09.642000Z","shell.execute_reply.started":"2022-05-24T08:53:09.542197Z","shell.execute_reply":"2022-05-24T08:53:09.641233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bn, _, parts, n_classes = preds_1.shape\nvis_preds = preds.reshape(bn, part, n_scored)\nvis_preds = vis_preds.transpose(0, 2, 1)  # (batch, class, parts)\nvis_preds.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:09.764912Z","iopub.execute_input":"2022-05-24T08:53:09.765408Z","iopub.status.idle":"2022-05-24T08:53:09.842321Z","shell.execute_reply.started":"2022-05-24T08:53:09.765368Z","shell.execute_reply":"2022-05-24T08:53:09.841582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"times = np.arange(5, 65, 5) - 2.5\nvis_df = pd.DataFrame(\n    vis_preds[0, :21], index=cfg.birds[:21], columns=times\n)  # (n_classes, parts)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:10.177711Z","iopub.execute_input":"2022-05-24T08:53:10.178275Z","iopub.status.idle":"2022-05-24T08:53:10.248310Z","shell.execute_reply.started":"2022-05-24T08:53:10.178238Z","shell.execute_reply":"2022-05-24T08:53:10.247618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rel_path = test_df.loc[0, \"filename\"]\nvisualize(rel_path, vis_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:10.590984Z","iopub.execute_input":"2022-05-24T08:53:10.591271Z","iopub.status.idle":"2022-05-24T08:53:12.170641Z","shell.execute_reply.started":"2022-05-24T08:53:10.591238Z","shell.execute_reply":"2022-05-24T08:53:12.169996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rel_path = test_df.loc[0, \"filename\"]\nvisualize(rel_path, vis_df > thresholds.reshape(-1, 1))","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:12.172194Z","iopub.execute_input":"2022-05-24T08:53:12.172880Z","iopub.status.idle":"2022-05-24T08:53:13.712535Z","shell.execute_reply.started":"2022-05-24T08:53:12.172842Z","shell.execute_reply":"2022-05-24T08:53:13.711925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rel_path = test_df.loc[0, \"filename\"]\nvisualize(\n    rel_path,\n    (vis_df - thresholds.reshape(-1, 1)) / np.sqrt(thresholds.reshape(-1, 1)) + 0.5,\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:13.713677Z","iopub.execute_input":"2022-05-24T08:53:13.714066Z","iopub.status.idle":"2022-05-24T08:53:15.239719Z","shell.execute_reply.started":"2022-05-24T08:53:13.714031Z","shell.execute_reply":"2022-05-24T08:53:15.239047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(vis_df > thresholds.reshape(-1, 1)).sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:53:15.241658Z","iopub.execute_input":"2022-05-24T08:53:15.242114Z","iopub.status.idle":"2022-05-24T08:53:15.315237Z","shell.execute_reply.started":"2022-05-24T08:53:15.242079Z","shell.execute_reply":"2022-05-24T08:53:15.314251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LB Probing I","metadata":{}},{"cell_type":"code","source":"def invert_target(input_df, birds_to_invert):\n    assert type(birds_to_invert) == list, type(birds)\n    tmp_df = input_df.copy()\n    prefixes, dates, birds, end_secs = zip(*tmp_df.row_id.str.split(\"_\"))\n    tmp_df[\"prefix\"] = prefixes\n    tmp_df[\"date\"] = dates\n    tmp_df[\"bird\"] = birds\n    tmp_df[\"end_sec\"] = end_secs\n\n    idxs = tmp_df.query(\"bird in @birds_to_invert\").index\n    tmp_df.loc[idxs, \"target\"] = tmp_df.loc[idxs, \"target\"].apply(lambda x: not (x))\n\n    return tmp_df[[\"row_id\", \"target\"]]","metadata":{"execution":{"iopub.status.busy":"2022-05-24T08:51:55.367217Z","iopub.execute_input":"2022-05-24T08:51:55.367737Z","iopub.status.idle":"2022-05-24T08:51:55.446825Z","shell.execute_reply.started":"2022-05-24T08:51:55.367697Z","shell.execute_reply":"2022-05-24T08:51:55.445851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scored_birds = pd.read_json(\"/kaggle/input/birdclef-2022/scored_birds.json\")[0].tolist()\ntrain = pd.read_csv(\"/kaggle/input/birdclef-2022/train_metadata.csv\")\nscored = train.query(\"primary_label in @scored_birds\")\nscored_count = scored[\"primary_label\"].value_counts()\ntop5 = scored_count[:5].index.tolist()\nmid_top5 = scored_count[5:10].index.tolist()\nmid_low5 = scored_count[10:15].index.tolist()\nlow6 = scored_count[15:].index.tolist()\ntop5, mid_top5, mid_low5, low6","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#inverted_df = invert_target(submit_df, low6)\n#inverted_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#inverted_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LB Probe II","metadata":{}},{"cell_type":"code","source":"def random_invert_pos_target(input_df, p=1.0):\n    assert (p >= 0.0) and (p <= 1.0), p\n    tmp_df = input_df.copy()\n    pos_df = tmp_df.query(\"target == True\").reset_index()\n    n_rows = len(pos_df)\n    n_inverted = int(n_rows * p)\n    idxs = np.random.permutation(n_rows)[:n_inverted]\n    pos_df.loc[idxs, \"target\"] = pos_df.loc[idxs, \"target\"].apply(lambda x: not (x))\n    pos_df = pos_df.set_index(\"index\")\n    pos_idxs = pos_df.index\n    tmp_df.loc[pos_idxs, \"target\"] = pos_df.loc[pos_idxs, \"target\"]\n\n    return tmp_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_positive_ratio_of_birds(input_df, target_birds):\n    assert type(target_birds) == list, type(target_birds)\n    tmp_df = input_df.copy()\n    prefixes, dates, birds, end_secs = zip(*tmp_df.row_id.str.split(\"_\"))\n    tmp_df[\"prefix\"] = prefixes\n    tmp_df[\"date\"] = dates\n    tmp_df[\"bird\"] = birds\n    tmp_df[\"end_sec\"] = end_secs\n\n    target_df = tmp_df.query(\"bird in @target_birds\")\n    pos_count = target_df[\"target\"].sum()\n    if len(target_df) == 0:\n        print(\"warning: no target birds detected\")\n        return 0.0\n\n    pos_ratio = pos_count / len(target_df)\n    return pos_ratio","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#p = get_positive_ratio_of_birds(submit_df, low6)\n#print(f\"p: {p}\")\n#probe_df = random_invert_pos_target(submit_df, p=p)\n#probe_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#probe_df.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}