{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on https://www.kaggle.com/code/julian3833/birdclef-21-2nd-place-model-submit-0-66","metadata":{}},{"cell_type":"code","source":"!pip install ../input/birds-inference-pip-wheels/torchaudio-0.8.1-cp37-cp37m-manylinux1_x86_64.whl ../input/birds-inference-pip-wheels/torch-1.8.1-cp37-cp37m-manylinux1_x86_64.whl\n!pip install ../input/timm-wheel/timm-0.5.4-py3-none-any.whl --no-index --no-deps\n!pip install ../input/birds-inference-pip-wheels/audiomentations-0.16.0-py3-none-any.whl --no-index --no-deps\n!pip install ../input/birds-inference-pip-wheels/torchlibrosa-0.0.9-py3-none-any.whl --no-index --no-deps","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-18T05:36:53.559718Z","iopub.execute_input":"2022-05-18T05:36:53.559988Z","iopub.status.idle":"2022-05-18T05:38:08.736582Z","shell.execute_reply.started":"2022-05-18T05:36:53.559905Z","shell.execute_reply":"2022-05-18T05:38:08.735716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ../input/audiofile-wheel/audeer-1.18.0-py3-none-any.whl --no-index --no-deps\n!pip install ../input/audiofile-wheel/audiofile-1.1.0-py3-none-any.whl --no-index --no-deps","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:08.740241Z","iopub.execute_input":"2022-05-18T05:38:08.740479Z","iopub.status.idle":"2022-05-18T05:38:12.370241Z","shell.execute_reply.started":"2022-05-18T05:38:08.740448Z","shell.execute_reply":"2022-05-18T05:38:12.369356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import timm\ntimm.__version__","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:12.372847Z","iopub.execute_input":"2022-05-18T05:38:12.373555Z","iopub.status.idle":"2022-05-18T05:38:13.927504Z","shell.execute_reply.started":"2022-05-18T05:38:12.373522Z","shell.execute_reply":"2022-05-18T05:38:13.926715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nimport os\nimport importlib\nimport multiprocessing as mp\n\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport glob\nimport torch\nfrom copy import copy\nimport math\n\nfrom torch.utils.data import DataLoader\n\nimport pandas as pd\nimport timm\nfrom torch import nn\nimport torch\nimport torchaudio as ta\nfrom torch.cuda.amp import autocast\nimport random\n\nfrom torch.nn import functional as F\nfrom torch.distributions import Beta\nfrom torch.nn.parameter import Parameter\nfrom torch.utils.data import Dataset\n\nimport numpy as np\nimport librosa\nimport ast\n\nimport torchvision\nimport torchvision.transforms as T\n\nimport os\nfrom types import SimpleNamespace\nimport numpy as np\n\nimport numpy as np\nimport pandas as pd\nimport importlib\nimport sys\nimport random\nfrom tqdm import tqdm\nimport gc\nimport argparse\nimport torch\nfrom torch import optim\nfrom torch.cuda.amp import GradScaler, autocast\nfrom collections import defaultdict\nimport cv2\nfrom copy import copy\nimport os\nfrom transformers import get_cosine_schedule_with_warmup\nfrom torch.utils.data import SequentialSampler, DataLoader\nimport audiofile","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:13.930307Z","iopub.execute_input":"2022-05-18T05:38:13.930842Z","iopub.status.idle":"2022-05-18T05:38:22.033682Z","shell.execute_reply.started":"2022-05-18T05:38:13.930796Z","shell.execute_reply":"2022-05-18T05:38:22.032919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=1234):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cudnn.benchmark = True","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.035Z","iopub.execute_input":"2022-05-18T05:38:22.03526Z","iopub.status.idle":"2022-05-18T05:38:22.042669Z","shell.execute_reply.started":"2022-05-18T05:38:22.035226Z","shell.execute_reply":"2022-05-18T05:38:22.041365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"cfg = SimpleNamespace()\n\n# paths\ncfg.data_folder = ''\ncfg.name = \"ari\"\ncfg.data_dir = \"../input/birdclef-2022/\"\ncfg.train_data_folder = cfg.data_dir + \"train_audio/\"\ncfg.val_data_folder = cfg.data_dir + \"train_audio/\"\ncfg.output_dir = \"first_model\"\n\n# dataset\ncfg.dataset = \"base_ds\"\ncfg.min_rating = 0\ncfg.val_df = None\ncfg.batch_size_val = 1\ncfg.train_aug = None\ncfg.val_aug = None\ncfg.test_augs = None\ncfg.wav_len_val = 5  # seconds\n\n# audio\ncfg.window_size = 2048\ncfg.hop_size = 512\ncfg.sample_rate = 32000\ncfg.fmin = 16\ncfg.fmax = 16386\ncfg.power = 2\ncfg.mel_bins = 256\ncfg.top_db = 80.0\ncfg.img_height = 256\ncfg.img_width = 512\n# img model\ncfg.backbone = \"resnet18\"\ncfg.pretrained = True\ncfg.pretrained_weights = None\ncfg.train = True\ncfg.val = False\ncfg.in_chans = 1\n\ncfg.alpha = 1\ncfg.eval_epochs = 1\ncfg.eval_train_epochs = 1\ncfg.warmup = 0\n\ncfg.mel_norm = False\n\ncfg.label_smoothing = 0\n\ncfg.remove_pretrained = []\n\n# training\ncfg.seed = 123\ncfg.save_val_data = True\n\n# ressources\ncfg.mixed_precision = True\ncfg.gpu = 0\ncfg.num_workers = 4 # 18\ncfg.drop_last = True \n\ncfg.mixup2 = 0\n\ncfg.label_smoothing = 0\n\ncfg.mixup_2x = False\n\n\ncfg.birds = np.array(['afrsil1', 'akekee', 'akepa1', 'akiapo', 'akikik', 'amewig',\n       'aniani', 'apapan', 'arcter', 'barpet', 'bcnher', 'belkin1',\n       'bkbplo', 'bknsti', 'bkwpet', 'blkfra', 'blknod', 'bongul',\n       'brant', 'brnboo', 'brnnod', 'brnowl', 'brtcur', 'bubsan',\n       'buffle', 'bulpet', 'burpar', 'buwtea', 'cacgoo1', 'calqua',\n       'cangoo', 'canvas', 'caster1', 'categr', 'chbsan', 'chemun',\n       'chukar', 'cintea', 'comgal1', 'commyn', 'compea', 'comsan',\n       'comwax', 'coopet', 'crehon', 'dunlin', 'elepai', 'ercfra',\n       'eurwig', 'fragul', 'gadwal', 'gamqua', 'glwgul', 'gnwtea',\n       'golphe', 'grbher3', 'grefri', 'gresca', 'gryfra', 'gwfgoo',\n       'hawama', 'hawcoo', 'hawcre', 'hawgoo', 'hawhaw', 'hawpet1',\n       'hoomer', 'houfin', 'houspa', 'hudgod', 'iiwi', 'incter1',\n       'jabwar', 'japqua', 'kalphe', 'kauama', 'laugul', 'layalb',\n       'lcspet', 'leasan', 'leater1', 'lessca', 'lesyel', 'lobdow',\n       'lotjae', 'madpet', 'magpet1', 'mallar3', 'masboo', 'mauala',\n       'maupar', 'merlin', 'mitpar', 'moudov', 'norcar', 'norhar2',\n       'normoc', 'norpin', 'norsho', 'nutman', 'oahama', 'omao', 'osprey',\n       'pagplo', 'palila', 'parjae', 'pecsan', 'peflov', 'perfal',\n       'pibgre', 'pomjae', 'puaioh', 'reccar', 'redava', 'redjun',\n       'redpha1', 'refboo', 'rempar', 'rettro', 'ribgul', 'rinduc',\n       'rinphe', 'rocpig', 'rorpar', 'rudtur', 'ruff', 'saffin', 'sander',\n       'semplo', 'sheowl', 'shtsan', 'skylar', 'snogoo', 'sooshe',\n       'sooter1', 'sopsku1', 'sora', 'spodov', 'sposan', 'towsol',\n       'wantat1', 'warwhe1', 'wesmea', 'wessan', 'wetshe', 'whfibi',\n       'whiter', 'whttro', 'wiltur', 'yebcar', 'yefcan', 'zebdov'])\n\n\ncfg.n_classes = len(cfg.birds)\n# dataset\ncfg.min_rating = 2.0\n\ncfg.wav_crop_len = 30  # seconds\n\ncfg.lr = 0.0001\ncfg.epochs = 5\ncfg.batch_size = 64\ncfg.batch_size_val = 64\ncfg.backbone = \"resnet34\"\n\n\ncfg.save_val_data = True\ncfg.mixed_precision = True\n\ncfg.mixup = True\ncfg.mix_beta = 1\n\n\ncfg.train_df1 = \"../input/birdclef-2022/train_metadata.csv\"\ncfg.train_df2 = \"../input/birdclef-2022-df-train-with-durations/df-with-durations.csv\"\n\n\ncfg.device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\ncfg.tr_collate_fn = None\ncfg.val_collate_fn = None\ncfg.val = False\n\ncfg.dev = False\n\ncfg.model = \"RN34\"\n\ncfg","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.057235Z","iopub.execute_input":"2022-05-18T05:38:22.057514Z","iopub.status.idle":"2022-05-18T05:38:22.143241Z","shell.execute_reply.started":"2022-05-18T05:38:22.057475Z","shell.execute_reply":"2022-05-18T05:38:22.142325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_AUDIO_ROOT = \"../input/birdclef-2022/test_soundscapes/\"\ncfg.val_data_folder = TEST_AUDIO_ROOT\ncfg.pretrained = False\n\n\nprint(cfg.model, cfg.dataset, cfg.backbone, cfg.pretrained_weights, cfg.mel_norm)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.144555Z","iopub.execute_input":"2022-05-18T05:38:22.145094Z","iopub.status.idle":"2022-05-18T05:38:22.155846Z","shell.execute_reply.started":"2022-05-18T05:38:22.145053Z","shell.execute_reply":"2022-05-18T05:38:22.154954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batch_to_device(batch, device):\n    batch_dict = {key: batch[key].to(device) for key in batch}\n    return batch_dict\n\n\n\nclass CustomDataset(Dataset):\n    def __init__(self, df, cfg, aug, mode=\"train\"):\n\n        self.cfg = cfg\n        self.mode = mode\n        self.df = df.copy()\n\n        self.bird2id = {bird: idx for idx, bird in enumerate(cfg.birds)}\n        if self.mode == \"train\":\n            self.data_folder = cfg.train_data_folder\n            self.df = self.df[self.df[\"rating\"] >= self.cfg.min_rating]\n        elif self.mode == \"val\":\n            self.data_folder = cfg.val_data_folder\n        elif self.mode == \"test\":\n            self.data_folder = cfg.test_data_folder\n\n        self.fns = self.df[\"filename\"].unique()\n\n        self.df = self.setup_df()\n\n        self.aug_audio = cfg.train_aug\n\n    def setup_df(self):\n        df = self.df.copy()\n\n        if self.mode == \"train\":\n\n            df[\"weight\"] = np.clip(df[\"rating\"] / df[\"rating\"].max(), 0.1, 1.0)\n            df['target'] = df['primary_label'].apply(self.bird2id.get)\n            labels = np.eye(self.cfg.n_classes)[df[\"target\"].astype(int).values]\n            label2 = df[\"secondary_labels\"].apply(lambda x: self.secondary2target(x)).values\n            for i, t in enumerate(label2):\n                labels[i, t] = 1\n        else:\n            targets = df[\"birds\"].apply(lambda x: self.birds2target(x)).values\n            labels = np.zeros((df.shape[0], self.cfg.n_classes))\n            # import pdb; pdb.set_trace()\n            for i, t in enumerate(targets):\n                labels[i, t] = 1\n\n        df[[f\"t{i}\" for i in range(self.cfg.n_classes)]] = labels\n\n        if self.mode != \"train\":\n            df = df.groupby(\"filename\")\n\n        return df\n\n    def __getitem__(self, idx):\n\n        if self.mode == \"train\":\n            row = self.df.iloc[idx]\n            fn = row[\"filename\"]\n            label = row[[f\"t{i}\" for i in range(self.cfg.n_classes)]].values\n            weight = row[\"weight\"]\n            #fold = row[\"fold\"]\n            fold = -1\n\n            #wav_len = row[\"length\"]\n            parts = 1\n        else:\n            fn = self.fns[idx]\n            row = self.df.get_group(fn)\n            label = row[[f\"t{i}\" for i in range(self.cfg.n_classes)]].values\n            wav_len = None\n            # Este es mi \"entrada\" a que un audio dure mucho\n            parts = label.shape[0]\n            fold = -1\n            weight = 1\n\n        if self.mode == \"train\":\n            #wav_len_sec = wav_len / self.cfg.sample_rate\n            wav_len_sec = row['duration']\n            duration = self.cfg.wav_crop_len\n            max_offset = wav_len_sec - duration\n            max_offset = max(max_offset, 1)\n            offset = np.random.randint(max_offset)\n        else:\n            offset = 0.0\n            duration = None\n\n        wav = self.load_one(fn, offset, duration)\n\n        if wav.shape[0] < (self.cfg.wav_crop_len * self.cfg.sample_rate):\n            pad = self.cfg.wav_crop_len * self.cfg.sample_rate - wav.shape[0]\n            wav = np.pad(wav, (0, pad))\n\n        if self.mode == \"train\":\n            if self.aug_audio:\n                wav = self.aug_audio(samples=wav, sample_rate=self.cfg.sample_rate)\n        else:\n            if self.cfg.val_aug:\n                wav = self.cfg.val_aug(samples=wav, sample_rate=self.cfg.sample_rate)\n\n        wav_tensor = torch.tensor(wav)  # (n_samples)\n        if parts > 1:\n            n_samples = wav_tensor.shape[0]\n            wav_tensor = wav_tensor[: n_samples // parts * parts].reshape(\n                parts, n_samples // parts\n            )\n\n        feature_dict = {\n            \"input\": wav_tensor,\n            \"target\": torch.tensor(label.astype(np.float32)),\n            \"weight\": torch.tensor(weight),\n            \"fold\": torch.tensor(fold),\n        }\n        return feature_dict\n\n    def __len__(self):\n        if cfg.dev:\n            return 256\n        return len(self.fns)\n\n    def load_one(self, id_, offset, duration):\n        fp = self.data_folder + id_\n        try:\n            wav, sr = librosa.load(fp, sr=None, offset=offset, duration=duration)\n        except:\n            print(\"FAIL READING rec\", fp)\n\n        return wav\n\n    def birds2target(self, birds):\n        #birds = birds.split()\n        target = [self.bird2id.get(item) for item in birds if not item == \"nocall\"]\n        return target\n\n    def secondary2target(self, secondary_label):\n        birds = ast.literal_eval(secondary_label)\n        target = [self.bird2id.get(item) for item in birds if not item == \"nocall\"]\n        return target\n","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.157534Z","iopub.execute_input":"2022-05-18T05:38:22.158159Z","iopub.status.idle":"2022-05-18T05:38:22.187803Z","shell.execute_reply.started":"2022-05-18T05:38:22.158115Z","shell.execute_reply":"2022-05-18T05:38:22.186824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gem(x, p=3, eps=1e-6):\n    return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1.0 / p)\n\n\nclass GeM(nn.Module):\n    # Generalized mean: https://arxiv.org/abs/1711.02512\n    def __init__(self, p=3, eps=1e-6):\n        super(GeM, self).__init__()\n        self.p = Parameter(torch.ones(1) * p)\n        self.eps = eps\n\n    def forward(self, x):\n        ret = gem(x, p=self.p, eps=self.eps)\n        return ret\n\n    def __repr__(self):\n        return (self.__class__.__name__+ \"(p=\"+ \"{:.4f}\".format(self.p.data.tolist()[0])+ \", eps=\"+ str(self.eps)+ \")\")\n\n\nclass Mixup(nn.Module):\n    def __init__(self, mix_beta):\n\n        super(Mixup, self).__init__()\n        self.beta_distribution = Beta(mix_beta, mix_beta)\n\n    def forward(self, X, Y, weight=None):\n\n        bs = X.shape[0]\n        n_dims = len(X.shape)\n        perm = torch.randperm(bs)\n        coeffs = self.beta_distribution.rsample(torch.Size((bs,))).to(X.device)\n\n        if n_dims == 2:\n            X = coeffs.view(-1, 1) * X + (1 - coeffs.view(-1, 1)) * X[perm]\n        elif n_dims == 3:\n            X = coeffs.view(-1, 1, 1) * X + (1 - coeffs.view(-1, 1, 1)) * X[perm]\n        else:\n            X = coeffs.view(-1, 1, 1, 1) * X + (1 - coeffs.view(-1, 1, 1, 1)) * X[perm]\n\n        Y = coeffs.view(-1, 1) * Y + (1 - coeffs.view(-1, 1)) * Y[perm]\n\n        if weight is None:\n            return X, Y\n        else:\n            weight = coeffs.view(-1) * weight + (1 - coeffs.view(-1)) * weight[perm]\n            return X, Y, weight\n\n        \n        \nclass Net(nn.Module):\n    def __init__(self, cfg):\n        super(Net, self).__init__()\n\n        self.cfg = cfg\n\n        self.n_classes = cfg.n_classes\n\n        self.mel_spec = ta.transforms.MelSpectrogram(\n            sample_rate=cfg.sample_rate,\n            n_fft=cfg.window_size,\n            win_length=cfg.window_size,\n            hop_length=cfg.hop_size,\n            f_min=cfg.fmin,\n            f_max=cfg.fmax,\n            pad=0,\n            n_mels=cfg.mel_bins,\n            power=cfg.power,\n            normalized=False,\n        )\n\n        self.amplitude_to_db = ta.transforms.AmplitudeToDB(top_db=cfg.top_db)\n        self.wav2img = torch.nn.Sequential(self.mel_spec, self.amplitude_to_db)\n        \n        self.resizeimg = T.Resize((cfg.img_height, cfg.img_width))\n\n        self.backbone = timm.create_model(\n            cfg.backbone,\n            pretrained=cfg.pretrained,\n            num_classes=0,\n            global_pool=\"\",\n            in_chans=cfg.in_chans,\n        )\n\n        if \"efficientnet\" in cfg.backbone:\n            backbone_out = self.backbone.num_features\n        else:\n            backbone_out = self.backbone.feature_info[-1][\"num_chs\"]\n\n        self.global_pool = GeM()\n\n        self.head = nn.Linear(backbone_out, self.n_classes)\n\n        if cfg.pretrained_weights is not None:\n            sd = torch.load(cfg.pretrained_weights, map_location=\"cpu\")[\"model\"]\n            sd = {k.replace(\"module.\", \"\"): v for k, v in sd.items()}\n            self.load_state_dict(sd, strict=True)\n            print(\"weights loaded from\", cfg.pretrained_weights)\n        self.loss_fn = nn.BCEWithLogitsLoss(reduction=\"none\")\n\n        self.mixup = Mixup(mix_beta=cfg.mix_beta)\n\n        self.factor = int(cfg.wav_crop_len / 5.0)\n\n    def forward(self, batch):\n\n        if not self.training:\n            x = batch[\"input\"]\n            bs, parts, time = x.shape\n            x = x.reshape(parts, time)\n            y = batch[\"target\"]\n            y = y[0]\n        else:\n            x = batch[\"input\"]\n            y = batch[\"target\"]\n            bs, time = x.shape\n            x = x.reshape(bs * self.factor, time // self.factor)\n\n        with autocast(enabled=False):\n            x = self.wav2img(x)  # (bs, mel, time)\n            x = self.resizeimg(x)\n            if self.cfg.mel_norm:\n                x = (x + 80) / 80\n\n        x = x.permute(0, 2, 1)\n        x = x[:, None, :, :]\n\n        weight = batch[\"weight\"]\n\n        if self.training:\n            b, c, t, f = x.shape\n            x = x.permute(0, 2, 1, 3)\n            x = x.reshape(b // self.factor, self.factor * t, c, f)\n\n            if self.cfg.mixup:\n                x, y, weight = self.mixup(x, y, weight)\n            if self.cfg.mixup2:\n                x, y, weight = self.mixup(x, y, weight)\n\n            x = x.reshape(b, t, c, f)\n            x = x.permute(0, 2, 1, 3)\n\n        x = self.backbone(x)\n\n        if self.training:\n            b, c, t, f = x.shape\n            x = x.permute(0, 2, 1, 3)\n            x = x.reshape(b // self.factor, self.factor * t, c, f)\n            x = x.permute(0, 2, 1, 3)\n        x = self.global_pool(x)\n        x = x[:, :, 0, 0]\n        logits = self.head(x)\n\n        loss = self.loss_fn(logits, y)\n        loss = (loss.mean(dim=1) * weight) / weight.sum()\n        loss = loss.sum()\n\n        return {\"loss\": loss, \"logits\": logits.sigmoid(), \"logits_raw\": logits, \"target\": y}\n","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.19165Z","iopub.execute_input":"2022-05-18T05:38:22.191916Z","iopub.status.idle":"2022-05-18T05:38:22.227907Z","shell.execute_reply.started":"2022-05-18T05:38:22.191871Z","shell.execute_reply":"2022-05-18T05:38:22.227086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_state_dict(sd_fp):\n    sd = torch.load(sd_fp, map_location=\"cpu\")['model']\n    sd = {k.replace(\"module.\", \"\"):v for k,v in sd.items()}\n    return sd\n\nfrom scipy.stats.mstats import gmean","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.232941Z","iopub.execute_input":"2022-05-18T05:38:22.233226Z","iopub.status.idle":"2022-05-18T05:38:22.239638Z","shell.execute_reply.started":"2022-05-18T05:38:22.233196Z","shell.execute_reply":"2022-05-18T05:38:22.238777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nTEST_AUDIO_PATH = '../input/birdclef-2022/test_soundscapes/'\n# TEST_AUDIO_PATH = '../input/birdclef-2021/train_soundscapes/'\n\nwith open('../input/birdclef-2022/scored_birds.json') as fp:\n    SCORED_BIRDS = json.load(fp)","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.2408Z","iopub.execute_input":"2022-05-18T05:38:22.244601Z","iopub.status.idle":"2022-05-18T05:38:22.251323Z","shell.execute_reply.started":"2022-05-18T05:38:22.244556Z","shell.execute_reply":"2022-05-18T05:38:22.250454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten(l):\n    return [item for sublist in l for item in sublist]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.254998Z","iopub.execute_input":"2022-05-18T05:38:22.255275Z","iopub.status.idle":"2022-05-18T05:38:22.259964Z","shell.execute_reply.started":"2022-05-18T05:38:22.255246Z","shell.execute_reply":"2022-05-18T05:38:22.259102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_df_test_from_path():\n    files = sorted(os.listdir(TEST_AUDIO_PATH))\n    data = []\n    for f in tqdm(files):\n#         wv, sr = librosa.load(TEST_AUDIO_PATH + f)\n        n_chunks = math.ceil(audiofile.duration(TEST_AUDIO_PATH + f)/ 5)\n#         print(\"Done with reading file\")\n        filename = f\n        row_prefix = f[:-4]\n        bird = SCORED_BIRDS[0]\n        for chunk in range(1, n_chunks + 1):\n            #for bird in SCORED_BIRDS:\n            #row_id = f\"{f[:-4]}_{bird}_{chunk*5}\"\n            \n            ending_second = chunk*5\n            data.append((filename, row_prefix, ending_second, [bird]))\n#         print(\"Done with reading chunk\")\n    return  pd.DataFrame(data, columns=['filename', 'row_prefix', 'ending_second', 'birds'])\n        \ntest_df = create_df_test_from_path()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.26206Z","iopub.execute_input":"2022-05-18T05:38:22.262341Z","iopub.status.idle":"2022-05-18T05:38:22.301699Z","shell.execute_reply.started":"2022-05-18T05:38:22.262304Z","shell.execute_reply":"2022-05-18T05:38:22.300941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.shape)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.303191Z","iopub.execute_input":"2022-05-18T05:38:22.3037Z","iopub.status.idle":"2022-05-18T05:38:22.322252Z","shell.execute_reply.started":"2022-05-18T05:38:22.303661Z","shell.execute_reply":"2022-05-18T05:38:22.321432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.tail()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.323955Z","iopub.execute_input":"2022-05-18T05:38:22.324488Z","iopub.status.idle":"2022-05-18T05:38:22.337375Z","shell.execute_reply.started":"2022-05-18T05:38:22.324433Z","shell.execute_reply":"2022-05-18T05:38:22.33654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_CORES = mp.cpu_count()\ncfg.batch_size = 1\n\naug = None\ntest_ds = CustomDataset(test_df, cfg, aug, mode=\"val\")\ntest_dl = DataLoader(test_ds, shuffle=False, batch_size = cfg.batch_size, num_workers = N_CORES)\n\ntest_ds[0]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.339388Z","iopub.execute_input":"2022-05-18T05:38:22.339935Z","iopub.status.idle":"2022-05-18T05:38:22.477927Z","shell.execute_reply.started":"2022-05-18T05:38:22.339861Z","shell.execute_reply":"2022-05-18T05:38:22.477147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = \"cuda\" if torch.cuda.is_available() else 'cpu'","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.479169Z","iopub.execute_input":"2022-05-18T05:38:22.480487Z","iopub.status.idle":"2022-05-18T05:38:22.485142Z","shell.execute_reply.started":"2022-05-18T05:38:22.48044Z","shell.execute_reply":"2022-05-18T05:38:22.484068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nstate_dicts = []\nbackbones = []\nfor filepath in glob.iglob('../input/seresnext50-256-512-best-models/*.pth'):\n    state_dicts.append(filepath)\n    backbones.append(\"seresnext50_32x4d\")\nprint(state_dicts)\n\nnets = []\n\nfor i,state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    net.load_state_dict(sd, strict=True)\n    nets += [net]\n    \nwith torch.no_grad():\n    preds_1 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)['logits']\n                preds_ += [out.cpu().numpy()]\n        preds_1 += [preds_]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:22.487962Z","iopub.execute_input":"2022-05-18T05:38:22.488163Z","iopub.status.idle":"2022-05-18T05:38:42.584603Z","shell.execute_reply.started":"2022-05-18T05:38:22.488135Z","shell.execute_reply":"2022-05-18T05:38:42.583016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_dicts = []\nbackbones = []\nfor filepath in glob.iglob('../input/eca-nfnet-l0-256-512-best-models/*.pth'):\n    state_dicts.append(filepath)\n    backbones.append(\"eca_nfnet_l0\")\nprint(state_dicts)\n\nnets = []\n\nfor i,state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    net.load_state_dict(sd, strict=True)\n    nets += [net]\n    \nwith torch.no_grad():\n    preds_2 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)['logits']\n                preds_ += [out.cpu().numpy()]\n        preds_2 += [preds_]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:42.586543Z","iopub.execute_input":"2022-05-18T05:38:42.586856Z","iopub.status.idle":"2022-05-18T05:38:57.525171Z","shell.execute_reply.started":"2022-05-18T05:38:42.586812Z","shell.execute_reply":"2022-05-18T05:38:57.52348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_dicts = []\nbackbones = []\nfor filepath in glob.iglob('../input/resnest50d-256-512-best-models/*.pth'):\n    state_dicts.append(filepath)\n    backbones.append(\"resnest50d_4s2x40d\")\nprint(state_dicts)\n\nnets = []\n\nfor i,state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    net.load_state_dict(sd, strict=True)\n    nets += [net]\n    \nwith torch.no_grad():\n    preds_3 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)['logits']\n                preds_ += [out.cpu().numpy()]\n        preds_3 += [preds_]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:38:57.526683Z","iopub.execute_input":"2022-05-18T05:38:57.529091Z","iopub.status.idle":"2022-05-18T05:39:16.607809Z","shell.execute_reply.started":"2022-05-18T05:38:57.529043Z","shell.execute_reply":"2022-05-18T05:39:16.606156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_dicts = []\nbackbones = []\nfor filepath in glob.iglob('../input/eca-nfnet-l1-256-512-best-models/*.pth'):\n    state_dicts.append(filepath)\n    backbones.append(\"eca_nfnet_l1\")\nprint(state_dicts)\n\nnets = []\n\nfor i,state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    net.load_state_dict(sd, strict=True)\n    nets += [net]\n    \nwith torch.no_grad():\n    preds_4 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)['logits']\n                preds_ += [out.cpu().numpy()]\n        preds_4 += [preds_]","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:16.609835Z","iopub.execute_input":"2022-05-18T05:39:16.610171Z","iopub.status.idle":"2022-05-18T05:39:39.587519Z","shell.execute_reply.started":"2022-05-18T05:39:16.610128Z","shell.execute_reply":"2022-05-18T05:39:39.585756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_dicts = []\nbackbones = []\nfor filepath in glob.iglob('../input/tf-efficientnet-v2m-256-512-best-models/*.pth'):\n    state_dicts.append(filepath)\n    backbones.append(\"tf_efficientnetv2_m_in21k\")\nprint(state_dicts)\n\nnets = []\n\nfor i,state_dict in enumerate(state_dicts):\n    cfg.backbone = backbones[i]\n    net = Net(cfg).eval().cuda()\n    sd = get_state_dict(state_dict)\n    print(\"loading dict\")\n    net.load_state_dict(sd, strict=True)\n    nets += [net]\n    \nwith torch.no_grad():\n    preds_5 = []\n    for batch in tqdm(test_dl):\n        batch = batch_to_device(batch, DEVICE)\n        with torch.cuda.amp.autocast():\n            preds_ = []\n            for net in nets:\n                out = net(batch)['logits']\n                preds_ += [out.cpu().numpy()]\n        preds_5 += [preds_]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_1 = np.array(preds_1).transpose(1, 0, 2, 3)\npreds_2 = np.array(preds_2).transpose(1, 0, 2, 3)\npreds_3 = np.array(preds_3).transpose(1, 0, 2, 3)\npreds_4 = np.array(preds_4).transpose(1, 0, 2, 3)\npreds_5 = np.array(preds_5).transpose(1, 0, 2, 3)","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:42:41.684744Z","iopub.execute_input":"2022-05-18T05:42:41.685457Z","iopub.status.idle":"2022-05-18T05:42:41.691023Z","shell.execute_reply.started":"2022-05-18T05:42:41.685417Z","shell.execute_reply":"2022-05-18T05:42:41.690254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_5.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:42:44.129279Z","iopub.execute_input":"2022-05-18T05:42:44.129813Z","iopub.status.idle":"2022-05-18T05:42:44.135615Z","shell.execute_reply.started":"2022-05-18T05:42:44.129772Z","shell.execute_reply":"2022-05-18T05:42:44.134915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_preds = pd.DataFrame(np.vstack(preds_1[0]), columns=test_ds.bird2id.keys())[SCORED_BIRDS]\ntest__df = test_df.copy()\ntest__df = test__df.join(df_preds).drop(['birds'], axis=1).reset_index()\ntest__df = pd.melt(test__df, id_vars=['filename', 'row_prefix', 'ending_second'], value_vars=SCORED_BIRDS, var_name=\"bird\", value_name=\"proba\")\ntest__df['row_id'] = test__df['row_prefix'] + \"_\" + test__df['bird'] + \"_\" + test__df['ending_second'].astype(str)\ntest__df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:39.6114Z","iopub.execute_input":"2022-05-18T05:39:39.611737Z","iopub.status.idle":"2022-05-18T05:39:39.646896Z","shell.execute_reply.started":"2022-05-18T05:39:39.611702Z","shell.execute_reply":"2022-05-18T05:39:39.646146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, model in enumerate([preds_1, preds_2, preds_3, preds_4, preds_5]):\n    for j, model_preds in enumerate(model):\n        df_preds = pd.DataFrame(np.vstack(model_preds), columns=test_ds.bird2id.keys())[SCORED_BIRDS]\n        model_df = test_df.copy()\n        model_df = model_df.join(df_preds).drop(['birds'], axis=1).reset_index()\n        model_df = pd.melt(model_df, id_vars=['filename', 'row_prefix', 'ending_second'], value_vars=SCORED_BIRDS, var_name=\"bird\", value_name=\"proba\")\n        model_df['row_id'] = model_df['row_prefix'] + \"_\" + model_df['bird'] + \"_\" + model_df['ending_second'].astype(str)\n        model_df['target'] = model_df['proba'] > 0.02\n        test__df[f'model_{i}_proba_{j}'] = model_df['proba']","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:39.648173Z","iopub.execute_input":"2022-05-18T05:39:39.649047Z","iopub.status.idle":"2022-05-18T05:39:39.79714Z","shell.execute_reply.started":"2022-05-18T05:39:39.649008Z","shell.execute_reply":"2022-05-18T05:39:39.796371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test__df","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:41:44.156807Z","iopub.execute_input":"2022-05-18T05:41:44.157582Z","iopub.status.idle":"2022-05-18T05:41:44.196619Z","shell.execute_reply.started":"2022-05-18T05:41:44.157544Z","shell.execute_reply":"2022-05-18T05:41:44.195787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test__df['final_proba']=(test__df['model_0_proba_0'] + test__df['model_0_proba_1'] + test__df['model_0_proba_2'] + test__df['model_0_proba_3'] + test__df['model_0_proba_4'] + test__df['model_1_proba_0'] + test__df['model_1_proba_1'] + test__df['model_1_proba_2'] + test__df['model_1_proba_3'] + test__df['model_1_proba_4'] + test__df['model_2_proba_0'] + test__df['model_2_proba_1'] + test__df['model_2_proba_2'] + test__df['model_2_proba_3'] + test__df['model_2_proba_4']+test__df['model_3_proba_0'] + test__df['model_3_proba_1'] + test__df['model_3_proba_2'] + test__df['model_3_proba_3'] + test__df['model_3_proba_4'] + test__df['model_4_proba_0'] + test__df['model_4_proba_1'] + test__df['model_4_proba_2'] + test__df['model_4_proba_3'] + test__df['model_4_proba_4']) / 25\ntest__df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:39.798679Z","iopub.execute_input":"2022-05-18T05:39:39.798953Z","iopub.status.idle":"2022-05-18T05:39:40.262075Z","shell.execute_reply.started":"2022-05-18T05:39:39.798917Z","shell.execute_reply":"2022-05-18T05:39:40.260933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def average_ensemble(row):\n    if (row.final_proba) > 0.02:\n        return True\n    \n    else:\n        return False","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:40.265311Z","iopub.status.idle":"2022-05-18T05:39:40.265921Z","shell.execute_reply.started":"2022-05-18T05:39:40.26565Z","shell.execute_reply":"2022-05-18T05:39:40.26568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test__df['target'] = test__df.apply(lambda row: average_ensemble(row), axis=1)\ntest__df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:40.267148Z","iopub.status.idle":"2022-05-18T05:39:40.26777Z","shell.execute_reply.started":"2022-05-18T05:39:40.267492Z","shell.execute_reply":"2022-05-18T05:39:40.267533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test__df[['row_id', 'target']]\nsub.to_csv(\"submission.csv\", index=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-18T05:39:40.268988Z","iopub.status.idle":"2022-05-18T05:39:40.269578Z","shell.execute_reply.started":"2022-05-18T05:39:40.269319Z","shell.execute_reply":"2022-05-18T05:39:40.269347Z"},"trusted":true},"execution_count":null,"outputs":[]}]}