{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import os\nimport sys\nimport time\nimport math\nimport random\nimport librosa\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport logging\n\nfrom pathlib import Path\nfrom fastprogress import progress_bar\nfrom contextlib import contextmanager\nfrom typing import Optional\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.nn.init as init\nfrom torchvision import transforms, utils\nfrom torch.utils.data import Dataset, DataLoader\n\n\n# Ignore warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# seeding function for reproducibility\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_logger(out_file=None):\n    logger = logging.getLogger()\n    formatter = logging.Formatter(\"%(asctime)s - %(levelname)s - %(message)s\")\n    logger.handlers = []\n    logger.setLevel(logging.INFO)\n\n    handler = logging.StreamHandler()\n    handler.setFormatter(formatter)\n    handler.setLevel(logging.INFO)\n    logger.addHandler(handler)\n\n    if out_file is not None:\n        fh = logging.FileHandler(out_file)\n        fh.setFormatter(formatter)\n        fh.setLevel(logging.INFO)\n        logger.addHandler(fh)\n    logger.info(\"logger set up\")\n    return logger\n\n@contextmanager\ndef timer(name: str, logger: Optional[logging.Logger] = None):\n    t0 = time.time()\n    msg = f\"[{name}] start\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)\n    yield\n\n    msg = f\"[{name}] done in {time.time() - t0:.2f} s\"\n    if logger is None:\n        print(msg)\n    else:\n        logger.info(msg)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ROOT = \"/kaggle/input/birdsong-recognition/\"\nos.listdir(ROOT)\ndf = pd.read_csv(os.path.join(ROOT, 'train.csv'))[['ebird_code', 'filename', 'duration']]\ndf['path'] = ROOT+'train_audio/' + df['ebird_code'] + \"/\" + df['filename']\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEED = 42\nFRAC = 0.2     # Validation fraction\nSR = 32000     # sampling rate\nMAXLEN = 60    # seconds\nN_MELS = 128\nbatch_size = 8\n\nseed_everything(SEED)\ndevice = torch.device('cuda')\n\n#Random sample of 10 birds to test code.\nclasses = sorted(set(df['ebird_code'].unique().tolist()))\nprint(classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TARGET_SR = 32000\nTEST = Path(\"../input/birdsong-recognition/test_audio\").exists()\nif TEST:\n    DATA_DIR = Path(\"../input/birdsong-recognition/\")\nelse:\n    # dataset created by @shonenkov, thanks!\n    DATA_DIR = Path(\"../input/birdcall-check/\")\n\ntest = pd.read_csv(DATA_DIR / \"test.csv\")\ntest_audio = DATA_DIR / \"test_audio\"\ntest.head()\n\n\nsub = pd.read_csv(\"../input/birdsong-recognition/sample_submission.csv\")\nsub.to_csv(\"submission.csv\", index=False)  # this will be overwritten if everything goes well","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"logger = get_logger(\"main.log\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[df.ebird_code.apply(lambda x: x in classes)].reset_index(drop=True)\nkeys = set(df.ebird_code)\nvalues = np.arange(0, len(keys))\ncode_dict = dict(zip(sorted(keys), values))\ndf['label'] = df['ebird_code'].apply(lambda x: code_dict[x])\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"code_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = code_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"INV_BIRD_CODE = {v: k for k, v in labels.items()}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class BirdSoundDataset(Dataset):\n    \"\"\"Bird Sound dataset.\"\"\"\n\n    def __init__(self, df):\n        \"\"\"\n        Args:\n            df (pd.DataFrame): must have ['path', 'label'] columns\n        \"\"\"\n        self.df = df\n\n    def __len__(self):\n        return len(self.df)\n    \n    \n    def loadMP3(self, path, duration):\n        \"\"\"\n        \"\"\"\n        try:\n            audio, sample_rate = librosa.load(path, \n                                              sr=SR, \n                                              mono=True, \n                                              offset=0.0,\n                                              duration=duration, \n                                              res_type='kaiser_fast')\n            mels = librosa.feature.melspectrogram(y=audio, sr=SR,n_mels=N_MELS)\n            return mels\n            # mels will be of shape (N_MELS, ceil(duration*SR/512)) \n            # 512 here is default hop length\n\n        except Exception as e:\n            print(\"Error encountered while parsing file: \", path, e)\n            mels = np.zeros((N_MELS, MAXLEN*SR//512), dtype=np.float32)\n            return mels\n            \n\n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        \n        path = self.df['path'].iloc[idx]\n        duration = self.df['duration'].iloc[idx]\n        if duration < MAXLEN:\n            duration = None # read entire file\n        else:\n            duration = MAXLEN\n        if os.path.exists(\"./\"+path.split('/')[-1]+\".npy\"):\n            mels = np.load(\"./\"+path.split('/')[-1]+\".npy\")\n        else:\n            mels = self.loadMP3(path, duration)\n            np.save(\"./\"+path.split('/')[-1]+\".npy\", mels)\n        label  = self.df['label'].iloc[idx]\n        sample = {'label':label, 'features': mels, 'duration': duration}\n        return sample","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Custom collect function to wrap sound features\n\nimport random\n\ndef collate_fn_wrap(batch):\n    '''\n    wraps batch of variable length\n    '''\n    \n    ## get sequence lengths\n    lengths = [t['features'].shape[1] for t in batch]\n    maxlen = max(312, random.choice(lengths))#max(lengths)\n    \n    for i in range(len(batch)):\n        batch[i]['features'] = torch.from_numpy(batch[i]['features'])\n        k = math.ceil(maxlen/lengths[i])\n        batch[i]['features'] = batch[i]['features'].repeat(1, k)[:, :maxlen]\n        # assert batch[i]['features'].shape[1] == maxlen\n        \n    labels = torch.tensor([i['label'] for i in batch])\n    features = torch.stack([i['features'] for i in batch])\n    return {'features':features, 'labels':labels}","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model \n### Conv2D's -> LSTM -> Classifier"},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\nimport random\n\n\nclass BirdModel(nn.Module):\n    \n    def __init__(self, outdim=len(classes)):\n        super(BirdModel, self).__init__()\n        \n        self.conv1 = torch.nn.Conv2d(1, 32, (16, 8), \n                                    stride=(8, 4),\n                                    padding=0, \n                                    dilation=1,\n                                    groups=1, \n                                    bias=True, \n                                    padding_mode='zeros')\n    \n        self.conv2 = torch.nn.Conv2d(32, 256, (8, 16), \n                                    stride=(1, 8),\n                                    padding=0, \n                                    dilation=1,\n                                    groups=1, \n                                    bias=True, \n                                    padding_mode='zeros')\n        \n        self.lstm = torch.nn.LSTM(input_size=256,\n                                  hidden_size=256,\n                                  num_layers=2, \n                                  dropout=0.2,\n                                  bidirectional=True)\n        \n        self.pool = torch.nn.MaxPool2d(kernel_size=(2,2),\n                                       stride=None,\n                                       padding=0,\n                                       dilation=1,\n                                       return_indices=False,\n                                       ceil_mode=True)\n        \n        self.fc = nn.Linear(2*256+128, outdim)\n        self.MaxPool1d = nn.AdaptiveMaxPool1d(1)\n        self.drop = nn.Dropout(p=0.2)\n        \n    def forward(self, input):\n        \n        avg_features = torch.mean(input, dim=2)\n        x = self.pool(self.conv1(input.unsqueeze(1)))\n        x = self.pool(self.conv2(x))\n        x = x.squeeze(2).permute(2, 0, 1)\n        print(len(x))\n        x, _ = self.lstm(x)\n        print(len(x))\n        x = self.MaxPool1d(x.permute(1, 2, 0)).squeeze(2)\n        x = torch.cat((x, avg_features), dim=1)\n        return self.fc(self.drop(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_model(weights_path: str):\n    model = model = BirdModel()\n    model.load_state_dict(torch.load(weights_path))\n    device = torch.device(\"cuda\")\n    model.to(device)\n    model.eval()\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"weights_path = \"../input/mymodels3/base_model/model_cifar.pt\"\nmodel = get_model(weights_path)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Mixup Augmentation"},{"metadata":{"trusted":true},"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, df: pd.DataFrame, clip: np.ndarray):\n        self.df = df\n        self.clip = clip\n        \n    def __len__(self):\n        return len(self.df)\n    \n    \n    def __getitem__(self, idx: int):\n        sample = self.df.loc[idx, :]\n        site = sample.site\n        row_id = sample.row_id\n        if site == \"site_3\":\n            y = self.clip.astype(np.float32)\n            len_y = len(y)\n            start = 0\n            end = SR * 5\n            images = []\n            while len_y > start:\n                y_batch = y[start:end].astype(np.float32)\n                if len(y_batch) != (SR * 5):\n                    break\n                start = end\n                end = end + SR * 5\n                \n                mels = librosa.feature.melspectrogram(y, sr=SR,n_mels=N_MELS)\n                images.append(mels)\n            mels = np.asarray(mels)\n            return mels, row_id, site\n        else:\n            end_seconds = int(sample.seconds)\n            start_seconds = int(end_seconds - 5)\n            \n            start_index = SR * start_seconds\n            end_index = SR * end_seconds\n            \n            y = self.clip[start_index:end_index].astype(np.float32)\n\n            mels = librosa.feature.melspectrogram(y, sr=SR,n_mels=N_MELS).astype(np.float32)\n            return mels, row_id, site","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prediction_for_clip(test_df: pd.DataFrame, \n                        clip: np.ndarray, \n                        model: BirdModel, \n                        threshold=0.5):\n    \n    dataset = TestDataset(df=test_df, \n                          clip=clip)\n    \n    loader = torch.utils.data.DataLoader(dataset,\n                             batch_size=1,  \n                             shuffle=False,\n                             drop_last = False)\n        \n    device = torch.device(\"cuda\" if torch.cuda.is_available() else\n    \"cpu\")\n\n    model.eval()\n    prediction_dict = {}\n    for image, row_id, site in progress_bar(loader):\n        site = site[0]\n        row_id = row_id[0]\n        if site in {\"site_1\", \"site_2\", \"site_3\"}:\n            image = image.to(device)\n\n            with torch.no_grad():\n                prediction = model(image)\n                #prediction = F.softmax(prediction,dim=1)\n                #prediction = F.sigmoid(prediction)\n                #proba = prediction.detach().cpu().numpy().reshape(-1)\n            #events = proba >= threshold\n            #labels = np.argwhere(events).reshape(-1).tolist()\n            proba = torch.argmax(prediction, dim=1)\n            labels = proba.tolist()\n\n        else:\n            # to avoid prediction on large batch\n            image = image.squeeze(0)\n            batch_size = 16\n            whole_size = image.size(0)\n            if whole_size % batch_size == 0:\n                n_iter = whole_size // batch_size\n            else:\n                n_iter = whole_size // batch_size + 1\n\n            all_events = set()\n            for batch_i in range(n_iter):\n                batch = image[batch_i * batch_size:(batch_i + 1) * batch_size]\n                if batch.ndim == 3:\n                    batch = batch.unsqueeze(0)\n\n                batch = batch.to(device)\n                with torch.no_grad():\n                    prediction = model(batch)\n                    prediction = F.softmax(prediction,dim=1)\n                    proba = prediction.detach().cpu().numpy().reshape(-1)\n\n                events = proba >= threshold\n                for i in range(len(events)):\n                    event = events[i, :]\n                    labels = np.argwhere(event).reshape(-1).tolist()\n                    for label in labels:\n                        all_events.add(label)\n\n            labels = list(all_events)\n        if len(labels) == 0:\n            prediction_dict[row_id] = \"nocall\"\n        else:\n            labels_str_list = list(map(lambda x: INV_BIRD_CODE[x], labels))\n            label_string = \" \".join(labels_str_list)\n            prediction_dict[row_id] = label_string\n    return prediction_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prediction(test_df: pd.DataFrame,\n               test_audio: Path,\n               model: BirdModel,\n               threshold=0.5):\n    #model = get_model(model_config, weights_path)\n    unique_audio_id = test_df.audio_id.unique()\n\n    warnings.filterwarnings(\"ignore\")\n    prediction_dfs = []\n    for audio_id in unique_audio_id:\n        clip, _ = librosa.load(test_audio / (audio_id + \".mp3\"),\n                               sr=SR, \n                               mono=True, \n                               offset=0.0,\n                               res_type='kaiser_fast')\n        \n        test_df_for_audio_id = test_df.query(\n            f\"audio_id == '{audio_id}'\").reset_index(drop=True)\n        print(f'Test_for_audio_id: {test_df_for_audio_id}')\n        with timer(f\"Prediction on {audio_id}\", logger):\n            prediction_dict = prediction_for_clip(test_df_for_audio_id,\n                                                  clip=clip,\n                                                  model=model,\n                                                  threshold=threshold)\n        row_id = list(prediction_dict.keys())\n        birds = list(prediction_dict.values())\n        prediction_df = pd.DataFrame({\n            \"row_id\": row_id,\n            \"birds\": birds\n        })\n        prediction_dfs.append(prediction_df)\n    \n    prediction_df = pd.concat(prediction_dfs, axis=0, sort=False).reset_index(drop=True)\n    return prediction_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = prediction(test_df=test,\n                        test_audio=test_audio,\n                        model=model,\n                        threshold=0.8)\nsubmission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!rm *.npy","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}