{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8359997,"sourceType":"datasetVersion","datasetId":4968282},{"sourceId":8378848,"sourceType":"datasetVersion","datasetId":4974563},{"sourceId":8231800,"sourceType":"datasetVersion","datasetId":4881848},{"sourceId":176542026,"sourceType":"kernelVersion"},{"sourceId":176701042,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **This is the inference notebook of my training notebook. In my training I make a custom small model that gives good results in about 5 minutes of training!** My training notebook: https://www.kaggle.com/code/max1mum/pytorch-full-model-train","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nfrom glob import glob\nimport re\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport librosa\nfrom scipy import signal as sci_signal\n\nimport torch\nimport torch.nn as nn\n\nimport kaggle_metric_utilities\n\n# import train_python\n\nimport os\n\ndataset_path = '/kaggle/input/noisereduce3'\nif os.path.exists(dataset_path):\n    print(\"Dataset found!\")\nelse:\n    print(\"Dataset not found, check the path.\")\n\nimport albumentations as albu","metadata":{"execution":{"iopub.status.busy":"2024-05-11T02:04:59.294765Z","iopub.execute_input":"2024-05-11T02:04:59.295470Z","iopub.status.idle":"2024-05-11T02:04:59.311145Z","shell.execute_reply.started":"2024-05-11T02:04:59.295407Z","shell.execute_reply":"2024-05-11T02:04:59.309581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/noisereduce3/noisereduce-3.0.2-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-05-11T02:06:25.865844Z","iopub.execute_input":"2024-05-11T02:06:25.866928Z","iopub.status.idle":"2024-05-11T02:06:35.651374Z","shell.execute_reply.started":"2024-05-11T02:06:25.866869Z","shell.execute_reply":"2024-05-11T02:06:35.649371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import noisereduce as nr","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    SEED = 14\n    DEVICE = 'cpu'\n    MIXED_PRECISION = False\n    OUTPUT_DIR = 'out_dir'\n    \n    DATA_ROOT = '/kaggle/input/birdclef-2024'\n    PREPROCESSED_DATA_ROOT = 'save_dir'\n    LOAD_DATA = True\n    FS = 32000\n    N_FFT = 1095\n    WIN_SIZE = 412\n    WIN_LAP = 100\n    MIN_FREQ = 40\n    MAX_FREQ = 15000\n    \n    BATCH_SIZE = 32\n    N_WORKERS = 12\n    \n    USE_XYMASKING = True\n    \n    FOLDS = 10\n    EPOCHS = 5\n    LR = 8e-5\n    WEIGHT_DECAY = 1e-5\n    \n    VISUALIZE = True\n\ndef set_seed(seed=14):\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    \nset_seed()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T02:04:21.654064Z","iopub.execute_input":"2024-05-11T02:04:21.654607Z","iopub.status.idle":"2024-05-11T02:04:21.667664Z","shell.execute_reply.started":"2024-05-11T02:04:21.654567Z","shell.execute_reply":"2024-05-11T02:04:21.666281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def SpectralNoiseReduction(audio_data, sr, min_length_sec=5):\n    if len(audio_data) < sr * min_length_sec:\n        return audio_data\n\n    hop_length = int(sr * 1.5)\n    win_length = int(sr * 3)\n    rms = librosa.feature.rms(y=audio_data, frame_length=win_length, hop_length=hop_length)\n\n    noise_sec = 1 \n    min_rms_idx = np.argmin(rms) \n    start_idx = min_rms_idx * hop_length\n    end_idx = start_idx + sr * noise_sec\n\n    start_idx = max(0, start_idx)\n    end_idx = min(len(audio_data), end_idx)\n\n    noise_data = audio_data[start_idx:end_idx]\n\n    return nr.reduce_noise(y=audio_data, sr=sr, y_noise=noise_data)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.020533Z","iopub.execute_input":"2024-05-11T01:59:08.021029Z","iopub.status.idle":"2024-05-11T01:59:08.032603Z","shell.execute_reply.started":"2024-05-11T01:59:08.020990Z","shell.execute_reply":"2024-05-11T01:59:08.031111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvBlock1(nn.Module):\n    def __init__(\n        self, in_channels, out_channels\n    ) -> None:\n        super().__init__()\n        self.same_channels = in_channels==out_channels\n\n        self.conv1 = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, 3, 1, 1),\n            nn.MaxPool2d(2),\n            nn.BatchNorm2d(out_channels),\n            nn.LeakyReLU(0.02),\n        )\n\n    def forward(self, x):\n        x = self.conv1(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.034446Z","iopub.execute_input":"2024-05-11T01:59:08.034911Z","iopub.status.idle":"2024-05-11T01:59:08.052388Z","shell.execute_reply.started":"2024-05-11T01:59:08.034864Z","shell.execute_reply":"2024-05-11T01:59:08.051025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvBlock2(nn.Module):\n    def __init__(\n        self, in_channels, out_channels\n    ) -> None:\n        super().__init__()\n        self.same_channels = in_channels==out_channels\n\n        self.conv1 = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, 3, 1, 1),\n            nn.AvgPool2d(2),\n            nn.BatchNorm2d(out_channels),\n            nn.LeakyReLU(0.02),\n        )\n\n    def forward(self, x):\n        x = self.conv1(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.055667Z","iopub.execute_input":"2024-05-11T01:59:08.056286Z","iopub.status.idle":"2024-05-11T01:59:08.064989Z","shell.execute_reply.started":"2024-05-11T01:59:08.056242Z","shell.execute_reply":"2024-05-11T01:59:08.063481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv = nn.Sequential(\n            ConvBlock1(1, 2),\n            ConvBlock1(2, 4),\n            ConvBlock1(4, 8),\n            ConvBlock1(8, 8),\n            ConvBlock2(8, 16),\n            ConvBlock2(16, 32),\n            ConvBlock2(32, 64),\n            ConvBlock2(64, 182),\n        )\n\n    def forward(self, x):\n        return nn.Softmax()(self.conv(x))","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.066890Z","iopub.execute_input":"2024-05-11T01:59:08.067433Z","iopub.status.idle":"2024-05-11T01:59:08.080349Z","shell.execute_reply.started":"2024-05-11T01:59:08.067386Z","shell.execute_reply":"2024-05-11T01:59:08.078919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model()\nmodel.load_state_dict(torch.load(\"/kaggle/input/5-min-checkpoint/model_save_new.ckpt\", map_location=torch.device('cpu')))","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.081914Z","iopub.execute_input":"2024-05-11T01:59:08.082414Z","iopub.status.idle":"2024-05-11T01:59:08.115699Z","shell.execute_reply.started":"2024-05-11T01:59:08.082360Z","shell.execute_reply":"2024-05-11T01:59:08.114201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transforms(_type):\n    if _type == 'train':\n        return albu.Compose([\n            albu.HorizontalFlip(p=0.5),\n            albu.XYMasking(\n                p=0.3,\n                num_masks_x=(1, 3),\n                num_masks_y=(1, 3),\n                mask_x_length=(1, 10),\n                mask_y_length=(1, 20),\n            ) if config.USE_XYMASKING else albu.NoOp()\n        ])\n    elif _type == 'valid' or _type == 'test':\n        return albu.Compose([])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.117615Z","iopub.execute_input":"2024-05-11T01:59:08.118723Z","iopub.status.idle":"2024-05-11T01:59:08.128089Z","shell.execute_reply.started":"2024-05-11T01:59:08.118671Z","shell.execute_reply":"2024-05-11T01:59:08.126721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scipy_spectro(audio_data):\n    mean_signal = np.nanmean(audio_data)\n    audio_data = np.nan_to_num(audio_data, nan=mean_signal) if np.isnan(audio_data).mean() < 1 else np.zeros_like(audio_data)\n    \n    frequencies, times, spec_data = sci_signal.spectrogram(\n        audio_data, \n        fs=config.FS, \n        nfft=config.N_FFT, \n        nperseg=config.WIN_SIZE, \n        noverlap=config.WIN_LAP, \n        window='hann'\n    )\n    \n    valid_freq = (frequencies >= config.MIN_FREQ) & (frequencies <= config.MAX_FREQ)\n    spec_data = spec_data[valid_freq, :]\n    \n    spec_data = np.log10(spec_data + 1e-20)\n    \n    spec_data = spec_data - spec_data.min()\n    spec_data = spec_data / spec_data.max()\n    \n    return spec_data","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.130013Z","iopub.execute_input":"2024-05-11T01:59:08.130479Z","iopub.status.idle":"2024-05-11T01:59:08.141624Z","shell.execute_reply.started":"2024-05-11T01:59:08.130438Z","shell.execute_reply":"2024-05-11T01:59:08.140212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_list = sorted(os.listdir(os.path.join(config.DATA_ROOT, 'train_audio')))\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.143675Z","iopub.execute_input":"2024-05-11T01:59:08.144741Z","iopub.status.idle":"2024-05-11T01:59:08.161471Z","shell.execute_reply.started":"2024-05-11T01:59:08.144684Z","shell.execute_reply":"2024-05-11T01:59:08.159710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(data_loader, model):\n    model.to(config.DEVICE)\n    model.eval()\n    pred = []\n    for batch in data_loader:\n        with torch.no_grad():\n            x, _ = batch\n            x = x.to(config.DEVICE)\n            outputs = model(x)\n        pred.append(outputs.detach().cpu())\n    \n    pred = torch.cat(pred, dim=0).cpu().detach()\n    \n    return pred.numpy().astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.163365Z","iopub.execute_input":"2024-05-11T01:59:08.163901Z","iopub.status.idle":"2024-05-11T01:59:08.173926Z","shell.execute_reply.started":"2024-05-11T01:59:08.163850Z","shell.execute_reply":"2024-05-11T01:59:08.172554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdDataset(torch.utils.data.Dataset):\n    def __init__(\n        self,\n        bird_data,\n        augmentation=None,\n    ):\n        super().__init__()\n        self.bird_data = bird_data\n        self.keys_list = list(bird_data.keys())\n        self.augmentation = augmentation\n    \n    def __len__(self):\n        return len(self.bird_data)\n    \n    def __getitem__(self, index):\n        _spec = self.bird_data[self.keys_list[index]]\n        \n        if self.augmentation is not None:\n            _spec = self.augmentation(image=_spec)['image'] \n        \n        # y is never used in the val/test prediction so 0 is a valid\n        # substitute to prevent \"ValueError: too many values to unpack (expected 2)\"\n        return torch.tensor(_spec, dtype=torch.float32).unsqueeze(0), 0","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.175776Z","iopub.execute_input":"2024-05-11T01:59:08.176212Z","iopub.status.idle":"2024-05-11T01:59:08.187774Z","shell.execute_reply.started":"2024-05-11T01:59:08.176173Z","shell.execute_reply":"2024-05-11T01:59:08.186448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_test_data = dict()\n\nif len(glob(f'{config.DATA_ROOT}/test_soundscapes/*.ogg')) > 0:\n    ogg_file_paths = glob(f'{config.DATA_ROOT}/test_soundscapes/*.ogg')\nelse:\n    ogg_file_paths = sorted(glob(f'{config.DATA_ROOT}/unlabeled_soundscapes/*.ogg'))[:10]\n\nfor i, file_path in tqdm(enumerate(ogg_file_paths)):\n    row_id = re.search(r'/([^/]+)\\.ogg$', file_path).group(1)\n    audio_data, _ = librosa.load(file_path, sr=config.FS)\n    \n    audio_data = SpectralNoiseReduction(audio_data, config.FS)\n    \n    spec = scipy_spectro(audio_data)\n    \n    pad = 512 - (spec.shape[1] % 512)\n    if pad > 0:\n        spec = np.pad(spec, ((0,0), (0,pad)))\n    \n    spec = spec.reshape(512,-1,512).transpose([0, 2, 1])\n    spec = cv2.resize(spec, (256, 256), interpolation=cv2.INTER_AREA)\n    \n    for j in range(48):\n        all_test_data[f'{row_id}_{(j+1)*5}'] = spec[:, :, j]","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:08.189754Z","iopub.execute_input":"2024-05-11T01:59:08.190238Z","iopub.status.idle":"2024-05-11T01:59:25.446066Z","shell.execute_reply.started":"2024-05-11T01:59:08.190194Z","shell.execute_reply":"2024-05-11T01:59:25.443811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_model = Model()\nbird_model.load_state_dict(torch.load(\"/kaggle/input/5-min-checkpoint/model_test_best.ckpt\", map_location=torch.device('cpu')))\n\ntest_dataset = BirdDataset(all_test_data, get_transforms('test'))\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset,\n    batch_size=1,\n    num_workers=config.N_WORKERS,\n    shuffle=False,\n    drop_last=False\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:25.448261Z","iopub.status.idle":"2024-05-11T01:59:25.448834Z","shell.execute_reply.started":"2024-05-11T01:59:25.448573Z","shell.execute_reply":"2024-05-11T01:59:25.448596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\n\npredictions.append(predict(test_loader, bird_model))\ngc.collect()\n\npredictions = np.mean(predictions, axis=0).squeeze()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:25.451033Z","iopub.status.idle":"2024-05-11T01:59:25.451599Z","shell.execute_reply.started":"2024-05-11T01:59:25.451355Z","shell.execute_reply":"2024-05-11T01:59:25.451378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:25.453066Z","iopub.status.idle":"2024-05-11T01:59:25.453615Z","shell.execute_reply.started":"2024-05-11T01:59:25.453380Z","shell.execute_reply":"2024-05-11T01:59:25.453402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pred = pd.DataFrame(predictions, columns=label_list)\nsub_id = pd.DataFrame({'row_id': list(all_test_data.keys())})\n\nsub = pd.concat([sub_id, sub_pred], axis=1)\n\nsub.to_csv('submission.csv',index=False)\nprint(f'Submission shape: {sub.shape}')\nsub.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:59:25.455669Z","iopub.status.idle":"2024-05-11T01:59:25.456348Z","shell.execute_reply.started":"2024-05-11T01:59:25.456001Z","shell.execute_reply":"2024-05-11T01:59:25.456030Z"},"trusted":true},"execution_count":null,"outputs":[]}]}