{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":33246,"databundleVersionId":3221581,"sourceType":"competition"},{"sourceId":10210406,"sourceType":"datasetVersion","datasetId":6310558}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <p style=\"background-color:#f3ab60;font-family:newtimeroman;color:#662e2e;font-size:130%;text-align:center;border-radius:40px 40px;\">BirdCLEF 2022</p>","metadata":{}},{"cell_type":"markdown","source":"<h1 align='center'>Table of Contents 📜</h1>\n<ul style=\"list-style-type:square\">\n    <li><a href=\"#1\">Importing Libraries</a></li>\n    <li><a href=\"#2\">Reading the data</a></li>\n    <li><a href=\"#3\">Quick EDA</a></li>\n    <ul style=\"list-style-type:disc\">\n        <li><a href=\"#3.1\">Train_Metadata</a></li>\n        <li><a href=\"#3.2\">Audio Files</a></li>\n    </ul>\n    <li><a href=\"#4\">Data Preprocessing</a></li>\n    <li><a href=\"#5\">Model</a></li>\n</ul>\n\n","metadata":{}},{"cell_type":"markdown","source":"<a id='1'>\n<h1 align='center'>Importing Libraries 📚</h1>","metadata":{}},{"cell_type":"code","source":"!pip install -q noisereduce efficientnet_pytorch\n!mkdir frozen_packages\n!pip download -q noisereduce efficientnet_pytorch --dest frozen_packages --prefer-binary\n!ls frozen_packages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T06:07:17.951106Z","iopub.execute_input":"2024-12-16T06:07:17.951446Z","iopub.status.idle":"2024-12-16T06:08:37.951010Z","shell.execute_reply.started":"2024-12-16T06:07:17.951416Z","shell.execute_reply":"2024-12-16T06:08:37.950127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport ast\nimport random\nimport numpy as np \nimport pandas as pd \nimport json\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom tqdm import tqdm\nimport noisereduce as nr\n\nimport torchaudio\nfrom efficientnet_pytorch import EfficientNet\nfrom torchvision import transforms\nimport IPython.display as ipd\nfrom collections import Counter\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\nimport torch\nimport torch.nn as nn\nfrom torch.optim import Adam\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader,IterableDataset\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-16T06:08:37.953224Z","iopub.execute_input":"2024-12-16T06:08:37.954035Z","iopub.status.idle":"2024-12-16T06:08:42.695577Z","shell.execute_reply.started":"2024-12-16T06:08:37.953988Z","shell.execute_reply":"2024-12-16T06:08:42.694794Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n<a id='1'>\n<h1 align='center'>Configs📚</h1>","metadata":{}},{"cell_type":"code","source":"class config:\n    seed=2022\n    num_fold = 5\n    sample_rate=32_000\n    n_fft=2048\n    hop_length=512\n    n_mels=512\n    duration=5\n    num_classes = 21\n    train_batch_size = 32\n    valid_batch_size = 64\n    epochs = 5 # change to 5\n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    learning_rate = 4e-3","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:08:42.696794Z","iopub.execute_input":"2024-12-16T06:08:42.697649Z","iopub.status.idle":"2024-12-16T06:08:42.772387Z","shell.execute_reply.started":"2024-12-16T06:08:42.697608Z","shell.execute_reply":"2024-12-16T06:08:42.771517Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\nseed_everything(config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:08:42.774882Z","iopub.execute_input":"2024-12-16T06:08:42.775498Z","iopub.status.idle":"2024-12-16T06:08:43.717315Z","shell.execute_reply.started":"2024-12-16T06:08:42.775456Z","shell.execute_reply":"2024-12-16T06:08:43.716403Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='2'>\n<h1 align='center'>Reading the data 📖</h1>","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/birdclef-2022/train_metadata.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:08:43.718260Z","iopub.execute_input":"2024-12-16T06:08:43.718519Z","iopub.status.idle":"2024-12-16T06:08:44.835153Z","shell.execute_reply.started":"2024-12-16T06:08:43.718494Z","shell.execute_reply":"2024-12-16T06:08:44.834242Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='4'>\n<h1 align='center'>Dataset Preprocessing 🛠️ </h1>","metadata":{}},{"cell_type":"markdown","source":"### First of all, as our target variable is in string format, we have to convert it to integer and here I have used LabelEncoder to perform this work.","metadata":{}},{"cell_type":"code","source":"!mkdir img\nbirds_path = \"/kaggle/input/birdclef-2022/scored_birds.json\"\nwith open(birds_path) as bf:\n    birds = json.load(bf)\nfor bird in birds:\n    path = f\"img/{bird}\"\n    if not os.path.exists(path):\n        os.makedirs(path)\n        \n        \ndf = df[df['primary_label'].isin(birds)].reset_index(drop=True)\n# df = df[df['secondary_labels']=='[]'].reset_index(drop=True)\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:08:44.836181Z","iopub.execute_input":"2024-12-16T06:08:44.836454Z","iopub.status.idle":"2024-12-16T06:08:47.173883Z","shell.execute_reply.started":"2024-12-16T06:08:44.836429Z","shell.execute_reply":"2024-12-16T06:08:47.172585Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# change any hyperparameter you like to achieve class-clear melspectogram image","metadata":{}},{"cell_type":"code","source":"mel_spectrogram = torchaudio.transforms.MelSpectrogram(sample_rate=config.sample_rate, \n                                                      n_fft=config.n_fft, \n                                                      hop_length=config.hop_length,                                                       \n                                                      n_mels=config.n_mels)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:08:47.175243Z","iopub.execute_input":"2024-12-16T06:08:47.175560Z","iopub.status.idle":"2024-12-16T06:08:47.282590Z","shell.execute_reply.started":"2024-12-16T06:08:47.175531Z","shell.execute_reply":"2024-12-16T06:08:47.281833Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"signal_1, sr = torchaudio.load(f\"../input/birdclef-2022/train_audio/puaioh/XC144892.ogg\")\nsignal_2, sr = torchaudio.load(f\"../input/birdclef-2022/train_audio/amewig/XC150015.ogg\")\nresampler = torchaudio.transforms.Resample(sr, config.sample_rate)\nsignal_1 = resampler(signal_1)\nnr_1=torch.tensor(nr.reduce_noise(y=signal_1, sr=config.sample_rate, n_fft=config.n_fft,use_tqdm=True))\n# nr_2 = torch.tensor(nr.reduce_noise(y=signal_2, sr=config.sample_rate, use_tqdm=True,win_length=config.n_fft))\nfig, ax = plt.subplots(1, 2, figsize=(20, 7))\nfig.suptitle(\"Mel Spectrogram\", fontsize=15)\nmel = mel_spectrogram(nr_1)\nmel = mel[:,93:406,:313].log1p()\ntfms = transforms.Compose([transforms.ToPILImage(),transforms.Resize([300,300]), transforms.ToTensor()])\nmel = tfms(mel)\nprint(torch.count_nonzero(mel))\nax[0].imshow(mel[0].numpy(), aspect='auto')\nax[0].set_title(\"Audio 1\")\n\nmel_2 = mel_spectrogram(signal_2)\nax[1].imshow(mel_2[0,93:406,:313].log1p().numpy(), aspect='auto')\nax[1].set_title(\"Audio 2\")\n\nplt.show()\nplt.figure(figsize=(15, 7))\n\nipd.Audio(nr_1,rate=config.sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:13:20.345194Z","iopub.execute_input":"2024-12-16T06:13:20.345811Z","iopub.status.idle":"2024-12-16T06:13:21.417803Z","shell.execute_reply.started":"2024-12-16T06:13:20.345775Z","shell.execute_reply":"2024-12-16T06:13:21.416754Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split in 5 seconds and save log1p-melspectogram as img","metadata":{}},{"cell_type":"code","source":"# from torchvision.utils import save_image\n# def trans(signal):\n#         mel = mel_spectrogram(signal)\n#         mel = mel[:,93:406,:].log1p()\n# #         print(mel.shape)\n#         tfms = transforms.Compose([transforms.ToPILImage(),transforms.Resize([300,300]), transforms.ToTensor()])\n#         mel = tfms(mel)\n#         return mel\n    \n    \n# def save_img(mels=config.n_mels,n_fft=config.n_fft,hop_length=config.hop_length,stride_sec=5):\n#     audio_paths = df['filename'].values\n#     labels = df['primary_label'].values\n#     num_samples = config.sample_rate*config.duration\n#     stride = config.sample_rate*stride_sec\n#     reduce_noise = True\n#     threshold=0.06\n#     img_list=[]\n    \n#     for path,label in tqdm(zip(audio_paths,labels),total=len(labels)):\n#         audio_path = f'../input/birdclef-2022/train_audio/{path}'\n#         signal, sr = torchaudio.load(audio_path)\n        \n#         # norm all signal to sr, post_nr, 3 channel, >=5sec\n#         if sr != config.sample_rate:\n#             resampler = torchaudio.transforms.Resample(sr, self.target_sample_rate)\n#             signal = resampler(signal)\n    \n#         if reduce_noise:\n#             signal = torch.tensor(nr.reduce_noise(y=signal, sr=config.sample_rate, n_fft=n_fft, use_tqdm=True, n_jobs=-1))\n#         if signal.shape[1] <= num_samples:\n#             num_missing_samples = num_samples - signal.shape[1]\n#             last_dim_padding = (0, num_missing_samples)\n#             signal = F.pad(signal, last_dim_padding)\n            \n#         if signal.shape[0]==1:\n#             signal=signal.repeat(3,1)\n#         elif signal.shape[0]==2:\n#             signal=torch.cat((signal,torch.mean(signal,dim=0,keepdim=True)),0)\n#         else:\n#             signal=signal[0:3]\n#         # split to 5sec frame\n#         length = signal.shape[1]\n#         end = 5\n#         start_index = 0 \n#         while start_index + num_samples <= length: # at least once\n#             frame = signal[:,start_index:start_index + num_samples]\n#             img = trans(frame)\n#             split_path=path.split('.')[0]+'_'+str(end)+\".png\"\n#             if torch.count_nonzero(img[0])>=int(threshold*img[0].numel()):\n#                 save_image(img,f\"img/{split_path}\")\n#                 img_dict={}\n#                 img_dict.update({\"path\":split_path,\"label\":label})\n#                 img_list.append(img_dict)\n            \n#             # update pointer\n#             end += stride_sec\n#             start_index += stride\n\n#     return img_list\n\n# img_df=pd.DataFrame(save_img())","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:56:59.861939Z","iopub.status.idle":"2022-04-04T13:56:59.862763Z","shell.execute_reply.started":"2022-04-04T13:56:59.862389Z","shell.execute_reply":"2022-04-04T13:56:59.862434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# you will directly read the image from /kaggle/input/augment-img-bird-call/birdclef-2022/noise (or cutout)\n# then create img_df with column: label(folder name), path\nbase_path = '/kaggle/input/augment-img-bird-call/birdclef-2022/cutout'  # Modify this path if needed\n\ndef create_img_df(base_path):\n    \"\"\"Creates a DataFrame containing image paths and their associated labels (folder names).\"\"\"\n    img_paths = []\n    labels = []\n\n    # Walk through the directory structure\n    for label in os.listdir(base_path):\n        label_path = os.path.join(base_path, label)\n        \n        # Ensure it's a directory\n        if os.path.isdir(label_path):\n            for file in os.listdir(label_path):\n                if file.endswith(\".png\"):  # Assuming images are in PNG format\n                    img_paths.append(os.path.join(label_path, file))\n                    labels.append(label)\n\n    # Create a DataFrame\n    img_df = pd.DataFrame({\n        'label': labels,\n        'path': img_paths\n    })\n    \n    return img_df\n\n# Create the DataFrame\nimg_df = create_img_df(base_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T06:52:40.617774Z","iopub.execute_input":"2024-12-16T06:52:40.618147Z","iopub.status.idle":"2024-12-16T06:52:40.877603Z","shell.execute_reply.started":"2024-12-16T06:52:40.618115Z","shell.execute_reply":"2024-12-16T06:52:40.876898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"total images:%d\" % len(img_df))\nprint(img_df[\"label\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:41.590095Z","iopub.execute_input":"2024-12-16T06:52:41.590447Z","iopub.status.idle":"2024-12-16T06:52:41.598647Z","shell.execute_reply.started":"2024-12-16T06:52:41.590414Z","shell.execute_reply":"2024-12-16T06:52:41.597753Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# do something to balance class","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:42.296807Z","iopub.execute_input":"2024-12-16T06:52:42.297507Z","iopub.status.idle":"2024-12-16T06:52:42.301086Z","shell.execute_reply.started":"2024-12-16T06:52:42.297473Z","shell.execute_reply":"2024-12-16T06:52:42.300138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=img_df\nencoder = LabelEncoder()\ndf['label_encoded'] = encoder.fit_transform(df['label'])\n\nskf = StratifiedKFold(n_splits=config.num_fold)\nfor k, (_, val_ind) in enumerate(skf.split(X=df, y=df['label_encoded'])):\n    df.loc[val_ind, 'fold'] = k\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:42.996042Z","iopub.execute_input":"2024-12-16T06:52:42.996755Z","iopub.status.idle":"2024-12-16T06:52:43.017425Z","shell.execute_reply.started":"2024-12-16T06:52:42.996720Z","shell.execute_reply":"2024-12-16T06:52:43.016744Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nclass BirdClefDataset(Dataset):\n    def __init__(self, df):\n        self.audio_paths = df['path'].values\n        self.labels = df['label_encoded'].values\n        self.tfms = transforms.Compose([transforms.ToTensor()])\n                     \n    def __len__(self):\n        return len(self.labels)\n\n    def __getitem__(self,index):\n        img_path, label = self.audio_paths[index], self.labels[index]\n        img = Image.open(f\"{img_path}\")\n        img = self.tfms(img)\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:43.942584Z","iopub.execute_input":"2024-12-16T06:52:43.942926Z","iopub.status.idle":"2024-12-16T06:52:43.948540Z","shell.execute_reply.started":"2024-12-16T06:52:43.942898Z","shell.execute_reply":"2024-12-16T06:52:43.947629Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to get data according to the folds\ndef get_data(fold):\n    train_df = df[df['fold'] != fold].reset_index(drop=True)\n    valid_df = df[df['fold'] == fold].reset_index(drop=True)\n    \n    train_dataset = BirdClefDataset(train_df)\n    valid_dataset = BirdClefDataset(valid_df)\n        \n    train_loader = torch.utils.data.DataLoader(train_dataset, batch_size=config.train_batch_size, shuffle=True)\n    valid_loader = torch.utils.data.DataLoader(valid_dataset, batch_size=config.valid_batch_size, shuffle=True)\n    \n    return train_loader, valid_loader","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:44.927527Z","iopub.execute_input":"2024-12-16T06:52:44.928132Z","iopub.status.idle":"2024-12-16T06:52:44.933578Z","shell.execute_reply.started":"2024-12-16T06:52:44.928094Z","shell.execute_reply":"2024-12-16T06:52:44.932665Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='5'>\n<h1 align='center'>Model 🤖</h1>","metadata":{}},{"cell_type":"markdown","source":"# EfficientNet cascade to use","metadata":{}},{"cell_type":"code","source":"# 修改模型\nmodel = EfficientNet.from_pretrained('efficientnet-b0').to(config.device)\n# model._conv_stem.in_channels = 1\n# model._conv_stem.weight = torch.nn.Parameter(torch.mean(model._conv_stem.weight, axis=1,keepdim=True))\nnum_ftrs = model._fc.in_features\nmodel._fc = nn.Linear(num_ftrs, config.num_classes).to(config.device)\ntorch.save(model.state_dict(), f'./model.bin')\n# print(model)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:46.329015Z","iopub.execute_input":"2024-12-16T06:52:46.329857Z","iopub.status.idle":"2024-12-16T06:52:46.529352Z","shell.execute_reply.started":"2024-12-16T06:52:46.329821Z","shell.execute_reply":"2024-12-16T06:52:46.528256Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samples = len(df)\nclass_weight = df.groupby('label').agg('count')\nclass_weight=np.log(samples/class_weight['label_encoded'].values)\nclass_weight=torch.tensor(class_weight).float().to(config.device)\nprint(class_weight)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:46.749858Z","iopub.execute_input":"2024-12-16T06:52:46.750506Z","iopub.status.idle":"2024-12-16T06:52:46.761832Z","shell.execute_reply.started":"2024-12-16T06:52:46.750470Z","shell.execute_reply":"2024-12-16T06:52:46.761108Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LabelSmoothingLoss(torch.nn.Module):\n    def __init__(self, smoothing: float = 0.1, \n                 reduction=\"mean\", weight=None):\n        super(LabelSmoothingLoss, self).__init__()\n        self.smoothing   = smoothing\n        self.reduction = reduction\n        self.weight    = weight\n    def reduce_loss(self, loss):\n        return loss.mean() if self.reduction == 'mean' else loss.sum() \\\n         if self.reduction == 'sum' else loss\n\n    def linear_combination(self, x, y):\n        return self.smoothing * x + (1 - self.smoothing) * y\n\n    def forward(self, preds, target):\n        assert 0 <= self.smoothing < 1\n\n        if self.weight is not None:\n            self.weight = self.weight.to(preds.device)\n\n        n = preds.size(-1)\n        log_preds = F.log_softmax(preds, dim=-1)\n        loss = self.reduce_loss(-log_preds.sum(dim=-1))\n        nll = F.nll_loss(\n            log_preds, target, reduction=self.reduction, weight=self.weight\n        )\n        return self.linear_combination(loss / n, nll)","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:47.865583Z","iopub.execute_input":"2024-12-16T06:52:47.866613Z","iopub.status.idle":"2024-12-16T06:52:47.873610Z","shell.execute_reply.started":"2024-12-16T06:52:47.866574Z","shell.execute_reply":"2024-12-16T06:52:47.872608Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def loss_fn(outputs, labels):     \n    return LabelSmoothingLoss(weight=class_weight)(outputs, labels)\n\ndef train(model, data_loader, optimizer, scheduler, device, epoch):\n    accumulation_steps=4\n    model.train()\n    loop = tqdm(data_loader, position=0)\n    running_loss = 0\n    count =0\n    losses=[]\n    for mels, labels in loop:\n        mels = mels.to(device)\n        labels = labels.to(device)\n\n    #         if count==0:\n    #             fig, ax = plt.subplots(8, 2, figsize=(20, 56))\n    #             fig.suptitle(\"Mel Spectrogram\", fontsize=15)\n    #             ax = ax.flatten()\n    #             for mel,labels,a in zip(mels,labels,ax):\n    #                 a.imshow(mel[0].detach().cpu().numpy(), aspect='auto')\n    #                 a.set_title(f\"{labels.detach().cpu().numpy()}\")\n    #             plt.show()\n    #             break\n        outputs = model(mels)\n        loss = loss_fn(outputs, labels)\n\n        if scheduler is not None:\n            scheduler.step(loss)\n\n        running_loss += loss.item()\n        \n        loop.set_description(f\"Epoch [{epoch+1}/{config.epochs}]\")\n        loop.set_postfix(loss=loss.item())\n        count+=1\n        \n        loss = loss/accumulation_steps\n        loss.backward()\n        if(count%accumulation_steps)==0:\n            # optimizer the net\n            optimizer.step()        # update parameters of net\n            optimizer.zero_grad()  \n\n    return running_loss/count","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:48.429109Z","iopub.execute_input":"2024-12-16T06:52:48.429812Z","iopub.status.idle":"2024-12-16T06:52:48.436882Z","shell.execute_reply.started":"2024-12-16T06:52:48.429778Z","shell.execute_reply":"2024-12-16T06:52:48.435874Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score,accuracy_score\ndef valid(model, data_loader, device, epoch):\n    model.eval()\n    loop = tqdm(data_loader, position=0)\n    running_loss = 0\n    count=0\n    counts=0\n    prediction=[]\n    all_labels=[]\n    for mels, labels in loop:\n        mels = mels.to(device)\n        labels = labels.to(device)\n        with torch.no_grad():\n            outputs = model(mels)\n        loss = loss_fn(outputs, labels)\n            \n        running_loss += loss.item()\n        \n        pred =outputs.argmax(1, True)\n        prediction.extend(pred)\n        all_labels.extend(labels)\n        correct=torch.eq(pred,labels.view_as(pred)).sum().item()\n        \n        F1_macro=f1_score(labels.cpu(),pred.cpu(),average='macro')\n        \n        loop.set_description(f\"Epoch [{epoch+1}/{config.epochs}]\")\n        loop.set_postfix(loss=loss.item(),acc=correct/pred.shape[0],F1_macro=F1_macro)\n        count+=1\n        counts+=pred.shape[0]\n    all_labels=torch.Tensor(all_labels).cpu()\n    prediction=torch.Tensor(prediction).cpu()\n    accuracy=accuracy_score(all_labels,prediction)\n    F1_macro=f1_score(all_labels,prediction,average='macro')\n#     F1_micro=f1_score(all_labels,prediction,average='micro')\n        \n    return running_loss/count,accuracy,F1_macro","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:48.915917Z","iopub.execute_input":"2024-12-16T06:52:48.916604Z","iopub.status.idle":"2024-12-16T06:52:48.923995Z","shell.execute_reply.started":"2024-12-16T06:52:48.916570Z","shell.execute_reply":"2024-12-16T06:52:48.923025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run(fold,best_fold_loss):\n    train_loader, valid_loader = get_data(fold)\n    \n#     ignored_params = list(map(id, model._fc.parameters()))\n#     base_params = filter(lambda p: id(p) not in ignored_params,model.parameters())\n#     optimizer = Adam([\n#                 {'params': base_params,'lr':0.1*config.learning_rate},\n#                 {'params': model._fc.parameters()}], \n#                  lr=config.learning_rate)\n    \n#     scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer)\n    optimizer = torch.optim.Adam(model.parameters(), lr=config.learning_rate)\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=10)\n    for epoch in range(config.epochs):\n        train_loss = train(model, train_loader, optimizer, scheduler, config.device, epoch)\n        valid_loss,valid_acc,F1a = valid(model, valid_loader, config.device, epoch)\n        if valid_loss < best_fold_loss:\n            print(f\"Validation Loss Improved - {best_fold_loss} ---> {valid_loss}\")\n            torch.save(model.state_dict(), f'./model_{fold}.bin')\n            print(f\"Saved model checkpoint at ./model_{fold}.bin,\\nacc: {valid_acc},\\nF1a: {F1a}\")\n            best_fold_loss = valid_loss\n    return best_fold_loss","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:49.418693Z","iopub.execute_input":"2024-12-16T06:52:49.419585Z","iopub.status.idle":"2024-12-16T06:52:49.425449Z","shell.execute_reply.started":"2024-12-16T06:52:49.419548Z","shell.execute_reply":"2024-12-16T06:52:49.424535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Run Training","metadata":{}},{"cell_type":"code","source":"best_loss = 100\nloss_list=[]\nfor fold in range(config.num_fold):\n    model.load_state_dict(torch.load(f'./model.bin')) \n    print(\"=\" * 30)\n    print(\"Training Fold - \", fold)\n    print(\"=\" * 30)\n    best_val_loss = run(fold,best_loss)\n#     if best_val_loss < best_loss:\n    print(f'Best Valid Loss: {best_val_loss:.5f}')\n#         best_loss = best_val_loss\n    loss_list.append(best_val_loss)\n    gc.collect()\n    torch.cuda.empty_cache()\n\n    break # To run for all the folds, just remove this break\n    \nprint(np.mean(loss_list))","metadata":{"execution":{"iopub.status.busy":"2024-12-16T06:52:51.194373Z","iopub.execute_input":"2024-12-16T06:52:51.194752Z","iopub.status.idle":"2024-12-16T07:03:22.363400Z","shell.execute_reply.started":"2024-12-16T06:52:51.194692Z","shell.execute_reply":"2024-12-16T07:03:22.362475Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def mel_transform(signal,sr):\n#     mel_spectrogram = torchaudio.transforms.MelSpectrogram(sample_rate=config.sample_rate, \n#                                                       n_fft=config.n_fft, \n#                                                       hop_length=config.hop_length, \n#                                                       n_mels=config.n_mels)\n#     num_samples = config.sample_rate*config.duration\n#     if sr != config.sample_rate:\n#         resampler = torchaudio.transforms.Resample(sr, config.sample_rate)\n#         signal = resampler(signal)\n\n#     start = int(signal.shape[1]*0.9)\n#     signal = signal[:,start:]\n    \n#     if signal.shape[1] > num_samples:\n#         signal = signal[:, :num_samples]\n\n#     if signal.shape[1]< num_samples:\n#         num_missing_samples = num_samples - signal.shape[1]\n#         last_dim_padding = (0, num_missing_samples)\n#         signal = F.pad(signal, last_dim_padding)\n        \n#     signal = torch.tensor(nr.reduce_noise(y=signal, sr=config.sample_rate, win_length=mel_spectrogram.win_length, use_tqdm=True, n_jobs=-1))\n#     mel = mel_spectrogram(signal)\n#     mel = mel[0,50:150,:].log1p()\n#     mel = torch.unsqueeze(mel,0)\n#     tfms = transforms.Compose([transforms.ToPILImage(),transforms.Resize([300,300]), transforms.ToTensor()])\n#     mel = tfms(mel)\n#     mel = tfms(mel).to(config.device)\n#     return mel","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:56:59.888751Z","iopub.status.idle":"2022-04-04T13:56:59.889484Z","shell.execute_reply.started":"2022-04-04T13:56:59.889187Z","shell.execute_reply":"2022-04-04T13:56:59.889231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model.load_state_dict(torch.load('./model.bin'))\n# model.eval()\n# data = {}\n# duration = 5\n# test_path = \"/kaggle/input/birdclef-2022/test_soundscapes/\"\n# files = [f.split('.')[0] for f in sorted(os.listdir(test_path))]\n# for f in files:\n#     file_path = test_path + f + '.ogg'\n#     audio, sr = torchaudio.load(file_path) # loaded the audio\n# #     audio, sr = torchaudio.load('../input/birdclef-2022/train_audio/apapan/XC256220.ogg')\n#     # Get number of samples for 5 seconds; replace 5 by any number\n#     buffer = duration * sr\n#     samples_total = audio.shape[1]\n#     samples_wrote = 0\n#     counter = 1\n    \n\n#     while samples_wrote+buffer <= samples_total:\n\n#         block = audio[:,samples_wrote : (samples_wrote + buffer)]\n#         feat = mel_transform(block, sr)        \n#         x = torch.unsqueeze(feat,dim=0)\n#         pred = model(x).detach().cpu()\n#         label_index = np.argmax(pred,axis=1)[0]\n#         for b in birds:\n#             segment_end = counter * duration   \n#             row_id = f + '_' + b + '_' + str(segment_end)\n#             target = False\n#             a = encoder.inverse_transform([label_index]) \n#             if a == b:\n#                 target = True\n#             data[row_id] = target\n#         counter += 1\n#         samples_wrote += buffer\n# data","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:56:59.890847Z","iopub.status.idle":"2022-04-04T13:56:59.89159Z","shell.execute_reply.started":"2022-04-04T13:56:59.89131Z","shell.execute_reply":"2022-04-04T13:56:59.891337Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sample_submission\n\n# sample_submission = pd.read_csv('../input/birdclef-2022/sample_submission.csv')\n# for i in range(len(sample_submission)):\n#     sample = sample_submission.row_id[i]\n#     sample_submission.iat[i,1]=True\n\n#     if sample in data:\n#         sample_submission.iat[i, 1] = data[sample]\n# sample_submission.to_csv(\"submission.csv\", index=False)\n# !head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-04-04T13:56:59.89408Z","iopub.status.idle":"2022-04-04T13:56:59.894813Z","shell.execute_reply.started":"2022-04-04T13:56:59.894555Z","shell.execute_reply":"2022-04-04T13:56:59.894582Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}