{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"machine_shape":"hm","gpuType":"T4"},"accelerator":"GPU","kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8553291,"sourceType":"datasetVersion","datasetId":5111393},{"sourceId":3729,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":2656}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport numpy as np\nimport librosa\nimport os\nfrom pathlib import Path\nimport math, random\nimport torchaudio\nfrom torchaudio import transforms\nfrom IPython.display import Audio\nimport random","metadata":{"id":"fe32599f-1998-403f-ba33-734aecd0ffa0","execution":{"iopub.status.busy":"2024-05-30T19:22:44.987579Z","iopub.execute_input":"2024-05-30T19:22:44.988002Z","iopub.status.idle":"2024-05-30T19:22:44.993547Z","shell.execute_reply.started":"2024-05-30T19:22:44.987975Z","shell.execute_reply":"2024-05-30T19:22:44.992615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/edited/dataset_edited.csv')","metadata":{"id":"JcGyGscrGSek","execution":{"iopub.status.busy":"2024-05-30T19:22:46.840487Z","iopub.execute_input":"2024-05-30T19:22:46.840906Z","iopub.status.idle":"2024-05-30T19:22:46.886828Z","shell.execute_reply.started":"2024-05-30T19:22:46.840874Z","shell.execute_reply":"2024-05-30T19:22:46.885622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioUtil:\n\n    @staticmethod\n    def open(audio_file):\n        sig, sr = torchaudio.load(audio_file)\n        return (sig, sr)\n\n    @staticmethod\n    def rechannel(aud, new_channel):\n        sig, sr = aud\n        if (sig.shape[0] == new_channel):\n          return aud\n        if (new_channel == 1):\n            resig = sig[:1, :]\n        else:\n            resig = torch.cat([sig, sig])\n        return ((resig, sr))\n\n    @staticmethod\n    def resample(aud, newsr):\n        sig, sr = aud\n        if (sr == newsr):\n            return aud\n        num_channels = sig.shape[0]\n        resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n        if (num_channels > 1):\n            retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n            resig = torch.cat([resig, retwo])\n            return (resig, newsr)\n        else:\n            return (resig, newsr)\n\n    @staticmethod\n    def pad_trunc(aud, max_ms):\n        sig, sr = aud\n        num_rows, sig_len = sig.shape\n        max_len = sr//1000 * max_ms\n\n        if (sig_len > max_len):\n          sig = sig[:,:max_len]\n\n        elif (sig_len < max_len):\n          pad_begin_len = random.randint(0, max_len - sig_len)\n          pad_end_len = max_len - sig_len - pad_begin_len\n\n          pad_begin = torch.zeros((num_rows, pad_begin_len))\n          pad_end = torch.zeros((num_rows, pad_end_len))\n\n          sig = torch.cat((pad_begin, sig, pad_end), 1)\n        return (sig, sr)\n\n    @staticmethod\n    def time_shift(aud, shift_limit):\n        sig,sr = aud\n        _, sig_len = sig.shape\n        shift_amt = int(random.random() * shift_limit * sig_len)\n        return (sig.roll(shift_amt), sr)\n\n    @staticmethod\n    def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n        sig,sr = aud\n        top_db = 80\n        spec = transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n        spec = transforms.AmplitudeToDB(top_db=top_db)(spec)\n        return (spec)\n\n    @staticmethod\n    def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n        _, n_mels, n_steps = spec.shape\n        mask_value = spec.mean()\n        aug_spec = spec\n\n        freq_mask_param = max_mask_pct * n_mels\n        for _ in range(n_freq_masks):\n          aug_spec = transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n        time_mask_param = max_mask_pct * n_steps\n        for _ in range(n_time_masks):\n          aug_spec = transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n        return aug_spec","metadata":{"id":"76bf60d8-f2de-42a8-9af7-11ee9acaadf4","execution":{"iopub.status.busy":"2024-05-30T19:22:47.786542Z","iopub.execute_input":"2024-05-30T19:22:47.787073Z","iopub.status.idle":"2024-05-30T19:22:47.804494Z","shell.execute_reply.started":"2024-05-30T19:22:47.787033Z","shell.execute_reply":"2024-05-30T19:22:47.802856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset, random_split\nimport torchaudio\n\nclass SoundDS(Dataset):\n\n    def __init__(self, df, data_path):\n        self.df = df\n        self.data_path = str(data_path)\n        self.duration = 4000\n        self.sr = 44100\n        self.channel = 2\n        self.shift_pct = 0.4\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        audio_file = self.data_path + self.df.loc[idx, 'filename']\n        class_id = self.df.loc[idx, 'label']\n\n        aud = AudioUtil.open(audio_file)\n        reaud = AudioUtil.resample(aud, self.sr)\n        rechan = AudioUtil.rechannel(reaud, self.channel)\n\n        dur_aud = AudioUtil.pad_trunc(rechan, self.duration)\n        shift_aud = AudioUtil.time_shift(dur_aud, self.shift_pct)\n        sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n        aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n        extra_channel = torch.zeros(1, 64, 344)\n        aug_sgram = torch.cat((aug_sgram, extra_channel), dim=0)\n        return aug_sgram, class_id","metadata":{"id":"ae69e683-1a8f-4b64-9950-6a4b5850a41f","execution":{"iopub.status.busy":"2024-05-30T19:22:50.032059Z","iopub.execute_input":"2024-05-30T19:22:50.032436Z","iopub.status.idle":"2024-05-30T19:22:50.042276Z","shell.execute_reply.started":"2024-05-30T19:22:50.032407Z","shell.execute_reply":"2024-05-30T19:22:50.041351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_labels = df['primary_label'].unique()\ndict = {unique_labels[i]: i for i in range(len(unique_labels))}\ndf['label'] = df['primary_label'].map(dict)","metadata":{"id":"9c311662-d77f-4472-84d3-71d591464aec","execution":{"iopub.status.busy":"2024-05-30T19:22:51.601200Z","iopub.execute_input":"2024-05-30T19:22:51.601748Z","iopub.status.idle":"2024-05-30T19:22:51.614488Z","shell.execute_reply.started":"2024-05-30T19:22:51.601702Z","shell.execute_reply":"2024-05-30T19:22:51.613288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['label'].unique())","metadata":{"id":"a6a8e719-02ee-4d30-9f22-6ecfb63821de","outputId":"431ccf4b-ba2b-46f2-edac-09b551d5ad53","execution":{"iopub.status.busy":"2024-05-30T19:22:52.803456Z","iopub.execute_input":"2024-05-30T19:22:52.803837Z","iopub.status.idle":"2024-05-30T19:22:52.810827Z","shell.execute_reply.started":"2024-05-30T19:22:52.803808Z","shell.execute_reply":"2024-05-30T19:22:52.809294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df.primary_label.unique()))","metadata":{"id":"7f4df7b9-868b-4924-a59e-f43b0a901fc6","outputId":"c129fb8d-deb3-4323-9054-c402dba317f5","execution":{"iopub.status.busy":"2024-05-30T19:22:54.887299Z","iopub.execute_input":"2024-05-30T19:22:54.887742Z","iopub.status.idle":"2024-05-30T19:22:54.895854Z","shell.execute_reply.started":"2024-05-30T19:22:54.887708Z","shell.execute_reply":"2024-05-30T19:22:54.894464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import random_split\ndata_path = '/kaggle/input/birdclef-2024/train_audio/'\nmyds = SoundDS(df, data_path)\n\nnum_items = len(myds)\nnum_train = round(num_items * 0.8)\nnum_val = num_items - num_train\ntrain_ds, val_ds = random_split(myds, [num_train, num_val])\n\ntrain_dl = torch.utils.data.DataLoader(train_ds, batch_size=16, shuffle=True)\nval_dl = torch.utils.data.DataLoader(val_ds, batch_size=16, shuffle=False)","metadata":{"id":"534c0398-9485-451b-9e67-625469494f1c","execution":{"iopub.status.busy":"2024-05-30T19:22:55.747472Z","iopub.execute_input":"2024-05-30T19:22:55.748605Z","iopub.status.idle":"2024-05-30T19:22:55.759921Z","shell.execute_reply.started":"2024-05-30T19:22:55.748562Z","shell.execute_reply":"2024-05-30T19:22:55.758125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"id":"c513292b-5706-4820-a1e1-10c3ec66240a","outputId":"5e82f171-7b63-4e87-d736-1329ff721f1d","execution":{"iopub.status.busy":"2024-05-30T19:22:57.348587Z","iopub.execute_input":"2024-05-30T19:22:57.349124Z","iopub.status.idle":"2024-05-30T19:22:57.367996Z","shell.execute_reply.started":"2024-05-30T19:22:57.349069Z","shell.execute_reply":"2024-05-30T19:22:57.367018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"myds[0]","metadata":{"id":"5364cd23-a3a2-4e6b-a530-e3fe3372f920","outputId":"b5124c37-bded-42f6-e2b2-7b36bbba1a91","execution":{"iopub.status.busy":"2024-05-30T19:22:58.871153Z","iopub.execute_input":"2024-05-30T19:22:58.871690Z","iopub.status.idle":"2024-05-30T19:22:58.959718Z","shell.execute_reply.started":"2024-05-30T19:22:58.871646Z","shell.execute_reply":"2024-05-30T19:22:58.958548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch import nn","metadata":{"id":"WAR4b8NocGur","execution":{"iopub.status.busy":"2024-05-30T19:23:00.877435Z","iopub.execute_input":"2024-05-30T19:23:00.878785Z","iopub.status.idle":"2024-05-30T19:23:00.884390Z","shell.execute_reply.started":"2024-05-30T19:23:00.878732Z","shell.execute_reply":"2024-05-30T19:23:00.883199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport pandas.api.types\n\n\nimport sklearn.metrics\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\ndef safe_call_score(metric_function, solution, submission, **metric_func_kwargs):\n    '''\n    Call score. If that raises an error and that already been specifically handled, just raise it.\n    Otherwise make a conservative attempt to identify potential participant visible errors.\n    '''\n    try:\n        score_result = metric_function(solution, submission, **metric_func_kwargs)\n    except Exception as err:\n        error_message = str(err)\n        if err.__class__.__name__ == 'ParticipantVisibleError':\n            raise ParticipantVisibleError(error_message)\n        elif err.__class__.__name__ == 'HostVisibleError':\n            raise HostVisibleError(error_message)\n        else:\n            if treat_as_participant_error(error_message, solution):\n                raise ParticipantVisibleError(error_message)\n            else:\n                raise err\n    return score_result\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Version of macro-averaged ROC-AUC score that ignores all classes that have no true positive labels.\n    '''\n    \n#     del solution[row_id_column_name]\n#     del submission[row_id_column_name]\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        bad_dtypes = {x: submission[x].dtype  for x in submission.columns if not pandas.api.types.is_numeric_dtype(submission[x])}\n        raise ParticipantVisibleError(f'Invalid submission data types found: {bad_dtypes}')\n\n    solution_sums = solution.sum(axis=0)\n    print(solution_sums[solution_sums > 0])\n    print(solution_sums[solution_sums > 0].index)\n    print(solution_sums[solution_sums > 0].index.values)\n    scored_columns = list(solution_sums[solution_sums > 0].index.values)\n    print(scored_columns)\n    assert len(scored_columns) > 0\n\n    return safe_call_score(sklearn.metrics.roc_auc_score, solution[scored_columns].values, submission[scored_columns].values, average='macro')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T19:23:02.421478Z","iopub.execute_input":"2024-05-30T19:23:02.422694Z","iopub.status.idle":"2024-05-30T19:23:02.433839Z","shell.execute_reply.started":"2024-05-30T19:23:02.422654Z","shell.execute_reply":"2024-05-30T19:23:02.432350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport torch\nimport timm\n\nNUM_FINETUNE_CLASSES = 182\n\nmodel = timm.create_model('tf_efficientnet_b0', pretrained=False, num_classes=1000)\n\ncheckpoint = torch.load('/kaggle/input/tf-efficientnet/pytorch/tf-efficientnet-b0/1/tf_efficientnet_b0_aa-827b6e33.pth', map_location='cpu')\n\nstate_dict = {k: v for k, v in checkpoint.items() if 'classifier' not in k}\nmodel.load_state_dict(state_dict, strict=False)\n\nin_features = model.classifier.in_features\nmodel.classifier = torch.nn.Linear(in_features, NUM_FINETUNE_CLASSES)\n\ntorch.nn.init.xavier_uniform_(model.classifier.weight)\ntorch.nn.init.zeros_(model.classifier.bias)\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = model.to(device)","metadata":{"id":"MkWKrH_JagX6","outputId":"18d66a8e-89e0-4763-8480-f15fe3878700","execution":{"iopub.status.busy":"2024-05-30T19:23:04.575892Z","iopub.execute_input":"2024-05-30T19:23:04.576279Z","iopub.status.idle":"2024-05-30T19:23:04.757968Z","shell.execute_reply.started":"2024-05-30T19:23:04.576250Z","shell.execute_reply":"2024-05-30T19:23:04.756825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference (model, val_dl):\n  correct_prediction = 0\n  total_prediction = 0\n\n  # Disable gradient updates\n  with torch.no_grad():\n    for data in val_dl:\n      # Get the input features and target labels, and put them on the GPU\n      inputs, labels = data[0].to(device), data[1].to(device)\n\n      # Normalize the inputs\n      inputs_m, inputs_s = inputs.mean(), inputs.std()\n      inputs = (inputs - inputs_m) / inputs_s\n\n      # Get predictions\n      outputs = model(inputs)\n\n      # Get the predicted class with the highest score\n      _, prediction = torch.max(outputs,1)\n      # Count of predictions that matched the target label\n      correct_prediction += (prediction == labels).sum().item()\n      total_prediction += prediction.shape[0]\n    \n  acc = correct_prediction/total_prediction\n  print(f'Accuracy: {acc:.2f}, Total items: {total_prediction}')\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T19:23:06.396537Z","iopub.execute_input":"2024-05-30T19:23:06.396951Z","iopub.status.idle":"2024-05-30T19:23:06.404804Z","shell.execute_reply.started":"2024-05-30T19:23:06.396919Z","shell.execute_reply":"2024-05-30T19:23:06.403328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def training(model, train_dl, num_epochs):\n  criterion = nn.CrossEntropyLoss()\n  optimizer = torch.optim.Adam(model.parameters(),lr=0.001)\n  scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, max_lr=0.001,\n                                                steps_per_epoch=int(len(train_dl)),\n                                                epochs=num_epochs,\n                                                anneal_strategy='linear')\n  scores = []\n\n  for epoch in range(num_epochs):\n    running_loss = 0.0\n    correct_prediction = 0\n    total_prediction = 0\n    \n    for i, data in enumerate(train_dl):\n        inputs, labels = data[0].to(device), data[1].to(device)\n\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        scheduler.step()\n\n        running_loss += loss.item()\n\n        _, prediction = torch.max(outputs,1)\n        correct_prediction += (prediction == labels).sum().item()\n        total_prediction += prediction.shape[0]\n\n\n    num_batches = len(train_dl)\n    avg_loss = running_loss / num_batches\n    acc = correct_prediction/total_prediction\n    inference(model, val_dl)\n    print(f'Epoch: {epoch}, Loss: {avg_loss:.2f}, Accuracy: {acc:.2f}')\n  print('Finished Training')\n  return scores\n\nnum_epochs = 2\nscores = training(model, train_dl, num_epochs)","metadata":{"id":"7d44b304-698b-4fb0-b8c0-ebffd6095e2e","outputId":"ba632aee-acaf-404e-f8ae-d6ae469b129c","execution":{"iopub.status.busy":"2024-05-30T19:23:08.296970Z","iopub.execute_input":"2024-05-30T19:23:08.297426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\np = Path('/kaggle/input/birdclef-2024/test_soundscapes/')\n\nrows_list = []\nfor x in p.rglob(\"*\"):\n    if '.ogg' in x.name:\n        aud = AudioUtil.open(x)\n        duration = 4000\n        sr = 44100\n        channel = 2\n        shift_pct = 0.4\n        reaud = AudioUtil.resample(aud, sr)\n        rechan = AudioUtil.rechannel(reaud, channel)\n        dur_aud = AudioUtil.pad_trunc(rechan, duration)\n        shift_aud = AudioUtil.time_shift(dur_aud, shift_pct)\n        sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n        aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n        extra_channel = torch.zeros(1, 64, 344)\n        aug_sgram = torch.cat((aug_sgram, extra_channel), dim=0).to(device)\n        inputs_m, inputs_s = aug_sgram.mean(), aug_sgram.std()\n        inputs = (aug_sgram - inputs_m) / inputs_s\n        line = {'row_id' : x.name}\n        for key, value in dict.items():\n            line[key] = F.softmax(model(inputs.unsqueeze(0)),dim=1)[0][value].cpu().item()\n        rows_list.append(line)\nsubmission = pd.DataFrame(rows_list)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T18:57:37.351236Z","iopub.status.idle":"2024-05-30T18:57:37.352854Z","shell.execute_reply.started":"2024-05-30T18:57:37.352552Z","shell.execute_reply":"2024-05-30T18:57:37.352579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('sumbission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T18:57:37.354393Z","iopub.status.idle":"2024-05-30T18:57:37.355631Z","shell.execute_reply.started":"2024-05-30T18:57:37.355290Z","shell.execute_reply":"2024-05-30T18:57:37.355317Z"},"trusted":true},"execution_count":null,"outputs":[]}]}