{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Before reading this Notebook, you can read my previous Notebook [[Fully Pipeline]|Stage 1: Training](https://www.kaggle.com/code/bibanh/0-71-fully-pipeline-resnet34-stage-1-training).","metadata":{}},{"cell_type":"markdown","source":"# 1. Imports","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset, random_split\nimport torch.nn.functional as F\nimport torchaudio\nfrom torchaudio import transforms\nfrom IPython.display import Audio\nimport torchvision","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-28T01:39:59.573206Z","iopub.execute_input":"2023-03-28T01:39:59.573823Z","iopub.status.idle":"2023-03-28T01:39:59.582195Z","shell.execute_reply.started":"2023-03-28T01:39:59.573758Z","shell.execute_reply":"2023-03-28T01:39:59.581155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\nimport pandas as pd\nimport os\nimport glob\nimport math, random\nimport timm\nimport librosa\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.584048Z","iopub.execute_input":"2023-03-28T01:39:59.585352Z","iopub.status.idle":"2023-03-28T01:39:59.598010Z","shell.execute_reply.started":"2023-03-28T01:39:59.585292Z","shell.execute_reply":"2023-03-28T01:39:59.596848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed = 1999\nseed_everything(seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.599724Z","iopub.execute_input":"2023-03-28T01:39:59.600383Z","iopub.status.idle":"2023-03-28T01:39:59.610686Z","shell.execute_reply.started":"2023-03-28T01:39:59.600345Z","shell.execute_reply":"2023-03-28T01:39:59.609715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    isOneHot = False\n    rate = 32000\n    num_classes = 264","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.612919Z","iopub.execute_input":"2023-03-28T01:39:59.613571Z","iopub.status.idle":"2023-03-28T01:39:59.621282Z","shell.execute_reply.started":"2023-03-28T01:39:59.613531Z","shell.execute_reply":"2023-03-28T01:39:59.620382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Define Audio Class to get Vector Embedding by MelSpectrogram","metadata":{}},{"cell_type":"markdown","source":"## 2.1 Audio Class with some necessary function","metadata":{}},{"cell_type":"code","source":"# class AudioUtil():\n#   @staticmethod\n#   def open(audio_file):\n#     sig, sr = torchaudio.load(audio_file)\n#     return (sig, sr)\n\n#   @staticmethod\n#   def rechannel(aud, new_channel):\n#     sig, sr = aud\n\n#     if (sig.shape[0] == new_channel):\n#       # Nothing to do\n#       return aud\n\n#     if (new_channel == 1):\n#       # Convert from stereo to mono by selecting only the first channel\n#       resig = sig[:1, :]\n#     else:\n#       # Convert from mono to stereo by duplicating the first channel\n#       resig = torch.cat([sig, sig, sig])\n\n#     return ((resig, sr))\n\n#   @staticmethod\n#   def resample(aud, newsr):\n#     sig, sr = aud\n\n#     if (sr == newsr):\n#       # Nothing to do\n#       return aud\n\n#     num_channels = sig.shape[0]\n#     # Resample first channel\n#     resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n#     if (num_channels > 1):\n#       # Resample the second channel and merge both channels\n#       retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n#       resig = torch.cat([resig, retwo])\n\n#     return ((resig, newsr))\n\n#   @staticmethod\n#   def pad_trunc(aud, max_ms):\n#     sig, sr = aud\n#     num_rows, sig_len = sig.shape\n#     max_len = sr//1000 * max_ms\n\n#     if (sig_len > max_len):\n#       # Truncate the signal to the given length\n#       sig = sig[:,:max_len]\n\n#     elif (sig_len < max_len):\n#       # Length of padding to add at the beginning and end of the signal\n#       pad_begin_len = random.randint(0, max_len - sig_len)\n#       pad_end_len = max_len - sig_len - pad_begin_len\n\n#       # Pad with 0s\n#       pad_begin = torch.zeros((num_rows, pad_begin_len))\n#       pad_end = torch.zeros((num_rows, pad_end_len))\n\n#       sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n#     return (sig, sr)\n\n#   @staticmethod\n#   def time_shift(aud, shift_limit):\n#     sig,sr = aud\n#     _, sig_len = sig.shape\n#     shift_amt = int(random.random() * shift_limit * sig_len)\n#     return (sig.roll(shift_amt), sr)\n\n#   @staticmethod\n#   def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n#     sig,sr = aud\n#     top_db = 80\n\n#     # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n#     spec = torchaudio.transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n#     # Convert to decibels\n#     spec = torchaudio.transforms.AmplitudeToDB(top_db=top_db)(spec)\n#     return (spec)\n\n#   @staticmethod\n#   def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n#     _, n_mels, n_steps = spec.shape\n#     mask_value = spec.mean()\n#     aug_spec = spec\n\n#     freq_mask_param = max_mask_pct * n_mels\n#     for _ in range(n_freq_masks):\n#       aug_spec = torchaudio.transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n#     time_mask_param = max_mask_pct * n_steps\n#     for _ in range(n_time_masks):\n#       aug_spec = torchaudio.transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n#     return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.622835Z","iopub.execute_input":"2023-03-28T01:39:59.623813Z","iopub.status.idle":"2023-03-28T01:39:59.634701Z","shell.execute_reply.started":"2023-03-28T01:39:59.623768Z","shell.execute_reply":"2023-03-28T01:39:59.633338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# librosa version\n\nclass AudioUtil():\n  @staticmethod\n  def open(audio_file):\n    sig, sr = librosa.load(audio_file, sr=32000)\n    return (sig, sr)\n\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = np.concatenate([sig, sig, sig])\n\n    return ((resig, sr))\n\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = librosa.resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = librosa.resample(sr, newsr)(sig[1:,:])\n      resig = np.concatenate([resig, retwo])\n\n    return ((resig, newsr))\n\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = np.zeros((num_rows, pad_begin_len))\n      pad_end = np.zeros((num_rows, pad_end_len))\n\n      sig = np.concatenate((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (np.roll(sig,shift_amt), sr)\n\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = librosa.feature.melspectrogram(y=sig, sr=sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)\n\n    # Convert to decibels\n    spec = librosa.amplitude_to_db(S=spec, top_db=top_db)\n    return (spec)\n\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n        f = np.random.randint(0, freq_mask_param)\n        mask_end = np.random.randint(f, n_mels)\n        mask_start = mask_end - f\n        aug_spec[mask_start:mask_end, :] = mask_value\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n        t = np.random.randint(0, time_mask_param)\n        mask_end = np.random.randint(t, n_steps)\n        mask_start = mask_end - t\n        aug_spec[:, mask_start:mask_end] = mask_value\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.636317Z","iopub.execute_input":"2023-03-28T01:39:59.637149Z","iopub.status.idle":"2023-03-28T01:39:59.663074Z","shell.execute_reply.started":"2023-03-28T01:39:59.637087Z","shell.execute_reply":"2023-03-28T01:39:59.661298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Preprocessing Data Before Put into Model","metadata":{}},{"cell_type":"code","source":"def preprocessing(aud):\n    duration = 8000\n    sr = 32000\n    channel = 3\n    shift_pct = 0.4\n    reaud = AudioUtil.resample(aud, sr)\n    rechan = AudioUtil.rechannel(reaud, channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=512)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n    aug_sgram_m, aug_sgram_s = aug_sgram.mean(), aug_sgram.std()\n    aug_sgram = (aug_sgram - aug_sgram_m) / aug_sgram_s\n    \n    return aug_sgram","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.665021Z","iopub.execute_input":"2023-03-28T01:39:59.665572Z","iopub.status.idle":"2023-03-28T01:39:59.683479Z","shell.execute_reply.started":"2023-03-28T01:39:59.665522Z","shell.execute_reply":"2023-03-28T01:39:59.682190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.3 Test function with a sample","metadata":{}},{"cell_type":"code","source":"# test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\n# test_samples\ntest_paths = glob.glob('/kaggle/input/birdclef-2023/test_soundscapes/*ogg')\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg',''))\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.684734Z","iopub.execute_input":"2023-03-28T01:39:59.685437Z","iopub.status.idle":"2023-03-28T01:39:59.708322Z","shell.execute_reply.started":"2023-03-28T01:39:59.685392Z","shell.execute_reply":"2023-03-28T01:39:59.706843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preprocessing(torchaudio.load(test_samples[0])).shape","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.709913Z","iopub.execute_input":"2023-03-28T01:39:59.711038Z","iopub.status.idle":"2023-03-28T01:39:59.716136Z","shell.execute_reply.started":"2023-03-28T01:39:59.710996Z","shell.execute_reply":"2023-03-28T01:39:59.714886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Load Model (CPU Only)","metadata":{}},{"cell_type":"code","source":"class GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6):\n        super(GeM, self).__init__()\n        self.p = nn.Parameter(torch.ones(1)*p)\n        self.eps = eps\n\n    def forward(self, x):\n        return self.gem(x, p=self.p, eps=self.eps)\n        \n    def gem(self, x, p=3, eps=1e-6):\n        return F.avg_pool2d(x.clamp(min=eps).pow(p), (x.size(-2), x.size(-1))).pow(1./p)\n        \n    def __repr__(self):\n        return self.__class__.__name__ + \\\n                '(' + 'p=' + '{:.4f}'.format(self.p.data.tolist()[0]) + \\\n                ', ' + 'eps=' + str(self.eps) + ')'","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.720508Z","iopub.execute_input":"2023-03-28T01:39:59.720997Z","iopub.status.idle":"2023-03-28T01:39:59.730331Z","shell.execute_reply.started":"2023-03-28T01:39:59.720957Z","shell.execute_reply":"2023-03-28T01:39:59.729366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BirdCLEFModel(nn.Module):\n    def __init__(self, model_name, embedding_size, pretrained=True):\n        super(BirdCLEFModel, self).__init__()\n        self.model = timm.create_model(model_name)\n        in_features = self.model.classifier.in_features\n        self.model.classifier = nn.Identity()\n        self.model.global_pool = nn.Identity()\n        self.pooling = GeM()\n        self.embedding = nn.Linear(in_features, embedding_size)\n        self.fc = nn.Linear(embedding_size, 264)\n\n    def forward(self, images):\n        features = self.model(images)\n        pooled_features = self.pooling(features).flatten(1)\n        embedding = self.embedding(pooled_features)\n        output = self.fc(embedding)\n        return output","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.731644Z","iopub.execute_input":"2023-03-28T01:39:59.732432Z","iopub.status.idle":"2023-03-28T01:39:59.742822Z","shell.execute_reply.started":"2023-03-28T01:39:59.732393Z","shell.execute_reply":"2023-03-28T01:39:59.741883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/baseline-inference-weight/CMAP0.6635_epoch8.bin'\nmodel = BirdCLEFModel(\"tf_efficientnet_b0_ns\", 768)\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel.load_state_dict(torch.load(PATH, map_location=torch.device('cpu') ))\nmyModel = model.to(device)\nnext(myModel.parameters()).device","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.744513Z","iopub.execute_input":"2023-03-28T01:39:59.745204Z","iopub.status.idle":"2023-03-28T01:39:59.974441Z","shell.execute_reply.started":"2023-03-28T01:39:59.745163Z","shell.execute_reply":"2023-03-28T01:39:59.972823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Mapping Classes to Column Name  ","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\nprint('data shape:',meta_df.shape)\nmeta_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:39:59.976130Z","iopub.execute_input":"2023-03-28T01:39:59.976900Z","iopub.status.idle":"2023-03-28T01:40:00.072388Z","shell.execute_reply.started":"2023-03-28T01:39:59.976849Z","shell.execute_reply":"2023-03-28T01:40:00.070992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder().fit(meta_df['primary_label'])\ncompetition_classes = le.classes_\ncompetition_classes","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.074669Z","iopub.execute_input":"2023-03-28T01:40:00.075582Z","iopub.status.idle":"2023-03-28T01:40:00.087755Z","shell.execute_reply.started":"2023-03-28T01:40:00.075526Z","shell.execute_reply":"2023-03-28T01:40:00.085944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"forced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.089141Z","iopub.execute_input":"2023-03-28T01:40:00.089578Z","iopub.status.idle":"2023-03-28T01:40:00.101317Z","shell.execute_reply.started":"2023-03-28T01:40:00.089521Z","shell.execute_reply":"2023-03-28T01:40:00.100198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.104472Z","iopub.execute_input":"2023-03-28T01:40:00.104973Z","iopub.status.idle":"2023-03-28T01:40:00.213468Z","shell.execute_reply.started":"2023-03-28T01:40:00.104911Z","shell.execute_reply":"2023-03-28T01:40:00.212082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, col in enumerate(sample_sub.drop(columns = ['row_id'], axis = 1).columns):\n    if col != competition_classes[idx]:\n        print('Not fit class!')\nprint('If dont have any log, all fit')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.215245Z","iopub.execute_input":"2023-03-28T01:40:00.216002Z","iopub.status.idle":"2023-03-28T01:40:00.227710Z","shell.execute_reply.started":"2023-03-28T01:40:00.215959Z","shell.execute_reply":"2023-03-28T01:40:00.226489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Make Predictions","metadata":{}},{"cell_type":"markdown","source":"## 5.1 Function to predict each data sample in test set","metadata":{}},{"cell_type":"code","source":"# from tqdm import tqdm \n# import numpy as np\n\n# def predict_for_sample(test_df, sample_submission):  \n#     ids = []\n#     preds = np.empty(shape=(0, 264), dtype='float32')\n#     start, end = 0, 5 \n#     for filepath in tqdm(test_df.filepath.tolist(), 'test'):\n#         filename = filepath.split('/')[-1].replace('.ogg','')\n#         file_id = filepath.split('.ogg')[0].split('/')[-1]\n\n#         data, rat = torchaudio.load(filepath)\n#         for i in range(int(data.shape[1]/(CFG.rate*5))):\n#             chunk = (data[:, start*rat:end*rat - 1], rat)\n# #             chunk = np.expand_dims(preprocessing(chunk), axis=0)\n#             chunk = preprocessing(chunk).unsqueeze(0)\n#             rec_preds = myModel(chunk)\n#             rec_preds = torch.nn.functional.softmax(rec_preds).detach().numpy()\n\n#             # set the appropriate row in the sample submission\n#             #  sample_submission.loc[sample_submission.row_id == file, competition_classes] = probabilities\n#             rec_ids = [f'{filename}_{(i+1)*5}']\n#             ids += rec_ids \n#             preds = np.concatenate([preds, rec_preds], axis=0)\n\n#             start += 5 \n#             end += 5\n    \n#     return preds, ids\n        ","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.229038Z","iopub.execute_input":"2023-03-28T01:40:00.229411Z","iopub.status.idle":"2023-03-28T01:40:00.238071Z","shell.execute_reply.started":"2023-03-28T01:40:00.229374Z","shell.execute_reply":"2023-03-28T01:40:00.236319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ids = []\n# preds = np.empty(shape=(0, 264), dtype='float32')\n# for filepath in test_df.filepath.tolist():\n#     filename = filepath.split('/')[-1].replace('.ogg','')\n#     file_id = filepath.split('.ogg')[0].split('/')[-1]\n\n#     data, rat = torchaudio.load(filepath)\n#     for i in tqdm(range(int(data.shape[1]/(CFG.rate*5)))):\n#         chunk = (data[:, 5*rat*i:5*rat*(i+1)], rat)\n# #         chunk = np.expand_dims(preprocessing(chunk), axis=0)\n#         chunk = preprocessing(chunk).unsqueeze(0)\n#         rec_preds = myModel(chunk)\n#         rec_preds = torch.nn.functional.softmax(rec_preds).detach().numpy()\n\n#         # set the appropriate row in the sample submission\n#         #  sample_submission.loc[sample_submission.row_id == file, competition_classes] = probabilities\n#         rec_ids = [f'{filename}_{(i+1)*5}']\n#         ids += rec_ids \n#         preds = np.concatenate([preds, rec_preds], axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.239946Z","iopub.execute_input":"2023-03-28T01:40:00.241384Z","iopub.status.idle":"2023-03-28T01:40:00.254958Z","shell.execute_reply.started":"2023-03-28T01:40:00.241335Z","shell.execute_reply":"2023-03-28T01:40:00.253628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\npreds = np.empty(shape=(0, 264), dtype='float32')\nstart, end = 0, 5 \nfor filepath in test_df.filepath.tolist():\n    filename = filepath.split('/')[-1].replace('.ogg','')\n    file_id = filepath.split('.ogg')[0].split('/')[-1]\n\n    data, rat = librosa.load(filepath, sr=32000)\n    for i in tqdm(range(int(data.shape[0]/(CFG.rate*5)))):\n        chunk = (np.expand_dims(data[5*rat*i:5*rat*(i+1)], axis=0), rat)\n        chunk = np.expand_dims(preprocessing(chunk), axis=0)\n#         chunk = preprocessing(chunk).unsqueeze(0)\n        rec_preds = myModel(torch.tensor(chunk, dtype=torch.float32))\n        rec_preds = torch.nn.functional.softmax(rec_preds).detach().numpy()\n\n        # set the appropriate row in the sample submission\n        #  sample_submission.loc[sample_submission.row_id == file, competition_classes] = probabilities\n        rec_ids = [f'{filename}_{(i+1)*5}']\n        ids += rec_ids \n        preds = np.concatenate([preds, rec_preds], axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:00.257662Z","iopub.execute_input":"2023-03-28T01:40:00.258099Z","iopub.status.idle":"2023-03-28T01:40:22.072300Z","shell.execute_reply.started":"2023-03-28T01:40:00.258058Z","shell.execute_reply":"2023-03-28T01:40:22.070190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5.2 PUT IT ALL TOGETHER","metadata":{}},{"cell_type":"code","source":"# for sample_filename in test_samples:\n#     predict_for_sample(sample_filename, sample_sub)\n# sample_sub\nimport pandas as pd \n\n# preds, ids = predict_for_sample(test_df, sample_sub)\npred_df = pd.DataFrame(ids, columns=['row_id'])\npred_df.loc[:, competition_classes.tolist()] = preds\npred_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:22.073744Z","iopub.execute_input":"2023-03-28T01:40:22.074118Z","iopub.status.idle":"2023-03-28T01:40:22.177750Z","shell.execute_reply.started":"2023-03-28T01:40:22.074082Z","shell.execute_reply":"2023-03-28T01:40:22.175797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:22.179949Z","iopub.execute_input":"2023-03-28T01:40:22.181002Z","iopub.status.idle":"2023-03-28T01:40:22.190210Z","shell.execute_reply.started":"2023-03-28T01:40:22.180953Z","shell.execute_reply":"2023-03-28T01:40:22.188857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df","metadata":{"execution":{"iopub.status.busy":"2023-03-28T01:40:22.191719Z","iopub.execute_input":"2023-03-28T01:40:22.192137Z","iopub.status.idle":"2023-03-28T01:40:22.226201Z","shell.execute_reply.started":"2023-03-28T01:40:22.192099Z","shell.execute_reply":"2023-03-28T01:40:22.224950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}