{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"[birdclef-2023-pytorch-lightning-inference](https://www.kaggle.com/code/nischaydnk/birdclef-2023-pytorch-lightning-inference)\n\n[birdclef-2023-pytorch-lightning-training-w-cmap](https://www.kaggle.com/code/nischaydnk/birdclef-2023-pytorch-lightning-training-w-cmap)\n\n[audio-deep-learning-made-simple-sound-classification-step-by-step](https://towardsdatascience.com/audio-deep-learning-made-simple-sound-classification-step-by-step-cebc936bbe5)\n\n[signal_framing](https://superkogito.github.io/blog/2020/01/25/signal_framing.html)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torchaudio\nfrom pathlib import Path\nfrom sklearn import preprocessing\nfrom torch.nn import init\nfrom torch.utils.data import DataLoader, Dataset, random_split\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:51.740632Z","iopub.execute_input":"2023-04-10T03:33:51.741379Z","iopub.status.idle":"2023-04-10T03:33:56.189922Z","shell.execute_reply.started":"2023-04-10T03:33:51.741334Z","shell.execute_reply":"2023-04-10T03:33:56.188372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"download_path = '/kaggle/input/'\n\nmetadata_file = download_path + '/birdclef-2023/train_metadata.csv'\ndf = pd.read_csv(metadata_file)\n\ndf['relative_path'] = 'train_audio/' + df['filename'].astype(str)\n\ndf = df[['relative_path', 'primary_label']]\n\nbirds = list(pd.get_dummies(df['primary_label']).columns)\n\nbirds = np.transpose(birds)\n\nle = preprocessing.LabelEncoder()\n\ndf[['classID']] = df[['primary_label']].apply(le.fit_transform)\ndel df['primary_label']\n\n#df.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-10T03:33:56.192732Z","iopub.execute_input":"2023-04-10T03:33:56.193860Z","iopub.status.idle":"2023-04-10T03:33:56.371202Z","shell.execute_reply.started":"2023-04-10T03:33:56.193816Z","shell.execute_reply":"2023-04-10T03:33:56.369813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#birds.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.373047Z","iopub.execute_input":"2023-04-10T03:33:56.373807Z","iopub.status.idle":"2023-04-10T03:33:56.379888Z","shell.execute_reply.started":"2023-04-10T03:33:56.373757Z","shell.execute_reply":"2023-04-10T03:33:56.378439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[['classID']].nunique(),df.classID.unique()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.383001Z","iopub.execute_input":"2023-04-10T03:33:56.384018Z","iopub.status.idle":"2023-04-10T03:33:56.390179Z","shell.execute_reply.started":"2023-04-10T03:33:56.383972Z","shell.execute_reply":"2023-04-10T03:33:56.388992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math, random\nimport torch\nimport torchaudio\nfrom torchaudio import transforms\nfrom IPython.display import Audio\n\nclass AudioUtil():\n  # ----------------------------\n  # Load an audio file. Return the signal as a tensor and the sample rate\n  # ----------------------------\n  @staticmethod\n  def open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)\n\n  # ----------------------------\n  # Convert the given audio to the desired number of channels\n  # ----------------------------\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = torch.cat([sig, sig])\n\n    return ((resig, sr))\n\n  # ----------------------------\n  # Since Resample applies to a single channel, we resample one channel at a time\n  # ----------------------------\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n      resig = torch.cat([resig, retwo])\n\n    return ((resig, newsr))\n\n# ----------------------------\n  # Pad (or truncate) the signal to a fixed length 'max_ms' in milliseconds\n  # ----------------------------\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = torch.zeros((num_rows, pad_begin_len))\n      pad_end = torch.zeros((num_rows, pad_end_len))\n\n      sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  # ----------------------------\n  # Shifts the signal to the left or right by some percent. Values at the end\n  # are 'wrapped around' to the start of the transformed signal.\n  # ----------------------------\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (sig.roll(shift_amt), sr)\n\n  # ----------------------------\n  # Generate a Spectrogram\n  # ----------------------------\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n    # Convert to decibels\n    spec = transforms.AmplitudeToDB(top_db=top_db)(spec)\n    return (spec)\n\n  # ----------------------------\n  # Augment the Spectrogram by masking out some sections of it in both the frequency\n  # dimension (ie. horizontal bars) and the time dimension (vertical bars) to prevent\n  # overfitting and to help the model generalise better. The masked sections are\n  # replaced with the mean value.\n  # ----------------------------\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n      aug_spec = transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n      aug_spec = transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.391851Z","iopub.execute_input":"2023-04-10T03:33:56.392596Z","iopub.status.idle":"2023-04-10T03:33:56.414454Z","shell.execute_reply.started":"2023-04-10T03:33:56.392557Z","shell.execute_reply":"2023-04-10T03:33:56.412956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let’s walk through the steps as our data gets transformed, starting with an audio file:\n\n* The audio from the file gets loaded into a Numpy array of shape (num_channels, num_samples). Most of the audio is sampled at 44.1kHz and is about 4 seconds in duration, resulting in 44,100 * 4 = 176,400 samples. If the audio has 1 channel, the shape of the array will be (1, 176,400). Similarly, audio of 4 seconds duration with 2 channels and sampled at 48kHz will have 192,000 samples and a shape of (2, 192,000).\n\n* Since the channels and sampling rates of each audio are different, the next two transforms resample the audio to a standard 44.1kHz and to a standard 2 channels.\n\n* Since some audio clips might be more or less than 4 seconds, we also standardize the audio duration to a fixed length of 4 seconds. Now arrays for all items have the same shape of (2, 176,400)\n\n* The Time Shift data augmentation now randomly shifts each audio sample forward or backward. The shapes are unchanged.\n\n* The augmented audio is now converted into a Mel Spectrogram, resulting in a shape of (num_channels, Mel freq_bands, time_steps) = (2, 64, 344)\n\n* The SpecAugment data augmentation now randomly applies Time and Frequency Masks to the Mel Spectrograms. The shapes are unchanged.\nThus, each batch will have two tensors, one for the X feature data containing the Mel Spectrograms and the other for the y target labels containing numeric Class IDs. The batches are picked randomly from the training data for each training epoch.\n\n* Each batch has a shape of (batch_sz, num_channels, Mel freq_bands, time_steps)","metadata":{}},{"cell_type":"code","source":"class SoundDS(Dataset):\n  def __init__(self, df, data_path):\n    self.df = df\n    self.data_path = str(data_path)\n    self.duration = 4000\n    self.sr = 44100\n    self.channel = 2\n    self.shift_pct = 0.4\n    \n  def __len__(self):\n    return len(self.df) \n\n  # self.audio_length = self.duration*self.sr\n    \n  def __getitem__(self, idx):\n    audio_file = self.data_path + self.df.loc[idx, 'relative_path']\n    class_id = self.df.loc[idx, 'classID']\n\n    aud = AudioUtil.open(audio_file)\n    reaud = AudioUtil.resample(aud, self.sr)\n    rechan = AudioUtil.rechannel(reaud, self.channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, self.duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, self.shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n\n    return aug_sgram, class_id","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.416253Z","iopub.execute_input":"2023-04-10T03:33:56.416764Z","iopub.status.idle":"2023-04-10T03:33:56.429357Z","shell.execute_reply.started":"2023-04-10T03:33:56.416721Z","shell.execute_reply":"2023-04-10T03:33:56.428403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = download_path + 'birdclef-2023/'\nmyds = SoundDS(df, data_path)\n\n\nnum_items = len(myds)\nnum_train = round(num_items * 0.8)\nnum_val = num_items - num_train\ntrain_ds, val_ds = random_split(myds, [num_train, num_val])\n\ntrain_dl = torch.utils.data.DataLoader(train_ds, batch_size=16, shuffle=True)\nval_dl = torch.utils.data.DataLoader(val_ds, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.433081Z","iopub.execute_input":"2023-04-10T03:33:56.433988Z","iopub.status.idle":"2023-04-10T03:33:56.456808Z","shell.execute_reply.started":"2023-04-10T03:33:56.433934Z","shell.execute_reply":"2023-04-10T03:33:56.455281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport sklearn.metrics\n\ndef padded_cmap(solution, submission, padding_factor=5):\n    solution = solution.drop(['row_id'], axis=1, errors='ignore')\n    submission = submission.drop(['row_id'], axis=1, errors='ignore')\n    new_rows = []\n    for i in range(padding_factor):\n        new_rows.append([1 for i in range(len(solution.columns))])\n    new_rows = pd.DataFrame(new_rows)\n    new_rows.columns = solution.columns\n    padded_solution = pd.concat([solution, new_rows]).reset_index(drop=True).copy()\n    padded_submission = pd.concat([submission, new_rows]).reset_index(drop=True).copy()\n    score = sklearn.metrics.average_precision_score(\n        padded_solution.values,\n        padded_submission.values,\n        average='macro',\n    )\n    return score","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.458605Z","iopub.execute_input":"2023-04-10T03:33:56.459244Z","iopub.status.idle":"2023-04-10T03:33:56.532491Z","shell.execute_reply.started":"2023-04-10T03:33:56.459190Z","shell.execute_reply":"2023-04-10T03:33:56.531029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioClassifier (nn.Module):\n    # ----------------------------\n    # Build the model architecture\n    # ----------------------------\n    def __init__(self):\n        super().__init__()\n        conv_layers = []\n\n        # First Convolution Block with Relu and Batch Norm. Use Kaiming Initialization\n        self.conv1 = nn.Conv2d(2, 8, kernel_size=(5, 5), stride=(2, 2), padding=(2, 2))\n        self.relu1 = nn.ReLU()\n        self.bn1 = nn.BatchNorm2d(8)\n        init.kaiming_normal_(self.conv1.weight, a=0.1)\n        self.conv1.bias.data.zero_()\n        conv_layers += [self.conv1, self.relu1, self.bn1]\n\n        # Second Convolution Block\n        self.conv2 = nn.Conv2d(8, 16, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu2 = nn.ReLU()\n        self.bn2 = nn.BatchNorm2d(16)\n        init.kaiming_normal_(self.conv2.weight, a=0.1)\n        self.conv2.bias.data.zero_()\n        conv_layers += [self.conv2, self.relu2, self.bn2]\n\n        # Second Convolution Block\n        self.conv3 = nn.Conv2d(16, 32, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu3 = nn.ReLU()\n        self.bn3 = nn.BatchNorm2d(32)\n        init.kaiming_normal_(self.conv3.weight, a=0.1)\n        self.conv3.bias.data.zero_()\n        conv_layers += [self.conv3, self.relu3, self.bn3]\n\n        # Second Convolution Block\n        self.conv4 = nn.Conv2d(32, 64, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu4 = nn.ReLU()\n        self.bn4 = nn.BatchNorm2d(64)\n        init.kaiming_normal_(self.conv4.weight, a=0.1)\n        self.conv4.bias.data.zero_()\n        conv_layers += [self.conv4, self.relu4, self.bn4]\n\n        # Linear Classifier\n        self.ap = nn.AdaptiveAvgPool2d(output_size=1)\n        self.lin = nn.Linear(in_features=64, out_features=264)\n\n        # Wrap the Convolutional Blocks\n        self.conv = nn.Sequential(*conv_layers)\n \n    # ----------------------------\n    # Forward pass computations\n    # ----------------------------\n    def forward(self, x):\n        # Run the convolutional blocks\n        x = self.conv(x)\n\n        # Adaptive pool and flatten for input to linear layer\n        x = self.ap(x)\n        x = x.view(x.shape[0], -1)\n\n        # Linear layer\n        x = self.lin(x)\n\n        # Final output\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.534359Z","iopub.execute_input":"2023-04-10T03:33:56.534756Z","iopub.status.idle":"2023-04-10T03:33:56.552491Z","shell.execute_reply.started":"2023-04-10T03:33:56.534715Z","shell.execute_reply":"2023-04-10T03:33:56.551129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"myModel = AudioClassifier()\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmyModel = myModel.to(device)\n\nnext(myModel.parameters()).device","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.555991Z","iopub.execute_input":"2023-04-10T03:33:56.556812Z","iopub.status.idle":"2023-04-10T03:33:56.614357Z","shell.execute_reply.started":"2023-04-10T03:33:56.556760Z","shell.execute_reply":"2023-04-10T03:33:56.612880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n# ----------------------------\n# Training Loop\n# ----------------------------\ndef training(model, train_dl, num_epochs):\n  # Loss Function, Optimizer and Scheduler\n  criterion = nn.CrossEntropyLoss()\n  optimizer = torch.optim.Adam(model.parameters(),lr=0.001)\n  scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, max_lr=0.001,\n                                                steps_per_epoch=int(len(train_dl)),\n                                                epochs=num_epochs,\n                                                anneal_strategy='linear')\n\n  # Repeat for each epoch\n  for epoch in range(num_epochs):\n    running_loss = 0.0\n    correct_prediction = 0\n    total_prediction = 0\n\n    # Repeat for each batch in the training set\n    for i, data in enumerate(train_dl):\n        # Get the input features and target labels, and put them on the GPU\n        inputs, labels = data[0].to(device), data[1].to(device)\n\n        # Normalize the inputs\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n\n        # Zero the parameter gradients\n        optimizer.zero_grad()\n\n        # forward + backward + optimize\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        scheduler.step()\n\n        # Keep stats for Loss and Accuracy\n        running_loss += loss.item()\n\n        # Get the predicted class with the highest score\n        _, prediction = torch.max(outputs,1)\n        # Count of predictions that matched the target label\n        correct_prediction += (prediction == labels).sum().item()\n        total_prediction += prediction.shape[0]\n\n        if i % 10 == 0:    # print every 10 mini-batches\n           print('[%d, %5d] loss: %.3f' % (epoch + 1, i + 1, running_loss / 10))\n    \n    # Print stats at the end of the epoch\n    num_batches = len(train_dl)\n    avg_loss = running_loss / num_batches\n    acc = correct_prediction/total_prediction\n    print(f'Epoch: {epoch}, Loss: {avg_loss:.2f}, Accuracy: {acc:.2f}')\n\n  print('Finished Training')\n  \nnum_epochs=2   # Just for demo, adjust this higher.\ntraining(myModel, train_dl, num_epochs)\n#23/03\n#Epoch: 0, Loss: 5.18, Accuracy: 0.04\n#Epoch: 1, Loss: 4.56, Accuracy: 0.09\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T03:33:56.616495Z","iopub.execute_input":"2023-04-10T03:33:56.617342Z","iopub.status.idle":"2023-04-10T04:18:28.491600Z","shell.execute_reply.started":"2023-04-10T03:33:56.617289Z","shell.execute_reply":"2023-04-10T04:18:28.490209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.save(myModel.state_dict(), \"r1001_birdclef2023.v.1.0.pth\")","metadata":{"execution":{"iopub.status.busy":"2023-04-10T04:35:15.061352Z","iopub.execute_input":"2023-04-10T04:35:15.062820Z","iopub.status.idle":"2023-04-10T04:35:15.079443Z","shell.execute_reply.started":"2023-04-10T04:35:15.062738Z","shell.execute_reply":"2023-04-10T04:35:15.077999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"myModel.load_state_dict(torch.load('r1001_birdclef2023.v.1.0.pth'))\n# evaluate on the validation set after each epoch\nmyModel.eval()\n\nwith torch.no_grad():\n\n    correct = 0\n    total = 0\n    # Repeat for each batch in the validation set\n    for i, data in enumerate(val_dl):\n        # Get the input features and target labels, and put them on the GPU\n        inputs, labels = data[0].to(device), data[1].to(device)\n        # Normalize the inputs\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n        \n        outputs = myModel(inputs)\n        _, predicted = torch.max(outputs, 1)\n        total += predicted.shape[0]\n        correct += (predicted == labels).sum().item()\n        \n        output_val = outputs.sigmoid().cpu().detach().numpy()\n        target_one_hot = torch.eye(264)[labels]\n        target_val = target_one_hot.numpy()\n\n        val_df = pd.DataFrame(target_val, columns = birds)\n        pred_df = pd.DataFrame(output_val, columns = birds)\n        \n        avg_score = padded_cmap(val_df, pred_df, padding_factor = 3)\n        print(f\"cmAP score pad 3: {avg_score}\")\n    accuracy = 100 * correct / total\n    print(f\"Validation Accuracy: {accuracy:.2f}%\")\nmyModel.train();\n# 23/03 - Validation Accuracy: 8.53%","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:03:17.840959Z","iopub.execute_input":"2023-04-07T07:03:17.841870Z","iopub.status.idle":"2023-04-07T07:09:59.085798Z","shell.execute_reply.started":"2023-04-07T07:03:17.841802Z","shell.execute_reply":"2023-04-07T07:09:59.084411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#(label_df == 1).any(axis=1).sum(), (predicted_df == 1).any(axis=1).sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:09:59.087715Z","iopub.execute_input":"2023-04-07T07:09:59.089716Z","iopub.status.idle":"2023-04-07T07:09:59.095780Z","shell.execute_reply.started":"2023-04-07T07:09:59.089665Z","shell.execute_reply":"2023-04-07T07:09:59.094673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label_df.columns[label_df.eq(1).any()],predicted_df.columns[predicted_df.eq(1).any()]","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:09:59.097463Z","iopub.execute_input":"2023-04-07T07:09:59.099648Z","iopub.status.idle":"2023-04-07T07:09:59.110686Z","shell.execute_reply.started":"2023-04-07T07:09:59.099566Z","shell.execute_reply":"2023-04-07T07:09:59.109697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_true, y_scores","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:09:59.112451Z","iopub.execute_input":"2023-04-07T07:09:59.113208Z","iopub.status.idle":"2023-04-07T07:09:59.133983Z","shell.execute_reply.started":"2023-04-07T07:09:59.113161Z","shell.execute_reply":"2023-04-07T07:09:59.132314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#labels, birds","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:09:59.135847Z","iopub.execute_input":"2023-04-07T07:09:59.136249Z","iopub.status.idle":"2023-04-07T07:09:59.149560Z","shell.execute_reply.started":"2023-04-07T07:09:59.136206Z","shell.execute_reply":"2023-04-07T07:09:59.148048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset\nimport numpy as np\n\nclass SoundTestDS(Dataset):\n  def __init__(self, df, data_path):\n    self.df = df\n    self.data_path = str(data_path)\n    self.duration = 4000\n    self.sr = 44100\n    self.channel = 2\n    self.shift_pct = 0.4\n    \n  def __len__(self):\n    return len(self.df) \n\n  def audio_to_image(self, aud):\n    reaud = AudioUtil.resample(aud, 44100)\n    rechan = AudioUtil.rechannel(reaud, 2)\n    dur_aud = AudioUtil.pad_trunc(rechan,4000)\n    shift_aud = AudioUtil.time_shift(dur_aud, 0.4)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n    return aug_sgram\n\n  def split_audio(self, aud, duration, step=None):\n    sig, sr = aud\n    audio_length = sr * 5\n    audios = []\n    step = None or audio_length\n    for i in range(audio_length, len(sig[0]) + step, step):\n        start = max(0, i - audio_length)\n        end = start + audio_length\n        audios.append(sig[0][start:end])\n\n    if len(audios[-1]) < audio_length:\n        audios = audios[:-1]\n        \n    images = [self.audio_to_image((audio.unsqueeze(0),sr)) for audio in audios]\n    images = np.stack(images)\n    return images\n\n  def __getitem__(self, idx):\n    audio_file = self.data_path + self.df.loc[idx, 'relative_path']\n    aud = AudioUtil.open(audio_file)\n    \n    return self.split_audio(aud,5)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T04:35:28.881328Z","iopub.execute_input":"2023-04-10T04:35:28.881766Z","iopub.status.idle":"2023-04-10T04:35:28.897172Z","shell.execute_reply.started":"2023-04-10T04:35:28.881732Z","shell.execute_reply":"2023-04-10T04:35:28.895379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import soundfile as sf\n# import os\n\n# # define the path to the audio file\n# audio_file_path = \"/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg\"\n\n# # define the duration of each segment in seconds\n# segment_duration = 60\n\n# # read the audio file\n# audio_data, samplerate = sf.read(audio_file_path)\n\n# # calculate the total number of segments\n# total_segments = int(len(audio_data) / (segment_duration * samplerate))\n\n# # create a directory to store the segments\n# output_dir = \"/kaggle/working/testaud\"\n# if not os.path.exists(output_dir):\n#     os.makedirs(output_dir)\n\n# # split the audio file into segments and save each segment as a separate file\n# for i in range(total_segments):\n#     segment_start = i * segment_duration * samplerate\n#     segment_end = segment_start + segment_duration * samplerate\n#     segment_data = audio_data[segment_start:segment_end]\n#     segment_file_name = os.path.join(output_dir, f\"segment_{i}.ogg\")\n#     sf.write(segment_file_name, segment_data, samplerate)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.671745Z","iopub.execute_input":"2023-04-10T07:29:45.672613Z","iopub.status.idle":"2023-04-10T07:29:45.679622Z","shell.execute_reply.started":"2023-04-10T07:29:45.672561Z","shell.execute_reply":"2023-04-10T07:29:45.677949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_test = pd.DataFrame(\n     [(path.stem, path.parent.absolute().stem + '/' + path.stem + '.ogg',-1) for path in Path(download_path + 'birdclef-2023/test_soundscapes/').glob(\"*.ogg\")],\n    columns = [\"filename\", \"relative_path\", \"classID\" ]\n)\n# df_test = pd.DataFrame(\n#      [(path.stem, path.parent.absolute().stem + '/' + path.stem + '.ogg',-1) for path in Path('/kaggle/working/testaud').glob(\"*.ogg\")],\n#     columns = [\"filename\", \"relative_path\", \"classID\" ]\n# )\n# print(df_test.shape)\n# df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.690832Z","iopub.execute_input":"2023-04-10T07:29:45.692129Z","iopub.status.idle":"2023-04-10T07:29:45.701986Z","shell.execute_reply.started":"2023-04-10T07:29:45.692079Z","shell.execute_reply":"2023-04-10T07:29:45.700968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_dataset = SoundTestDS(df_test[['relative_path','classID']], '/kaggle/working/')#data_path)\ntest_dataset = SoundTestDS(df_test[['relative_path','classID']], data_path)\ntest_dataloader = DataLoader(test_dataset, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.709007Z","iopub.execute_input":"2023-04-10T07:29:45.709910Z","iopub.status.idle":"2023-04-10T07:29:45.716879Z","shell.execute_reply.started":"2023-04-10T07:29:45.709837Z","shell.execute_reply":"2023-04-10T07:29:45.715855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset = train_ds\n# # Iterate through the dataset\n# for i in range(10):\n#     # Get the i-th sample from the dataset\n#     sample = dataset[i]\n\n#     # Extract the input (audio data) and target (label) from the sample\n#     input, target = sample\n\n#     # Do something with the input and target\n#     print(f'Sample {i}: Input shape: {input.shape}, Target: {target}')","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.729277Z","iopub.execute_input":"2023-04-10T07:29:45.730150Z","iopub.status.idle":"2023-04-10T07:29:45.736168Z","shell.execute_reply.started":"2023-04-10T07:29:45.730091Z","shell.execute_reply":"2023-04-10T07:29:45.734800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset = test_dataset\n# # Iterate through the dataset\n# for i in range(len(dataset)):\n#     # Get the i-th sample from the dataset\n#     sample = dataset[i]\n\n#     for j in range(len(sample)):\n    \n#         aud = sample[j]\n#         # Extract the input (audio data) and target (label) from the sample\n#         input, _ = aud\n\n#         # Do something with the input and target\n#         print(f'Sample {i}: Audion {j}: Input shape: {input.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.755067Z","iopub.execute_input":"2023-04-10T07:29:45.755967Z","iopub.status.idle":"2023-04-10T07:29:45.761364Z","shell.execute_reply.started":"2023-04-10T07:29:45.755860Z","shell.execute_reply":"2023-04-10T07:29:45.760148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load your trained model\nmyModel.load_state_dict(torch.load('r1001_birdclef2023.v.1.0.pth'))\n\n# Put the model in evaluation mode\nmyModel.eval()\n\n# Iterate over the test set and make predictions\npredictions = []\nwith torch.no_grad():\n    # Repeat for each batch in the test set\n    for  i, data in enumerate(test_dataloader):\n        # Get the input features and target labels, and put them on the GPU\n        row = []\n        for j in range(len(data)):\n            inputs = data[j].to(device)\n            # Normalize the inputs\n            inputs_m, inputs_s = inputs.mean(), inputs.std()\n            inputs = (inputs - inputs_m) / inputs_s\n\n            # Make a prediction on the waveform tensor\n            outputs = myModel(inputs)\n            row.append(outputs.sigmoid().cpu().detach().numpy())\n        predictions.append(row)\n        \nmyModel.train();","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:45.778431Z","iopub.execute_input":"2023-04-10T07:29:45.779247Z","iopub.status.idle":"2023-04-10T07:29:48.212530Z","shell.execute_reply.started":"2023-04-10T07:29:45.779204Z","shell.execute_reply":"2023-04-10T07:29:48.211311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(filenames), len(predictions[0]),len(predictions[0][0]), predictions[0][0].shape,len(predictions[0][0][1])","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:48.215037Z","iopub.execute_input":"2023-04-10T07:29:48.215554Z","iopub.status.idle":"2023-04-10T07:29:48.222017Z","shell.execute_reply.started":"2023-04-10T07:29:48.215500Z","shell.execute_reply":"2023-04-10T07:29:48.220717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(bird_cols)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:48.224071Z","iopub.execute_input":"2023-04-10T07:29:48.224602Z","iopub.status.idle":"2023-04-10T07:29:48.233667Z","shell.execute_reply.started":"2023-04-10T07:29:48.224534Z","shell.execute_reply":"2023-04-10T07:29:48.232222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = df_test.filename.values.tolist()\nbird_cols = list(birds)\nsub_df = pd.DataFrame(columns=['row_id']+bird_cols)\nsub_df[bird_cols] = sub_df[bird_cols].astype(np.float32)\nfor i, file in enumerate(filenames):\n    pred = predictions[0][i]\n    num_rows = len(pred)\n    row_ids = [f'{file}_{(j+1)*5}' for j in range(num_rows)]\n    df = pd.DataFrame(columns=['row_id']+bird_cols)\n    df['row_id'] = row_ids\n    df[bird_cols] = pred\n\n    sub_df = pd.concat([sub_df,df]).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:48.236266Z","iopub.execute_input":"2023-04-10T07:29:48.236798Z","iopub.status.idle":"2023-04-10T07:29:48.435047Z","shell.execute_reply.started":"2023-04-10T07:29:48.236756Z","shell.execute_reply":"2023-04-10T07:29:48.433312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len(sub_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:48.437231Z","iopub.execute_input":"2023-04-10T07:29:48.438756Z","iopub.status.idle":"2023-04-10T07:29:48.444430Z","shell.execute_reply.started":"2023-04-10T07:29:48.438702Z","shell.execute_reply":"2023-04-10T07:29:48.442965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub_df[0:13]","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:30:04.444851Z","iopub.execute_input":"2023-04-10T07:30:04.446268Z","iopub.status.idle":"2023-04-10T07:30:04.451860Z","shell.execute_reply.started":"2023-04-10T07:30:04.446202Z","shell.execute_reply":"2023-04-10T07:30:04.450250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-10T07:29:48.485714Z","iopub.execute_input":"2023-04-10T07:29:48.486496Z","iopub.status.idle":"2023-04-10T07:29:48.527521Z","shell.execute_reply.started":"2023-04-10T07:29:48.486451Z","shell.execute_reply":"2023-04-10T07:29:48.526150Z"},"trusted":true},"execution_count":null,"outputs":[]}]}