{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we'll use a pre-trained machine learning model to generate a submission to the [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023).  The goal of the competition is to identify Eastern African bird species by sound.","metadata":{}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport math, random\nimport librosa\nimport glob\nimport torch\nimport torchaudio\n\n\nimport csv\nimport io\nimport torch.nn as nn\n\nfrom torchaudio import transforms\nfrom IPython.display import Audio\nfrom pathlib import Path\nfrom sklearn import preprocessing\nfrom torch.nn import init\nfrom torch.utils.data import DataLoader, Dataset, random_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-26T03:50:15.945811Z","iopub.execute_input":"2023-04-26T03:50:15.946210Z","iopub.status.idle":"2023-04-26T03:50:15.954296Z","shell.execute_reply.started":"2023-04-26T03:50:15.946171Z","shell.execute_reply":"2023-04-26T03:50:15.952721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Explore the training data\n\nWe'll start by loading a couple of training examples and using the IPython.display.Audio module to play them!","metadata":{}},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:15.957124Z","iopub.execute_input":"2023-04-26T03:50:15.957918Z","iopub.status.idle":"2023-04-26T03:50:16.141065Z","shell.execute_reply.started":"2023-04-26T03:50:15.957867Z","shell.execute_reply":"2023-04-26T03:50:16.139503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.143545Z","iopub.execute_input":"2023-04-26T03:50:16.144322Z","iopub.status.idle":"2023-04-26T03:50:16.208711Z","shell.execute_reply.started":"2023-04-26T03:50:16.144266Z","shell.execute_reply":"2023-04-26T03:50:16.207350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.210287Z","iopub.execute_input":"2023-04-26T03:50:16.211481Z","iopub.status.idle":"2023-04-26T03:50:16.291300Z","shell.execute_reply.started":"2023-04-26T03:50:16.211408Z","shell.execute_reply":"2023-04-26T03:50:16.289750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Match the model's output with the bird species in the competition\n\nThe competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{}},{"cell_type":"code","source":"download_path = '/kaggle/input/'\n\nmetadata_file = download_path + '/birdclef-2023/train_metadata.csv'\ndf = pd.read_csv(metadata_file, nrows=1395)\n\ndf['relative_path'] = 'train_audio/' + df['filename'].astype(str)\n\ndf = df[['relative_path', 'primary_label']]\n\nclasses = sorted(df.primary_label.unique())\n# classes = list(pd.get_dummies(df['primary_label']).columns)\n\n# classes = np.transpose(classes)\n\nle = preprocessing.LabelEncoder()\n\ndf[['classID']] = df[['primary_label']].apply(le.fit_transform)\ndel df['primary_label']","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.294355Z","iopub.execute_input":"2023-04-26T03:50:16.294876Z","iopub.status.idle":"2023-04-26T03:50:16.318259Z","shell.execute_reply.started":"2023-04-26T03:50:16.294838Z","shell.execute_reply":"2023-04-26T03:50:16.317329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\n# train_metadata.head()\n# competition_classes = sorted(train_metadata.primary_label.unique())\n\ncompetition_classes = classes\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.319870Z","iopub.execute_input":"2023-04-26T03:50:16.320533Z","iopub.status.idle":"2023-04-26T03:50:16.329495Z","shell.execute_reply.started":"2023-04-26T03:50:16.320498Z","shell.execute_reply":"2023-04-26T03:50:16.328231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{}},{"cell_type":"code","source":"import torch.nn.functional as F\n\ndef frame(signal, frame_length, frame_step, pad_end=False, pad_value=1000, axis=-1):\n    \"\"\"\n    equivalent of tf.signal.frame\n    \"\"\"\n    signal_length = signal.shape[axis]\n    if pad_end:\n        frames_overlap = frame_length - frame_step\n        rest_samples = np.abs(signal_length - frames_overlap) % np.abs(frame_length - frames_overlap)\n        pad_size = int(frame_length - rest_samples)\n        if pad_size != 0:\n            pad_axis = [0] * signal.ndim\n            pad_axis[axis] = pad_size\n            # calculate the padding size needed for both ends\n            left_pad = pad_size // 2\n            right_pad = pad_size - left_pad\n            signal = F.pad(signal, (left_pad, right_pad), \"constant\",pad_value)\n    frames=signal.unfold(axis, frame_length, frame_step)\n    return frames","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.331256Z","iopub.execute_input":"2023-04-26T03:50:16.332040Z","iopub.status.idle":"2023-04-26T03:50:16.345671Z","shell.execute_reply.started":"2023-04-26T03:50:16.332003Z","shell.execute_reply":"2023-04-26T03:50:16.344621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    max_len = sample_rate//1000 * window_size_s\n\n    framed_audio = frame(audio_array, frame_length, hop_length,pad_value=max_len, pad_end=True)\n    #framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        # convert the NumPy array to a PyTorch tensor with a floating-point data type\n        audio_tensor = torch.from_numpy(waveform).float()\n        waveform = torchaudio.transforms.Resample(original_sample_rate, desired_sample_rate)(audio_tensor)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.347491Z","iopub.execute_input":"2023-04-26T03:50:16.347865Z","iopub.status.idle":"2023-04-26T03:50:16.359040Z","shell.execute_reply.started":"2023-04-26T03:50:16.347830Z","shell.execute_reply":"2023-04-26T03:50:16.358058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below we load one training sample - use the Audio function to listen to the samples inside the notebook!","metadata":{}},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.360419Z","iopub.execute_input":"2023-04-26T03:50:16.361589Z","iopub.status.idle":"2023-04-26T03:50:16.533625Z","shell.execute_reply.started":"2023-04-26T03:50:16.361550Z","shell.execute_reply":"2023-04-26T03:50:16.531877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Build the Model","metadata":{}},{"cell_type":"code","source":"class AudioUtil():\n  # ----------------------------\n  # Load an audio file. Return the signal as a tensor and the sample rate\n  # ----------------------------\n  @staticmethod\n  def open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)\n\n  # ----------------------------\n  # Convert the given audio to the desired number of channels\n  # ----------------------------\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = torch.cat([sig, sig])\n\n    return ((resig, sr))\n\n  # ----------------------------\n  # Since Resample applies to a single channel, we resample one channel at a time\n  # ----------------------------\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n      resig = torch.cat([resig, retwo])\n\n    return ((resig, newsr))\n\n# ----------------------------\n  # Pad (or truncate) the signal to a fixed length 'max_ms' in milliseconds\n  # ----------------------------\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = torch.zeros((num_rows, pad_begin_len))\n      pad_end = torch.zeros((num_rows, pad_end_len))\n\n      sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  # ----------------------------\n  # Shifts the signal to the left or right by some percent. Values at the end\n  # are 'wrapped around' to the start of the transformed signal.\n  # ----------------------------\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (sig.roll(shift_amt), sr)\n\n  # ----------------------------\n  # Generate a Spectrogram\n  # ----------------------------\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n    # Convert to decibels\n    spec = transforms.AmplitudeToDB(top_db=top_db)(spec)\n    return (spec)\n\n  # ----------------------------\n  # Augment the Spectrogram by masking out some sections of it in both the frequency\n  # dimension (ie. horizontal bars) and the time dimension (vertical bars) to prevent\n  # overfitting and to help the model generalise better. The masked sections are\n  # replaced with the mean value.\n  # ----------------------------\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n      aug_spec = transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n      aug_spec = transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.535672Z","iopub.execute_input":"2023-04-26T03:50:16.536787Z","iopub.status.idle":"2023-04-26T03:50:16.565504Z","shell.execute_reply.started":"2023-04-26T03:50:16.536738Z","shell.execute_reply":"2023-04-26T03:50:16.564352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SoundDS(Dataset):\n  def __init__(self, df, data_path):\n    self.df = df\n    self.data_path = str(data_path)\n    self.duration = 4000\n    self.sr = 44100\n    self.channel = 2\n    self.shift_pct = 0.4\n    \n  def __len__(self):\n    return len(self.df) \n\n  # self.audio_length = self.duration*self.sr\n    \n  def __getitem__(self, idx):\n    audio_file = self.data_path + self.df.loc[idx, 'relative_path']\n    class_id = self.df.loc[idx, 'classID']\n   \n    aud = AudioUtil.open(audio_file)\n    reaud = AudioUtil.resample(aud, self.sr)\n    rechan = AudioUtil.rechannel(reaud, self.channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, self.duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, self.shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n\n    return aug_sgram, class_id","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.572253Z","iopub.execute_input":"2023-04-26T03:50:16.573519Z","iopub.status.idle":"2023-04-26T03:50:16.586372Z","shell.execute_reply.started":"2023-04-26T03:50:16.573446Z","shell.execute_reply":"2023-04-26T03:50:16.585153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioClassifier (nn.Module):\n    # ----------------------------\n    # Build the model architecture\n    # ----------------------------\n    def __init__(self):\n        super().__init__()\n        conv_layers = []\n\n        # First Convolution Block with Relu and Batch Norm. Use Kaiming Initialization\n        self.conv1 = nn.Conv2d(2, 8, kernel_size=(5, 5), stride=(2, 2), padding=(2, 2))\n        self.relu1 = nn.ReLU()\n        self.bn1 = nn.BatchNorm2d(8)\n        init.kaiming_normal_(self.conv1.weight, a=0.1)\n        self.conv1.bias.data.zero_()\n        conv_layers += [self.conv1, self.relu1, self.bn1]\n\n        # Second Convolution Block\n        self.conv2 = nn.Conv2d(8, 16, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu2 = nn.ReLU()\n        self.bn2 = nn.BatchNorm2d(16)\n        init.kaiming_normal_(self.conv2.weight, a=0.1)\n        self.conv2.bias.data.zero_()\n        conv_layers += [self.conv2, self.relu2, self.bn2]\n\n        # Second Convolution Block\n        self.conv3 = nn.Conv2d(16, 32, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu3 = nn.ReLU()\n        self.bn3 = nn.BatchNorm2d(32)\n        init.kaiming_normal_(self.conv3.weight, a=0.1)\n        self.conv3.bias.data.zero_()\n        conv_layers += [self.conv3, self.relu3, self.bn3]\n\n        # Second Convolution Block\n        self.conv4 = nn.Conv2d(32, 64, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu4 = nn.ReLU()\n        self.bn4 = nn.BatchNorm2d(64)\n        init.kaiming_normal_(self.conv4.weight, a=0.1)\n        self.conv4.bias.data.zero_()\n        conv_layers += [self.conv4, self.relu4, self.bn4]\n\n        # Linear Classifier\n        self.ap = nn.AdaptiveAvgPool2d(output_size=1)\n        self.lin = nn.Linear(in_features=64, out_features=264)\n\n        # Wrap the Convolutional Blocks\n        self.conv = nn.Sequential(*conv_layers)\n \n    # ----------------------------\n    # Forward pass computations\n    # ----------------------------\n    def forward(self, x):\n        # Run the convolutional blocks\n        x = self.conv(x)\n\n        # Adaptive pool and flatten for input to linear layer\n        x = self.ap(x)\n        x = x.view(x.shape[0], -1)\n\n        # Linear layer\n        x = self.lin(x)\n\n        # Final output\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.588411Z","iopub.execute_input":"2023-04-26T03:50:16.589273Z","iopub.status.idle":"2023-04-26T03:50:16.615348Z","shell.execute_reply.started":"2023-04-26T03:50:16.589218Z","shell.execute_reply":"2023-04-26T03:50:16.614138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = download_path + 'birdclef-2023/'\nmyds = SoundDS(df, data_path)\n\n\nnum_items = len(myds)\nnum_train = round(num_items * 0.8)\nnum_val = num_items - num_train\ntrain_ds, val_ds = random_split(myds, [num_train, num_val])\n\ntrain_dl = torch.utils.data.DataLoader(train_ds, batch_size=16, shuffle=True)\nval_dl = torch.utils.data.DataLoader(val_ds, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.617363Z","iopub.execute_input":"2023-04-26T03:50:16.618221Z","iopub.status.idle":"2023-04-26T03:50:16.632941Z","shell.execute_reply.started":"2023-04-26T03:50:16.618175Z","shell.execute_reply":"2023-04-26T03:50:16.631723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = train_ds\n# Iterate through the dataset\nfor i in range(10):\n    # Get the i-th sample from the dataset\n    sample = dataset[i]\n\n    # Extract the input (audio data) and target (label) from the sample\n    input, target = sample\n\n    # Do something with the input and target\n    print(f'Sample {i}: Input shape: {input.shape}, Target: {target}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:16.634905Z","iopub.execute_input":"2023-04-26T03:50:16.636586Z","iopub.status.idle":"2023-04-26T03:50:17.781665Z","shell.execute_reply.started":"2023-04-26T03:50:16.636491Z","shell.execute_reply":"2023-04-26T03:50:17.780648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audioModel = AudioClassifier()\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\naudioModel = audioModel.to(device)\nnext(audioModel.parameters()).device","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:17.783338Z","iopub.execute_input":"2023-04-26T03:50:17.784069Z","iopub.status.idle":"2023-04-26T03:50:17.795806Z","shell.execute_reply.started":"2023-04-26T03:50:17.784029Z","shell.execute_reply":"2023-04-26T03:50:17.794796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Train the Model","metadata":{}},{"cell_type":"code","source":"import gc\n# ----------------------------\n# Training Loop\n# ----------------------------\ndef training(model, train_dl, num_epochs):\n  # Loss Function, Optimizer and Scheduler\n  criterion = nn.CrossEntropyLoss()\n  optimizer = torch.optim.Adam(model.parameters(),lr=0.001)\n  scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, max_lr=0.001,\n                                                steps_per_epoch=int(len(train_dl)),\n                                                epochs=num_epochs,\n                                                anneal_strategy='linear')\n\n  # Repeat for each epoch\n  for epoch in range(num_epochs):\n    running_loss = 0.0\n    correct_prediction = 0\n    total_prediction = 0\n\n    # Repeat for each batch in the training set\n    for i, data in enumerate(train_dl):\n        # Get the input features and target labels, and put them on the GPU\n        inputs, labels = data[0].to(device), data[1].to(device)\n\n        # Normalize the inputs\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n\n        # Zero the parameter gradients\n        optimizer.zero_grad()\n\n        # forward + backward + optimize\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        scheduler.step()\n\n        # Keep stats for Loss and Accuracy\n        running_loss += loss.item()\n\n        # Get the predicted class with the highest score\n        _, prediction = torch.max(outputs,1)\n        # Count of predictions that matched the target label\n        correct_prediction += (prediction == labels).sum().item()\n        total_prediction += prediction.shape[0]\n\n        if i % 10 == 0:    # print every 10 mini-batches\n           print('[%d, %5d] loss: %.3f' % (epoch + 1, i + 1, running_loss / 10))\n    \n    # Print stats at the end of the epoch\n    num_batches = len(train_dl)\n    avg_loss = running_loss / num_batches\n    acc = correct_prediction/total_prediction\n    print(f'Epoch: {epoch}, Loss: {avg_loss:.2f}, Accuracy: {acc:.2f}')\n\n  print('Finished Training')\n  \nnum_epochs=200   # Just for demo, adjust this higher.\ntraining(audioModel, train_dl, num_epochs)\n#23/03\n#Epoch: 0, Loss: 5.18, Accuracy: 0.04\n#Epoch: 1, Loss: 4.56, Accuracy: 0.09\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T03:50:17.797286Z","iopub.execute_input":"2023-04-26T03:50:17.798339Z","iopub.status.idle":"2023-04-26T09:01:33.545973Z","shell.execute_reply.started":"2023-04-26T03:50:17.798298Z","shell.execute_reply":"2023-04-26T09:01:33.544526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 7: Make predictions\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nfixed_tm = frame_audio(wav_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.547813Z","iopub.execute_input":"2023-04-26T09:01:33.548556Z","iopub.status.idle":"2023-04-26T09:01:33.656748Z","shell.execute_reply.started":"2023-04-26T09:01:33.548502Z","shell.execute_reply":"2023-04-26T09:01:33.655389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def audio_transform(aud, sr, channel, duration, shift_pct):\n    reaud = AudioUtil.resample(aud, sr)\n    rechan = AudioUtil.rechannel(reaud, channel)\n    dur_aud = AudioUtil.pad_trunc(rechan, duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n\n    return aug_sgram","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.658770Z","iopub.execute_input":"2023-04-26T09:01:33.659304Z","iopub.status.idle":"2023-04-26T09:01:33.685677Z","shell.execute_reply.started":"2023-04-26T09:01:33.659254Z","shell.execute_reply":"2023-04-26T09:01:33.684351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = audio_transform((fixed_tm[:1], sample_rate), 44100, 2, 4000, 0.4)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.687330Z","iopub.execute_input":"2023-04-26T09:01:33.688094Z","iopub.status.idle":"2023-04-26T09:01:33.728633Z","shell.execute_reply.started":"2023-04-26T09:01:33.688043Z","shell.execute_reply":"2023-04-26T09:01:33.727253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixed_tm[:1].shape, inputs.shape, inputs[1].unsqueeze(0).shape","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.730530Z","iopub.execute_input":"2023-04-26T09:01:33.731517Z","iopub.status.idle":"2023-04-26T09:01:33.741777Z","shell.execute_reply.started":"2023-04-26T09:01:33.731465Z","shell.execute_reply":"2023-04-26T09:01:33.740528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outputs = audioModel(inputs.unsqueeze(0))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.743420Z","iopub.execute_input":"2023-04-26T09:01:33.744708Z","iopub.status.idle":"2023-04-26T09:01:33.781554Z","shell.execute_reply.started":"2023-04-26T09:01:33.744656Z","shell.execute_reply":"2023-04-26T09:01:33.780514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixed_tm.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.783103Z","iopub.execute_input":"2023-04-26T09:01:33.783894Z","iopub.status.idle":"2023-04-26T09:01:33.795950Z","shell.execute_reply.started":"2023-04-26T09:01:33.783843Z","shell.execute_reply":"2023-04-26T09:01:33.794855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fixed_tm[:1].squeeze(0).shape, fixed_tm[:1].shape, fixed_tm[1:].shape","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.798888Z","iopub.execute_input":"2023-04-26T09:01:33.800222Z","iopub.status.idle":"2023-04-26T09:01:33.806655Z","shell.execute_reply.started":"2023-04-26T09:01:33.800171Z","shell.execute_reply":"2023-04-26T09:01:33.804846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aud = AudioUtil.open(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\n# inputs = audio_transform(aud, 44100, 2, 4000, 0.4, 15)\n# outputs = audioModel(inputs[0:2].unsqueeze(0))\nprobabilities = outputs.sigmoid().cpu().detach().numpy()\nargmax = np.argmax(probabilities)\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.809299Z","iopub.execute_input":"2023-04-26T09:01:33.810383Z","iopub.status.idle":"2023-04-26T09:01:33.823042Z","shell.execute_reply.started":"2023-04-26T09:01:33.810313Z","shell.execute_reply":"2023-04-26T09:01:33.821523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    fixed_tm = frame_audio(wav_data)\n    inputs = audio_transform((fixed_tm[:1], sample_rate), 44100, 2, 4000, 0.4)\n    \n    frame = 5\n    all_logits = audioModel(inputs.unsqueeze(0)).cpu().detach().numpy()\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        inputs = audio_transform((window[np.newaxis, :], sample_rate), 44100, 2, 4000, 0.4)\n        logits = audioModel(inputs.unsqueeze(0)).cpu().detach().numpy()\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = torch.nn.functional.softmax(torch.from_numpy(frame_logits), dim=0).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.824669Z","iopub.execute_input":"2023-04-26T09:01:33.825549Z","iopub.status.idle":"2023-04-26T09:01:33.837492Z","shell.execute_reply.started":"2023-04-26T09:01:33.825510Z","shell.execute_reply":"2023-04-26T09:01:33.836015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.839881Z","iopub.execute_input":"2023-04-26T09:01:33.840862Z","iopub.status.idle":"2023-04-26T09:01:33.878002Z","shell.execute_reply.started":"2023-04-26T09:01:33.840822Z","shell.execute_reply":"2023-04-26T09:01:33.876681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.879502Z","iopub.execute_input":"2023-04-26T09:01:33.880785Z","iopub.status.idle":"2023-04-26T09:01:33.951855Z","shell.execute_reply.started":"2023-04-26T09:01:33.880733Z","shell.execute_reply":"2023-04-26T09:01:33.950515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:33.953676Z","iopub.execute_input":"2023-04-26T09:01:33.954121Z","iopub.status.idle":"2023-04-26T09:01:35.547238Z","shell.execute_reply.started":"2023-04-26T09:01:33.954082Z","shell.execute_reply":"2023-04-26T09:01:35.546177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:35.551741Z","iopub.execute_input":"2023-04-26T09:01:35.552323Z","iopub.status.idle":"2023-04-26T09:01:35.575388Z","shell.execute_reply.started":"2023-04-26T09:01:35.552284Z","shell.execute_reply":"2023-04-26T09:01:35.573959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T09:01:35.577298Z","iopub.execute_input":"2023-04-26T09:01:35.577835Z","iopub.status.idle":"2023-04-26T09:01:35.594333Z","shell.execute_reply.started":"2023-04-26T09:01:35.577786Z","shell.execute_reply":"2023-04-26T09:01:35.592996Z"},"trusted":true},"execution_count":null,"outputs":[]}]}