{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Learning how to apply transformation to audio data before model traning\n\nMany new concepts to digest\n\nStill working on it...\n\nResources: \n\n- https://towardsdatascience.com/audio-deep-learning-made-simple-sound-classification-step-by-step-cebc936bbe5\n- https://pytorch.org/tutorials/beginner/audio_preprocessing_tutorial.html#","metadata":{}},{"cell_type":"markdown","source":"## Prepare traning data from metadata file","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\n\n# Read in the CSV files.\nBASE_DIR = '../input/birdclef-2024/'\ntrain = pd.read_csv(f'{BASE_DIR}/train_metadata.csv')\n#ss = pd.read_csv(f'{BASE_DIR}/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:32.800015Z","iopub.execute_input":"2024-04-13T16:28:32.800438Z","iopub.status.idle":"2024-04-13T16:28:34.111659Z","shell.execute_reply.started":"2024-04-13T16:28:32.800405Z","shell.execute_reply":"2024-04-13T16:28:34.110302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = '/kaggle/input/birdclef-2024/train_audio/'\n#train['file_path'] = TRAIN_DIR + train['filename']\ntrain = train[['filename','primary_label']]","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:34.113842Z","iopub.execute_input":"2024-04-13T16:28:34.114287Z","iopub.status.idle":"2024-04-13T16:28:34.130077Z","shell.execute_reply.started":"2024-04-13T16:28:34.114247Z","shell.execute_reply":"2024-04-13T16:28:34.128960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:34.131355Z","iopub.execute_input":"2024-04-13T16:28:34.131723Z","iopub.status.idle":"2024-04-13T16:28:34.154769Z","shell.execute_reply.started":"2024-04-13T16:28:34.131689Z","shell.execute_reply":"2024-04-13T16:28:34.153913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:34.156793Z","iopub.execute_input":"2024-04-13T16:28:34.157302Z","iopub.status.idle":"2024-04-13T16:28:35.525227Z","shell.execute_reply.started":"2024-04-13T16:28:34.157274Z","shell.execute_reply":"2024-04-13T16:28:35.524069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\n\ntrain['primary_label'] = le.fit_transform(train['primary_label'])","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:35.526473Z","iopub.execute_input":"2024-04-13T16:28:35.526850Z","iopub.status.idle":"2024-04-13T16:28:35.541306Z","shell.execute_reply.started":"2024-04-13T16:28:35.526816Z","shell.execute_reply":"2024-04-13T16:28:35.540068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Audio pre-processing: transformations\n- File loading function\n- Resample and convert to two channels (stereo)\n- Resize\n- Apply time shift\n- Convert to Mel spectrogram\n- Spectrogram augmentation","metadata":{}},{"cell_type":"code","source":"import math, random\nimport torch\nimport torchaudio\nfrom torchaudio import transforms\n\nclass AudioUtil():\n  # ----------------------------\n  # Load an audio file. Return the signal as a tensor and the sample rate\n  # ----------------------------\n  @staticmethod\n  def open(audio_file):\n    sig, sr = torchaudio.load(audio_file)\n    return (sig, sr)\n\n  # ----------------------------\n  # Convert the given audio to the desired number of channels\n  # ----------------------------\n  @staticmethod\n  def rechannel(aud, new_channel):\n    sig, sr = aud\n\n    if (sig.shape[0] == new_channel):\n      # Nothing to do\n      return aud\n\n    if (new_channel == 1):\n      # Convert from stereo to mono by selecting only the first channel\n      resig = sig[:1, :]\n    else:\n      # Convert from mono to stereo by duplicating the first channel\n      resig = torch.cat([sig, sig])\n\n    return ((resig, sr))\n\n  # ----------------------------\n  # Since Resample applies to a single channel, we resample one channel at a time\n  # ----------------------------\n  @staticmethod\n  def resample(aud, newsr):\n    sig, sr = aud\n\n    if (sr == newsr):\n      # Nothing to do\n      return aud\n\n    num_channels = sig.shape[0]\n    # Resample first channel\n    resig = torchaudio.transforms.Resample(sr, newsr)(sig[:1,:])\n    if (num_channels > 1):\n      # Resample the second channel and merge both channels\n      retwo = torchaudio.transforms.Resample(sr, newsr)(sig[1:,:])\n      resig = torch.cat([resig, retwo])\n\n    return ((resig, newsr))\n\n  # ----------------------------\n  # Pad (or truncate) the signal to a fixed length 'max_ms' in milliseconds\n  # ----------------------------\n  @staticmethod\n  def pad_trunc(aud, max_ms):\n    sig, sr = aud\n    num_rows, sig_len = sig.shape\n    max_len = sr//1000 * max_ms\n\n    if (sig_len > max_len):\n      # Truncate the signal to the given length\n      sig = sig[:,:max_len]\n\n    elif (sig_len < max_len):\n      # Length of padding to add at the beginning and end of the signal\n      pad_begin_len = random.randint(0, max_len - sig_len)\n      pad_end_len = max_len - sig_len - pad_begin_len\n\n      # Pad with 0s\n      pad_begin = torch.zeros((num_rows, pad_begin_len))\n      pad_end = torch.zeros((num_rows, pad_end_len))\n\n      sig = torch.cat((pad_begin, sig, pad_end), 1)\n      \n    return (sig, sr)\n\n  # ----------------------------\n  # Shifts the signal to the left or right by some percent. Values at the end\n  # are 'wrapped around' to the start of the transformed signal.\n  # ----------------------------\n  @staticmethod\n  def time_shift(aud, shift_limit):\n    sig,sr = aud\n    _, sig_len = sig.shape\n    shift_amt = int(random.random() * shift_limit * sig_len)\n    return (sig.roll(shift_amt), sr)\n\n  # ----------------------------\n  # Generate a Spectrogram\n  # ----------------------------\n  @staticmethod\n  def spectro_gram(aud, n_mels=64, n_fft=1024, hop_len=None):\n    sig,sr = aud\n    top_db = 80\n\n    # spec has shape [channel, n_mels, time], where channel is mono, stereo etc\n    spec = transforms.MelSpectrogram(sr, n_fft=n_fft, hop_length=hop_len, n_mels=n_mels)(sig)\n\n    # Convert to decibels\n    spec = transforms.AmplitudeToDB(top_db=top_db)(spec)\n    return (spec)\n\n  # ----------------------------\n  # Augment the Spectrogram by masking out some sections of it in both the frequency\n  # dimension (ie. horizontal bars) and the time dimension (vertical bars) to prevent\n  # overfitting and to help the model generalise better. The masked sections are\n  # replaced with the mean value.\n  # ----------------------------\n  @staticmethod\n  def spectro_augment(spec, max_mask_pct=0.1, n_freq_masks=1, n_time_masks=1):\n    _, n_mels, n_steps = spec.shape\n    mask_value = spec.mean()\n    aug_spec = spec\n\n    freq_mask_param = max_mask_pct * n_mels\n    for _ in range(n_freq_masks):\n      aug_spec = transforms.FrequencyMasking(freq_mask_param)(aug_spec, mask_value)\n\n    time_mask_param = max_mask_pct * n_steps\n    for _ in range(n_time_masks):\n      aug_spec = transforms.TimeMasking(time_mask_param)(aug_spec, mask_value)\n\n    return aug_spec","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:35.542833Z","iopub.execute_input":"2024-04-13T16:28:35.543834Z","iopub.status.idle":"2024-04-13T16:28:40.271378Z","shell.execute_reply.started":"2024-04-13T16:28:35.543803Z","shell.execute_reply":"2024-04-13T16:28:40.270101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader, Dataset, random_split\nimport torchaudio\n\n# ----------------------------\n# Sound Dataset\n# ----------------------------\nclass SoundDS(Dataset):\n  def __init__(self, df, data_path):\n    self.df = df\n    self.data_path = str(data_path)\n    self.duration = 5000\n    self.sr = 44100\n    self.channel = 2\n    self.shift_pct = 0.4\n            \n  # ----------------------------\n  # Number of items in dataset\n  # ----------------------------\n  def __len__(self):\n    return len(self.df)    \n    \n  # ----------------------------\n  # Get i'th item in dataset\n  # ----------------------------\n  def __getitem__(self, idx):\n    # Absolute file path of the audio file - concatenate the audio directory with\n    # the relative path\n    audio_file = self.data_path + self.df.loc[idx, 'filename']\n    # Get the Class ID\n    primary_label = self.df.loc[idx, 'primary_label']\n\n    aud = AudioUtil.open(audio_file)\n    # Some sounds have a higher sample rate, or fewer channels compared to the\n    # majority. So make all sounds have the same number of channels and same \n    # sample rate. Unless the sample rate is the same, the pad_trunc will still\n    # result in arrays of different lengths, even though the sound duration is\n    # the same.\n    reaud = AudioUtil.resample(aud, self.sr)\n    rechan = AudioUtil.rechannel(reaud, self.channel)\n\n    dur_aud = AudioUtil.pad_trunc(rechan, self.duration)\n    shift_aud = AudioUtil.time_shift(dur_aud, self.shift_pct)\n    sgram = AudioUtil.spectro_gram(shift_aud, n_mels=64, n_fft=1024, hop_len=None)\n    aug_sgram = AudioUtil.spectro_augment(sgram, max_mask_pct=0.1, n_freq_masks=2, n_time_masks=2)\n\n    return aug_sgram, primary_label","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:40.272830Z","iopub.execute_input":"2024-04-13T16:28:40.273365Z","iopub.status.idle":"2024-04-13T16:28:40.286529Z","shell.execute_reply.started":"2024-04-13T16:28:40.273331Z","shell.execute_reply":"2024-04-13T16:28:40.284432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import random_split\n\nmyds = SoundDS(train, TRAIN_DIR)\n\n# Random split of 80:20 between training and validation\nnum_items = len(myds)\nnum_train = round(num_items * 0.8)\nnum_val = num_items - num_train\ntrain_ds, val_ds = random_split(myds, [num_train, num_val])\n\n# Create training and validation data loaders\ntrain_dl = torch.utils.data.DataLoader(train_ds, batch_size=16, shuffle=True)\nval_dl = torch.utils.data.DataLoader(val_ds, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:40.288568Z","iopub.execute_input":"2024-04-13T16:28:40.289107Z","iopub.status.idle":"2024-04-13T16:28:40.326950Z","shell.execute_reply.started":"2024-04-13T16:28:40.289063Z","shell.execute_reply":"2024-04-13T16:28:40.325784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn import init\n\n# ----------------------------\n# Audio Classification Model\n# ----------------------------\nclass AudioClassifier (nn.Module):\n    # ----------------------------\n    # Build the model architecture\n    # ----------------------------\n    def __init__(self):\n        super().__init__()\n        conv_layers = []\n\n        # First Convolution Block with Relu and Batch Norm. Use Kaiming Initialization\n        self.conv1 = nn.Conv2d(2, 8, kernel_size=(5, 5), stride=(2, 2), padding=(2, 2))\n        self.relu1 = nn.ReLU()\n        self.bn1 = nn.BatchNorm2d(8)\n        init.kaiming_normal_(self.conv1.weight, a=0.1)\n        self.conv1.bias.data.zero_()\n        conv_layers += [self.conv1, self.relu1, self.bn1]\n\n        # Second Convolution Block\n        self.conv2 = nn.Conv2d(8, 16, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu2 = nn.ReLU()\n        self.bn2 = nn.BatchNorm2d(16)\n        init.kaiming_normal_(self.conv2.weight, a=0.1)\n        self.conv2.bias.data.zero_()\n        conv_layers += [self.conv2, self.relu2, self.bn2]\n\n        # Third Convolution Block\n        self.conv3 = nn.Conv2d(16, 32, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu3 = nn.ReLU()\n        self.bn3 = nn.BatchNorm2d(32)\n        init.kaiming_normal_(self.conv3.weight, a=0.1)\n        self.conv3.bias.data.zero_()\n        conv_layers += [self.conv3, self.relu3, self.bn3]\n\n        # Fourth Convolution Block\n        self.conv4 = nn.Conv2d(32, 64, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1))\n        self.relu4 = nn.ReLU()\n        self.bn4 = nn.BatchNorm2d(64)\n        init.kaiming_normal_(self.conv4.weight, a=0.1)\n        self.conv4.bias.data.zero_()\n        conv_layers += [self.conv4, self.relu4, self.bn4]\n\n        # Linear Classifier\n        self.ap = nn.AdaptiveAvgPool2d(output_size=1)\n        self.lin = nn.Linear(in_features=64, out_features=182)\n\n        # Wrap the Convolutional Blocks\n        self.conv = nn.Sequential(*conv_layers)\n \n    # ----------------------------\n    # Forward pass computations\n    # ----------------------------\n    def forward(self, x):\n        # Run the convolutional blocks\n        x = self.conv(x)\n\n        # Adaptive pool and flatten for input to linear layer\n        x = self.ap(x)\n        x = x.view(x.shape[0], -1)\n\n        # Linear layer\n        x = self.lin(x)\n\n        # Final output\n        return x\n\n# Create the model and put it on the GPU if available\nmyModel = AudioClassifier()\n#device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n#myModel = myModel.to(device)\n# Check that it is on Cuda\n#next(myModel.parameters()).device","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:40.328402Z","iopub.execute_input":"2024-04-13T16:28:40.328780Z","iopub.status.idle":"2024-04-13T16:28:40.388233Z","shell.execute_reply.started":"2024-04-13T16:28:40.328749Z","shell.execute_reply":"2024-04-13T16:28:40.387302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ----------------------------\n# Training Loop\n# ----------------------------\ndef training(model, train_dl, num_epochs):\n  # Loss Function, Optimizer and Scheduler\n  criterion = nn.CrossEntropyLoss()\n  optimizer = torch.optim.Adam(model.parameters(),lr=0.001)\n  scheduler = torch.optim.lr_scheduler.OneCycleLR(optimizer, max_lr=0.001,\n                                                steps_per_epoch=int(len(train_dl)),\n                                                epochs=num_epochs,\n                                                anneal_strategy='linear')\n\n  # Repeat for each epoch\n  for epoch in range(num_epochs):\n    running_loss = 0.0\n    correct_prediction = 0\n    total_prediction = 0\n\n    # Repeat for each batch in the training set\n    for i, data in enumerate(train_dl):\n        # Get the input features and target labels, and put them on the GPU\n        #inputs, labels = data[0].to(device), data[1].to(device)\n        inputs, labels = data[0], data[1]\n\n        # Normalize the inputs\n        inputs_m, inputs_s = inputs.mean(), inputs.std()\n        inputs = (inputs - inputs_m) / inputs_s\n\n        # Zero the parameter gradients\n        optimizer.zero_grad()\n\n        # forward + backward + optimize\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        scheduler.step()\n\n        # Keep stats for Loss and Accuracy\n        running_loss += loss.item()\n\n        # Get the predicted class with the highest score\n        _, prediction = torch.max(outputs,1)\n        # Count of predictions that matched the target label\n        correct_prediction += (prediction == labels).sum().item()\n        total_prediction += prediction.shape[0]\n\n        #if i % 10 == 0:    # print every 10 mini-batches\n        #    print('[%d, %5d] loss: %.3f' % (epoch + 1, i + 1, running_loss / 10))\n    \n    # Print stats at the end of the epoch\n    num_batches = len(train_dl)\n    avg_loss = running_loss / num_batches\n    acc = correct_prediction/total_prediction\n    print(f'Epoch: {epoch}, Loss: {avg_loss:.2f}, Accuracy: {acc:.2f}')\n\n  print('Finished Training')\n  \nnum_epochs=2   # Just for demo, adjust this higher.\ntraining(myModel, train_dl, num_epochs)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T16:28:40.390631Z","iopub.execute_input":"2024-04-13T16:28:40.391171Z","iopub.status.idle":"2024-04-13T16:30:55.425255Z","shell.execute_reply.started":"2024-04-13T16:28:40.391139Z","shell.execute_reply":"2024-04-13T16:30:55.423832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference comes next","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}