{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### In this notebook we take a look at \n\n1. how to generate mel spectrograms for audio data. \n2. how to augment audio data by apply audio transforms using the audiomentations library (which is very similar to the famous albumentations library in computer vision)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom IPython import display as ipd\nfrom glob import glob\nimport torch\nimport torchaudio\nimport torchaudio.transforms as T\nimport seaborn as sns\nimport skimage.io\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:21:04.519457Z","iopub.execute_input":"2022-04-22T12:21:04.519757Z","iopub.status.idle":"2022-04-22T12:21:04.525096Z","shell.execute_reply.started":"2022-04-22T12:21:04.519725Z","shell.execute_reply":"2022-04-22T12:21:04.523903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/kaggle-pog-series-s01e02/\"\nRESAMPLED_AUDIO_PATH = \"/kaggle/input/pogmusicclassification/\"\nPROCESSED_TRAIN_DATA_PATH = \"/kaggle/processed_train/\"\nPROCESSED_TEST_DATA_PATH = \"/kaggle/processed_test/\"\n# Number of mel specs to create for each audio using random clipping and audio transforms\nNUM_MEL_SPECS = 5\n\nclass AudioConfig:\n    # settings\n    # number of samples per time-step in spectrogram. Defaults to win_length / 4\n    # Also the step or stride between windows. If the step is smaller than the window length, the windows will overlap\n    hop_length = 512 \n    # number of bins in spectrogram. Height of image\n    n_mels = 224 \n    # number of time-steps. Width of image\n    time_steps = 512 \n    # number of samples per second\n    sampling_rate = 44100\n    # sec\n    duration = 10 \n    fmin = 20\n    fmax = sampling_rate // 2\n    # FFT window size or length of the windowed signal after padding with zeros. Default value = 2048 ( for music signals)    \n    n_fft = hop_length * 4\n    # Each frame of audio is windowed by window of length win_length and then padded with zeros to match n_fft. Defaults to n_fft\n    win_length = hop_length * 4    \n    padmode = 'constant'\n    samples = sampling_rate * duration","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:35:53.575698Z","iopub.execute_input":"2022-04-22T12:35:53.576Z","iopub.status.idle":"2022-04-22T12:35:53.582641Z","shell.execute_reply.started":"2022-04-22T12:35:53.575967Z","shell.execute_reply":"2022-04-22T12:35:53.581577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load train audio metadata","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(DATA_PATH + \"train.csv\")\ndf_train[\"file_exists\"] = df_train.filepath.map(lambda fp: os.path.exists(DATA_PATH + fp))\ndf_train[~df_train.file_exists]\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:21:04.56122Z","iopub.execute_input":"2022-04-22T12:21:04.561789Z","iopub.status.idle":"2022-04-22T12:21:12.639505Z","shell.execute_reply.started":"2022-04-22T12:21:04.561742Z","shell.execute_reply":"2022-04-22T12:21:12.638703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load test audio metadata","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(DATA_PATH + \"test.csv\")\ndf_test[\"file_exists\"] = df_test.filepath.map(lambda fp: os.path.exists(DATA_PATH + fp))\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:21:12.641081Z","iopub.execute_input":"2022-04-22T12:21:12.641308Z","iopub.status.idle":"2022-04-22T12:21:14.798287Z","shell.execute_reply.started":"2022-04-22T12:21:12.641282Z","shell.execute_reply":"2022-04-22T12:21:14.797367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Crop audio of fixed duration randomly","metadata":{}},{"cell_type":"code","source":"def crop_or_pad(y, length, is_train=True):\n    \"\"\"\n    Crops an array to a chosen length\n    Arguments:\n        y {1D np array} -- Array to crop\n        length {int} -- Length of the crop\n        sr {int} -- Sampling rate\n    Keyword Arguments:\n        train {bool} -- Whether we are at train time. If so, crop randomly, else return the beginning of y (default: {True})        \n    Returns:\n        1D np array -- Cropped array\n    \"\"\"    \n    if length > len(y):\n        # if length of array is less than the length to be cropped, we need to pad \n        padding = length - len(y)    # add padding at both ends\n        offset = padding // 2\n        y = np.pad(y, (offset, padding - offset), AudioConfig.padmode)\n    else:\n        if not is_train:\n            start = 0\n        else:\n            start = np.random.randint(len(y) - length)            \n        y = y[start: start + length]\n    return y.astype(np.float32)        ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:21:14.79951Z","iopub.execute_input":"2022-04-22T12:21:14.79975Z","iopub.status.idle":"2022-04-22T12:21:14.805075Z","shell.execute_reply.started":"2022-04-22T12:21:14.799721Z","shell.execute_reply":"2022-04-22T12:21:14.804541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import soundfile\n\ndef read_audio(audio_path, is_train=True, is_crop_or_pad=True):    \n    y, sr = torchaudio.load(audio_path)\n    y = y.numpy()\n    if len(y.shape) > 1:\n        y = np.mean(y, axis=1)        \n    if is_crop_or_pad:\n        audio_samples_length = AudioConfig.sampling_rate * AudioConfig.duration\n        y = crop_or_pad(y, audio_samples_length, is_train=is_train)\n    return (torch.from_numpy(y), sr)\n\ndef read_audio_librosa(audio_path, sr, is_train=True, is_crop_or_pad=True):    \n    y, sr = librosa.load(path=audio_path, sr=sr)    \n    if len(y.shape) > 1:\n        y = np.mean(y, axis=1)        \n    if is_crop_or_pad:\n        audio_samples_length = AudioConfig.sampling_rate * AudioConfig.duration\n        y = crop_or_pad(y, audio_samples_length, is_train=is_train)\n    return (y, sr)","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:41:54.250032Z","iopub.execute_input":"2022-04-22T13:41:54.250821Z","iopub.status.idle":"2022-04-22T13:41:54.258121Z","shell.execute_reply.started":"2022-04-22T13:41:54.250783Z","shell.execute_reply":"2022-04-22T13:41:54.25726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mono_to_color(X, eps=1e-6, mean=None, std=None):\n    \"\"\"\n    Converts a one channel array to a 3 channel one in [0, 255]\n    Arguments:\n        X {numpy array [H x W]} -- 2D array to convert\n    Keyword Arguments:\n        eps {float} -- To avoid dividing by 0 (default: {1e-6})\n        mean {None or np array} -- Mean for normalization (default: {None})\n        std {None or np array} -- Std for normalization (default: {None})\n    Returns:\n        numpy array [3 x H x W] -- RGB numpy array\n    \"\"\"\n    X = np.stack([X, X, X], axis=-1)\n\n    # Standardize\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n\n    # Normalize to [0, 255]\n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:21:14.837046Z","iopub.execute_input":"2022-04-22T12:21:14.837439Z","iopub.status.idle":"2022-04-22T12:21:14.848995Z","shell.execute_reply.started":"2022-04-22T12:21:14.837288Z","shell.execute_reply":"2022-04-22T12:21:14.848287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert the audio clipping to mel spectrogram","metadata":{}},{"cell_type":"code","source":"def create_mel_spec(y, sr):\n    \"\"\"\n    Computes a mel-spectrogram and puts it at decibel scale\n    Arguments:\n        y {np array} -- signal\n        sr {int} -- audio sample rate\n        conf {AudioParams} -- Parameters to use for the spectrogram. Expected to have the attributes sr, n_mels, n_fft, hop_length\n    Returns:\n        np array -- Mel-spectrogram\n    \"\"\"\n    mel_spectrogram = T.MelSpectrogram(\n        sample_rate=sr,\n        n_fft=AudioConfig.n_fft,\n        win_length=AudioConfig.win_length,\n        hop_length=AudioConfig.hop_length,\n        center=True,\n        pad_mode=\"reflect\",\n        power=2.0,\n        norm='slaney',\n        onesided=True,\n        n_mels=AudioConfig.n_mels,\n        mel_scale=\"htk\",\n    )\n    mel_s = mel_spectrogram(y)\n    print(f\"mel_s.shape={mel_s.shape}\")\n    return mel_s ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:52:32.047974Z","iopub.execute_input":"2022-04-22T12:52:32.048763Z","iopub.status.idle":"2022-04-22T12:52:32.054929Z","shell.execute_reply.started":"2022-04-22T12:52:32.048714Z","shell.execute_reply":"2022-04-22T12:52:32.054036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_mel_spec_librosa(y, sr, conf):\n    \"\"\"\n    Computes a mel-spectrogram and puts it at decibel scale\n    Arguments:\n        y {np array} -- signal\n        sr {int} -- audio sample rate\n        conf {AudioParams} -- Parameters to use for the spectrogram. Expected to have the attributes sr, n_mels, n_fft, hop_length\n    Returns:\n        np array -- Mel-spectrogram\n    \"\"\"\n    mel_s = librosa.feature.melspectrogram(\n        y=y, \n        sr=sr, \n        n_mels=conf.n_mels,\n        n_fft=conf.n_fft, \n        hop_length=conf.hop_length,\n        fmin=conf.fmin,\n        fmax=conf.fmax\n    )\n    # convert amplitude to decibels\n    #mel_s = librosa.power_to_db(mel_s, ref=np.max)            \n    return mel_s ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:44:40.092668Z","iopub.execute_input":"2022-04-22T13:44:40.093268Z","iopub.status.idle":"2022-04-22T13:44:40.09781Z","shell.execute_reply.started":"2022-04-22T13:44:40.093223Z","shell.execute_reply":"2022-04-22T13:44:40.097247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display sample spectrogram generated","metadata":{}},{"cell_type":"code","source":"import librosa\n\ndef plot_spectrogram(spec, title=None, ylabel='freq_bin', aspect='auto', xmax=None):\n    fig, axs = plt.subplots(1, 1)\n    axs.set_title(title or 'Spectrogram (db)')\n    axs.set_ylabel(ylabel)\n    axs.set_xlabel('frame')\n    im = axs.imshow(librosa.power_to_db(spec), origin='lower', aspect=aspect)\n    if xmax:\n        axs.set_xlim((0, xmax))\n    fig.colorbar(im, ax=axs)\n    plt.show(block=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:22:30.360999Z","iopub.execute_input":"2022-04-22T12:22:30.361297Z","iopub.status.idle":"2022-04-22T12:22:32.110551Z","shell.execute_reply.started":"2022-04-22T12:22:30.361266Z","shell.execute_reply":"2022-04-22T12:22:32.109652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa.display\n\ndef show_spectrogram(spec, title, sr, hop_length, y_axis='log', x_axis='time'):\n    librosa.display.specshow(spec, sr=sr, y_axis=y_axis, x_axis=x_axis, hop_length=hop_length)\n    plt.title(title)\n    plt.colorbar(format='%+2.0f dB')\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:40:30.509802Z","iopub.execute_input":"2022-04-22T13:40:30.51047Z","iopub.status.idle":"2022-04-22T13:40:30.517862Z","shell.execute_reply.started":"2022-04-22T13:40:30.51043Z","shell.execute_reply":"2022-04-22T13:40:30.516922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_stats(waveform, sample_rate=None, src=None):\n    if src:\n        print(\"-\" * 10)\n        print(\"Source:\", src)\n        print(\"-\" * 10)\n    if sample_rate:\n        print(\"Sample Rate:\", sample_rate)\n    print(\"Shape:\", tuple(waveform.shape))\n    print(\"Dtype:\", waveform.dtype)\n    print(f\" - Max:     {waveform.max().item():6.3f}\")\n    print(f\" - Min:     {waveform.min().item():6.3f}\")\n    print(f\" - Mean:    {waveform.mean().item():6.3f}\")\n    print(f\" - Std Dev: {waveform.std().item():6.3f}\")\n    print()\n    print(waveform)\n    print()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test by generating and displaying some mel spectrograms\naudio_path = DATA_PATH + \"train/000002.ogg\"\naudio_metadata = torchaudio.info(audio_path)\nprint(audio_metadata)\ny, sr = read_audio_librosa(audio_path, sr=AudioConfig.sampling_rate)    \nprint(f\"For {audio_path} sr={sr} and y.shape={y.shape}\")\nsample_mels = create_mel_spec_librosa(y, sr, AudioConfig)\nshow_spectrogram(sample_mels, title=\"Sample mel spectrogram\", sr=AudioConfig.sampling_rate, hop_length=AudioConfig.hop_length)","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:44:49.238192Z","iopub.execute_input":"2022-04-22T13:44:49.238777Z","iopub.status.idle":"2022-04-22T13:44:49.807578Z","shell.execute_reply.started":"2022-04-22T13:44:49.238739Z","shell.execute_reply":"2022-04-22T13:44:49.806669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_pct = (sample_mels == 0).sum()/(sample_mels.shape[0]*sample_mels.shape[1])\nzero_pct","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:47:06.647315Z","iopub.execute_input":"2022-04-22T13:47:06.647988Z","iopub.status.idle":"2022-04-22T13:47:06.655148Z","shell.execute_reply.started":"2022-04-22T13:47:06.647944Z","shell.execute_reply":"2022-04-22T13:47:06.654237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test by generating and displaying some mel spectrograms\n#audio_path = RESAMPLED_AUDIO_PATH + \"resampled_train/000040_16k.wav\"\naudio_path = DATA_PATH + \"train/000002.ogg\"\naudio_metadata = torchaudio.info(audio_path)\nprint(audio_metadata)\ny, sr = read_audio(audio_path)    \nprint(f\"For {audio_path} sr={sr} and y.shape={y.shape}\")\nsample_mels = create_mel_spec(y, sr)\nplot_spectrogram(sample_mels, title=\"Sample mel spectrogram\")","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:49:53.253875Z","iopub.execute_input":"2022-04-22T13:49:53.254198Z","iopub.status.idle":"2022-04-22T13:49:53.6356Z","shell.execute_reply.started":"2022-04-22T13:49:53.25417Z","shell.execute_reply":"2022-04-22T13:49:53.634597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us also see how the spectrogram image we generated looks like","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\nsample_mels = mono_to_color(sample_mels)\nskimage.io.imsave(\"000002_1.jpg\", sample_mels)\nimg = Image.open(\"000002_1.jpg\")\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2022-04-22T13:58:03.275126Z","iopub.execute_input":"2022-04-22T13:58:03.275424Z","iopub.status.idle":"2022-04-22T13:58:03.457717Z","shell.execute_reply.started":"2022-04-22T13:58:03.275393Z","shell.execute_reply":"2022-04-22T13:58:03.456579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef create_processing_dirs():\n    if not os.path.isdir(PROCESSED_TRAIN_DATA_PATH):\n        os.mkdir(PROCESSED_TRAIN_DATA_PATH)\n    if not os.path.isdir(PROCESSED_TEST_DATA_PATH):\n        os.mkdir(PROCESSED_TEST_DATA_PATH)            ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.931912Z","iopub.status.idle":"2022-04-22T12:18:36.932187Z","shell.execute_reply.started":"2022-04-22T12:18:36.932043Z","shell.execute_reply":"2022-04-22T12:18:36.932059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create audio augmentations","metadata":{}},{"cell_type":"code","source":"import audiomentations\nfrom audiomentations import AddGaussianSNR, TimeStretch, PitchShift, Shift\n\naudio_transforms = audiomentations.Compose([\n    AddGaussianSNR(min_snr_in_db=5, max_snr_in_db=40.0, p=0.5),\n    TimeStretch(min_rate=0.8, max_rate=1.25, p=0.5),\n    PitchShift(min_semitones=-4, max_semitones=4, p=0.5),\n    Shift(min_fraction=-0.5, max_fraction=0.5, p=0.5),\n])\n\ndef apply_audio_transforms(y, sr):        \n    y = audio_transforms(samples=y, sample_rate=sr)        \n    return y","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.933223Z","iopub.status.idle":"2022-04-22T12:18:36.933498Z","shell.execute_reply.started":"2022-04-22T12:18:36.933355Z","shell.execute_reply":"2022-04-22T12:18:36.93337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Process audio to create mel spectrograms","metadata":{}},{"cell_type":"code","source":"def audio_to_melspec(file_path, folder_path):\n    filename_no_ext = file_path.split(\"/\")[-1].split(\".\")[0]\n    file_path = DATA_PATH + file_path    \n    if os.path.exists(file_path):\n        for i in range(NUM_MEL_SPECS):            \n            y, sr = read_audio(file_path)\n            y = apply_audio_transforms(y, sr)\n            mel_s = create_mel_spec(y, sr, AudioConfig)\n            mel_s = mono_to_color(mel_s)        \n            if not os.path.isdir(folder_path + \"mel_spec/\"):\n                os.mkdir(folder_path + \"mel_spec/\")\n            mel_s_img_name = filename_no_ext + f\"_{i}.jpg\"                \n            mels_img_path = folder_path + \"mel_spec/\" + mel_s_img_name       \n            # save as PNG\n            skimage.io.imsave(mels_img_path, mel_s)            \n    return file_path","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.934837Z","iopub.status.idle":"2022-04-22T12:18:36.935138Z","shell.execute_reply.started":"2022-04-22T12:18:36.93498Z","shell.execute_reply":"2022-04-22T12:18:36.934996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Parallel processing using joblib to speed up spectrogram generation","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom joblib import delayed, Parallel\n\ncreate_processing_dirs()\n# for testing we process just 100 rows of the data\ndf_train = df_train.head(100)\ndelayed_funcs_train = [delayed(audio_to_melspec)(row[\"filepath\"], PROCESSED_TRAIN_DATA_PATH) \n                       for i, row in df_train[df_train.file_exists].iterrows()]\nresults_train = Parallel(n_jobs=-1, verbose=5)(delayed_funcs_train)    ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.936022Z","iopub.status.idle":"2022-04-22T12:18:36.936335Z","shell.execute_reply.started":"2022-04-22T12:18:36.936151Z","shell.execute_reply":"2022-04-22T12:18:36.936203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for testing we process just 100 rows of the data\ndf_test = df_test.head(100)\ndelayed_funcs_test = [delayed(audio_to_melspec)(row[\"filepath\"], PROCESSED_TEST_DATA_PATH) \n                      for i, row in df_test[df_test.file_exists].iterrows()]\nresults_test = Parallel(n_jobs=-1, verbose=5)(delayed_funcs_test)    ","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.937107Z","iopub.status.idle":"2022-04-22T12:18:36.937414Z","shell.execute_reply.started":"2022-04-22T12:18:36.93727Z","shell.execute_reply":"2022-04-22T12:18:36.937286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# zip the processed_train and processed_test folders\n!zip -rq mel_spec_train.zip /kaggle/processed_train/mel_spec\n!zip -rq mel_spec_test.zip /kaggle/processed_test/mel_spec","metadata":{"execution":{"iopub.status.busy":"2022-04-22T12:18:36.938512Z","iopub.status.idle":"2022-04-22T12:18:36.938848Z","shell.execute_reply.started":"2022-04-22T12:18:36.938681Z","shell.execute_reply":"2022-04-22T12:18:36.938704Z"},"trusted":true},"execution_count":null,"outputs":[]}]}