{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom IPython import display as ipd\nfrom glob import glob\nimport librosa\nimport seaborn as sns\nimport librosa.display\nimport skimage.io\nfrom tqdm.notebook import tqdm\nfrom joblib import delayed, Parallel\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:22.280557Z","iopub.execute_input":"2022-03-22T06:53:22.280865Z","iopub.status.idle":"2022-03-22T06:53:22.288002Z","shell.execute_reply.started":"2022-03-22T06:53:22.280834Z","shell.execute_reply":"2022-03-22T06:53:22.287164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/kaggle-pog-series-s01e02/\"\n\nclass conf:\n    # settings\n    # number of samples per time-step in spectrogram. Defaults to win_length / 4\n    hop_length = 512 \n    # number of bins in spectrogram. Height of image\n    n_mels = 224 \n    # number of time-steps. Width of image\n    time_steps = 223 \n    # number of samples per second\n    sampling_rate = 22050\n    # sec\n    duration = 10 \n    fmin = 20\n    fmax = sampling_rate // 2\n    # length of the windowed signal after padding with zeros. Default value = 2048 ( for music signals)    \n    n_fft = hop_length * 4\n    # Each frame of audio is windowed by window of length win_length and then padded with zeros to match n_fft. Defaults to n_fft\n    win_length = hop_length * 4    \n    padmode = 'constant'\n    samples = sampling_rate * duration","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:22.290013Z","iopub.execute_input":"2022-03-22T06:53:22.290526Z","iopub.status.idle":"2022-03-22T06:53:22.302529Z","shell.execute_reply.started":"2022-03-22T06:53:22.290483Z","shell.execute_reply":"2022-03-22T06:53:22.301695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(DATA_PATH + \"train.csv\")\ndf_train[\"file_exists\"] = df_train.filepath.map(lambda fp: os.path.exists(DATA_PATH + fp))\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:22.303740Z","iopub.execute_input":"2022-03-22T06:53:22.303965Z","iopub.status.idle":"2022-03-22T06:53:50.657126Z","shell.execute_reply.started":"2022-03-22T06:53:22.303930Z","shell.execute_reply":"2022-03-22T06:53:50.656244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[~df_train.file_exists]","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:50.659515Z","iopub.execute_input":"2022-03-22T06:53:50.659821Z","iopub.status.idle":"2022-03-22T06:53:50.673646Z","shell.execute_reply.started":"2022-03-22T06:53:50.659771Z","shell.execute_reply":"2022-03-22T06:53:50.672745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(DATA_PATH + \"test.csv\")\ndf_test[\"file_exists\"] = df_test.filepath.map(lambda fp: os.path.exists(DATA_PATH + fp))\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:50.676161Z","iopub.execute_input":"2022-03-22T06:53:50.676428Z","iopub.status.idle":"2022-03-22T06:53:58.357689Z","shell.execute_reply.started":"2022-03-22T06:53:50.676399Z","shell.execute_reply":"2022-03-22T06:53:58.356783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bar_plot(values, labels, ylabel, xlabel, title):    \n    fig, ax = plt.subplots(figsize=(14, 6))\n    sns.set_theme(style=\"whitegrid\")\n    bars = ax.bar(labels, values)\n    ax.set_title(title)\n    ax.set_xlabel(xlabel)\n    ax.set_ylabel(ylabel)\n    plt.xticks(rotation=90)\n    # add the y value on top of each bar\n    for bar in bars:\n        y_val = bar.get_height()\n        plt.text(bar.get_x(), y_val+0.05, y_val)    \n    ax = sns.barplot(y=values, x=labels, ax=ax)    ","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:58.358894Z","iopub.execute_input":"2022-03-22T06:53:58.359214Z","iopub.status.idle":"2022-03-22T06:53:58.366571Z","shell.execute_reply.started":"2022-03-22T06:53:58.359187Z","shell.execute_reply":"2022-03-22T06:53:58.365866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_counts = df_train.genre.value_counts()\nbar_plot(\n            target_counts.values, \n            target_counts.index, \n            ylabel=\"count\", \n            xlabel=\"genre\", \n            title=\"target distribution\"\n        )","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:58.367675Z","iopub.execute_input":"2022-03-22T06:53:58.368027Z","iopub.status.idle":"2022-03-22T06:53:58.896493Z","shell.execute_reply.started":"2022-03-22T06:53:58.367997Z","shell.execute_reply":"2022-03-22T06:53:58.895411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_audio(conf, pathname, trim_long_data):\n    y, sr = librosa.load(pathname, sr=None)\n    # trim silence\n    if 0 < len(y): # workaround: 0 length causes error\n        y, _ = librosa.effects.trim(y) # trim, top_db=default(60)\n    # extract a fixed length window\n    start_sample = 0 # starting at beginning\n    length_samples = conf.time_steps * conf.hop_length    \n    # make it unified length to conf.samples\n    if len(y) > conf.samples: # long enough\n        if trim_long_data:\n            y = y[start_sample : start_sample+length_samples]        \n    else: # pad blank\n        padding = length_samples - len(y)    # add padding at both ends\n        offset = padding // 2\n        y = np.pad(y, (offset, conf.samples - len(y) - offset), conf.padmode)\n    return y, sr","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:58.897939Z","iopub.execute_input":"2022-03-22T06:53:58.898248Z","iopub.status.idle":"2022-03-22T06:53:58.907157Z","shell.execute_reply.started":"2022-03-22T06:53:58.898206Z","shell.execute_reply":"2022-03-22T06:53:58.906362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_minmax(X, min=0.0, max=1.0):\n    X_std = (X - X.min()) / (X.max() - X.min())\n    X_scaled = X_std * (max - min) + min\n    return X_scaled\n\ndef save_melspec_image(y, img_path):\n    if not os.path.exists(img_path):                                \n        # use log-melspectrogram\n        mels = librosa.feature.melspectrogram(\n            y=y, \n            sr=conf.sampling_rate, \n            n_mels=conf.n_mels,\n            n_fft=conf.n_fft, \n            hop_length=conf.hop_length\n        )\n        mels = np.log(mels + 1e-9) # add small number to avoid log(0)\n        # min-max scale to fit inside 8-bit range\n        img = scale_minmax(mels, 0, 255).astype(np.uint8)\n        img = np.flip(img, axis=0) # put low frequencies at the bottom in image\n        img = 255-img # invert. make black==more energy\n        img = np.stack([img]*3, axis=-1)\n        # save as PNG\n        skimage.io.imsave(img_path, img)\n        return img","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:58.908772Z","iopub.execute_input":"2022-03-22T06:53:58.909275Z","iopub.status.idle":"2022-03-22T06:53:58.923828Z","shell.execute_reply.started":"2022-03-22T06:53:58.909234Z","shell.execute_reply":"2022-03-22T06:53:58.923145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# smel_arr = save_melspec_image(\n#     \"/kaggle/input/kaggle-pog-series-s01e02/train/002549.ogg\", \n#     \"/kaggle/working/002549.jpg\")\n# mfcc_arr = librosa.feature.mfcc(S=librosa.power_to_db(smel_arr))\n# print(f\"smel_arr.shape = {smel_arr.shape}\")\n# print(f\"mfcc_arr.shape = {mfcc_arr.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-22T06:53:58.924825Z","iopub.execute_input":"2022-03-22T06:53:58.925374Z","iopub.status.idle":"2022-03-22T06:53:58.934507Z","shell.execute_reply.started":"2022-03-22T06:53:58.925343Z","shell.execute_reply":"2022-03-22T06:53:58.933760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\ndef create_processing_dirs():\n    if not os.path.isdir(\"/kaggle/train\"):\n        os.mkdir(\"/kaggle/train\")\n    if not os.path.isdir(\"/kaggle/test\"):\n        os.mkdir(\"/kaggle/test\")            ","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:12.921215Z","iopub.execute_input":"2022-03-22T07:05:12.921524Z","iopub.status.idle":"2022-03-22T07:05:12.926204Z","shell.execute_reply.started":"2022-03-22T07:05:12.921494Z","shell.execute_reply":"2022-03-22T07:05:12.925461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_audio(file_path, folder_path):\n    filename_no_ext = file_path.split(\"/\")[-1].split(\".\")[0]\n    file_path = DATA_PATH + file_path    \n    if os.path.exists(file_path):\n        y, sr = read_audio(conf, file_path, True)\n        if not os.path.isdir(folder_path + \"mel_spec_224/\"):\n            os.mkdir(folder_path + \"mel_spec_224/\")\n        smel_img_path = folder_path + \"mel_spec_224/\" + filename_no_ext + \".jpg\"\n        if not os.path.isdir(folder_path + \"mfcc/\"):\n            os.mkdir(folder_path + \"mfcc/\")\n        mfcc_path = folder_path + \"mfcc/\" + filename_no_ext + \".npy\"\n        smel = save_melspec_image(y, smel_img_path)         \n        mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=20)        \n        np.save(mfcc_path, mfcc)\n    return file_path","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:18.999594Z","iopub.execute_input":"2022-03-22T07:05:18.999879Z","iopub.status.idle":"2022-03-22T07:05:19.009249Z","shell.execute_reply.started":"2022-03-22T07:05:18.999847Z","shell.execute_reply":"2022-03-22T07:05:19.008245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#result = process_audio(\"train/002549.ogg\", \"/kaggle/train/\")","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:19.010734Z","iopub.execute_input":"2022-03-22T07:05:19.011418Z","iopub.status.idle":"2022-03-22T07:05:19.022666Z","shell.execute_reply.started":"2022-03-22T07:05:19.011359Z","shell.execute_reply":"2022-03-22T07:05:19.021803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create mel spectrograms for train data\ncreate_processing_dirs()\ndelayed_funcs_train = [delayed(process_audio)(row[\"filepath\"], \"/kaggle/train/\") \n                       for i, row in df_train[df_train.file_exists].iterrows()]\nresults_train = Parallel(n_jobs=-1, verbose=5)(delayed_funcs_train)    ","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:19.023772Z","iopub.execute_input":"2022-03-22T07:05:19.024379Z","iopub.status.idle":"2022-03-22T07:05:53.905737Z","shell.execute_reply.started":"2022-03-22T07:05:19.024344Z","shell.execute_reply":"2022-03-22T07:05:53.904716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create mel spectrograms for test data\ndelayed_funcs_test = [delayed(process_audio)(row[\"filepath\"], \"/kaggle/test/\") \n                      for i, row in df_test[df_test.file_exists].iterrows()]\nresults_test = Parallel(n_jobs=-1, verbose=5)(delayed_funcs_test)    ","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:53.907044Z","iopub.status.idle":"2022-03-22T07:05:53.907586Z","shell.execute_reply.started":"2022-03-22T07:05:53.907391Z","shell.execute_reply":"2022-03-22T07:05:53.907412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# zip the processed_train and processed_test folders\n!zip -rq processed_train.zip /kaggle/train/\n!zip -rq processed_test.zip /kaggle/test/","metadata":{"execution":{"iopub.status.busy":"2022-03-22T07:05:53.908641Z","iopub.status.idle":"2022-03-22T07:05:53.909131Z","shell.execute_reply.started":"2022-03-22T07:05:53.908949Z","shell.execute_reply":"2022-03-22T07:05:53.908969Z"},"trusted":true},"execution_count":null,"outputs":[]}]}