{"cells":[{"metadata":{},"cell_type":"markdown","source":"---\n# Init"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport librosa\nimport librosa.display\n\nimport wave\nfrom matplotlib import pyplot as plt\nfrom IPython import display as ipd\nfrom scipy.io import wavfile","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Script containing lwlrap functions\n%run ../usr/lib/2019_fat_lwlrap/2019_fat_lwlrap.py","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SAMPLE_RATE = 44100\nN_MFCC = 50\nFIGSIZE = (16,9)\nWIDTH = 1.0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"DATA_PATH = '../input/freesound-audio-tagging-2019/'\n\nCURATED_CSV = DATA_PATH + 'train_curated.csv'\nCURATED_DIR = DATA_PATH + 'train_curated/'\nNOISY_CSV = DATA_PATH + 'train_noisy.csv'\nNOISY_DIR = DATA_PATH + 'train_noisy/'\nSAMPLE_CSV = DATA_PATH + 'sample_submission.csv'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preprocessing"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"def gen_duration(path):\n    def f(file):\n        with wave.open(path+file) as data:\n            return data.getnframes() / SAMPLE_RATE\n    return f\n\ntrain_curated = pd.read_csv(CURATED_CSV)\ntrain_curated['curated'] = True\ntrain_curated['duration'] = train_curated.fname.apply(gen_duration(CURATED_DIR))\n\ntrain_noisy = pd.DataFrame()\ntrain_noisy = pd.read_csv(NOISY_CSV)\ntrain_noisy['curated'] = False\ntrain_noisy['duration'] = train_noisy.fname.apply(gen_duration(NOISY_DIR))\n\ntrain = pd.concat([train_curated, train_noisy], ignore_index=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"---\n# Analysis"},{"metadata":{},"cell_type":"markdown","source":"## Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"# N-hot for labels\ncurated_labels = train[train.curated==True].labels.str.get_dummies(sep=',')\nnoisy_labels = train[train.curated==False].labels.str.get_dummies(sep=',')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- File with N labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"# plt.figure(figsize=FIGSIZE)\n\nplt.subplot(121)\nax = curated_labels.sum(axis=1).value_counts().sort_index().plot.bar()\nax.set_title('File with N labels : Curated')\nax.set_xlabel('label count')\nax.set_ylabel('count')\nplt.tight_layout()\n\nplt.subplot(122)\nax = noisy_labels.sum(axis=1).value_counts().sort_index().plot.bar()\nax.set_title('File with N labels : Noisy')\nax.set_xlabel('label count')\nax.set_ylabel('count')\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"- Samples per label(1-label)"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=FIGSIZE)\n\nax = curated_labels.sum().plot.bar(color='r', width=WIDTH)\nax = curated_labels[curated_labels.sum(axis=1)==1].sum().plot.bar(color='b', width=WIDTH)\nax.set_title('samples per label : Curated');\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=FIGSIZE)\n\nax = noisy_labels.sum().plot.bar(color='r', width=WIDTH)\nax = noisy_labels[noisy_labels.sum(axis=1)==1].sum().plot.bar(color='b', width=WIDTH)\nax.set_title('samples per label : Noisy')\nplt.tight_layout();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=FIGSIZE)\n\nlabels = train.labels.str.get_dummies(sep=',')\nax = labels.sum().plot.bar(color='r', width=WIDTH)\nax = labels[labels.sum(axis=1)==1].sum().plot.bar(color='b', width=WIDTH)\nax.set_title('samples per label : Combined')\nplt.tight_layout();","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Audio"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=FIGSIZE)\n\nN_BIN = 50\n\n# upper half\nplt.subplot(211)\nax1 = train.duration.hist(bins=N_BIN, color='r')\nax1 = train[train.curated==True].duration.hist(bins=N_BIN)\nplt.ylim([17000,18500])\n\n# lower half\nplt.subplot(212)\nax2 = train.duration.hist(bins=N_BIN, color='r')\nax2 = train[train.curated==True].duration.hist(bins=N_BIN)\nax2.set_xlabel('Duration (sec)')\nax2.set_ylabel('count');\nplt.ylim([0,1500]);\n\n\n# hide the spines between ax and ax2\nax1.spines['bottom'].set_visible(False)\nax2.spines['top'].set_visible(False)\nax1.xaxis.tick_top()\nax1.tick_params(labeltop='off')  # don't put tick labels at the top\nax2.xaxis.tick_bottom()\n\nd = .01  # how big to make the diagonal lines in axes coordinates\n# arguments to pass to plot, just so we don't keep repeating them\nkwargs = dict(transform=ax1.transAxes, color='k', clip_on=False)\nax1.plot((-d, +d), (-d, +d), **kwargs)        # top-left diagonal\nax1.plot((1 - d, 1 + d), (-d, +d), **kwargs)  # top-right diagonal\n\nkwargs.update(transform=ax2.transAxes)  # switch to the bottom axes\nax2.plot((-d, +d), (1 - d, 1 + d), **kwargs)  # bottom-left diagonal\nax2.plot((1 - d, 1 + d), (1 - d, 1 + d), **kwargs); # bottom-right diagonal","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"durations = []\nlabels = []\nfor index, row in train[(train.curated==True)].iterrows():\n    arr = row.labels.split(',')\n    durations.extend([row.duration]*len(arr))\n    labels.extend(arr)\n\nplt.figure(figsize=FIGSIZE)\nplt.grid()\nax = sns.boxplot(x=np.array(labels), y=np.array(durations));\nax.set_title('Duration distribution per class')\nplt.xticks(rotation=90);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def audio_analysis(x, draw_plot=True):\n    \n    if type(x) is str:\n        rows = train[train.fname==x]\n        assert len(rows)>0, f'\"{x}\" not found in training dataframe.'\n        x = rows.iloc[0]\n    \n    \n    DIR = CURATED_DIR if x.curated else NOISY_DIR\n    \n    # Obtain audio data\n    rate, data = wavfile.read(DIR + x.fname)\n    data = data.astype(float)\n    data /= data.max()\n    \n    if draw_plot:\n        # Plot\n        plt.figure(figsize=FIGSIZE)\n\n        # Waveform\n        ax = plt.subplot(311)\n        ax.set_title('{} / {} / {}'.format(x.fname, x.labels, f'Curated={x.curated}'))\n        ax.set_ylabel('Amplitude')    \n        librosa.display.waveplot(data, sr=rate)\n\n        # Spectrogram\n        ax = plt.subplot(312)\n        S = librosa.feature.melspectrogram(data, sr=rate, n_mels=128)\n        log_S = librosa.power_to_db(S, ref=np.max)\n        librosa.display.specshow(log_S, sr=rate, x_axis='time', y_axis='mel')\n        ax.set_title('Mel power spectrogram ')\n        plt.colorbar(format='%+02.0f dB')\n        plt.tight_layout()\n\n        # MFCC coefficient\n        ax = plt.subplot(313)\n        mfcc = librosa.feature.mfcc(S=log_S, n_mfcc=13)\n        delta2_mfcc = librosa.feature.delta(mfcc, order=2)\n        librosa.display.specshow(delta2_mfcc)\n        plt.ylabel('MFCC coeffs')\n        plt.xlabel('Time')\n        plt.title('MFCC')\n        plt.colorbar()\n        plt.tight_layout()\n    \n    return ipd.Audio(filename=DIR+x.fname)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# numerous labels\naudio_analysis('4dad3998.wav')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Distinct labels\naudio_analysis('d828ffce.wav')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Mixed with other sounds\naudio_analysis('8250b48f.wav')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Incorrect label '6a1f682a.wav'\naudio_analysis('f76181c4.wav')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}