{"metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"mimetype": "text/x-python", "file_extension": ".py", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", "name": "python", "version": "3.6.3", "codemirror_mode": {"name": "ipython", "version": 3}}}, "nbformat_minor": 0, "nbformat": 4, "cells": [{"cell_type": "code", "metadata": {"_uuid": "05a7e0382f0469b8a1ffea641af517af5d1464c6", "collapsed": false, "_cell_guid": "22d1443c-d944-4606-b9b3-c0d62ba57686"}, "outputs": [], "execution_count": null, "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "# Input data files are available in the \"../input/\" directory.\n", "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n", "\n", "# Any results you write to the current directory are saved as output."]}, {"cell_type": "markdown", "metadata": {"_uuid": "9d5c385c33d14f1cc1fa313cabfffa4b1deae47f", "_cell_guid": "48e96efd-60ce-46f7-a6e8-cc16aba4c9ca"}, "source": [" ## Data augmentation definition :\n", "* Data augmentation is the process by which we create new synthetic training samples by adding small perturbations on our initial training set.\n", "* The objective is to make our model invariant to those perturbations and enhace its ability to generalize.\n", "* In order to this to work adding the perturbations must conserve the same label as the original training sample.\n", "* In images data augmention can be performed by shifting the image, zooming, rotating ... \n", "* In our case we will add noise, stretch and roll, pitch shift ... "]}, {"cell_type": "code", "metadata": {"_uuid": "5afa490db6d79308f6de183ec0be35304ff97c5c", "collapsed": true, "_cell_guid": "69d385d7-44b0-4b14-a406-e78ee04b1b4b"}, "outputs": [], "execution_count": null, "source": ["#Import stuff\n", "\n", "import numpy as np\n", "import random\n", "import itertools\n", "import librosa\n", "import IPython.display as ipd\n", "import matplotlib.pyplot as plt\n", "\n", "%matplotlib inline"]}, {"cell_type": "code", "metadata": {"_uuid": "b232201620caaa349d7f07d74780d1242ab7a4fc", "collapsed": false, "_cell_guid": "06dfa34d-cf83-4eb7-855c-b298bb4daed0"}, "outputs": [], "execution_count": null, "source": ["def load_audio_file(file_path):\n", "    input_length = 16000\n", "    data = librosa.core.load(file_path)[0] #, sr=16000\n", "    if len(data)>input_length:\n", "        data = data[:input_length]\n", "    else:\n", "        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n", "    return data\n", "def plot_time_series(data):\n", "    fig = plt.figure(figsize=(14, 8))\n", "    plt.title('Raw wave ')\n", "    plt.ylabel('Amplitude')\n", "    plt.plot(np.linspace(0, 1, len(data)), data)\n", "    plt.show()"]}, {"cell_type": "code", "metadata": {"_uuid": "9d28c1dfbb211c875f026ab73dba231ea572e238", "collapsed": false, "_cell_guid": "b9db3341-0145-4f9c-a33f-7c10951ff6c5"}, "outputs": [], "execution_count": null, "source": ["data = load_audio_file(\"../input/train/audio/off/1df483c0_nohash_0.wav\")\n", "plot_time_series(data)"]}, {"cell_type": "code", "metadata": {"_uuid": "c698649bd96c4ccc232526ea9b9681f3d358369f", "collapsed": false, "_cell_guid": "6f70aea2-f984-4aad-ba54-5861134bc117"}, "outputs": [], "execution_count": null, "source": ["#Hear it ! \n", "ipd.Audio(data, rate=16000)"]}, {"cell_type": "code", "metadata": {"_uuid": "f45ff5f8132b8516522464d3f90277f0197fa71c", "collapsed": false, "_cell_guid": "8a346d98-18c4-4be8-802d-a94a8c3fd088"}, "outputs": [], "execution_count": null, "source": ["# Adding white noise \n", "wn = np.random.randn(len(data))\n", "data_wn = data + 0.005*wn\n", "plot_time_series(data_wn)\n", "# We limited the amplitude of the noise so we can still hear the word even with the noise, \n", "#which is the objective\n", "ipd.Audio(data_wn, rate=16000)"]}, {"cell_type": "code", "metadata": {"_uuid": "875d393f58b7948752802d897783d054dc6b676f", "collapsed": false, "_cell_guid": "3dd85e75-3d49-4187-9e2f-14b2638a13c2"}, "outputs": [], "execution_count": null, "source": ["# Shifting the sound\n", "data_roll = np.roll(data, 1600)\n", "plot_time_series(data_roll)\n", "ipd.Audio(data_roll, rate=16000)"]}, {"cell_type": "code", "metadata": {"_uuid": "f786049cbbd24ee4fbfd8beb0fc3601e8d2d2051", "collapsed": false, "_cell_guid": "128047ec-5389-4a41-a37d-3b973ce0daeb"}, "outputs": [], "execution_count": null, "source": ["# stretching the sound\n", "def stretch(data, rate=1):\n", "    input_length = 16000\n", "    data = librosa.effects.time_stretch(data, rate)\n", "    if len(data)>input_length:\n", "        data = data[:input_length]\n", "    else:\n", "        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n", "\n", "    return data\n", "\n", "\n", "data_stretch =stretch(data, 0.8)\n", "print(\"This makes the sound deeper but we can still hear 'off' \")\n", "plot_time_series(data_stretch)\n", "ipd.Audio(data_stretch, rate=16000)\n", "\n", "data_stretch =stretch(data, 1.2)\n", "print(\"Higher frequencies  \")\n", "plot_time_series(data_stretch)\n", "ipd.Audio(data_stretch, rate=16000)"]}, {"cell_type": "code", "metadata": {"_uuid": "872e68bc7de37759466996b48fad430fae7243bc", "collapsed": true, "_cell_guid": "bf197741-38a1-487c-b99c-2585dde6f311"}, "outputs": [], "execution_count": null, "source": ["# You can now plug all those transformations in your keras data generator and see your LB rank go up :D"]}]}