{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"05a7e0382f0469b8a1ffea641af517af5d1464c6","collapsed":false,"_cell_guid":"22d1443c-d944-4606-b9b3-c0d62ba57686","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T14:25:42.851783Z","iopub.execute_input":"2022-07-29T14:25:42.852355Z","iopub.status.idle":"2022-07-29T14:25:42.873337Z","shell.execute_reply.started":"2022-07-29T14:25:42.852304Z","shell.execute_reply":"2022-07-29T14:25:42.871874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Data augmentation definition :\n* Data augmentation is the process by which we create new synthetic training samples by adding small perturbations on our initial training set.\n* The objective is to make our model invariant to those perturbations and enhace its ability to generalize.\n* In order to this to work adding the perturbations must conserve the same label as the original training sample.\n* In images data augmention can be performed by shifting the image, zooming, rotating ... \n* In our case we will add noise, stretch and roll, pitch shift ... ","metadata":{"_uuid":"9d5c385c33d14f1cc1fa313cabfffa4b1deae47f","_cell_guid":"48e96efd-60ce-46f7-a6e8-cc16aba4c9ca"}},{"cell_type":"code","source":"#Import stuff\n\nimport numpy as np\nimport random\nimport itertools\nimport librosa\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\n\n%matplotlib inline","metadata":{"_uuid":"5afa490db6d79308f6de183ec0be35304ff97c5c","_cell_guid":"69d385d7-44b0-4b14-a406-e78ee04b1b4b","collapsed":true,"jupyter":{"outputs_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio_file(file_path):\n    input_length = 16000\n    data = librosa.core.load(file_path)[0] #, sr=16000\n    if len(data)>input_length:\n        data = data[:input_length]\n    else:\n        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n    return data\ndef plot_time_series(data):\n    fig = plt.figure(figsize=(14, 8))\n    plt.title('Raw wave ')\n    plt.ylabel('Amplitude')\n    plt.plot(np.linspace(0, 1, len(data)), data)\n    plt.show()","metadata":{"_uuid":"b232201620caaa349d7f07d74780d1242ab7a4fc","collapsed":false,"_cell_guid":"06dfa34d-cf83-4eb7-855c-b298bb4daed0","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/train/audio/off/1df483c0_nohash_0.wav\")\nplot_time_series(data)","metadata":{"_uuid":"9d28c1dfbb211c875f026ab73dba231ea572e238","collapsed":false,"_cell_guid":"b9db3341-0145-4f9c-a33f-7c10951ff6c5","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"_uuid":"c698649bd96c4ccc232526ea9b9681f3d358369f","collapsed":false,"_cell_guid":"6f70aea2-f984-4aad-ba54-5861134bc117","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"_uuid":"f45ff5f8132b8516522464d3f90277f0197fa71c","collapsed":false,"_cell_guid":"8a346d98-18c4-4be8-802d-a94a8c3fd088","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"_uuid":"875d393f58b7948752802d897783d054dc6b676f","collapsed":false,"_cell_guid":"3dd85e75-3d49-4187-9e2f-14b2638a13c2","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stretching the sound\ndef stretch(data, rate=1):\n    input_length = 16000\n    data = librosa.effects.time_stretch(data, rate)\n    if len(data)>input_length:\n        data = data[:input_length]\n    else:\n        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n\n    return data\n\n\ndata_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"_uuid":"f786049cbbd24ee4fbfd8beb0fc3601e8d2d2051","collapsed":false,"_cell_guid":"128047ec-5389-4a41-a37d-3b973ce0daeb","jupyter":{"outputs_hidden":false}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can now plug all those transformations in your keras data generator and see your LB rank go up :D","metadata":{"_uuid":"872e68bc7de37759466996b48fad430fae7243bc","_cell_guid":"bf197741-38a1-487c-b99c-2585dde6f311","collapsed":true,"jupyter":{"outputs_hidden":true}},"execution_count":null,"outputs":[]}]}