{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"05a7e0382f0469b8a1ffea641af517af5d1464c6","collapsed":false,"_cell_guid":"22d1443c-d944-4606-b9b3-c0d62ba57686","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:47:05.223880Z","iopub.execute_input":"2022-07-29T09:47:05.224827Z","iopub.status.idle":"2022-07-29T09:47:05.267409Z","shell.execute_reply.started":"2022-07-29T09:47:05.224730Z","shell.execute_reply":"2022-07-29T09:47:05.266236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ## Data augmentation definition :\n* Data augmentation is the process by which we create new synthetic training samples by adding small perturbations on our initial training set.\n* The objective is to make our model invariant to those perturbations and enhace its ability to generalize.\n* In order to this to work adding the perturbations must conserve the same label as the original training sample.\n* In images data augmention can be performed by shifting the image, zooming, rotating ... \n* In our case we will add noise, stretch and roll, pitch shift ... ","metadata":{"_uuid":"9d5c385c33d14f1cc1fa313cabfffa4b1deae47f","_cell_guid":"48e96efd-60ce-46f7-a6e8-cc16aba4c9ca"}},{"cell_type":"code","source":"#Import stuff\n\nimport numpy as np\nimport random\nimport itertools\nimport librosa\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\n\n%matplotlib inline","metadata":{"_uuid":"5afa490db6d79308f6de183ec0be35304ff97c5c","_cell_guid":"69d385d7-44b0-4b14-a406-e78ee04b1b4b","execution":{"iopub.status.busy":"2022-07-29T09:47:08.618303Z","iopub.execute_input":"2022-07-29T09:47:08.618905Z","iopub.status.idle":"2022-07-29T09:47:19.977300Z","shell.execute_reply.started":"2022-07-29T09:47:08.618856Z","shell.execute_reply":"2022-07-29T09:47:19.976301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio_file(file_path):\n    input_length = 16000\n    data = librosa.core.load(file_path)[0] #, sr=16000\n    if len(data)>input_length:\n        data = data[:input_length]\n    else:\n        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n    return data\ndef plot_time_series(data):\n    fig = plt.figure(figsize=(14, 8))\n    plt.title('Raw wave ')\n    plt.ylabel('Amplitude')\n    plt.plot(np.linspace(0, 1, len(data)), data)\n    plt.show()","metadata":{"_uuid":"b232201620caaa349d7f07d74780d1242ab7a4fc","collapsed":false,"_cell_guid":"06dfa34d-cf83-4eb7-855c-b298bb4daed0","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:47:51.225787Z","iopub.execute_input":"2022-07-29T09:47:51.226261Z","iopub.status.idle":"2022-07-29T09:47:51.237598Z","shell.execute_reply.started":"2022-07-29T09:47:51.226217Z","shell.execute_reply":"2022-07-29T09:47:51.236470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_30999159.wav\")\nplot_time_series(data)\n\n#aud_01","metadata":{"_uuid":"9d28c1dfbb211c875f026ab73dba231ea572e238","collapsed":false,"_cell_guid":"b9db3341-0145-4f9c-a33f-7c10951ff6c5","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:53:43.201424Z","iopub.execute_input":"2022-07-29T09:53:43.202469Z","iopub.status.idle":"2022-07-29T09:53:43.686903Z","shell.execute_reply.started":"2022-07-29T09:53:43.202419Z","shell.execute_reply":"2022-07-29T09:53:43.685707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"_uuid":"c698649bd96c4ccc232526ea9b9681f3d358369f","collapsed":false,"_cell_guid":"6f70aea2-f984-4aad-ba54-5861134bc117","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:54:09.199199Z","iopub.execute_input":"2022-07-29T09:54:09.200218Z","iopub.status.idle":"2022-07-29T09:54:09.211112Z","shell.execute_reply.started":"2022-07-29T09:54:09.200175Z","shell.execute_reply":"2022-07-29T09:54:09.209853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"_uuid":"f45ff5f8132b8516522464d3f90277f0197fa71c","collapsed":false,"_cell_guid":"8a346d98-18c4-4be8-802d-a94a8c3fd088","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:54:16.596212Z","iopub.execute_input":"2022-07-29T09:54:16.597334Z","iopub.status.idle":"2022-07-29T09:54:16.806211Z","shell.execute_reply.started":"2022-07-29T09:54:16.597287Z","shell.execute_reply":"2022-07-29T09:54:16.805288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"_uuid":"875d393f58b7948752802d897783d054dc6b676f","collapsed":false,"_cell_guid":"3dd85e75-3d49-4187-9e2f-14b2638a13c2","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:54:20.648821Z","iopub.execute_input":"2022-07-29T09:54:20.649177Z","iopub.status.idle":"2022-07-29T09:54:20.845979Z","shell.execute_reply.started":"2022-07-29T09:54:20.649147Z","shell.execute_reply":"2022-07-29T09:54:20.845051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stretching the sound\ndef stretch(data, rate=1):\n    input_length = 16000\n    data = librosa.effects.time_stretch(data, rate)\n    if len(data)>input_length:\n        data = data[:input_length]\n    else:\n        data = np.pad(data, (0, max(0, input_length - len(data))), \"constant\")\n\n    return data\n\n\ndata_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"_uuid":"f786049cbbd24ee4fbfd8beb0fc3601e8d2d2051","collapsed":false,"_cell_guid":"128047ec-5389-4a41-a37d-3b973ce0daeb","jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-07-29T09:54:25.638065Z","iopub.execute_input":"2022-07-29T09:54:25.639090Z","iopub.status.idle":"2022-07-29T09:54:26.462472Z","shell.execute_reply.started":"2022-07-29T09:54:25.639049Z","shell.execute_reply":"2022-07-29T09:54:26.461409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can now plug all those transformations in your keras data generator and see your LB rank go up :D","metadata":{"_uuid":"872e68bc7de37759466996b48fad430fae7243bc","_cell_guid":"bf197741-38a1-487c-b99c-2585dde6f311"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31565582.wav\")\nplot_time_series(data)\n\n#2nd aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:57:14.057561Z","iopub.execute_input":"2022-07-29T09:57:14.058351Z","iopub.status.idle":"2022-07-29T09:57:14.525436Z","shell.execute_reply.started":"2022-07-29T09:57:14.058311Z","shell.execute_reply":"2022-07-29T09:57:14.524354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:57:18.690070Z","iopub.execute_input":"2022-07-29T09:57:18.690484Z","iopub.status.idle":"2022-07-29T09:57:18.699206Z","shell.execute_reply.started":"2022-07-29T09:57:18.690448Z","shell.execute_reply":"2022-07-29T09:57:18.698246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:57:23.520374Z","iopub.execute_input":"2022-07-29T09:57:23.521048Z","iopub.status.idle":"2022-07-29T09:57:23.850705Z","shell.execute_reply.started":"2022-07-29T09:57:23.521013Z","shell.execute_reply":"2022-07-29T09:57:23.849790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:57:30.457525Z","iopub.execute_input":"2022-07-29T09:57:30.458515Z","iopub.status.idle":"2022-07-29T09:57:30.660761Z","shell.execute_reply.started":"2022-07-29T09:57:30.458478Z","shell.execute_reply":"2022-07-29T09:57:30.659778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T09:57:36.342130Z","iopub.execute_input":"2022-07-29T09:57:36.342537Z","iopub.status.idle":"2022-07-29T09:57:36.752326Z","shell.execute_reply.started":"2022-07-29T09:57:36.342503Z","shell.execute_reply":"2022-07-29T09:57:36.751417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31547085.wav\")\nplot_time_series(data)\n\n#3rd aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:00:02.954008Z","iopub.execute_input":"2022-07-29T10:00:02.954375Z","iopub.status.idle":"2022-07-29T10:00:03.451397Z","shell.execute_reply.started":"2022-07-29T10:00:02.954345Z","shell.execute_reply":"2022-07-29T10:00:03.450340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:00:10.665033Z","iopub.execute_input":"2022-07-29T10:00:10.665460Z","iopub.status.idle":"2022-07-29T10:00:10.675510Z","shell.execute_reply.started":"2022-07-29T10:00:10.665415Z","shell.execute_reply":"2022-07-29T10:00:10.674431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:00:24.080333Z","iopub.execute_input":"2022-07-29T10:00:24.081415Z","iopub.status.idle":"2022-07-29T10:00:24.450361Z","shell.execute_reply.started":"2022-07-29T10:00:24.081345Z","shell.execute_reply":"2022-07-29T10:00:24.449294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:00:39.127214Z","iopub.execute_input":"2022-07-29T10:00:39.128233Z","iopub.status.idle":"2022-07-29T10:00:39.361578Z","shell.execute_reply.started":"2022-07-29T10:00:39.128196Z","shell.execute_reply":"2022-07-29T10:00:39.360628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:00:44.047124Z","iopub.execute_input":"2022-07-29T10:00:44.049677Z","iopub.status.idle":"2022-07-29T10:00:44.524480Z","shell.execute_reply.started":"2022-07-29T10:00:44.049638Z","shell.execute_reply":"2022-07-29T10:00:44.523428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31000407.wav\")\nplot_time_series(data)\n\n#4th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:02:38.329347Z","iopub.execute_input":"2022-07-29T10:02:38.330403Z","iopub.status.idle":"2022-07-29T10:02:38.829124Z","shell.execute_reply.started":"2022-07-29T10:02:38.330349Z","shell.execute_reply":"2022-07-29T10:02:38.828009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:02:43.702165Z","iopub.execute_input":"2022-07-29T10:02:43.702568Z","iopub.status.idle":"2022-07-29T10:02:43.712174Z","shell.execute_reply.started":"2022-07-29T10:02:43.702535Z","shell.execute_reply":"2022-07-29T10:02:43.711114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:02:46.450045Z","iopub.execute_input":"2022-07-29T10:02:46.450455Z","iopub.status.idle":"2022-07-29T10:02:47.022220Z","shell.execute_reply.started":"2022-07-29T10:02:46.450422Z","shell.execute_reply":"2022-07-29T10:02:47.021123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:02:54.160933Z","iopub.execute_input":"2022-07-29T10:02:54.161308Z","iopub.status.idle":"2022-07-29T10:02:54.378596Z","shell.execute_reply.started":"2022-07-29T10:02:54.161279Z","shell.execute_reply":"2022-07-29T10:02:54.377662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:02:58.383532Z","iopub.execute_input":"2022-07-29T10:02:58.383907Z","iopub.status.idle":"2022-07-29T10:02:59.060477Z","shell.execute_reply.started":"2022-07-29T10:02:58.383878Z","shell.execute_reply":"2022-07-29T10:02:59.059264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31563142.wav\")\nplot_time_series(data)\n\n#5th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:05:58.277906Z","iopub.execute_input":"2022-07-29T10:05:58.278275Z","iopub.status.idle":"2022-07-29T10:05:58.740658Z","shell.execute_reply.started":"2022-07-29T10:05:58.278246Z","shell.execute_reply":"2022-07-29T10:05:58.739122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31604699.wav\")\nplot_time_series(data)\n\n#6th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:07:17.362776Z","iopub.execute_input":"2022-07-29T10:07:17.363159Z","iopub.status.idle":"2022-07-29T10:07:17.806447Z","shell.execute_reply.started":"2022-07-29T10:07:17.363131Z","shell.execute_reply":"2022-07-29T10:07:17.805328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:07:22.951679Z","iopub.execute_input":"2022-07-29T10:07:22.952068Z","iopub.status.idle":"2022-07-29T10:07:22.962312Z","shell.execute_reply.started":"2022-07-29T10:07:22.952036Z","shell.execute_reply":"2022-07-29T10:07:22.960969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:07:28.164899Z","iopub.execute_input":"2022-07-29T10:07:28.165266Z","iopub.status.idle":"2022-07-29T10:07:28.510813Z","shell.execute_reply.started":"2022-07-29T10:07:28.165235Z","shell.execute_reply":"2022-07-29T10:07:28.508547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:07:32.729920Z","iopub.execute_input":"2022-07-29T10:07:32.730292Z","iopub.status.idle":"2022-07-29T10:07:32.930626Z","shell.execute_reply.started":"2022-07-29T10:07:32.730263Z","shell.execute_reply":"2022-07-29T10:07:32.929648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:07:40.855121Z","iopub.execute_input":"2022-07-29T10:07:40.855521Z","iopub.status.idle":"2022-07-29T10:07:41.299825Z","shell.execute_reply.started":"2022-07-29T10:07:40.855490Z","shell.execute_reply":"2022-07-29T10:07:41.298910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31531430.wav\")\nplot_time_series(data)\n\n#7th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:09:40.656714Z","iopub.execute_input":"2022-07-29T10:09:40.657096Z","iopub.status.idle":"2022-07-29T10:09:41.192736Z","shell.execute_reply.started":"2022-07-29T10:09:40.657067Z","shell.execute_reply":"2022-07-29T10:09:41.191603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:09:47.022859Z","iopub.execute_input":"2022-07-29T10:09:47.023243Z","iopub.status.idle":"2022-07-29T10:09:47.034435Z","shell.execute_reply.started":"2022-07-29T10:09:47.023211Z","shell.execute_reply":"2022-07-29T10:09:47.032567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:09:52.164516Z","iopub.execute_input":"2022-07-29T10:09:52.165256Z","iopub.status.idle":"2022-07-29T10:09:52.542534Z","shell.execute_reply.started":"2022-07-29T10:09:52.165219Z","shell.execute_reply":"2022-07-29T10:09:52.541558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:09:58.202440Z","iopub.execute_input":"2022-07-29T10:09:58.203491Z","iopub.status.idle":"2022-07-29T10:09:58.403349Z","shell.execute_reply.started":"2022-07-29T10:09:58.203442Z","shell.execute_reply":"2022-07-29T10:09:58.402314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:10:01.438206Z","iopub.execute_input":"2022-07-29T10:10:01.438916Z","iopub.status.idle":"2022-07-29T10:10:01.868883Z","shell.execute_reply.started":"2022-07-29T10:10:01.438874Z","shell.execute_reply":"2022-07-29T10:10:01.867731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31556731.wav\")\nplot_time_series(data)\n\n#8th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:11:31.205012Z","iopub.execute_input":"2022-07-29T10:11:31.205440Z","iopub.status.idle":"2022-07-29T10:11:31.704537Z","shell.execute_reply.started":"2022-07-29T10:11:31.205402Z","shell.execute_reply":"2022-07-29T10:11:31.703437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:11:35.065657Z","iopub.execute_input":"2022-07-29T10:11:35.066059Z","iopub.status.idle":"2022-07-29T10:11:35.075629Z","shell.execute_reply.started":"2022-07-29T10:11:35.066025Z","shell.execute_reply":"2022-07-29T10:11:35.074329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:11:39.726681Z","iopub.execute_input":"2022-07-29T10:11:39.727065Z","iopub.status.idle":"2022-07-29T10:11:40.038790Z","shell.execute_reply.started":"2022-07-29T10:11:39.727031Z","shell.execute_reply":"2022-07-29T10:11:40.037713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:11:59.119517Z","iopub.execute_input":"2022-07-29T10:11:59.120224Z","iopub.status.idle":"2022-07-29T10:11:59.341176Z","shell.execute_reply.started":"2022-07-29T10:11:59.120188Z","shell.execute_reply":"2022-07-29T10:11:59.340252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:12:05.261979Z","iopub.execute_input":"2022-07-29T10:12:05.262351Z","iopub.status.idle":"2022-07-29T10:12:05.724720Z","shell.execute_reply.started":"2022-07-29T10:12:05.262319Z","shell.execute_reply":"2022-07-29T10:12:05.723839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31565674.wav\")\nplot_time_series(data)\n\n#9th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:13:55.741369Z","iopub.execute_input":"2022-07-29T10:13:55.741915Z","iopub.status.idle":"2022-07-29T10:13:56.225004Z","shell.execute_reply.started":"2022-07-29T10:13:55.741872Z","shell.execute_reply":"2022-07-29T10:13:56.223344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = load_audio_file(\"../input/dl-sprint-buet-cse-fest-2022-processed-dataset/new_train/common_voice_bn_31546405.wav\")\nplot_time_series(data)\n\n#10th aud","metadata":{"execution":{"iopub.status.busy":"2022-07-29T10:16:04.090851Z","iopub.execute_input":"2022-07-29T10:16:04.091241Z","iopub.status.idle":"2022-07-29T10:16:04.571978Z","shell.execute_reply.started":"2022-07-29T10:16:04.091210Z","shell.execute_reply":"2022-07-29T10:16:04.570341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Hear it ! \nipd.Audio(data, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding white noise \nwn = np.random.randn(len(data))\ndata_wn = data + 0.005*wn\nplot_time_series(data_wn)\n# We limited the amplitude of the noise so we can still hear the word even with the noise, \n#which is the objective\nipd.Audio(data_wn, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shifting the sound\ndata_roll = np.roll(data, 1600)\nplot_time_series(data_roll)\nipd.Audio(data_roll, rate=16000)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_stretch =stretch(data, 0.8)\nprint(\"This makes the sound deeper but we can still hear 'off' \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)\n\ndata_stretch =stretch(data, 1.2)\nprint(\"Higher frequencies  \")\nplot_time_series(data_stretch)\nipd.Audio(data_stretch, rate=16000)","metadata":{},"execution_count":null,"outputs":[]}]}