{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pyunpack\n!pip install patool\n!pip install py7zr\n!pip install sounddevice\n!pip install noisereduce\n!pip install librosa\n! pip install python_speech_features\n! pip install tensorflow==2.4\n! pip install malaya_speech\n! pip install webrtcvad","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:58:21.891105Z","iopub.execute_input":"2022-01-01T20:58:21.891428Z","iopub.status.idle":"2022-01-01T21:00:07.091488Z","shell.execute_reply.started":"2022-01-01T20:58:21.891399Z","shell.execute_reply":"2022-01-01T21:00:07.090392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Import the libraries**\n\nFirst, import all the necessary libraries into our notebook. LibROSA and SciPy are the Python libraries used for processing audio signals.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom py7zr import unpack_7zarchive\nimport shutil\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nimport librosa\nimport IPython.display as ipd\nfrom scipy.io import wavfile\n\nimport noisereduce as nr\nimport tensorflow \nfrom malaya_speech import Pipeline\n\nimport malaya_speech\nimport os\n\nfrom python_speech_features import mfcc\n\nfrom sklearn.preprocessing import LabelEncoder\nimport seaborn as sn","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:00:07.093989Z","iopub.execute_input":"2022-01-01T21:00:07.094363Z","iopub.status.idle":"2022-01-01T21:00:11.891822Z","shell.execute_reply.started":"2022-01-01T21:00:07.09432Z","shell.execute_reply":"2022-01-01T21:00:11.890819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.register_unpack_format('7zip', ['.7z'], unpack_7zarchive)\nshutil.unpack_archive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z', '/kaggle/working/tensorflow-speech-recognition-challenge/train/')","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:00:11.893723Z","iopub.execute_input":"2022-01-01T21:00:11.894072Z","iopub.status.idle":"2022-01-01T21:02:36.547356Z","shell.execute_reply.started":"2022-01-01T21:00:11.894033Z","shell.execute_reply":"2022-01-01T21:02:36.54637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from pyunpack import Archive\n# import shutil\n# if not os.path.exists('/kaggle/working/tensorflow-speech-recognition-challenge/train/'):\n#     os.makedirs('/kaggle/working/tensorflow-speech-recognition-challenge/train/')\n# Archive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z').extractall('/kaggle/working/tensorflow-speech-recognition-challenge/train/')\n\n#for dirname, _, filenames in os.walk('/kaggle/working/tensorflow-speech-recognition-challenge/train/train/audio'):\n #   for filename in filename[:5]:\n  #      print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2021-12-31T13:41:37.431539Z","iopub.execute_input":"2021-12-31T13:41:37.431891Z","iopub.status.idle":"2021-12-31T13:41:37.437226Z","shell.execute_reply.started":"2021-12-31T13:41:37.431857Z","shell.execute_reply":"2021-12-31T13:41:37.436335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <center> Implementing the Speech-to-Text Model in Python\n**Understanding the Problem Statement for our Speech-to-Text Project**\n\nLet’s understand the problem statement of our project before we move into the implementation part.\n\nWe might be on the verge of having too many screens around us. It seems like every day, new versions of common objects are “re-invented” with built-in wifi and bright touchscreens. A promising antidote to our screen addiction is voice interfaces. \n\n__You can download the dataset from__ [here](https://www.kaggle.com/c/tensorflow-speech-recognition-challenge).\n    \nTensorFlow recently released the Speech Commands Datasets. It includes 65,000 one-second long utterances of 30 short words, by thousands of different people. We’ll build a speech recognition system that understands simple spoken commands. <br>    ","metadata":{}},{"cell_type":"markdown","source":"**Data Exploration and Visualization**\n\nData Exploration and Visualization helps us to understand the data as well as pre-processing steps in a better way. \n\n**Visualization of Audio signal in time series domain**\n\nNow, we’ll visualize the audio signal in the time series domain:","metadata":{}},{"cell_type":"code","source":"train_audio_path = '/kaggle/working/tensorflow-speech-recognition-challenge/train/train/audio/'","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:02:36.549643Z","iopub.execute_input":"2022-01-01T21:02:36.55013Z","iopub.status.idle":"2022-01-01T21:02:36.55442Z","shell.execute_reply.started":"2022-01-01T21:02:36.550091Z","shell.execute_reply":"2022-01-01T21:02:36.553616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Accessing each file in data**","metadata":{}},{"cell_type":"code","source":"#!apt-get install -y p7zip-full\n#!7z x ../input/tensorflow-speech-recognition-challenge/train.7z","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:02:36.555628Z","iopub.execute_input":"2022-01-01T21:02:36.556108Z","iopub.status.idle":"2022-01-01T21:02:37.746786Z","shell.execute_reply.started":"2022-01-01T21:02:36.556072Z","shell.execute_reply":"2022-01-01T21:02:37.745777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples, sample_rate = librosa.load('../input/onnnnnnn/on1.wav', sr = 16000)\nfig = plt.figure(figsize=(14, 8))\nax1 = fig.add_subplot(211)\nax1.set_title('Raw wave of ' + '../input/train/audio/on/0a7c2a8d_nohash_0.wav')\nax1.set_xlabel('time')\nax1.set_ylabel('Amplitude')\nax1.plot(np.linspace(0, sample_rate/len(samples), sample_rate), samples)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:02:59.176424Z","iopub.execute_input":"2022-01-01T21:02:59.176762Z","iopub.status.idle":"2022-01-01T21:02:59.698349Z","shell.execute_reply.started":"2022-01-01T21:02:59.17673Z","shell.execute_reply":"2022-01-01T21:02:59.696524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sampling rate **\n\nLet us now look at the sampling rate of the audio signals","metadata":{}},{"cell_type":"code","source":"ipd.Audio(samples, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:03:14.149065Z","iopub.execute_input":"2022-01-01T21:03:14.149417Z","iopub.status.idle":"2022-01-01T21:03:14.160397Z","shell.execute_reply.started":"2022-01-01T21:03:14.149381Z","shell.execute_reply":"2022-01-01T21:03:14.159505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sample_rate)\nsig1=samples\nfs=sample_rate\nsr=fs","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:07.682294Z","iopub.execute_input":"2022-01-01T19:27:07.682611Z","iopub.status.idle":"2022-01-01T19:27:07.69282Z","shell.execute_reply.started":"2022-01-01T19:27:07.68258Z","shell.execute_reply":"2022-01-01T19:27:07.691796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time = np.linspace(0, len(sig1 - 1) / fs, len(sig1 - 1))\nreduced_noise1 = nr.reduce_noise(y=sig1, sr=fs,stationary=True)\nplt.plot(time, reduced_noise1)  # plot in seconds\n#reduced_noise2 = nr.reduce_noise(y=sig2, sr=fs,stationary=True)\n#plt.plot(time, reduced_noise2)  # plot in seconds\n#plt.title(\"Voice Signal\")\nplt.xlabel(\"Time [seconds]\")\nplt.ylabel(\"Voice amplitude\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:11.174197Z","iopub.execute_input":"2022-01-01T19:27:11.174519Z","iopub.status.idle":"2022-01-01T19:27:11.342239Z","shell.execute_reply.started":"2022-01-01T19:27:11.174487Z","shell.execute_reply":"2022-01-01T19:27:11.341258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ipd.Audio(reduced_noise1, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:16.066124Z","iopub.execute_input":"2022-01-01T19:27:16.066449Z","iopub.status.idle":"2022-01-01T19:27:16.073994Z","shell.execute_reply.started":"2022-01-01T19:27:16.066418Z","shell.execute_reply":"2022-01-01T19:27:16.073143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Silence Removal\nvad = malaya_speech.vad.webrtc()\ny=reduced_noise1\ny_= malaya_speech.resample(y, sr, 16000)\ny_ = malaya_speech.astype.float_to_int(y_)\nframes = malaya_speech.generator.frames(y, 30, sr)\nframes_ = list(malaya_speech.generator.frames(y_, 30, 16000, append_ending_trail = False))\nframes_webrtc = [(frames[no], vad(frame)) for no, frame in enumerate(frames_)]\ny_ = malaya_speech.combine.without_silent(frames_webrtc)\ny_","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:08:40.154413Z","iopub.execute_input":"2022-01-01T21:08:40.154768Z","iopub.status.idle":"2022-01-01T21:08:40.327814Z","shell.execute_reply.started":"2022-01-01T21:08:40.154736Z","shell.execute_reply":"2022-01-01T21:08:40.326039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ipd.Audio(y_, rate = sr )","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:27.491448Z","iopub.execute_input":"2022-01-01T19:27:27.491798Z","iopub.status.idle":"2022-01-01T19:27:27.498695Z","shell.execute_reply.started":"2022-01-01T19:27:27.491755Z","shell.execute_reply":"2022-01-01T19:27:27.49789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero = np.zeros((1*sr+4000-y_.shape[0]))\nsignal = np.concatenate((y_,zero))\nsignal.shape\ntime = np.linspace(0, len(signal - 1) / fs, len(signal - 1))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:30.479274Z","iopub.execute_input":"2022-01-01T19:27:30.479663Z","iopub.status.idle":"2022-01-01T19:27:30.484687Z","shell.execute_reply.started":"2022-01-01T19:27:30.479621Z","shell.execute_reply":"2022-01-01T19:27:30.483874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(time,signal)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:33.205995Z","iopub.execute_input":"2022-01-01T19:27:33.206322Z","iopub.status.idle":"2022-01-01T19:27:33.351382Z","shell.execute_reply.started":"2022-01-01T19:27:33.206278Z","shell.execute_reply":"2022-01-01T19:27:33.350557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=os.listdir(train_audio_path)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:35.091403Z","iopub.execute_input":"2022-01-01T19:27:35.091757Z","iopub.status.idle":"2022-01-01T19:27:35.097421Z","shell.execute_reply.started":"2022-01-01T19:27:35.091725Z","shell.execute_reply":"2022-01-01T19:27:35.096386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:37.859338Z","iopub.execute_input":"2022-01-01T19:27:37.859766Z","iopub.status.idle":"2022-01-01T19:27:38.20136Z","shell.execute_reply.started":"2022-01-01T19:27:37.85972Z","shell.execute_reply":"2022-01-01T19:27:38.200573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=[\"on\",\"off\",\"up\",\"down\"]","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:42.093451Z","iopub.execute_input":"2022-01-01T19:27:42.093792Z","iopub.status.idle":"2022-01-01T19:27:42.099821Z","shell.execute_reply.started":"2022-01-01T19:27:42.093754Z","shell.execute_reply":"2022-01-01T19:27:42.098611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Duration of recordings**\n\nWhat’s next? A look at the distribution of the duration of recordings:","metadata":{}},{"cell_type":"markdown","source":"**Preprocessing the audio waves**\n\nIn the data exploration part earlier, we have seen that the duration of a few recordings is less than 1 second and the sampling rate is too high. So, let us read the audio waves and use the below-preprocessing steps to deal with this.\n\nHere are the two steps we’ll follow:\n\n* Noise Reduction\n* Silence Removal\n\nLet us define these preprocessing steps in the below code snippet:","metadata":{}},{"cell_type":"code","source":"sr=16000\nvad = malaya_speech.vad.webrtc()\nall_wave = []\nall_label = []\nfor label in labels:\n    print(label)\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        samples, sample_rate = librosa.load(train_audio_path + '/' + label + '/' + wav, sr = 16000)\n        samples = nr.reduce_noise(y=samples, sr=sr,stationary=True)\n        y_= malaya_speech.resample(samples, sr, 16000)\n        y_ = malaya_speech.astype.float_to_int(y_)\n        frames = malaya_speech.generator.frames(samples, 30, sr)\n        frames_ = list(malaya_speech.generator.frames(y_, 30, 16000, append_ending_trail = False))\n        frames_webrtc = [(frames[no], vad(frame)) for no, frame in enumerate(frames_)]\n        y_ = malaya_speech.combine.without_silent(frames_webrtc)\n        zero = np.zeros(((1*sr+4000)-y_.shape[0]))\n        signal = np.concatenate((y_,zero))\n        all_wave.append(signal)\n        all_label.append(label)\n        \n","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:27:46.05538Z","iopub.execute_input":"2022-01-01T19:27:46.055713Z","iopub.status.idle":"2022-01-01T19:32:04.308886Z","shell.execute_reply.started":"2022-01-01T19:27:46.055662Z","shell.execute_reply":"2022-01-01T19:32:04.307901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#samples, sr = librosa.load('../input/onnnnnnn/on1.wav', sr = 16000)\n#fs=sr\n#samples = nr.reduce_noise(y=samples, sr=sr,stationary=True)\n#y_= malaya_speech.resample(samples, sr, 16000)\n#y_ = malaya_speech.astype.float_to_int(y_)\n#frames = malaya_speech.generator.frames(samples, 30, sr)\n#frames_ = list(malaya_speech.generator.frames(y_, 30, 16000, append_ending_trail = False))\n#frames_webrtc = [(frames[no], vad(frame)) for no, frame in enumerate(frames_)]\n#y_ = malaya_speech.combine.without_silent(frames_webrtc)\n#zero = np.zeros(((1*sr+4000)-y_.shape[0]))\n#signal = np.concatenate((y_,zero))\n#mfcc_feat = mfcc(signal , fs, winlen=256/fs, winstep=256/(2*fs), numcep=13, nfilt=26, nfft=256,\n#             lowfreq=0, highfreq=fs/2, preemph=0.97, ceplifter=22, appendEnergy=True, winfunc=np.hamming)\n#mfcc_feat= np.transpose(mfcc_feat)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-01T21:09:11.914326Z","iopub.execute_input":"2022-01-01T21:09:11.914671Z","iopub.status.idle":"2022-01-01T21:09:11.971669Z","shell.execute_reply.started":"2022-01-01T21:09:11.914638Z","shell.execute_reply":"2022-01-01T21:09:11.970704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.array(all_wave).shape)\nprint(np.array(all_label).shape)\ntime = np.linspace(0, len(signal - 1) / fs, len(signal - 1))\nplt.plot(time,np.array(all_wave)[2000,:])\nprint(np.array(all_label)[2000])\nipd.Audio(np.array(all_wave)[2000,:], rate = sr )","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:32:04.310712Z","iopub.execute_input":"2022-01-01T19:32:04.311112Z","iopub.status.idle":"2022-01-01T19:32:08.731121Z","shell.execute_reply.started":"2022-01-01T19:32:04.311072Z","shell.execute_reply":"2022-01-01T19:32:08.729945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_mfcc=[]\nfor wave in all_wave:\n    i=0\n    mfcc_feat = mfcc(wave , fs, winlen=256/fs, winstep=256/(2*fs), numcep=13, nfilt=26, nfft=256,\n                 lowfreq=0, highfreq=fs/2, preemph=0.97, ceplifter=22, appendEnergy=True, winfunc=np.hamming)\n    mfcc_feat= np.transpose(mfcc_feat)\n    all_mfcc.append(mfcc_feat)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:32:08.73313Z","iopub.execute_input":"2022-01-01T19:32:08.733488Z","iopub.status.idle":"2022-01-01T19:32:38.039418Z","shell.execute_reply.started":"2022-01-01T19:32:08.733448Z","shell.execute_reply":"2022-01-01T19:32:38.037359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.array(all_mfcc).shape)\nprint(np.array(all_label).shape)\nd1=np.array(all_mfcc).shape[1]\nd2=np.array(all_mfcc).shape[2]\nd=d1*d2\nprint(d)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T19:33:52.868454Z","iopub.execute_input":"2022-01-01T19:33:52.868819Z","iopub.status.idle":"2022-01-01T19:33:53.068649Z","shell.execute_reply.started":"2022-01-01T19:33:52.868786Z","shell.execute_reply":"2022-01-01T19:33:53.06774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"op_mfcc=np.array(all_mfcc)\nop_mfcc=op_mfcc.reshape(np.array(all_mfcc).shape[0],-1)\nop_mfcc.shape","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:05:38.021004Z","iopub.execute_input":"2022-01-01T20:05:38.021435Z","iopub.status.idle":"2022-01-01T20:05:38.159078Z","shell.execute_reply.started":"2022-01-01T20:05:38.021391Z","shell.execute_reply":"2022-01-01T20:05:38.158286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#all_label = all_label.tolist()\n\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)\nclasses","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:11:01.79625Z","iopub.execute_input":"2022-01-01T20:11:01.796596Z","iopub.status.idle":"2022-01-01T20:11:01.806386Z","shell.execute_reply.started":"2022-01-01T20:11:01.796563Z","shell.execute_reply":"2022-01-01T20:11:01.805478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model based on CNN** ","metadata":{}},{"cell_type":"code","source":"! pip install --upgrade tensorflow\n! pip install --upgrade tensorflow-gpu\n! pip install keras==2.3.1","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:05:43.917016Z","iopub.execute_input":"2022-01-01T20:05:43.91734Z","iopub.status.idle":"2022-01-01T20:06:05.553764Z","shell.execute_reply.started":"2022-01-01T20:05:43.917309Z","shell.execute_reply":"2022-01-01T20:06:05.552578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.optimizers import SGD\nfrom keras.constraints import maxnorm\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Conv2D, Flatten, Dense,Dropout\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:06:06.998192Z","iopub.execute_input":"2022-01-01T20:06:06.998542Z","iopub.status.idle":"2022-01-01T20:06:07.004944Z","shell.execute_reply.started":"2022-01-01T20:06:06.998508Z","shell.execute_reply":"2022-01-01T20:06:07.004071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=tensorflow.keras.utils.to_categorical(y, num_classes=len(labels), dtype='float32')\nprint(y)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:06:10.310502Z","iopub.execute_input":"2022-01-01T20:06:10.311109Z","iopub.status.idle":"2022-01-01T20:06:10.320479Z","shell.execute_reply.started":"2022-01-01T20:06:10.311065Z","shell.execute_reply":"2022-01-01T20:06:10.31947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_tr, x_test, y_tr, y_test= train_test_split(op_mfcc,np.array(y),test_size = 0.3,random_state=777,shuffle=True)\nx_tr, x_val, y_tr, y_val= train_test_split(x_tr,y_tr,test_size = 0.25,random_state=777,shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:07:03.928713Z","iopub.execute_input":"2022-01-01T20:07:03.929049Z","iopub.status.idle":"2022-01-01T20:07:04.021841Z","shell.execute_reply.started":"2022-01-01T20:07:03.92902Z","shell.execute_reply":"2022-01-01T20:07:04.020964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_tr.shape)\nprint(y_tr.shape)\nprint(x_test.shape)\nprint(y_test.shape)\nprint(x_val.shape)\nprint(y_val.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:07:05.053936Z","iopub.execute_input":"2022-01-01T20:07:05.054262Z","iopub.status.idle":"2022-01-01T20:07:05.062941Z","shell.execute_reply.started":"2022-01-01T20:07:05.054231Z","shell.execute_reply":"2022-01-01T20:07:05.061706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Model Architecture**","metadata":{}},{"cell_type":"code","source":"#from keras.models import Sequential\n#from keras.layers import Dense, Dropout, Activation\n\n#Model Architecture\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom tensorflow.keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\n\n\nmodel = Sequential()\nmodel.add(Dense(100, activation='relu', input_shape=(d,), kernel_constraint=maxnorm(3)))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(80, activation='relu', kernel_constraint=maxnorm(3)))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(len(classes), activation='softmax' , kernel_constraint=maxnorm(3)))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:12:56.35446Z","iopub.execute_input":"2022-01-01T20:12:56.354809Z","iopub.status.idle":"2022-01-01T20:12:56.399192Z","shell.execute_reply.started":"2022-01-01T20:12:56.354779Z","shell.execute_reply":"2022-01-01T20:12:56.398416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensorflow.keras.utils.plot_model(model, 'model.png',show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:13:00.336108Z","iopub.execute_input":"2022-01-01T20:13:00.336433Z","iopub.status.idle":"2022-01-01T20:13:00.546621Z","shell.execute_reply.started":"2022-01-01T20:13:00.336402Z","shell.execute_reply":"2022-01-01T20:13:00.545635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adamax',metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:13:07.513625Z","iopub.execute_input":"2022-01-01T20:13:07.513999Z","iopub.status.idle":"2022-01-01T20:13:07.530491Z","shell.execute_reply.started":"2022-01-01T20:13:07.513965Z","shell.execute_reply":"2022-01-01T20:13:07.529588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=10, min_delta=0.0001) \nmc = ModelCheckpoint('best_model.hdf5', monitor='val_acc', verbose=1, save_best_only=True, mode='max')","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:13:09.379476Z","iopub.execute_input":"2022-01-01T20:13:09.379832Z","iopub.status.idle":"2022-01-01T20:13:09.385758Z","shell.execute_reply.started":"2022-01-01T20:13:09.379783Z","shell.execute_reply":"2022-01-01T20:13:09.384797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#x_tr=np.expand_dims(x_tr,axis=0)\n#y_tr=np.expand_dims(y_tr,axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T14:53:13.923405Z","iopub.execute_input":"2021-12-31T14:53:13.923766Z","iopub.status.idle":"2021-12-31T14:53:13.927064Z","shell.execute_reply.started":"2021-12-31T14:53:13.923729Z","shell.execute_reply":"2021-12-31T14:53:13.926133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr,validation_data=(x_val,y_val), epochs=120, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:13:13.068744Z","iopub.execute_input":"2022-01-01T20:13:13.06911Z","iopub.status.idle":"2022-01-01T20:14:19.89255Z","shell.execute_reply.started":"2022-01-01T20:13:13.069077Z","shell.execute_reply":"2022-01-01T20:14:19.89178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_score = model.evaluate(x_tr, y_tr, batch_size=12)\nprint(train_score)\n\nprint('----------------Training Complete-----------------')\n\ntest_score = model.evaluate(x_val, y_val, batch_size = 12)\nprint(test_score)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:14:27.817415Z","iopub.execute_input":"2022-01-01T20:14:27.817761Z","iopub.status.idle":"2022-01-01T20:14:28.904069Z","shell.execute_reply.started":"2022-01-01T20:14:27.817727Z","shell.execute_reply":"2022-01-01T20:14:28.902522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history.history.keys()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:15:58.154001Z","iopub.execute_input":"2022-01-01T20:15:58.154352Z","iopub.status.idle":"2022-01-01T20:15:58.160898Z","shell.execute_reply.started":"2022-01-01T20:15:58.154319Z","shell.execute_reply":"2022-01-01T20:15:58.159902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot\npyplot.plot(history.history['loss'], label='train')\npyplot.plot(history.history['val_loss'], label='test')\npyplot.legend()\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:16:00.20615Z","iopub.execute_input":"2022-01-01T20:16:00.20647Z","iopub.status.idle":"2022-01-01T20:16:00.362888Z","shell.execute_reply.started":"2022-01-01T20:16:00.20644Z","shell.execute_reply":"2022-01-01T20:16:00.362053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:16:05.716638Z","iopub.execute_input":"2022-01-01T20:16:05.716997Z","iopub.status.idle":"2022-01-01T20:16:05.853028Z","shell.execute_reply.started":"2022-01-01T20:16:05.716966Z","shell.execute_reply":"2022-01-01T20:16:05.852188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict=model.predict(x_test)\nconf_mat=tensorflow.math.confusion_matrix(np.argmax(y_test,axis=1) , np.argmax(y_predict,axis=1))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:16:58.072078Z","iopub.execute_input":"2022-01-01T20:16:58.072503Z","iopub.status.idle":"2022-01-01T20:16:58.188975Z","shell.execute_reply.started":"2022-01-01T20:16:58.072453Z","shell.execute_reply":"2022-01-01T20:16:58.188161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cm = pd.DataFrame(np.array(conf_mat), index = [i for i in classes],\n                  columns = [i for i in classes])\nplt.figure(figsize = (13,7))\nax = sn.heatmap(df_cm, annot=True)\nplt.title(\"Confusion Matrix\", fontsize=20)\nplt.ylabel(\"True Class\"     , fontsize=20)\nplt.xlabel(\"Predicted Class\", fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:16:59.832339Z","iopub.execute_input":"2022-01-01T20:16:59.832668Z","iopub.status.idle":"2022-01-01T20:17:00.185365Z","shell.execute_reply.started":"2022-01-01T20:16:59.832634Z","shell.execute_reply":"2022-01-01T20:17:00.184537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test[1].shape\nmodel.predict(x_test[1].reshape((1,d)))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:17:38.220763Z","iopub.execute_input":"2022-01-01T20:17:38.221114Z","iopub.status.idle":"2022-01-01T20:17:38.270693Z","shell.execute_reply.started":"2022-01-01T20:17:38.221083Z","shell.execute_reply":"2022-01-01T20:17:38.2697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define the function that predicts text for the given audio:","metadata":{}},{"cell_type":"code","source":"def predict(audio):\n    print(samples.shape)\n    prob=model.predict(audio)\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:19:35.528508Z","iopub.execute_input":"2022-01-01T20:19:35.528857Z","iopub.status.idle":"2022-01-01T20:19:35.533159Z","shell.execute_reply.started":"2022-01-01T20:19:35.528824Z","shell.execute_reply":"2022-01-01T20:19:35.532062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prediction time! Make predictions on the validation data:","metadata":{}},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_test)-1)\nprint(index)\nsamples=x_test[index]\nprint(\"Audio:\",classes[np.argmax(y_test[index])])\n#ipd.Audio(np.array(all_wave)[index,:], rate=16000)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:22:29.601913Z","iopub.execute_input":"2022-01-01T20:22:29.602328Z","iopub.status.idle":"2022-01-01T20:22:29.609932Z","shell.execute_reply.started":"2022-01-01T20:22:29.602278Z","shell.execute_reply":"2022-01-01T20:22:29.608848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Text:\",predict(samples.reshape(1,d)))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:22:32.282551Z","iopub.execute_input":"2022-01-01T20:22:32.282916Z","iopub.status.idle":"2022-01-01T20:22:32.328696Z","shell.execute_reply.started":"2022-01-01T20:22:32.282884Z","shell.execute_reply":"2022-01-01T20:22:32.327817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\nmodel.save(\"final.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:22:02.723504Z","iopub.execute_input":"2022-01-01T20:22:02.723875Z","iopub.status.idle":"2022-01-01T20:22:02.768141Z","shell.execute_reply.started":"2022-01-01T20:22:02.723842Z","shell.execute_reply":"2022-01-01T20:22:02.76735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:20:19.602966Z","iopub.execute_input":"2022-01-01T20:20:19.603298Z","iopub.status.idle":"2022-01-01T20:20:19.608548Z","shell.execute_reply.started":"2022-01-01T20:20:19.603267Z","shell.execute_reply":"2022-01-01T20:20:19.607783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_path='../input/onnnnnnn/'\nfilename = 'on1.wav'\n#filename2='clip_499bf869f.wav'\nsamples, sample_rate = librosa.load(test_audio_path+filename, sr = 16000)\ntime = np.linspace(0, len(samples - 1) / fs, len(samples - 1))\nreduced_noise1 = nr.reduce_noise(y=samples, sr=fs,stationary=True)\nplt.plot(time, reduced_noise1)  # plot in seconds\n#reduced_noise2 = nr.reduce_noise(y=sig2, sr=fs,stationary=True)\n#plt.plot(time, reduced_noise2)  # plot in seconds\n#plt.title(\"Voice Signal\")\nplt.xlabel(\"Time [seconds]\")\nplt.ylabel(\"Voice amplitude\")\nplt.show()\n#Silence Removal\nvad = malaya_speech.vad.webrtc()\nsr=16000\ny=reduced_noise1\n#Silence Removal\ny_= malaya_speech.resample(y, sr, 16000)\ny_ = malaya_speech.astype.float_to_int(y_)\nframes = malaya_speech.generator.frames(y, 30, sr)\nframes_ = list(malaya_speech.generator.frames(y_, 30, 16000, append_ending_trail = False))\nframes_webrtc = [(frames[no], vad(frame)) for no, frame in enumerate(frames_)]\ny_ = malaya_speech.combine.without_silent(frames_webrtc)\nipd.Audio(y_, rate = sr )\nzero = np.zeros(((1*sr+4000)-y_.shape[0]))\nsignal = np.concatenate((y_,zero))\ntime = np.linspace(0, len(signal - 1) / fs, len(signal - 1))\nprint(signal.shape)\ntime = np.linspace(0, len(signal - 1) / fs, len(signal - 1))\nmfcc_feat1 = mfcc(signal , fs, winlen=256/fs, winstep=256/(2*fs), numcep=13, nfilt=26, nfft=256,\n                 lowfreq=0, highfreq=fs/2, preemph=0.97, ceplifter=22, appendEnergy=True, winfunc=np.hamming)\nmfcc_feat1.shape\nmfcc_feat1=mfcc_feat1.T\nu=np.max(mfcc_feat1)\nmu=np.mean(mfcc_feat1)\nmfcc_feat1 = (mfcc_feat1-mu)/u\nd1=np.array(mfcc_feat1).shape[0]\nd2=np.array(mfcc_feat1).shape[1]\nd=d1*d2\nprint(d)","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:42:37.035465Z","iopub.execute_input":"2022-01-01T20:42:37.035818Z","iopub.status.idle":"2022-01-01T20:42:37.266314Z","shell.execute_reply.started":"2022-01-01T20:42:37.035785Z","shell.execute_reply":"2022-01-01T20:42:37.265318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Text:\",predict(mfcc_feat1.reshape(1,d)))","metadata":{"execution":{"iopub.status.busy":"2022-01-01T20:42:41.888158Z","iopub.execute_input":"2022-01-01T20:42:41.888497Z","iopub.status.idle":"2022-01-01T20:42:41.946141Z","shell.execute_reply.started":"2022-01-01T20:42:41.888463Z","shell.execute_reply":"2022-01-01T20:42:41.945178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}