{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-12-29T17:05:38.67908Z","iopub.execute_input":"2022-12-29T17:05:38.679464Z","iopub.status.idle":"2022-12-29T17:05:38.734469Z","shell.execute_reply.started":"2022-12-29T17:05:38.679379Z","shell.execute_reply":"2022-12-29T17:05:38.73351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyunpack\n!pip install patool","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:05:38.736477Z","iopub.execute_input":"2022-12-29T17:05:38.737177Z","iopub.status.idle":"2022-12-29T17:06:02.856779Z","shell.execute_reply.started":"2022-12-29T17:05:38.737141Z","shell.execute_reply":"2022-12-29T17:06:02.855496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom pyunpack import Archive\nimport shutil\nif not os.path.exists('/kaggle/working/train/'):\n    os.makedirs('/kaggle/working/train/')\nArchive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z').extractall('/kaggle/working/train/')\nfor dirname, _, filenames in os.walk('/kaggle/working/train/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:06:02.859825Z","iopub.execute_input":"2022-12-29T17:06:02.860277Z","iopub.status.idle":"2022-12-29T17:07:49.865777Z","shell.execute_reply.started":"2022-12-29T17:06:02.860213Z","shell.execute_reply":"2022-12-29T17:07:49.864293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words=os.listdir('/kaggle/working/train/train/audio')\nwords","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:07:49.869079Z","iopub.execute_input":"2022-12-29T17:07:49.86993Z","iopub.status.idle":"2022-12-29T17:07:49.884503Z","shell.execute_reply.started":"2022-12-29T17:07:49.869883Z","shell.execute_reply":"2022-12-29T17:07:49.883567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\ntrain_audio_path='/kaggle/working/train/train/audio'\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:07:49.886177Z","iopub.execute_input":"2022-12-29T17:07:49.886781Z","iopub.status.idle":"2022-12-29T17:07:50.382288Z","shell.execute_reply.started":"2022-12-29T17:07:49.886746Z","shell.execute_reply":"2022-12-29T17:07:50.381304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nall_wave = []\nall_label = []\n\nfor label in labels[:6]:\n    print(label)\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        samples, sample_rate = librosa.load(train_audio_path + '/' + label + '/' + wav, sr = 8000)\n        #samples = librosa.resample(samples, sample_rate, 8000)\n        if(len(samples)== 8000) : \n            all_wave.append(samples)\n            all_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:07:50.386143Z","iopub.execute_input":"2022-12-29T17:07:50.38645Z","iopub.status.idle":"2022-12-29T17:11:27.197834Z","shell.execute_reply.started":"2022-12-29T17:07:50.386422Z","shell.execute_reply":"2022-12-29T17:11:27.196665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:27.199726Z","iopub.execute_input":"2022-12-29T17:11:27.200278Z","iopub.status.idle":"2022-12-29T17:11:27.210249Z","shell.execute_reply.started":"2022-12-29T17:11:27.200215Z","shell.execute_reply":"2022-12-29T17:11:27.209305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import np_utils\ny=np_utils.to_categorical(y, num_classes=len(labels))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:27.21184Z","iopub.execute_input":"2022-12-29T17:11:27.212216Z","iopub.status.idle":"2022-12-29T17:11:37.018894Z","shell.execute_reply.started":"2022-12-29T17:11:27.212178Z","shell.execute_reply":"2022-12-29T17:11:37.017919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:37.020589Z","iopub.execute_input":"2022-12-29T17:11:37.021328Z","iopub.status.idle":"2022-12-29T17:11:37.294499Z","shell.execute_reply.started":"2022-12-29T17:11:37.021287Z","shell.execute_reply":"2022-12-29T17:11:37.293514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave = np.array(all_wave).reshape(-1,8000)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:37.299231Z","iopub.execute_input":"2022-12-29T17:11:37.300192Z","iopub.status.idle":"2022-12-29T17:11:37.424576Z","shell.execute_reply.started":"2022-12-29T17:11:37.300155Z","shell.execute_reply":"2022-12-29T17:11:37.42344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:37.426073Z","iopub.execute_input":"2022-12-29T17:11:37.427315Z","iopub.status.idle":"2022-12-29T17:11:37.437397Z","shell.execute_reply.started":"2022-12-29T17:11:37.427272Z","shell.execute_reply":"2022-12-29T17:11:37.435668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave),np.array(y),stratify=y,test_size = 0.2,random_state=777,shuffle=True)\nx_te, x_val, y_te, y_val = train_test_split(x_val,y_val,stratify=y_val,test_size = 0.5,random_state=777,shuffle=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:19:49.99801Z","iopub.execute_input":"2022-12-29T17:19:49.998554Z","iopub.status.idle":"2022-12-29T17:19:52.463886Z","shell.execute_reply.started":"2022-12-29T17:19:49.998509Z","shell.execute_reply":"2022-12-29T17:19:52.462731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\nK.clear_session()\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(8,13, padding='valid', activation='relu', strides=1)(inputs)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Second Conv1D layer\nconv = Conv1D(16, 11, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Third Conv1D layer\nconv = Conv1D(32, 9, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Fourth Conv1D layer\nconv = Conv1D(64, 7, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Flatten layer\nconv = Flatten()(conv)\n\n#Dense Layer 1\nconv = Dense(256, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\n#Dense Layer 2\nconv = Dense(128, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\noutputs = Dense(len(labels), activation='softmax')(conv)\n\nmodel = Model(inputs, outputs)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:37.989878Z","iopub.execute_input":"2022-12-29T17:11:37.990279Z","iopub.status.idle":"2022-12-29T17:11:43.590938Z","shell.execute_reply.started":"2022-12-29T17:11:37.990226Z","shell.execute_reply":"2022-12-29T17:11:43.589948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=10, min_delta=0.0001) \n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:11:43.592209Z","iopub.execute_input":"2022-12-29T17:11:43.592573Z","iopub.status.idle":"2022-12-29T17:11:43.605446Z","shell.execute_reply.started":"2022-12-29T17:11:43.592537Z","shell.execute_reply":"2022-12-29T17:11:43.604579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr ,epochs=10,callbacks=es,  batch_size=32, validation_data=(x_val,y_val))\n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:24.748224Z","iopub.execute_input":"2022-12-29T17:14:24.748823Z","iopub.status.idle":"2022-12-29T17:14:49.027192Z","shell.execute_reply.started":"2022-12-29T17:14:24.748778Z","shell.execute_reply":"2022-12-29T17:14:49.026176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot \npyplot.plot(history.history['loss'], label='train') \npyplot.plot(history.history['val_loss'], label='test') \npyplot.legend()\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:13:47.825464Z","iopub.execute_input":"2022-12-29T17:13:47.825857Z","iopub.status.idle":"2022-12-29T17:14:07.205406Z","shell.execute_reply.started":"2022-12-29T17:13:47.825816Z","shell.execute_reply":"2022-12-29T17:14:07.204267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import save\nmodel.save('best_model.hdf5')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:15:19.035216Z","iopub.execute_input":"2022-12-29T17:15:19.035816Z","iopub.status.idle":"2022-12-29T17:15:19.122657Z","shell.execute_reply.started":"2022-12-29T17:15:19.035777Z","shell.execute_reply":"2022-12-29T17:15:19.121423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(audio):\n    prob=model.predict(audio.reshape(1,8000))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.290351Z","iopub.execute_input":"2022-12-29T17:14:07.29084Z","iopub.status.idle":"2022-12-29T17:14:07.29688Z","shell.execute_reply.started":"2022-12-29T17:14:07.290798Z","shell.execute_reply":"2022-12-29T17:14:07.295743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_val)-1)\nsamples=x_val[index]\nprint(\"Audio:\",classes[np.argmax(y_val[index])])\n#ipd.Audio(samples, rate=8000)\nprint(\"Text:\",predict(samples))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.298335Z","iopub.execute_input":"2022-12-29T17:14:07.29937Z","iopub.status.idle":"2022-12-29T17:14:07.52205Z","shell.execute_reply.started":"2022-12-29T17:14:07.299332Z","shell.execute_reply":"2022-12-29T17:14:07.520942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x=x_te, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.524016Z","iopub.execute_input":"2022-12-29T17:14:07.524444Z","iopub.status.idle":"2022-12-29T17:14:07.870776Z","shell.execute_reply.started":"2022-12-29T17:14:07.524407Z","shell.execute_reply":"2022-12-29T17:14:07.869678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_te.shape,predictions.shape)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.872142Z","iopub.execute_input":"2022-12-29T17:14:07.872721Z","iopub.status.idle":"2022-12-29T17:14:07.883749Z","shell.execute_reply.started":"2022-12-29T17:14:07.872683Z","shell.execute_reply":"2022-12-29T17:14:07.882581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\nconfusion_matrix = confusion_matrix(y_te.argmax(axis=1),predictions.argmax(axis=1))\nconfusion_matrix\nprint(classification_report(y_te.argmax(axis=1), predictions.argmax(axis=1)))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.887434Z","iopub.execute_input":"2022-12-29T17:14:07.888171Z","iopub.status.idle":"2022-12-29T17:14:07.919808Z","shell.execute_reply.started":"2022-12-29T17:14:07.888134Z","shell.execute_reply":"2022-12-29T17:14:07.91895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\n\nsn.heatmap(confusion_matrix, annot=True, cmap=plt.cm.Blues)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:14:07.923633Z","iopub.execute_input":"2022-12-29T17:14:07.927809Z","iopub.status.idle":"2022-12-29T17:14:08.743313Z","shell.execute_reply.started":"2022-12-29T17:14:07.927774Z","shell.execute_reply":"2022-12-29T17:14:08.742211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt install libasound2-dev portaudio19-dev libportaudio2 libportaudiocpp0 ffmpeg","metadata":{"execution":{"iopub.status.busy":"2022-12-29T17:49:18.849836Z","iopub.execute_input":"2022-12-29T17:49:18.850245Z","iopub.status.idle":"2022-12-29T18:06:44.48276Z","shell.execute_reply.started":"2022-12-29T17:49:18.85021Z","shell.execute_reply":"2022-12-29T18:06:44.481555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install speechRecognition\nimport speech_recognition as sr\n \n\ndef audio_to_text():\n\n  r = sr.Recognizer()\n\n  with sr.Microphone() as source:\n\n    print('Speak now:')\n\n    audio = r.listen(source)\n\n  try:\n\n    text = r.recognize_google(audio)\n\n    print(f'You said: {text}')\n\n  except sr.UnknownValueError:\n\n    print('Sorry, I could not understand what you said.')\n\n  except sr.RequestError as e:\n\n    print(f'Error while requesting results from Google Speech Recognition service: {e}')\n\n \n\naudio_to_text()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T18:15:40.157204Z","iopub.execute_input":"2022-12-29T18:15:40.158552Z","iopub.status.idle":"2022-12-29T18:15:50.405754Z","shell.execute_reply.started":"2022-12-29T18:15:40.158514Z","shell.execute_reply":"2022-12-29T18:15:50.404117Z"},"trusted":true},"execution_count":null,"outputs":[]}]}