{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":7634,"databundleVersionId":46676,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-17T23:57:31.785754Z","iopub.execute_input":"2023-05-17T23:57:31.786530Z","iopub.status.idle":"2023-05-17T23:57:31.828412Z","shell.execute_reply.started":"2023-05-17T23:57:31.786483Z","shell.execute_reply":"2023-05-17T23:57:31.827572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyunpack\n!pip install patool","metadata":{"execution":{"iopub.status.busy":"2023-05-17T23:58:00.979059Z","iopub.execute_input":"2023-05-17T23:58:00.979454Z","iopub.status.idle":"2023-05-17T23:58:26.742536Z","shell.execute_reply.started":"2023-05-17T23:58:00.979417Z","shell.execute_reply":"2023-05-17T23:58:26.741128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom pyunpack import Archive\nimport shutil\nif not os.path.exists('/kaggle/working/train/'):\n    os.makedirs('/kaggle/working/train/')\nArchive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z').extractall('/kaggle/working/train/')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T23:58:38.761815Z","iopub.execute_input":"2023-05-17T23:58:38.762266Z","iopub.status.idle":"2023-05-18T00:00:23.095334Z","shell.execute_reply.started":"2023-05-17T23:58:38.762224Z","shell.execute_reply":"2023-05-18T00:00:23.094333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nimport matplotlib.style as ms","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:00:31.680105Z","iopub.execute_input":"2023-05-18T00:00:31.680495Z","iopub.status.idle":"2023-05-18T00:00:31.685454Z","shell.execute_reply.started":"2023-05-18T00:00:31.680465Z","shell.execute_reply":"2023-05-18T00:00:31.684523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/down/3bfd30e6_nohash_3.wav')\nplt.figure(figsize=(12,4))\nlibrosa.display.waveshow(data,sr=sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:00:35.032544Z","iopub.execute_input":"2023-05-18T00:00:35.033043Z","iopub.status.idle":"2023-05-18T00:00:47.621398Z","shell.execute_reply.started":"2023-05-18T00:00:35.033003Z","shell.execute_reply":"2023-05-18T00:00:47.620196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/down/3bfd30e6_nohash_3.wav')#data,samples taken per second\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:00:54.333155Z","iopub.execute_input":"2023-05-18T00:00:54.333570Z","iopub.status.idle":"2023-05-18T00:00:54.343124Z","shell.execute_reply.started":"2023-05-18T00:00:54.333536Z","shell.execute_reply":"2023-05-18T00:00:54.341784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Desired sampling rate.","metadata":{}},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/zero/327289eb_nohash_0.wav',sr=None)\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:00:57.597986Z","iopub.execute_input":"2023-05-18T00:00:57.598406Z","iopub.status.idle":"2023-05-18T00:00:57.607732Z","shell.execute_reply.started":"2023-05-18T00:00:57.598374Z","shell.execute_reply":"2023-05-18T00:00:57.606427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/zero/b93528e3_nohash_0.wav',sr=8000)\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:02.921092Z","iopub.execute_input":"2023-05-18T00:01:02.921504Z","iopub.status.idle":"2023-05-18T00:01:02.931037Z","shell.execute_reply.started":"2023-05-18T00:01:02.921471Z","shell.execute_reply":"2023-05-18T00:01:02.929677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ms.use('seaborn-muted')","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:05.789252Z","iopub.execute_input":"2023-05-18T00:01:05.789644Z","iopub.status.idle":"2023-05-18T00:01:05.795945Z","shell.execute_reply.started":"2023-05-18T00:01:05.789616Z","shell.execute_reply":"2023-05-18T00:01:05.794844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words=os.listdir('/kaggle/working/train/train/audio')\nwords","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:10.703967Z","iopub.execute_input":"2023-05-18T00:01:10.704385Z","iopub.status.idle":"2023-05-18T00:01:10.713047Z","shell.execute_reply.started":"2023-05-18T00:01:10.704337Z","shell.execute_reply":"2023-05-18T00:01:10.711907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import matplotlib.pyplot as plt\nimport numpy as np\ntrain_audio_path='/kaggle/working/train/train/audio'\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T14:30:48.479089Z","iopub.execute_input":"2023-05-17T14:30:48.479621Z","iopub.status.idle":"2023-05-17T14:30:49.078110Z","shell.execute_reply.started":"2023-05-17T14:30:48.479583Z","shell.execute_reply":"2023-05-17T14:30:49.076565Z"}}},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/down/3bfd30e6_nohash_3.wav')\nmelspec=librosa.feature.melspectrogram(y=data,sr=sampling_rate)\nplt.figure()\nlibrosa.display.specshow(melspec,y_axis='mel',x_axis='time')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:16.491922Z","iopub.execute_input":"2023-05-18T00:01:16.492319Z","iopub.status.idle":"2023-05-18T00:01:18.410961Z","shell.execute_reply.started":"2023-05-18T00:01:16.492289Z","shell.execute_reply":"2023-05-18T00:01:18.409789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_audio_path='/kaggle/working/train/train/audio'\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:23.646659Z","iopub.execute_input":"2023-05-18T00:01:23.647044Z","iopub.status.idle":"2023-05-18T00:01:24.205292Z","shell.execute_reply.started":"2023-05-18T00:01:23.647015Z","shell.execute_reply":"2023-05-18T00:01:24.204159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave = []\nall_label = []\n\nfor label in labels[:6]:\n    print(label)\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        samples, sample_rate = librosa.load(train_audio_path + '/' + label + '/' + wav, sr = 8000)\n        #samples = librosa.resample(samples, sample_rate, 8000)\n        if(len(samples)== 8000) : \n            all_wave.append(samples)\n            all_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:29.546014Z","iopub.execute_input":"2023-05-18T00:01:29.546428Z","iopub.status.idle":"2023-05-18T00:01:37.547706Z","shell.execute_reply.started":"2023-05-18T00:01:29.546396Z","shell.execute_reply":"2023-05-18T00:01:37.546716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:42.112854Z","iopub.execute_input":"2023-05-18T00:01:42.114152Z","iopub.status.idle":"2023-05-18T00:01:42.176453Z","shell.execute_reply.started":"2023-05-18T00:01:42.114107Z","shell.execute_reply":"2023-05-18T00:01:42.175416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:01:54.566584Z","iopub.execute_input":"2023-05-18T00:01:54.566968Z","iopub.status.idle":"2023-05-18T00:01:54.574038Z","shell.execute_reply.started":"2023-05-18T00:01:54.566940Z","shell.execute_reply":"2023-05-18T00:01:54.572746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import np_utils\ny=np_utils.to_categorical(y, num_classes=len(labels))","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:09.488726Z","iopub.execute_input":"2023-05-18T00:03:09.489115Z","iopub.status.idle":"2023-05-18T00:03:17.561137Z","shell.execute_reply.started":"2023-05-18T00:03:09.489088Z","shell.execute_reply":"2023-05-18T00:03:17.559943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:20.795239Z","iopub.execute_input":"2023-05-18T00:03:20.796976Z","iopub.status.idle":"2023-05-18T00:03:20.809434Z","shell.execute_reply.started":"2023-05-18T00:03:20.796932Z","shell.execute_reply":"2023-05-18T00:03:20.806731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave = np.array(all_wave).reshape(-1,8000)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:29.355166Z","iopub.execute_input":"2023-05-18T00:03:29.355614Z","iopub.status.idle":"2023-05-18T00:03:29.580118Z","shell.execute_reply.started":"2023-05-18T00:03:29.355579Z","shell.execute_reply":"2023-05-18T00:03:29.578859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:33.101833Z","iopub.execute_input":"2023-05-18T00:03:33.102256Z","iopub.status.idle":"2023-05-18T00:03:33.109999Z","shell.execute_reply.started":"2023-05-18T00:03:33.102222Z","shell.execute_reply":"2023-05-18T00:03:33.108859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave),np.array(y),stratify=y,test_size = 0.2,random_state=777,shuffle=True)\nx_te, x_val, y_te, y_val = train_test_split(x_val,y_val,stratify=y_val,test_size = 0.5,random_state=777,shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:35.554504Z","iopub.execute_input":"2023-05-18T00:03:35.555029Z","iopub.status.idle":"2023-05-18T00:03:36.607208Z","shell.execute_reply.started":"2023-05-18T00:03:35.554982Z","shell.execute_reply":"2023-05-18T00:03:36.605904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\nK.clear_session()\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(8,13, padding='valid', activation='relu', strides=1)(inputs)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Second Conv1D layer\nconv = Conv1D(16, 11, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Third Conv1D layer\nconv = Conv1D(32, 9, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Fourth Conv1D layer\nconv = Conv1D(64, 7, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Flatten layer\nconv = Flatten()(conv)\n\n#Dense Layer 1\nconv = Dense(256, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\n#Dense Layer 2\nconv = Dense(128, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\noutputs = Dense(len(labels), activation='softmax')(conv)\n\nmodel = Model(inputs, outputs)\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:38.377651Z","iopub.execute_input":"2023-05-18T00:03:38.378158Z","iopub.status.idle":"2023-05-18T00:03:38.813636Z","shell.execute_reply.started":"2023-05-18T00:03:38.378106Z","shell.execute_reply":"2023-05-18T00:03:38.812281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=10, min_delta=0.0001) ","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:46.558132Z","iopub.execute_input":"2023-05-18T00:03:46.558588Z","iopub.status.idle":"2023-05-18T00:03:46.590581Z","shell.execute_reply.started":"2023-05-18T00:03:46.558553Z","shell.execute_reply":"2023-05-18T00:03:46.589437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr ,epochs=20,callbacks=es,  batch_size=32, validation_data=(x_val,y_val))","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:03:48.743303Z","iopub.execute_input":"2023-05-18T00:03:48.743707Z","iopub.status.idle":"2023-05-18T00:17:14.983000Z","shell.execute_reply.started":"2023-05-18T00:03:48.743677Z","shell.execute_reply":"2023-05-18T00:17:14.981901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot \npyplot.plot(history.history['loss'], label='train') \npyplot.plot(history.history['val_loss'], label='test') \npyplot.legend()\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:22:27.850452Z","iopub.execute_input":"2023-05-18T00:22:27.850900Z","iopub.status.idle":"2023-05-18T00:22:28.094496Z","shell.execute_reply.started":"2023-05-18T00:22:27.850864Z","shell.execute_reply":"2023-05-18T00:22:28.093430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\nmodel.save('best_model.hdf5')","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:22:35.633433Z","iopub.execute_input":"2023-05-18T00:22:35.633825Z","iopub.status.idle":"2023-05-18T00:22:35.743889Z","shell.execute_reply.started":"2023-05-18T00:22:35.633795Z","shell.execute_reply":"2023-05-18T00:22:35.742663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(audio):\n    prob=model.predict(audio.reshape(1,8000))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:22:47.717237Z","iopub.execute_input":"2023-05-18T00:22:47.717636Z","iopub.status.idle":"2023-05-18T00:22:47.723802Z","shell.execute_reply.started":"2023-05-18T00:22:47.717605Z","shell.execute_reply":"2023-05-18T00:22:47.722398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_val)-1)\nsamples=x_val[index]\nprint(\"Audio:\",classes[np.argmax(y_val[index])])\n#ipd.Audio(samples, rate=8000)\nprint(\"Text:\",predict(samples))","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:22:55.480438Z","iopub.execute_input":"2023-05-18T00:22:55.480820Z","iopub.status.idle":"2023-05-18T00:22:55.739986Z","shell.execute_reply.started":"2023-05-18T00:22:55.480793Z","shell.execute_reply":"2023-05-18T00:22:55.739185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x=x_te, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:23:00.412953Z","iopub.execute_input":"2023-05-18T00:23:00.413694Z","iopub.status.idle":"2023-05-18T00:23:01.594206Z","shell.execute_reply.started":"2023-05-18T00:23:00.413656Z","shell.execute_reply":"2023-05-18T00:23:01.593084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_te.shape,predictions.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:23:02.735876Z","iopub.execute_input":"2023-05-18T00:23:02.736278Z","iopub.status.idle":"2023-05-18T00:23:02.742056Z","shell.execute_reply.started":"2023-05-18T00:23:02.736247Z","shell.execute_reply":"2023-05-18T00:23:02.740693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\nconfusion_matrix = confusion_matrix(y_te.argmax(axis=1),predictions.argmax(axis=1))\nconfusion_matrix\nprint(classification_report(y_te.argmax(axis=1), predictions.argmax(axis=1)))","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:23:04.789312Z","iopub.execute_input":"2023-05-18T00:23:04.789730Z","iopub.status.idle":"2023-05-18T00:23:04.809790Z","shell.execute_reply.started":"2023-05-18T00:23:04.789699Z","shell.execute_reply":"2023-05-18T00:23:04.808945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\n\nsn.heatmap(confusion_matrix, annot=True, cmap=plt.cm.Blues)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-18T00:23:10.253720Z","iopub.execute_input":"2023-05-18T00:23:10.254295Z","iopub.status.idle":"2023-05-18T00:23:10.991177Z","shell.execute_reply.started":"2023-05-18T00:23:10.254251Z","shell.execute_reply":"2023-05-18T00:23:10.990080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}