{"metadata":{"colab":{"provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install kaggle","metadata":{"id":"IyqCbpztB4Vk","outputId":"f64e72f9-0aef-4c4e-b48e-ab1c0c4ad4eb","execution":{"iopub.status.busy":"2023-08-01T12:50:25.12341Z","iopub.execute_input":"2023-08-01T12:50:25.12448Z","iopub.status.idle":"2023-08-01T12:50:40.521954Z","shell.execute_reply.started":"2023-08-01T12:50:25.124443Z","shell.execute_reply":"2023-08-01T12:50:40.520389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install kaggle --user","metadata":{"id":"VAxtZM4LB8l5","outputId":"fefdca5e-28d9-40e1-8e5a-aca78fd6d5f3","execution":{"iopub.status.busy":"2023-08-01T12:53:17.523171Z","iopub.execute_input":"2023-08-01T12:53:17.523632Z","iopub.status.idle":"2023-08-01T12:53:30.711452Z","shell.execute_reply.started":"2023-08-01T12:53:17.523601Z","shell.execute_reply":"2023-08-01T12:53:30.709831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install -q kaggle","metadata":{"id":"balf2jKlC2AM","execution":{"iopub.status.busy":"2023-08-01T12:53:44.149509Z","iopub.execute_input":"2023-08-01T12:53:44.149909Z","iopub.status.idle":"2023-08-01T12:53:57.406226Z","shell.execute_reply.started":"2023-08-01T12:53:44.149874Z","shell.execute_reply":"2023-08-01T12:53:57.404679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install google-colab","metadata":{"execution":{"iopub.status.busy":"2023-08-01T12:54:36.304125Z","iopub.execute_input":"2023-08-01T12:54:36.304612Z","iopub.status.idle":"2023-08-01T12:56:33.768838Z","shell.execute_reply.started":"2023-08-01T12:54:36.304575Z","shell.execute_reply":"2023-08-01T12:56:33.767381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp kaggle.json ~/.kaggle/","metadata":{"id":"KW_LJ0K5B_qt"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!chmod 600 ~/.kaggle/kaggle.json","metadata":{"id":"peOfAjAvDvBv"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! kaggle datasets list","metadata":{"id":"52DPumGUD30T","outputId":"97c15048-6671-4992-a039-d54fa5965234"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle competitions download -c tensorflow-speech-recognition-challenge","metadata":{"id":"2z8zyWFHD6du","outputId":"8ddeb845-85b7-491d-b93e-9b0e9bce586c","execution":{"iopub.status.busy":"2023-08-01T12:50:48.076416Z","iopub.execute_input":"2023-08-01T12:50:48.077522Z","iopub.status.idle":"2023-08-01T12:50:49.580121Z","shell.execute_reply.started":"2023-08-01T12:50:48.077469Z","shell.execute_reply":"2023-08-01T12:50:49.578882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"quiXQLoWIJej"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir train","metadata":{"id":"r74NlBMIEbOm","outputId":"1249b2f2-9094-4f06-d29d-deb0e18dfc4c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! unzip tensorflow-speech-recognition-challenge","metadata":{"id":"riRRtNLlE3Be","outputId":"e14b66d8-e9aa-4ead-cf61-c1eab1e5d15b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get install p7zip-full","metadata":{"id":"Jm001FZ8HLEd","outputId":"2e33699d-a80f-499a-b827-a9453848e58d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!7z x train.7z -otrain","metadata":{"id":"P-suw43RHVd8","outputId":"a6ee2cf8-e729-4101-d5ea-b7abf575a9ed"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!7z x test.7z -otest","metadata":{"id":"vo6fINxkILNs","outputId":"33b771d0-b1a9-4f4c-a07d-399f14317e08"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from google.colab import drive\n# drive.mount('/content/drive')","metadata":{"id":"EUxsbr2p0hZf","outputId":"f8611f47-b261-4c17-db0f-3e915fa4b1f7"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from google.colab import files\n# uploaded = files.upload()","metadata":{"id":"i7n4aVUV0lO7","outputId":"68e26ae4-d982-4250-9d0b-36ac9fe4ef14"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n# shutil.move(list(uploaded.keys())[0], '/content/drive/My Drive/train')","metadata":{"id":"CtHckSjU0ogO"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"main code","metadata":{"id":"yaqytO31JOG_"}},{"cell_type":"code","source":"import os\nimport librosa\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.io import wavfile\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"id":"IPwCdKvj6KgL","execution":{"iopub.status.busy":"2023-08-01T12:52:41.833817Z","iopub.execute_input":"2023-08-01T12:52:41.834256Z","iopub.status.idle":"2023-08-01T12:52:41.892466Z","shell.execute_reply.started":"2023-08-01T12:52:41.834224Z","shell.execute_reply":"2023-08-01T12:52:41.891274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/input/tensorflow-speech-recognition-challenge/test.7z')","metadata":{"id":"UjkZLA--6KgN","outputId":"274f6e7e-95d4-45f7-be4b-91ac6785e89f","execution":{"iopub.status.busy":"2023-08-01T12:52:49.620603Z","iopub.execute_input":"2023-08-01T12:52:49.620997Z","iopub.status.idle":"2023-08-01T12:52:49.665708Z","shell.execute_reply.started":"2023-08-01T12:52:49.620968Z","shell.execute_reply":"2023-08-01T12:52:49.663915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# New Section","metadata":{"id":"GrS56vjYztmv"}},{"cell_type":"code","source":"pip install samplerate resampy","metadata":{"id":"YTVDhnrJ6KgP","outputId":"02026e9d-b46b-44b0-f1a7-aa2304577acf"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Exploration and Visualization**\n\nData Exploration and Visualization helps us to understand the data as well as pre-processing steps in a better way.\n\n**Visualization of Audio signal in time series domain**\n\nNow, we’ll visualize the audio signal in the time series domain:","metadata":{"id":"IW4Hn70m6KgQ"}},{"cell_type":"code","source":"# samples =s , smaple_rate = sr","metadata":{"id":"o900rt6w6KgS"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_audio_path = '/kaggle/input/tensorflow-speech-recognition-challenge/test.7z/audio/'\ns, sr = librosa.load(train_audio_path+'yes/00f0204f_nohash_0.wav', sr = 16000)\nfig = plt.figure(figsize=(14, 8))\nax1 = fig.add_subplot(211)\nax1.set_title('Raw wave of ' + '../input/train/audio/yes/0a7c2a8d_nohash_0.wav')\nax1.set_xlabel('time')\nax1.set_ylabel('Amplitude')\nax1.plot(np.linspace(0, sr/len(s), sr), s)","metadata":{"id":"wZ8PhoCE6KgS","outputId":"c18d0952-a214-4fbf-fcd9-a13bab4cd904","execution":{"iopub.status.busy":"2023-08-01T12:52:35.607482Z","iopub.execute_input":"2023-08-01T12:52:35.608238Z","iopub.status.idle":"2023-08-01T12:52:35.672554Z","shell.execute_reply.started":"2023-08-01T12:52:35.608203Z","shell.execute_reply":"2023-08-01T12:52:35.668075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sampling rate **\n\nLet us now look at the sampling rate of the audio signals","metadata":{"id":"ak0B2u316KgT"}},{"cell_type":"code","source":"ipd.Audio(s, rate=sr)","metadata":{"id":"efO3Vj8f6KgT","outputId":"3767bffc-79c3-4129-8385-03ce7deb72a7"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sr)","metadata":{"id":"QxCGm6N-6KgU","outputId":"d9656ad9-4141-4a90-d504-77eb548e2655"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Resampling**\n\nFrom the above, we can understand that the sampling rate of the signal is 16000 hz. Let us resample it to 8000 hz since most of the speech related frequencies are present in 8000z","metadata":{"id":"vLmehfYf6KgV"}},{"cell_type":"code","source":"s = librosa.resample(s, orig_sr=sr,target_sr=8000)\nipd.Audio(s, rate=8000)","metadata":{"id":"NJCi6LyM6KgX","outputId":"f3889f78-c499-4ffd-ca08-5500ab218894"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s, sr =librosa.load(train_audio_path+'yes/00f0204f_nohash_0.wav', sr = 16000)\ns_8k = librosa.resample(s, orig_sr=sr, target_sr=8000)\nipd.Audio(s, rate=8000)\ns.shape, s_8k.shape","metadata":{"id":"yWhTOD2m6KgY","outputId":"6d0e5cbe-db89-4fae-f49e-9948fdda47ed"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, let’s understand the number of recordings for each voice command:","metadata":{"id":"rHat3XAo6KgY"}},{"cell_type":"code","source":"labels=os.listdir(train_audio_path)","metadata":{"id":"qosGW-wS6KgY"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n\n\n\nprint(no_of_recordings)\n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()\n","metadata":{"id":"Cm-muS7M6KgY","outputId":"fbaf827c-a13d-4807-de0f-894e66846441"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=[\"yes\", \"no\", \"up\", \"down\", \"left\", \"right\", \"on\", \"off\", \"stop\", \"go\"]","metadata":{"id":"ftoAPA306KgZ"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Duration of recordings**\n\nWhat’s next? A look at the distribution of the duration of recordings:","metadata":{"id":"tBnfRkJv6KgZ"}},{"cell_type":"code","source":"duration_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        sr, y = wavfile.read(train_audio_path + '/' + label + '/' + wav)\n        duration_of_recordings.append(float(len(y)/sr))\n\nplt.hist(np.array(duration_of_recordings))\n","metadata":{"id":"iSvFU1-s6KgZ","outputId":"a4f1c423-fb62-4e10-8340-a07795dda15d"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"from sklearn.model_selection import train_test_split**Preprocessing the audio waves**\n\nIn the data exploration part earlier, we have seen that the duration of a few recordings is less than 1 second and the sampling rate is too high. So, let us read the audio waves and use the below-preprocessing steps to deal with this.\n\nHere are the two steps we’ll follow:\n\n* Resampling\n* Removing shorter commands of less than 1 second\n\nLet us define these preprocessing steps in the below code snippet:","metadata":{"id":"PGsRpo0z6KgZ"}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_audio_path = '/kaggle/input/tensorflow-speech-recognition-challenge/test.7z/'\nall_wave = []\nall_label = []\nfor label in labels:\n    print(label)\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        s,sr= librosa.load(train_audio_path + '/' + label + '/' + wav, sr = 16000)\n        s = librosa.resample(s, orig_sr=sr, target_sr=8000)\n        if(len(s)== 8000) :\n            all_wave.append(s)\n            all_label.append(label)\n\n\n# Convert the lists to numpy arrays\nall_wave = np.array(all_wave)\ny = np.array(all_label)\n\n# Check the length of the arrays\nprint(\"Number of samples in all_wave:\", len(all_wave))\nprint(\"Number of labels in S:\", len(y))\n\nall_wave = np.array(all_wave)\ny = np.array(y)\n\n# Check the length of the arrays\nprint(\"Number of samples in all_wave:\", len(all_wave))\nprint(\"Number of labels in y:\", len(y))\n\n# Split the dataset into training and validation sets\nx_tr, x_val, y_tr, y_val = train_test_split(all_wave, y, stratify=y, test_size=0.2, random_state=42, shuffle=True)\n\n# Print the shape of each split\nprint(\"Shape of x_tr:\", x_tr.shape)\nprint(\"Shape of x_val:\", x_val.shape)\nprint(\"Shape of y_tr:\", y_tr.shape)\nprint(\"Shape of y_val:\", y_val.shape)\n","metadata":{"id":"yhQoCGq66KgZ","outputId":"a5096aa0-4828-41fd-88be-58838a851c70","execution":{"iopub.status.busy":"2023-08-01T12:52:11.812758Z","iopub.execute_input":"2023-08-01T12:52:11.813238Z","iopub.status.idle":"2023-08-01T12:52:12.946225Z","shell.execute_reply.started":"2023-08-01T12:52:11.813191Z","shell.execute_reply":"2023-08-01T12:52:12.944269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert the output labels to integer encoded:","metadata":{"id":"eybl-8ch6KgZ"}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)","metadata":{"id":"-GUBzo806KgZ"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, convert the integer encoded labels to a one-hot vector since it is a multi-classification problem:","metadata":{"id":"Aw-pDYkz6Kga"}},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\n\n# Assuming y is your target variable and labels contains the unique class labels\ny = to_categorical(y, num_classes=len(labels))\n","metadata":{"id":"Vb9hxnZN6Kga"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reshape the 2D array to 3D since the input to the conv1d must be a 3D array:","metadata":{"id":"p6GmOHfw6Kga"}},{"cell_type":"code","source":"all_wave = np.array(all_wave).reshape(-1,8000,1)","metadata":{"id":"Y9xEN1bC6Kga"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split into train and validation set**\n\nNext, we will train the model on 80% of the data and validate on the remaining 20%:\n","metadata":{"id":"uWHX4YYz6Kga"}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# all_wave = np.array(waves)\n# y = np.array(labels)\n\n# # Check the length of the arrays\n# print(\"Number of samples in all_wave:\", len(all_wave))\n# print(\"Number of labels in y:\", len(y))\n\n# x_tr, x_val, y_tr, y_val = train_test_split(all_wave,y,stratify=y,test_size = 0.2,train_size=0.8,random_state=42,shuffle=True)","metadata":{"id":"Pxh8y4A_6Kga"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# # Assuming you already have the 'all_wave' and 'y' arrays\n# #\n# # Split the dataset into training and validation sets\n# x_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave), np.array(y), stratify=y, test_size=0.2, random_state=777, shuffle=True)\n\n# # Print the shape of each split\n# print(\"Shape of x_tr:\", x_tr.shape)\n# print(\"Shape of x_val:\", x_val.shape)\n# print(\"Shape of y_tr:\", y_tr.shape)\n# print(\"Shape of y_val:\", y_val.shape)\n","metadata":{"id":"NvMrntmm6Kga"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Architecture for this problem**\n\nWe will build the speech-to-text model using conv1d. Conv1d is a convolutional neural network which performs the convolution along only one dimension.","metadata":{"id":"WXdTMD-06Kgd"}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Assuming you already have the 'all_wave' and 'y' arrays\n\n# Split the dataset into training and validation sets\nx_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave), np.array(y), stratify=y, test_size=0.2, random_state=777, shuffle=True)\n\n# Print the shape of each split\nprint(\"Shape of x_tr:\", x_tr.shape)\nprint(\"Shape of x_val:\", x_val.shape)\nprint(\"Shape of y_tr:\", y_tr.shape)\nprint(\"Shape of y_val:\", y_val.shape)","metadata":{"id":"xdpbJpu36Kgd","outputId":"e7a1cbca-d3da-4042-887b-2f6c5095b0a2"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Assuming you already have the 'x_tr', 'x_val', 'y_tr', and 'y_val' arrays\n\n# Get the unique labels and their counts in the training set\nunique_labels_tr, counts_tr = np.unique(y_tr, return_counts=True)\n\n# Get the unique labels and their counts in the validation set\nunique_labels_val, counts_val = np.unique(y_val, return_counts=True)\n\n# Set the figure size\nplt.figure(figsize=(12, 6))\n\n# Plot the bar graph for training set\nplt.subplot(1, 2, 1)\nplt.bar(unique_labels_tr, counts_tr)\nplt.title('Training Set')\nplt.xlabel('Labels')\nplt.ylabel('Number of Samples')\n\n# Plot the bar graph for validation set\nplt.subplot(1, 2, 2)\nplt.bar(unique_labels_val, counts_val)\nplt.title('Validation Set')\nplt.xlabel('Labels')\nplt.ylabel('Number of Samples')\n\n# Adjust the layout\nplt.tight_layout()\n\n# Show the plots\nplt.show()\n","metadata":{"id":"3OzidL3t6Kge","outputId":"33151867-d28e-4242-ef8f-f8b8d0d9749a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Assuming you already have the 'x_tr', 'x_val', 'y_tr', and 'y_val' arrays\n\n# Get the unique labels and their counts in the training set\nunique_labels_tr, counts_tr = np.unique(y_tr, return_counts=True)\n\n# Get the unique labels and their counts in the validation set\nunique_labels_val, counts_val = np.unique(y_val, return_counts=True)\n\n# Set the figure size\nplt.figure(figsize=(12, 6))\n\n# Plot the line graph for training set\nplt.subplot(1, 2, 1)\nplt.plot(unique_labels_tr, counts_tr, marker='o')\nplt.title('Training Set')\nplt.xlabel('Labels')\nplt.ylabel('Number of Samples')\nplt.xticks(rotation=45)\n\n# Plot the line graph for validation set\nplt.subplot(1, 2, 2)\nplt.plot(unique_labels_val, counts_val, marker='o')\nplt.title('Validation Set')\nplt.xlabel('Labels')\nplt.ylabel('Number of Samples')\nplt.xticks(rotation=45)\n\n# Adjust the layout\nplt.tight_layout()\n\n# Show the plots\nplt.show()\n","metadata":{"id":"CC0HYDz16Kge","outputId":"b890918a-b7a3-435d-c2bd-d61443ea0367"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have already loaded and preprocessed the data\n# Assuming you have 'x_tr', 'x_val', 'y' as numpy arrays\n\n# Reshape the input data to include the channel dimension\nx_tr = x_tr.reshape(x_tr.shape[0], x_tr.shape[1], 1)\nx_val = x_val.reshape(x_val.shape[0], x_val.shape[1], 1)\n\n# Rest of the model remains the same as before\n# Define the input shape\n\n\n# Define the CNN model\n# (rest of the model definition remains unchanged)\n","metadata":{"id":"35OTQFv56Kge"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cnn model","metadata":{"id":"G2IIFFnHJteF"}},{"cell_type":"markdown","source":"**Model building**\n\nLet us implement the model using Keras functional API.","metadata":{"id":"1HV669tb6Kge"}},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\nK.clear_session()\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(8,13, padding='valid', activation='relu', strides=1)(inputs)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Second Conv1D layer\nconv = Conv1D(16, 11, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Third Conv1D layer\nconv = Conv1D(32, 9, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Fourth Conv1D layer\nconv = Conv1D(64, 7, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Flatten layer\nconv = Flatten()(conv)\n\n#Dense Layer 1\nconv = Dense(256, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\n#Dense Layer 2\nconv = Dense(128, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\noutputs = Dense(len(labels), activation='softmax')(conv)\n\nmodel = Model(inputs, outputs)\nmodel.summary()\n\n\n# Compile the model\nmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nsave_folder = '/content/tensorflow-speech-recognition-challenge/model'\n\n# Set up callbacks for early stopping and model checkpointing\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=5)\nmc = ModelCheckpoint(os.path.join(save_folder, 'best_model.h5'), monitor='val_accuracy', mode='max', verbose=1, save_best_only=True)\n\n# Train the model on the training data with a batch size of 32\nhistory = model.fit(x_tr, y_tr, epochs=50, callbacks=[es, mc], batch_size=32, validation_data=(x_val, y_val))\n\n# Evaluate the model on the holdout set\nloss, accuracy = model.evaluate(x_val, y_val)\n\nprint(\"Loss on the holdout set:\", loss)\nprint(\"Accuracy on the holdout set:\", accuracy)","metadata":{"id":"jznMMwOm6Kgf","outputId":"47e53c27-ee3c-4431-81e3-c6303fdb02b3"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rnn model","metadata":{"id":"T-9KTeZ-J5JH"}},{"cell_type":"markdown","source":"Define the loss function to be categorical cross-entropy since it is a multi-classification problem:","metadata":{"id":"PpxWXx6u6Kgf"}},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])","metadata":{"id":"Pz0FIsTa6Kgf"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Early stopping and model checkpoints are the callbacks to stop training the neural network at the right time and to save the best model after every epoch:","metadata":{"id":"dcImReYc6Kgk"}},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=10, min_delta=0.0001)\nmc = ModelCheckpoint('best_model.hdf5', monitor='val_acc', verbose=1, save_best_only=True, mode='max')","metadata":{"id":"H-ct6sJz6Kgl"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us train the model on a batch size of 32 and evaluate the performance on the holdout set:","metadata":{"id":"rYEgbpLn6Kgl"}},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr ,epochs=50, callbacks=[es,mc], batch_size=32, validation_data=(x_val,y_val))","metadata":{"id":"zrTohf7f6Kgl","outputId":"39dea065-c5ca-4e65-cb4d-e6dcd24626a4"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Diagnostic plot**\n\nI’m going to lean on visualization again to understand the performance of the model over a period of time:","metadata":{"id":"Pk8MWcEc6Kgl"}},{"cell_type":"code","source":"from matplotlib import pyplot\npyplot.plot(history.history['loss'], label='train')\npyplot.plot(history.history['val_loss'], label='test')\npyplot.legend()\npyplot.show()","metadata":{"id":"5lnzGAJS6Kgl","outputId":"7a529f32-2b5d-4de7-9f8b-4ce0f2c2bfe0"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\n\nfile_path ='/content/tensorflow-speech-recognition-challenge/model/best_model.h5'\nif os.path.exists(file_path):\n    model = load_model(file_path)\n    print(\"succesfully loaded\")\nelse:\n    print(\"Model file not found.\")\n","metadata":{"id":"4128_r986Kgl","outputId":"8d918980-1a65-4f4f-e071-e20a655ba24f"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading the best model**","metadata":{"id":"mfnLv_IW6Kgl"}},{"cell_type":"code","source":"import os\nfrom keras.models import load_model\nmodel=load_model(\"/content/tensorflow-speech-recognition-challenge/model/best_model.h5\")","metadata":{"id":"UlPET3iu6Kgm"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define the function that predicts text for the given audio:","metadata":{"id":"gOudTBrc6Kgm"}},{"cell_type":"code","source":"def predict(audio):\n    prob=model.predict(audio.reshape(1,8000,1))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"id":"e8nuvI4w6Kgm"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prediction time! Make predictions on the validation data:","metadata":{"id":"d2pMOZD76Kgm"}},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_val)-1)\nsamples=x_val[index].ravel()\nprint(\"Audio:\",classes[np.argmax(y_val[index])])\nipd.Audio(samples, rate=8000)","metadata":{"id":"pbSij7y76Kgm","outputId":"9bd99b0e-3d38-4f35-fa6e-cc1efa122f5a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Text:\",predict(samples))","metadata":{"id":"hwibovu56Kgm","outputId":"b2f60686-282c-4aa2-82da-d19a95dc83b1"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us now read the saved voice command and convert it to text:","metadata":{"id":"V5MjIArP6Kgn"}},{"cell_type":"code","source":"import librosa\nimport IPython.display as ipd\nimport scipy.signal\n\n# Replace 'your_pre_recorded_filepath.wav' with the actual path to your audio file\nyour_pre_recorded_filepath = '/content/LJ001-0001.wav'\n\n# Load the audio file\ns, sr = librosa.load(your_pre_recorded_filepath, sr=16000)\n\n# Normalize the audio to bring the amplitudes within a suitable range (-1 to 1)\nnormalized_samples = librosa.util.normalize(s)\n\n# Resample the audio to 8000 Hz using scipy.signal.resample\ntarget_sample_rate = 8000\nresampled_samples = scipy.signal.resample(normalized_samples, int(len(normalized_samples) * target_sample_rate / sr))\n\n# Play the resampled audio\nipd.Audio(resampled_samples, rate=8000)\n","metadata":{"id":"YODp7Obv6Kgn","outputId":"03de8e60-aa14-4b02-8e03-5509ea2a84aa"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport numpy as np\n\n# Function to predict the label for audio chunks\n#sample_rate = 44100  # Replace this value with the actual sample rate of your audio files\n\ndef predict(audio_chunks):\n    predictions = []\n    for chunk in audio_chunks:\n        # Resample the chunk to the model's required sample rate (8000 Hz)\n        chunk = librosa.resample(chunk, orig_sr=sr, target_sr=8000)\n\n        # Ensure the chunk has at least 8000 samples\n        if len(chunk) < 8000:\n            chunk = np.pad(chunk, (0, 8000 - len(chunk)), 'constant')\n        else:\n            chunk = chunk[:8000]\n\n        # Ensure the chunk has the correct shape (1, 8000, 1)\n        chunk = chunk.reshape(1, 8000, 1)\n\n        # Make the prediction using the model\n        prob = model.predict(chunk)\n        index = np.argmax(prob[0])\n\n        # Append the predicted label to the list\n        predictions.append(classes[index])\n\n    return predictions\n\n# Usage example\nyour_pre_recorded_filepath ='/content/LJ001-0001.wav'\ns, sr = librosa.load(your_pre_recorded_filepath, sr=16000)\n\n# Divide the audio into smaller chunks of size 8000 samples\nchunk_size = 8000\nnum_chunks = len(s) // chunk_size\naudio_chunks = [s[i * chunk_size : (i + 1) * chunk_size] for i in range(num_chunks)]\n\n# Predict the labels for the audio chunks\npredictions = predict(audio_chunks)\nprint(predictions)","metadata":{"id":"W2D4Ihkn6Kgn","outputId":"72dea0f8-e155-4d79-f1a7-61ad23dc4e1a"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**text to other  language **","metadata":{"id":"Hylat4bjgryO"}},{"cell_type":"markdown","source":"","metadata":{"id":"x66VbIxzggZL"}},{"cell_type":"code","source":"from keras.models import Model\nfrom keras.layers import Input, LSTM, Dense\nimport numpy as np\n\n# Assuming you have parallel data in the form of (source_sentences, target_sentences)\n\n# Tokenize and preprocess the data (convert words to integers, pad sequences, etc.)\n\n# Define the input and output language vocabularies (source_vocab_size, target_vocab_size)\n\n# Define the encoder\nencoder_input = Input(shape=(max_source_sequence_length,))\nencoder_lstm = LSTM(hidden_size)\nencoder_output, state_h, state_c = encoder_lstm(encoder_input)\nencoder_states = [state_h, state_c]\n\n# Define the decoder\ndecoder_input = Input(shape=(max_target_sequence_length,))\ndecoder_lstm = LSTM(hidden_size, return_sequences=True)\ndecoder_output = decoder_lstm(decoder_input, initial_state=encoder_states)\ndecoder_dense = Dense(target_vocab_size, activation='softmax')\ndecoder_output = decoder_dense(decoder_output)\n\n# Define the seq2seq model\nmodel = Model([encoder_input, decoder_input], decoder_output)\nmodel.summary()\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='categorical_crossentropy')\n\n# Train the model\nmodel.fit([source_data, target_data[:, :-1]], target_data[:, 1:], epochs=num_epochs, batch_size=batch_size)\n\n# Inference mode (encoder and decoder separate from the training model)\nencoder_model = Model(encoder_input, encoder_states)\n\ndecoder_state_input_h = Input(shape=(hidden_size,))\ndecoder_state_input_c = Input(shape=(hidden_size,))\ndecoder_states_input = [decoder_state_input_h, decoder_state_input_c]\n\ndecoder_output, state_h, state_c = decoder_lstm(decoder_input, initial_state=decoder_states_input)\ndecoder_states = [state_h, state_c]\ndecoder_output = decoder_dense(decoder_output)\n\ndecoder_model = Model([decoder_input] + decoder_states_input, [decoder_output] + decoder_states)\n\n# Implement decoding algorithm (e.g., greedy search, beam search) for translation\n\n# Example usage for translation:\nsource_sequence = preprocess_source_sequence('Your source sentence in the source language')\ncontext_vector = encoder_model.predict(source_sequence)\ntarget_sequence = translate_sequence(context_vector, target_language_vocab, max_target_sequence_length)\n\n# Convert the target_sequence back to the target language sentence and display the translation\n","metadata":{"id":"FF_Wo5zoTbeu"},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}