{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":7634,"databundleVersionId":46676,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:32:08.825167Z","iopub.execute_input":"2025-11-09T17:32:08.826033Z","iopub.status.idle":"2025-11-09T17:32:11.357211Z","shell.execute_reply.started":"2025-11-09T17:32:08.825982Z","shell.execute_reply":"2025-11-09T17:32:11.356166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import the libraries\nimport os\nimport librosa   #for audio processing\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom scipy.io import wavfile #for audio processing\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:50:39.865227Z","iopub.execute_input":"2025-11-09T17:50:39.865579Z","iopub.status.idle":"2025-11-09T17:50:39.871656Z","shell.execute_reply.started":"2025-11-09T17:50:39.865554Z","shell.execute_reply":"2025-11-09T17:50:39.870266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install py7zr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:51:52.055015Z","iopub.execute_input":"2025-11-09T17:51:52.055526Z","iopub.status.idle":"2025-11-09T17:52:00.286648Z","shell.execute_reply.started":"2025-11-09T17:51:52.055491Z","shell.execute_reply":"2025-11-09T17:52:00.285417Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import py7zr\nimport os\n\n# Define paths\narchive_path = '/kaggle/input/tensorflow-speech-recognition-challenge/train.7z'\nextract_path = '/kaggle/working/train'  # Extract into a writable directory\n\n# Extract files\nwith py7zr.SevenZipFile(archive_path, mode='r') as z:\n    z.extractall(path=extract_path)\n\nprint(\"✅ Extraction complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:52:32.894608Z","iopub.execute_input":"2025-11-09T17:52:32.894965Z","iopub.status.idle":"2025-11-09T17:55:37.190221Z","shell.execute_reply.started":"2025-11-09T17:52:32.894936Z","shell.execute_reply":"2025-11-09T17:55:37.189198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !ls /kaggle/input/tensorflow-speech-recognition-challenge\n!ls /kaggle/working/train/train/audio","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:56:39.043137Z","iopub.execute_input":"2025-11-09T17:56:39.045938Z","iopub.status.idle":"2025-11-09T17:56:39.238588Z","shell.execute_reply.started":"2025-11-09T17:56:39.045846Z","shell.execute_reply":"2025-11-09T17:56:39.236929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yes_path = '/kaggle/working/train/train/audio/yes'\nall_files = sorted([f for f in os.listdir(yes_path) if f.endswith('.wav')])\n\nfirst_three = all_files[:3]\nfirst_three_paths = [os.path.join(yes_path, f) for f in first_three]\nprint(first_three_paths)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T17:57:46.352925Z","iopub.execute_input":"2025-11-09T17:57:46.353416Z","iopub.status.idle":"2025-11-09T17:57:46.365774Z","shell.execute_reply.started":"2025-11-09T17:57:46.353386Z","shell.execute_reply":"2025-11-09T17:57:46.364049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualization of Audio signal in time series domain\ntrain_audio_path = '/kaggle/working/train/train/audio/'\nsamples, sample_rate = librosa.load(train_audio_path+'yes/004ae714_nohash_0.wav', sr = 16000)\nfig = plt.figure(figsize=(14, 8))\nax1 = fig.add_subplot(211)\nax1.set_title('Raw wave of ' + '../input/train/train/audio/yes/0a7c2a8d_nohash_0.wav')\nax1.set_xlabel('time')\nax1.set_ylabel('Amplitude')\nax1.plot(np.linspace(0, sample_rate/len(samples), sample_rate), samples);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:00:00.930748Z","iopub.execute_input":"2025-11-09T18:00:00.93119Z","iopub.status.idle":"2025-11-09T18:00:01.414132Z","shell.execute_reply.started":"2025-11-09T18:00:00.931161Z","shell.execute_reply":"2025-11-09T18:00:01.411658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sampling rate\nipd.Audio(samples, rate=sample_rate)\nprint(sample_rate)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:00:51.72384Z","iopub.execute_input":"2025-11-09T18:00:51.724282Z","iopub.status.idle":"2025-11-09T18:00:51.741844Z","shell.execute_reply.started":"2025-11-09T18:00:51.724253Z","shell.execute_reply":"2025-11-09T18:00:51.740513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samples = librosa.resample(y=samples, orig_sr=sample_rate, target_sr=8000)\nipd.Audio(samples, rate=8000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:05:35.910524Z","iopub.execute_input":"2025-11-09T18:05:35.911633Z","iopub.status.idle":"2025-11-09T18:05:35.924801Z","shell.execute_reply.started":"2025-11-09T18:05:35.911598Z","shell.execute_reply":"2025-11-09T18:05:35.92371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# No. of recordings\n\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:08:55.673695Z","iopub.execute_input":"2025-11-09T18:08:55.674166Z","iopub.status.idle":"2025-11-09T18:08:56.156975Z","shell.execute_reply.started":"2025-11-09T18:08:55.674139Z","shell.execute_reply":"2025-11-09T18:08:56.155649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels=[\"yes\", \"no\", \"up\", \"down\", \"left\", \"right\", \"on\", \"off\", \"stop\", \"go\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:13:02.020926Z","iopub.execute_input":"2025-11-09T18:13:02.022154Z","iopub.status.idle":"2025-11-09T18:13:02.028041Z","shell.execute_reply.started":"2025-11-09T18:13:02.022086Z","shell.execute_reply":"2025-11-09T18:13:02.026961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Duration of recordings\nduration_of_recordings = []\n\nfor label in labels:\n    folder = os.path.join(train_audio_path, label)\n    waves = [f for f in os.listdir(folder) if f.endswith('.wav')]\n    \n    for wav in waves:\n        file_path = os.path.join(folder, wav)\n        sample_rate, samples = wavfile.read(file_path)\n        duration = len(samples) / sample_rate  # duration in seconds\n        duration_of_recordings.append(duration)\n\n# Plot histogram of durations\nplt.figure(figsize=(10, 5))\nplt.hist(duration_of_recordings, bins=50, color='skyblue', edgecolor='black')\nplt.xlabel('Duration (seconds)', fontsize=12)\nplt.ylabel('Number of Recordings', fontsize=12)\nplt.title('Distribution of Audio Recording Durations', fontsize=14)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:15:37.71039Z","iopub.execute_input":"2025-11-09T18:15:37.710725Z","iopub.status.idle":"2025-11-09T18:15:39.743905Z","shell.execute_reply.started":"2025-11-09T18:15:37.710703Z","shell.execute_reply":"2025-11-09T18:15:39.741951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing the audio waves\nfrom tqdm import tqdm\ntrain_audio_path = '/kaggle/working/train/train/audio/'\n\nall_wave = []\nall_label = []\nfor label in tqdm(labels):\n    folder = os.path.join(train_audio_path, label)\n    waves = [f for f in os.listdir(folder) if f.endswith('.wav')]\n    \n    for wav in waves:\n        file_path = os.path.join(folder, wav)\n        samples, sample_rate = librosa.load(file_path, sr = 16000)\n        samples = librosa.resample(samples, orig_sr=sample_rate, target_sr=8000)\n        # Normalize clip length to exactly 1 second\n        samples = librosa.util.fix_length(samples, size=8000)\n        all_wave.append(samples)\n        all_label.append(label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:27:15.761821Z","iopub.execute_input":"2025-11-09T18:27:15.762196Z","iopub.status.idle":"2025-11-09T18:27:34.778979Z","shell.execute_reply.started":"2025-11-09T18:27:15.76217Z","shell.execute_reply":"2025-11-09T18:27:34.776501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding the labels\nfrom sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:43:15.25178Z","iopub.execute_input":"2025-11-09T18:43:15.252314Z","iopub.status.idle":"2025-11-09T18:43:15.392607Z","shell.execute_reply.started":"2025-11-09T18:43:15.252284Z","shell.execute_reply":"2025-11-09T18:43:15.39145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\n\ny=to_categorical(y, num_classes=len(labels))\n\n# Convert to NumPy array, reshape and normalize\nall_wave = np.array(all_wave, dtype=np.float32).reshape(-1, 8000, 1)\nall_wave /= np.max(np.abs(all_wave))  # normalize between -1 and 1\n\nfrom sklearn.model_selection import train_test_split\nx_tr, x_val, y_tr, y_val = train_test_split(\n    all_wave,\n    y,\n    stratify=y,\n    test_size=0.2,\n    random_state=777,\n    shuffle=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T18:55:53.520239Z","iopub.execute_input":"2025-11-09T18:55:53.520627Z","iopub.status.idle":"2025-11-09T18:55:55.413606Z","shell.execute_reply.started":"2025-11-09T18:55:53.520603Z","shell.execute_reply":"2025-11-09T18:55:55.412138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**- I will build the speech-to-text model using conv1d. Conv1d is a convolutional neural network which performs the convolution along only one dimension.**","metadata":{}},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\nK.clear_session()\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(8,13, padding='valid', activation='relu', strides=1)(inputs)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Second Conv1D layer\nconv = Conv1D(16, 11, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Third Conv1D layer\nconv = Conv1D(32, 9, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Fourth Conv1D layer\nconv = Conv1D(64, 7, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Flatten layer\nconv = Flatten()(conv)\n\n#Dense Layer 1\nconv = Dense(256, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\n#Dense Layer 2\nconv = Dense(128, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\noutputs = Dense(len(labels), activation='softmax')(conv)\n\nmodel = Model(inputs, outputs)\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T19:10:13.801983Z","iopub.execute_input":"2025-11-09T19:10:13.803778Z","iopub.status.idle":"2025-11-09T19:10:14.4513Z","shell.execute_reply.started":"2025-11-09T19:10:13.803725Z","shell.execute_reply":"2025-11-09T19:10:14.450367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T19:11:17.266574Z","iopub.execute_input":"2025-11-09T19:11:17.267068Z","iopub.status.idle":"2025-11-09T19:11:17.287667Z","shell.execute_reply.started":"2025-11-09T19:11:17.267033Z","shell.execute_reply":"2025-11-09T19:11:17.286448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\nes = EarlyStopping(\n    monitor='val_loss', \n    mode='min', \n    verbose=1, \n    patience=10, \n    min_delta=0.0001\n)\n\nmc = ModelCheckpoint(\n    'best_model.keras',\n    monitor='val_accuracy',\n    verbose=1,\n    save_best_only=True,\n    mode='max'\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T19:14:30.22657Z","iopub.execute_input":"2025-11-09T19:14:30.227007Z","iopub.status.idle":"2025-11-09T19:14:30.23742Z","shell.execute_reply.started":"2025-11-09T19:14:30.226979Z","shell.execute_reply":"2025-11-09T19:14:30.235815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr ,epochs=100, callbacks=[es,mc], batch_size=32, validation_data=(x_val,y_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T19:14:48.832478Z","iopub.execute_input":"2025-11-09T19:14:48.832862Z","iopub.status.idle":"2025-11-09T20:13:01.405993Z","shell.execute_reply.started":"2025-11-09T19:14:48.832821Z","shell.execute_reply":"2025-11-09T20:13:01.403039Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot accuracy and loss\nplt.figure(figsize=(12,5))\n\nplt.subplot(1,2,1)\nplt.plot(history.history['accuracy'], label='train')\nplt.plot(history.history['val_accuracy'], label='val')\nplt.title('Model Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.subplot(1,2,2)\nplt.plot(history.history['loss'], label='train')\nplt.plot(history.history['val_loss'], label='val')\nplt.title('Model Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T20:14:38.680793Z","iopub.execute_input":"2025-11-09T20:14:38.68157Z","iopub.status.idle":"2025-11-09T20:14:39.173968Z","shell.execute_reply.started":"2025-11-09T20:14:38.681518Z","shell.execute_reply":"2025-11-09T20:14:39.172896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.models import load_model\nmodel=load_model('best_model.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T20:15:52.042001Z","iopub.execute_input":"2025-11-09T20:15:52.043035Z","iopub.status.idle":"2025-11-09T20:15:52.361541Z","shell.execute_reply.started":"2025-11-09T20:15:52.043001Z","shell.execute_reply":"2025-11-09T20:15:52.360266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# prediction function\ndef predict(audio):\n    prob=model.predict(audio.reshape(1,8000,1))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T20:21:58.842028Z","iopub.execute_input":"2025-11-09T20:21:58.842447Z","iopub.status.idle":"2025-11-09T20:21:58.848098Z","shell.execute_reply.started":"2025-11-09T20:21:58.842421Z","shell.execute_reply":"2025-11-09T20:21:58.847135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_val)-1)\nsamples=x_val[index].ravel()\nprint(\"Audio:\",classes[np.argmax(y_val[index])])\nipd.Audio(samples, rate=8000)\nprint(\"Text:\",predict(samples))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-09T20:22:08.098302Z","iopub.execute_input":"2025-11-09T20:22:08.098688Z","iopub.status.idle":"2025-11-09T20:22:08.230448Z","shell.execute_reply.started":"2025-11-09T20:22:08.098661Z","shell.execute_reply":"2025-11-09T20:22:08.229381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}