{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **SpeechPilot**\nCode licensed under GNU General Public License v3.0","metadata":{}},{"cell_type":"markdown","source":"## Unpacking\nTo use the dataset provided by Google, we have to extract it. With pyunpack we can unpack the provided .7z file to our `/kaggle/working/` directory.","metadata":{}},{"cell_type":"code","source":"!pip install pyunpack\n!pip install patool","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyunpack import Archive\nArchive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z').extractall(\"/kaggle/working/\")","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:49:50.695617Z","iopub.execute_input":"2023-06-20T07:49:50.696042Z","iopub.status.idle":"2023-06-20T07:51:36.490159Z","shell.execute_reply.started":"2023-06-20T07:49:50.696004Z","shell.execute_reply":"2023-06-20T07:51:36.489130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initialize the libraries and lists\nHere we import the libraries we'll need and the variables like the ```PATH``` and our ```LABELS``` which we need","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom scipy.io import wavfile\nimport os\nfrom scipy.signal import resample\nfrom sklearn.preprocessing import LabelEncoder\nfrom keras.utils import np_utils\nfrom sklearn.model_selection import train_test_split\nfrom keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D\nfrom keras.models import Model, load_model, save_model\nfrom keras import backend as K\nfrom keras.callbacks import EarlyStopping\nfrom time import sleep\nfrom IPython.display import clear_output\n\nTRAIN_PATH = \"/kaggle/working/train/audio/\"\nLABELS = [\"_background_noise_\", \"eight\", \"five\", \"four\", \"go\", \"left\", \"nine\", \"one\", \"right\", \"seven\", \"six\", \"stop\", \"three\", \"two\", \"zero\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:51:54.028368Z","iopub.execute_input":"2023-06-20T07:51:54.028741Z","iopub.status.idle":"2023-06-20T07:51:54.036912Z","shell.execute_reply.started":"2023-06-20T07:51:54.028711Z","shell.execute_reply":"2023-06-20T07:51:54.035876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert audio\nTo train the algorithm, we have to convert every audio file into 8 kHz. To do this, we iterate through the files and check if they fulfill this.\\\nIf they do, their name gets added to the `all_wave` list and their label gets added to the `all_label` list.\\\nIf they don't, they get resampled to 8 kHz using the `resample` function from the SciPy library.","metadata":{}},{"cell_type":"code","source":"all_wave = []\nall_label = []\n\nfor label in LABELS:\n\n    print(label)\n    waves = [f for f in os.listdir(TRAIN_PATH + label) if f.endswith('.wav')]\n\n    for wav in waves:\n\n        sample_rate, sample = wavfile.read(TRAIN_PATH + label + '/' + wav)\n\n        if sample_rate != 8000 or len(sample) != 8000:\n\n            sample = resample(sample, 8000)\n            wavfile.write(TRAIN_PATH + label + '/' + wav, 8000, sample)\n            sample_rate, sample = wavfile.read(TRAIN_PATH + label + '/' + wav)\n        \n        if sample_rate == 8000 and len(sample) == 8000:\n\n            all_wave.append(sample)\n            all_label.append(label)\n\ntrain_size=len(all_wave)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:51:58.571851Z","iopub.execute_input":"2023-06-20T07:51:58.572248Z","iopub.status.idle":"2023-06-20T07:52:32.706381Z","shell.execute_reply.started":"2023-06-20T07:51:58.572217Z","shell.execute_reply":"2023-06-20T07:52:32.705434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initialize and configure Keras\nBeschreibung kommt bald","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)\nprint(classes)\ny=np_utils.to_categorical(y, num_classes=len(LABELS))\nall_wave = np.array(all_wave).reshape(-1,8000,1)\nx_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave),np.array(y),stratify=y,test_size = 0.2,random_state=777,shuffle=True)\n\nK.clear_session()\n\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(filters=8,kernel_size=13, padding='valid', activation='relu', strides=1, input_shape=(8000,1))(inputs)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Second Conv1D layer\nconv = Conv1D(16, 11, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Third Conv1D layer\nconv = Conv1D(32, 9, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Fourth Conv1D layer\nconv = Conv1D(64, 7, padding='valid', activation='relu', strides=1)(conv)\nconv = MaxPooling1D(3)(conv)\nconv = Dropout(0.3)(conv)\n\n#Flatten layer\nconv = Flatten()(conv)\n\n#Dense Layer 1\nconv = Dense(256, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\n#Dense Layer 2\nconv = Dense(128, activation='relu')(conv)\nconv = Dropout(0.3)(conv)\n\noutputs = Dense(len(LABELS), activation='softmax')(conv)\n\nmodel = Model(inputs, outputs)\n#model.summary()\n\nmodel.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=100, min_delta=0.001) ","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:52:35.653641Z","iopub.execute_input":"2023-06-20T07:52:35.654008Z","iopub.status.idle":"2023-06-20T07:52:47.888354Z","shell.execute_reply.started":"2023-06-20T07:52:35.653971Z","shell.execute_reply":"2023-06-20T07:52:47.887366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cleanup\nNow we just do some variable cleanup to free up 2-4gb RAM","metadata":{}},{"cell_type":"code","source":"conv = None\noutputs = None\ninputs = None\nall_wave = None\ny = None\nle = None\nwaves = None\nprint(classes)","metadata":{"execution":{"iopub.status.busy":"2023-03-23T13:52:33.030367Z","iopub.execute_input":"2023-03-23T13:52:33.031414Z","iopub.status.idle":"2023-03-23T13:52:33.115088Z","shell.execute_reply.started":"2023-03-23T13:52:33.031351Z","shell.execute_reply":"2023-03-23T13:52:33.113851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training\nNow that we have enough RAM, let's train.\\\nWe specify `generation`: the amount of times it should go through the entire set.\\\nAswell as `steps`: To determine how many samples should be processed all at once. This may be off by one due to pythons rounding.\nGenerations can be chosen freely, steps should be around 100 on kaggle (If it's too high, performance will decline, if it's too small the algorithm trains slower)","metadata":{}},{"cell_type":"code","source":"generations = 8\nsteps = 200\ntry:\n    model = load_model(\"/kaggle/working/model.h5\")\n    print(\"Model loaded\")\nexcept:\n    print(\"Model not found!\") \n    \nfor i in range(generations):\n    print(f\"Generation: {i+1}\")\n    history = model.fit(x_tr, y_tr ,epochs=50, validation_data=(x_val,y_val), batch_size=train_size//steps)\n    save_model(model, \"model.h5\")\n    clear_output()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nmodel = load_model(\"/kaggle/working/model.h5\")\ntf.saved_model.save(model, \"SavedModel\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T10:12:12.703484Z","iopub.execute_input":"2023-05-30T10:12:12.703856Z","iopub.status.idle":"2023-05-30T10:12:15.329326Z","shell.execute_reply.started":"2023-05-30T10:12:12.703827Z","shell.execute_reply":"2023-05-30T10:12:15.328353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def zip_folder(folder_path, zip_path):\n    # Create a ZipFile object with write permission\n    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zip_obj:\n        # Iterate over all the files in the folder\n        for foldername, subfolders, filenames in os.walk(folder_path):\n            for filename in filenames:\n                # Create the full file path by joining the folder path and file name\n                file_path = os.path.join(foldername, filename)\n                # Add the file to the ZIP archive\n                zip_obj.write(file_path, os.path.relpath(file_path, folder_path))\n\n# Example usage\nfolder_path = '/kaggle/working/SavedModel'\nzip_path = '/kaggle/working/archive.zip'\nzip_folder(folder_path, zip_path)","metadata":{"execution":{"iopub.status.busy":"2023-03-23T14:42:26.430155Z","iopub.execute_input":"2023-03-23T14:42:26.430599Z","iopub.status.idle":"2023-03-23T14:42:27.447072Z","shell.execute_reply.started":"2023-03-23T14:42:26.430562Z","shell.execute_reply":"2023-03-23T14:42:27.445970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Unpacking the testing dataset","metadata":{}},{"cell_type":"code","source":"Archive('/kaggle/input/tensorflow-speech-recognition-challenge/test.7z').extractall(\"/kaggle/working/\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting\nWe can predict what an audiofile means by simply asking the model. We reshape the audio into the same shape as our testing models and the predict using `model.predict(sample)`. with `np.argmax()` we take the highest value and in the `classes` list we look up which category is meant.","metadata":{}},{"cell_type":"code","source":"model = load_model('model.h5')\n\ndef prediction(audio):\n    sample_rate, sample = wavfile.read(audio)\n    sample = sample.reshape(-1,8000,1)\n    prob=model.predict(sample, verbose = 0)\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"execution":{"iopub.status.busy":"2023-03-17T13:45:03.923952Z","iopub.execute_input":"2023-03-17T13:45:03.924740Z","iopub.status.idle":"2023-03-17T13:45:04.598291Z","shell.execute_reply.started":"2023-03-17T13:45:03.924695Z","shell.execute_reply":"2023-03-17T13:45:04.597180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Score calculation\nWe use the training set to evaluate how accurate the model is at predicting. This is due to the testing set not having an index, so we can't validate the output of the model","metadata":{}},{"cell_type":"code","source":"x_tr, x_val, y_tr, y_val = None,None,None,None\nscore = 0\nerrors = 0\ncheckpoint = 0\nfor label in LABELS:\n    if label == \"_background_noise_\":\n        continue\n    else:\n        for file in os.listdir(f\"{TRAIN_PATH}{label}/\"):\n            if checkpoint % 100 == 0:\n                clear_output()\n                print(checkpoint)\n            value = prediction(f\"{TRAIN_PATH}{label}/{file}\")\n            if value == label:\n                score += 1\n            else:\n                errors += 1\n            checkpoint += 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(str(score) +\"/\"+ str(errors))\nprint(f\"{round((score/(score+errors))*100,2)}% Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2023-03-17T14:19:30.218308Z","iopub.execute_input":"2023-03-17T14:19:30.219088Z","iopub.status.idle":"2023-03-17T14:19:30.224979Z","shell.execute_reply.started":"2023-03-17T14:19:30.219027Z","shell.execute_reply":"2023-03-17T14:19:30.223684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Export of the trained model\nIn this step we export the model into a .tflite as this is compatible with Tensorflow Lite (A Tensorflow version for low power devices, like Raspberry Pi's)","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nconverter = tf.lite.TFLiteConverter.from_saved_model(\"/kaggle/working/SavedModel/\")\n\nconverter.optimizations = [tf.lite.Optimize.DEFAULT]\n\ntflite_model = converter.convert()\n\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T10:13:22.515913Z","iopub.execute_input":"2023-05-30T10:13:22.516271Z","iopub.status.idle":"2023-05-30T10:13:23.991774Z","shell.execute_reply.started":"2023-05-30T10:13:22.516243Z","shell.execute_reply":"2023-05-30T10:13:23.990808Z"},"trusted":true},"execution_count":null,"outputs":[]}]}