{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import IPython.display as ipd\nimport librosa as lb\nimport librosa.display as ld\nimport sklearn as sk\nimport seaborn as sb\nimport plotly as ply\nimport scipy as sp\nfrom scipy import signal\nfrom scipy.fftpack import fft","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-28T02:53:44.241736Z","iopub.execute_input":"2021-08-28T02:53:44.24218Z","iopub.status.idle":"2021-08-28T02:53:46.618773Z","shell.execute_reply.started":"2021-08-28T02:53:44.242087Z","shell.execute_reply":"2021-08-28T02:53:46.618005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%pylab inline\nimport os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2021-08-28T02:53:49.645252Z","iopub.execute_input":"2021-08-28T02:53:49.645851Z","iopub.status.idle":"2021-08-28T02:53:49.658472Z","shell.execute_reply.started":"2021-08-28T02:53:49.645818Z","shell.execute_reply":"2021-08-28T02:53:49.657585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = r'..'\nout_path = r'.'\nmodel_path = r'.'\ntrain_path = os.path.join(root_path, 'input', 'tfsr1', 'train', 'audio')\n#test_path = os.path.join(root_path, 'kaggle', 'working', 'test', 'test', 'audio')","metadata":{"_uuid":"a8caae6852bb5b99071df37fdb676f7616a8711b","execution":{"iopub.status.busy":"2021-08-28T02:54:38.710229Z","iopub.execute_input":"2021-08-28T02:54:38.710736Z","iopub.status.idle":"2021-08-28T02:54:38.714879Z","shell.execute_reply.started":"2021-08-28T02:54:38.710706Z","shell.execute_reply":"2021-08-28T02:54:38.714018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(train_path))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T02:54:51.733201Z","iopub.execute_input":"2021-08-28T02:54:51.733707Z","iopub.status.idle":"2021-08-28T02:54:51.758118Z","shell.execute_reply.started":"2021-08-28T02:54:51.733678Z","shell.execute_reply":"2021-08-28T02:54:51.757133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = lb.load(train_path + '/_background_noise_/dude_miaowing.wav')","metadata":{"_uuid":"0218ecad9647987dba9c6f854087a85564e66301","execution":{"iopub.status.busy":"2021-08-28T02:54:53.83505Z","iopub.execute_input":"2021-08-28T02:54:53.835611Z","iopub.status.idle":"2021-08-28T02:54:56.020545Z","shell.execute_reply.started":"2021-08-28T02:54:53.83558Z","shell.execute_reply":"2021-08-28T02:54:56.019714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data, sampling_rate)","metadata":{"_uuid":"afe9c86a23bc9f482b8d31267d8c30558a75879e","execution":{"iopub.status.busy":"2021-08-28T02:54:58.16833Z","iopub.execute_input":"2021-08-28T02:54:58.168696Z","iopub.status.idle":"2021-08-28T02:54:58.174362Z","shell.execute_reply.started":"2021-08-28T02:54:58.168662Z","shell.execute_reply":"2021-08-28T02:54:58.173223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\ndef log_spectrogram(audio, sampling_rate, window_size=20, step_size=10, eps=1e-10):\n    nps = int(round(window_size * sampling_rate / 1e3))\n    nol = int(round(step_size * sampling_rate / 1e3))\n    frequencies, times, specs = signal.spectrogram(audio, fs=sampling_rate, window='hann', nperseg=nps, noverlap=nol, detrend=False)\n    return frequencies, times, np.log(specs.T.astype(np.float32) + eps)","metadata":{"_uuid":"cd6aea35218b45167f0edc074e278be83ed9eff2","execution":{"iopub.status.busy":"2021-08-28T02:55:02.903745Z","iopub.execute_input":"2021-08-28T02:55:02.904134Z","iopub.status.idle":"2021-08-28T02:55:02.910265Z","shell.execute_reply.started":"2021-08-28T02:55:02.904099Z","shell.execute_reply":"2021-08-28T02:55:02.909139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.io import wavfile\nimport scipy.io\n\nfile = train_path + '/tree/0e4d22f1_nohash_0.wav'\nsampling_rate, samples = wavfile.read(file)\n\nfrequencies, times, spectrogram = log_spectrogram(samples, sampling_rate)\n\nfig = plt.figure(figsize=(14,8))\nax1 = fig.add_subplot(211)\nax1.set_title('Raw Wave of ' + file)\nax1.set_ylabel('Amplitude')\nax1.plot(np.linspace(0, sampling_rate/len(samples), sampling_rate), samples)\nax2 = fig.add_subplot(212)\nax2.imshow(spectrogram.T, aspect='auto', origin='lower', extent=[times.min(), times.max(), frequencies.min(), frequencies.max()])\nax2.set_yticks(frequencies[::16])\nax2.set_xticks(times[::16])\nax2.set_title('Spectrogram of ' + file)\nax2.set_ylabel('Frequencies in Hz')\nax2.set_xlabel('Seconds')","metadata":{"_uuid":"21799e7c77608e1e4828857ec2aee56954648ca4","execution":{"iopub.status.busy":"2021-08-28T02:55:09.567473Z","iopub.execute_input":"2021-08-28T02:55:09.567969Z","iopub.status.idle":"2021-08-28T02:55:10.145077Z","shell.execute_reply.started":"2021-08-28T02:55:09.56794Z","shell.execute_reply":"2021-08-28T02:55:10.144455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = np.mean(spectrogram, axis=0)\nstd = np.std(spectrogram, axis=0)\nspectrogram = (spectrogram - mean) / std\nprint(spectrogram)","metadata":{"_uuid":"d61fd568d8caa6cc56bc5289f7771248aab1d7ee","execution":{"iopub.status.busy":"2021-08-28T02:55:16.588757Z","iopub.execute_input":"2021-08-28T02:55:16.58915Z","iopub.status.idle":"2021-08-28T02:55:16.595948Z","shell.execute_reply.started":"2021-08-28T02:55:16.589121Z","shell.execute_reply":"2021-08-28T02:55:16.59493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pic_path = os.path.join(root_path, 'input', 'train')\ntest_pic_path = os.path.join(root_path, 'input', 'test')\n","metadata":{"_uuid":"840f0dd3d7fa3bf28e6fab14b1e62d6f737dd9d1","execution":{"iopub.status.busy":"2021-08-28T02:55:19.609423Z","iopub.execute_input":"2021-08-28T02:55:19.609776Z","iopub.status.idle":"2021-08-28T02:55:19.614006Z","shell.execute_reply.started":"2021-08-28T02:55:19.609749Z","shell.execute_reply":"2021-08-28T02:55:19.613245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib as mpl\nfrom keras import backend as K\nfrom keras.datasets import mnist\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation\nfrom keras.layers import Flatten, Conv2D, MaxPooling2D, GRU\nfrom keras.optimizers import SGD, Adam, RMSprop, Adadelta\nfrom keras.utils import np_utils, plot_model\nfrom keras.layers.normalization import BatchNormalization\nfrom keras.layers.advanced_activations import LeakyReLU, PReLU\nfrom keras.callbacks import LearningRateScheduler\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers.recurrent import SimpleRNN, LSTM \nfrom keras.layers.embeddings import Embedding\n","metadata":{"_uuid":"532ab78759e83533754034d0938f548eb1f1e1f4","execution":{"iopub.status.busy":"2021-08-28T02:55:21.810888Z","iopub.execute_input":"2021-08-28T02:55:21.811667Z","iopub.status.idle":"2021-08-28T02:55:26.863579Z","shell.execute_reply.started":"2021-08-28T02:55:21.811621Z","shell.execute_reply":"2021-08-28T02:55:26.862728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rc('font', family = 'serif', size = 17)\nmpl.rcParams['xtick.major.size'] = 5\nmpl.rcParams['xtick.minor.size'] = 2\nmpl.rcParams['ytick.major.size'] = 5\nmpl.rcParams['ytick.minor.size'] = 2","metadata":{"_uuid":"3b0ef8d642781f7027f59c912f22fe94cd088d6b","execution":{"iopub.status.busy":"2021-08-28T02:55:28.782134Z","iopub.execute_input":"2021-08-28T02:55:28.78247Z","iopub.status.idle":"2021-08-28T02:55:28.787619Z","shell.execute_reply.started":"2021-08-28T02:55:28.782443Z","shell.execute_reply":"2021-08-28T02:55:28.786597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hyper_pwr = 0.5\nhyper_train_ratio = 0.9\nhyper_n = 25\nhyper_m = 15\nhyper_NR = 208\nhyper_NC = 112\nhyper_delta = 0.3\nhyper_dropout0 = 0.2\nhyper_dropout1 = 0.4\nhyper_dropout2 = 0.6\nhyper_dropout3 = 0.6\nhyper_dropout4 = 0.4\nhyper_dropout5 = 0.7\n\ntarget_labels = ['yes', 'no', 'up', 'down', 'left', 'right', 'on', 'off', 'stop', 'go', 'silence', 'unknown']","metadata":{"_uuid":"95519f8a05a070d2acf3fa2e85ecf6501e568fbf","execution":{"iopub.status.busy":"2021-08-28T02:55:31.289691Z","iopub.execute_input":"2021-08-28T02:55:31.290073Z","iopub.status.idle":"2021-08-28T02:55:31.295966Z","shell.execute_reply.started":"2021-08-28T02:55:31.290043Z","shell.execute_reply":"2021-08-28T02:55:31.294917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"L = 16000\ndef load_audio_data(path, ltoi):\n    x = []\n    y = []\n    for i, folder in enumerate(os.listdir(path)):\n        for filename in os.listdir(path + '/' + folder):\n            if filename == 'README.md':\n                continue\n            rate, sample = wavfile.read(train_path + '/' + folder + '/' + filename)\n            assert(rate == L)\n            if folder == '_background_noise_':\n                length = len(sample)\n                for j in range(int(length/rate)):\n                    x.append(np.array(sample[j*rate: (j+1)*rate]))\n                    y.append(ltoi['silence'])\n            else:\n                x.append(np.array(sample))\n                label = folder\n                if folder not in target_labels:\n                    label = 'unknown'\n                y.append(ltoi[label])\n    x = np.array(pad_sequences(x, maxlen=L))\n    y = np.array(y)\n    df = pd.DataFrame()\n    df['x'] = list(x)\n    df['y'] = list(y)\n    return df","metadata":{"_uuid":"92f4b245eb639cdfa19d32450f149eff574a8922","execution":{"iopub.status.busy":"2021-08-28T02:55:34.891561Z","iopub.execute_input":"2021-08-28T02:55:34.891916Z","iopub.status.idle":"2021-08-28T02:55:34.901793Z","shell.execute_reply.started":"2021-08-28T02:55:34.891888Z","shell.execute_reply":"2021-08-28T02:55:34.900815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.listdir('{0}/_background_noise_'.format(train_path))","metadata":{"_uuid":"0bba05312b9ac994f1b6156988c1bfbc62b8405f","execution":{"iopub.status.busy":"2021-08-28T02:55:38.411909Z","iopub.execute_input":"2021-08-28T02:55:38.412285Z","iopub.status.idle":"2021-08-28T02:55:38.433686Z","shell.execute_reply.started":"2021-08-28T02:55:38.41226Z","shell.execute_reply":"2021-08-28T02:55:38.432734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.io import wavfile\nprint(\"LOADING RAW DATA!\")\nlabel2idx = {}\nidmap = {}\nfor i,lab in enumerate(target_labels):\n    label2idx[lab] = i\n    idmap[i] = lab\nraw_df = load_audio_data(train_path, label2idx)\nprint(label2idx)\nprint(idmap)\n#print(raw_df.x.to_numpy().shape)\n#print(raw_df.y.to_numpy().shape)","metadata":{"_uuid":"2d2786472c10f48343022e5b76e24d16daab76d2","execution":{"iopub.status.busy":"2021-08-28T02:55:41.094883Z","iopub.execute_input":"2021-08-28T02:55:41.095231Z","iopub.status.idle":"2021-08-28T03:01:50.763951Z","shell.execute_reply.started":"2021-08-28T02:55:41.095203Z","shell.execute_reply":"2021-08-28T03:01:50.762757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.utils import shuffle\n","metadata":{"_uuid":"aa4311bacc098c52c45b758d49974927842fda9a","execution":{"iopub.status.busy":"2021-08-28T03:01:57.779279Z","iopub.execute_input":"2021-08-28T03:01:57.779687Z","iopub.status.idle":"2021-08-28T03:01:57.784637Z","shell.execute_reply.started":"2021-08-28T03:01:57.779643Z","shell.execute_reply":"2021-08-28T03:01:57.783655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split train, test sets, and also return label_map\ndef train_test_split(df, train_ratio = 0.2, test_ratio = 0.1):\n    \n    test_x = []\n    test_y = []\n    train_x = []\n    train_y = []\n    for i in set(df.y.tolist()):\n        tmp_df = df[df.y == i]\n        tmp_df = shuffle(tmp_df)\n        tmp_n = int(len(tmp_df)*train_ratio)\n        tmp_m = int(len(tmp_df)*test_ratio)\n        train_x += tmp_df.x.tolist()[: tmp_n]\n        test_x += tmp_df.x.tolist()[tmp_n: tmp_n + tmp_m]\n        train_y += tmp_df.y.tolist()[: tmp_n]\n        test_y += tmp_df.y.tolist()[tmp_n: tmp_n + tmp_m]\n    return np.array(train_x), np.array(train_y), np.array(test_x), np.array(test_y)\n## Parsing the data Frame into train and test sets\nprint(\"SPLITTING DATA INTO TRAIN AND TEST SETS!\")\ntr_x, tr_y, ts_x, ts_y = train_test_split(raw_df, 0.3, 0.1)\nprint(tr_x.shape)\nprint(tr_y.shape)\nprint(ts_x.shape)\nprint(ts_y.shape)\ndel raw_df","metadata":{"_uuid":"0cddf21e23a1bf2139ab4b8e568a3493f03e7ab3","execution":{"iopub.status.busy":"2021-08-28T03:01:59.917513Z","iopub.execute_input":"2021-08-28T03:01:59.918064Z","iopub.status.idle":"2021-08-28T03:02:00.527554Z","shell.execute_reply.started":"2021-08-28T03:01:59.91803Z","shell.execute_reply":"2021-08-28T03:02:00.526337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base = min(tr_x.min(), ts_x.min())\nprint(base)\ntr_x = tr_x - base\nts_x = ts_x - base\nUPPER_X = max(tr_x.max(), ts_x.max()) + 1\nprint(UPPER_X)\nprint(tr_x[0])","metadata":{"_uuid":"228aedda1c037c3701c446622a567296db88265f","execution":{"iopub.status.busy":"2021-08-28T03:02:04.313421Z","iopub.execute_input":"2021-08-28T03:02:04.313763Z","iopub.status.idle":"2021-08-28T03:02:05.51914Z","shell.execute_reply.started":"2021-08-28T03:02:04.313736Z","shell.execute_reply":"2021-08-28T03:02:05.51798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tr_x[0])\nprint(tr_x.max())\nprint(ts_x.max())\nprint(tr_x.min())\nprint(ts_x.min())\n","metadata":{"_uuid":"cfaff2c4ec72c51b3514aa0dfddef6d3521beecc","execution":{"iopub.status.busy":"2021-08-28T03:02:07.352957Z","iopub.execute_input":"2021-08-28T03:02:07.353304Z","iopub.status.idle":"2021-08-28T03:02:08.052073Z","shell.execute_reply.started":"2021-08-28T03:02:07.353271Z","shell.execute_reply":"2021-08-28T03:02:08.051074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def comp_cls_wts(y, pwr = 0.5):\n    dic = {}\n    for x in set(y):\n        dic[x] = len(y)**pwr/list(y).count(x)**pwr\n    return dic","metadata":{"_uuid":"db618e04e4357cc02dd213a2285ab80cd8643fb1","execution":{"iopub.status.busy":"2021-08-28T03:02:09.402054Z","iopub.execute_input":"2021-08-28T03:02:09.402426Z","iopub.status.idle":"2021-08-28T03:02:09.407527Z","shell.execute_reply.started":"2021-08-28T03:02:09.402393Z","shell.execute_reply":"2021-08-28T03:02:09.406671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_wts = comp_cls_wts(tr_y)\nprint(cls_wts)","metadata":{"_uuid":"1169066165342c1e43ca2baead3e5476c8459c56","execution":{"iopub.status.busy":"2021-08-28T03:02:11.5506Z","iopub.execute_input":"2021-08-28T03:02:11.550957Z","iopub.status.idle":"2021-08-28T03:02:11.619675Z","shell.execute_reply.started":"2021-08-28T03:02:11.550928Z","shell.execute_reply":"2021-08-28T03:02:11.618751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_CLS = len(target_labels)\ntr_y = np_utils.to_categorical(tr_y, num_classes=NUM_CLS)\nts_y = np_utils.to_categorical(ts_y, num_classes=NUM_CLS)","metadata":{"_uuid":"e90e00b66a8a3c2edc6f5bef79e52f1d8d6ddb7f","execution":{"iopub.status.busy":"2021-08-28T03:02:13.697573Z","iopub.execute_input":"2021-08-28T03:02:13.697948Z","iopub.status.idle":"2021-08-28T03:02:13.703411Z","shell.execute_reply.started":"2021-08-28T03:02:13.697917Z","shell.execute_reply":"2021-08-28T03:02:13.702453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Embedding(UPPER_X, 128, input_length=L))\nmodel.add(SimpleRNN(512))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(NUM_CLS, activation='softmax'))\nmodel.summary()","metadata":{"_uuid":"01f032cd8a4121c0a233f350cfc9a8769d6c0399","execution":{"iopub.status.busy":"2021-08-28T03:02:16.570815Z","iopub.execute_input":"2021-08-28T03:02:16.571192Z","iopub.status.idle":"2021-08-28T03:02:16.839191Z","shell.execute_reply.started":"2021-08-28T03:02:16.571159Z","shell.execute_reply":"2021-08-28T03:02:16.838369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = SGD()\nmetrics = ['accuracy']\nloss = 'categorical_crossentropy'\nmodel.compile(optimizer = optimizer, loss = loss, metrics = metrics)","metadata":{"_uuid":"fbe046b863da0ad79ab5acfd2ab4b096e7c16f1f","execution":{"iopub.status.busy":"2021-08-28T03:02:19.79191Z","iopub.execute_input":"2021-08-28T03:02:19.7924Z","iopub.status.idle":"2021-08-28T03:02:19.809389Z","shell.execute_reply.started":"2021-08-28T03:02:19.792367Z","shell.execute_reply":"2021-08-28T03:02:19.808385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(tr_x[:1000], tr_y[:1000], batch_size = 64,epochs = 5, validation_data = (ts_x[:330], ts_y[:330]), class_weight = cls_wts)","metadata":{"_uuid":"4ce49dbc96d7461fe7ed1fb94bf30c2f75654788","execution":{"iopub.status.busy":"2021-08-28T03:02:22.759261Z","iopub.execute_input":"2021-08-28T03:02:22.75961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot \npyplot.plot(history.history['loss'], label='train') \npyplot.plot(history.history['val_loss'], label='test') \npyplot.legend() pyplot.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('saved_model/my_model')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(audio):\n    prob=model.predict(audio.reshape(1,8000,1))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sounddevice as sd\nimport soundfile as sf\n\nsamplerate = 16000  \nduration = 1 # seconds\nfilename = 'yes.wav'\nprint(\"start\")\nmydata = sd.rec(int(samplerate * duration), samplerate=samplerate,\n    channels=1, blocking=True)\nprint(\"end\")\nsd.wait()\nsf.write(filename, mydata, samplerate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('../input/voice-commands/prateek_voice_v2')\nfilepath='../input/voice-commands/prateek_voice_v2'\n\n#reading the voice commands\nsamples, sample_rate = librosa.load(filepath + '/' + 'stop.wav', sr = 16000)\nsamples = librosa.resample(samples, sample_rate, 8000)\nipd.Audio(samples,rate=8000)  \n\npredict(samples)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"_uuid":"1392ffe0e10f953e0b01a574f68383a101b72017","trusted":true},"execution_count":null,"outputs":[]}]}