{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":7634,"databundleVersionId":46676,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        #In ra đường dẫn đầy đủ của mỗi tập tin được tìm thấy trong thư mục đầu vào.\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-09T08:23:03.534279Z","iopub.execute_input":"2024-06-09T08:23:03.535118Z","iopub.status.idle":"2024-06-09T08:23:03.548127Z","shell.execute_reply.started":"2024-06-09T08:23:03.535080Z","shell.execute_reply":"2024-06-09T08:23:03.547118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyunpack\n!pip install patool","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:23:03.550007Z","iopub.execute_input":"2024-06-09T08:23:03.550299Z","iopub.status.idle":"2024-06-09T08:23:27.148543Z","shell.execute_reply.started":"2024-06-09T08:23:03.550275Z","shell.execute_reply":"2024-06-09T08:23:27.147173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom pyunpack import Archive\nimport shutil\nif not os.path.exists('/kaggle/working/train/'):\n    os.makedirs('/kaggle/working/train/')\nArchive('/kaggle/input/tensorflow-speech-recognition-challenge/train.7z').extractall('/kaggle/working/train/')","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:23:27.150218Z","iopub.execute_input":"2024-06-09T08:23:27.150640Z","iopub.status.idle":"2024-06-09T08:24:54.448625Z","shell.execute_reply.started":"2024-06-09T08:23:27.150603Z","shell.execute_reply":"2024-06-09T08:24:54.447446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nimport matplotlib.style as ms\nimport IPython.display as ipd","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:24:54.450090Z","iopub.execute_input":"2024-06-09T08:24:54.451519Z","iopub.status.idle":"2024-06-09T08:24:54.493074Z","shell.execute_reply.started":"2024-06-09T08:24:54.451484Z","shell.execute_reply":"2024-06-09T08:24:54.492165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/working/train/train/audio/cat/caf9fceb_nohash_1.wav'\ndata, sampling_rate = librosa.load(file_path)\nplt.figure(figsize=(12,4))\nlibrosa.display.waveshow(data,sr=sampling_rate)\nipd.Audio(file_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:24:54.495744Z","iopub.execute_input":"2024-06-09T08:24:54.496031Z","iopub.status.idle":"2024-06-09T08:25:03.363674Z","shell.execute_reply.started":"2024-06-09T08:24:54.496006Z","shell.execute_reply":"2024-06-09T08:25:03.362637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/down/3bfd30e6_nohash_3.wav')#data,samples taken per second\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.365128Z","iopub.execute_input":"2024-06-09T08:25:03.365734Z","iopub.status.idle":"2024-06-09T08:25:03.377169Z","shell.execute_reply.started":"2024-06-09T08:25:03.365705Z","shell.execute_reply":"2024-06-09T08:25:03.375301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Desired sampling rate.","metadata":{}},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/zero/327289eb_nohash_0.wav',sr=None)\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.378393Z","iopub.execute_input":"2024-06-09T08:25:03.378719Z","iopub.status.idle":"2024-06-09T08:25:03.389690Z","shell.execute_reply.started":"2024-06-09T08:25:03.378684Z","shell.execute_reply":"2024-06-09T08:25:03.388757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sampling_rate = librosa.load('/kaggle/working/train/train/audio/zero/b93528e3_nohash_0.wav',sr=8000)\nprint('Data : ', data)\nprint('number of Data : ', len(data))\nprint('sampling_rate : ', sampling_rate)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.391217Z","iopub.execute_input":"2024-06-09T08:25:03.391568Z","iopub.status.idle":"2024-06-09T08:25:03.404389Z","shell.execute_reply.started":"2024-06-09T08:25:03.391535Z","shell.execute_reply":"2024-06-09T08:25:03.403293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ms.use('seaborn-muted')","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.407637Z","iopub.execute_input":"2024-06-09T08:25:03.407931Z","iopub.status.idle":"2024-06-09T08:25:03.413754Z","shell.execute_reply.started":"2024-06-09T08:25:03.407897Z","shell.execute_reply":"2024-06-09T08:25:03.412726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words=os.listdir('/kaggle/working/train/train/audio')\nwords","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.415409Z","iopub.execute_input":"2024-06-09T08:25:03.415713Z","iopub.status.idle":"2024-06-09T08:25:03.426609Z","shell.execute_reply.started":"2024-06-09T08:25:03.415687Z","shell.execute_reply":"2024-06-09T08:25:03.425637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import matplotlib.pyplot as plt\nimport numpy as np\ntrain_audio_path='/kaggle/working/train/train/audio'\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T14:30:48.479089Z","iopub.execute_input":"2023-05-17T14:30:48.479621Z","iopub.status.idle":"2023-05-17T14:30:49.07811Z","shell.execute_reply.started":"2023-05-17T14:30:48.479583Z","shell.execute_reply":"2023-05-17T14:30:49.076565Z"}}},{"cell_type":"code","source":"data, sampling_rate = librosa.load(file_path)\nmelspec=librosa.feature.melspectrogram(y=data,sr=sampling_rate)\nplt.figure()\nlibrosa.display.specshow(melspec,y_axis='mel',x_axis='time')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:03.427938Z","iopub.execute_input":"2024-06-09T08:25:03.428183Z","iopub.status.idle":"2024-06-09T08:25:04.850147Z","shell.execute_reply.started":"2024-06-09T08:25:03.428162Z","shell.execute_reply":"2024-06-09T08:25:04.849090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_audio_path='/kaggle/working/train/train/audio'\nlabels=os.listdir(train_audio_path)\n\n#find count of each label and plot bar graph\nno_of_recordings=[]\nfor label in labels:\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    no_of_recordings.append(len(waves))\n    \n#plot\nplt.figure(figsize=(30,5))\nindex = np.arange(len(labels))\nplt.bar(index, no_of_recordings)\nplt.xlabel('Commands', fontsize=12)\nplt.ylabel('No of recordings', fontsize=12)\nplt.xticks(index, labels, fontsize=15, rotation=60)\nplt.title('No. of recordings for each command')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:04.851390Z","iopub.execute_input":"2024-06-09T08:25:04.851681Z","iopub.status.idle":"2024-06-09T08:25:05.410866Z","shell.execute_reply.started":"2024-06-09T08:25:04.851655Z","shell.execute_reply":"2024-06-09T08:25:05.409939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave = []\nall_label = []\n\nfor label in labels:\n    if label[0]=='_':\n        continue\n    print(label)\n    waves = [f for f in os.listdir(train_audio_path + '/'+ label) if f.endswith('.wav')]\n    for wav in waves:\n        samples, sample_rate = librosa.load(train_audio_path + '/' + label + '/' + wav, sr = 8000)\n        #samples = librosa.resample(samples, sample_rate, 8000)\n        if(len(samples)== 8000) : \n            all_wave.append(samples)\n            all_label.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:05.414938Z","iopub.execute_input":"2024-06-09T08:25:05.415219Z","iopub.status.idle":"2024-06-09T08:25:46.016281Z","shell.execute_reply.started":"2024-06-09T08:25:05.415195Z","shell.execute_reply":"2024-06-09T08:25:46.015012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ny=le.fit_transform(all_label)\nclasses= list(le.classes_)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:46.017826Z","iopub.execute_input":"2024-06-09T08:25:46.018272Z","iopub.status.idle":"2024-06-09T08:25:46.298245Z","shell.execute_reply.started":"2024-06-09T08:25:46.018205Z","shell.execute_reply":"2024-06-09T08:25:46.297447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:46.299448Z","iopub.execute_input":"2024-06-09T08:25:46.299761Z","iopub.status.idle":"2024-06-09T08:25:46.306357Z","shell.execute_reply.started":"2024-06-09T08:25:46.299733Z","shell.execute_reply":"2024-06-09T08:25:46.305490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.utils import np_utils\ny=np_utils.to_categorical(y, num_classes=len(labels))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:46.307508Z","iopub.execute_input":"2024-06-09T08:25:46.307830Z","iopub.status.idle":"2024-06-09T08:25:52.259824Z","shell.execute_reply.started":"2024-06-09T08:25:46.307799Z","shell.execute_reply":"2024-06-09T08:25:52.258771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave[0:5]","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:52.261064Z","iopub.execute_input":"2024-06-09T08:25:52.261689Z","iopub.status.idle":"2024-06-09T08:25:52.269258Z","shell.execute_reply.started":"2024-06-09T08:25:52.261659Z","shell.execute_reply":"2024-06-09T08:25:52.268380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave = np.array(all_wave).reshape(-1,8000)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:52.270508Z","iopub.execute_input":"2024-06-09T08:25:52.270872Z","iopub.status.idle":"2024-06-09T08:25:52.878231Z","shell.execute_reply.started":"2024-06-09T08:25:52.270839Z","shell.execute_reply":"2024-06-09T08:25:52.877447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_wave.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:52.881270Z","iopub.execute_input":"2024-06-09T08:25:52.881651Z","iopub.status.idle":"2024-06-09T08:25:52.887672Z","shell.execute_reply.started":"2024-06-09T08:25:52.881617Z","shell.execute_reply":"2024-06-09T08:25:52.886710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_tr, x_val, y_tr, y_val = train_test_split(np.array(all_wave),np.array(y),stratify=y,test_size = 0.2,random_state=0,shuffle=True)\nx_te, x_val, y_te, y_val = train_test_split(x_val,y_val,stratify=y_val,test_size = 0.5,random_state=777,shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:52.888780Z","iopub.execute_input":"2024-06-09T08:25:52.889107Z","iopub.status.idle":"2024-06-09T08:25:56.622285Z","shell.execute_reply.started":"2024-06-09T08:25:52.889083Z","shell.execute_reply":"2024-06-09T08:25:56.621106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.layers import Dense, Dropout, Flatten, Conv1D, Input, MaxPooling1D, Bidirectional, LSTM, BatchNormalization, GlobalAveragePooling1D, AveragePooling1D\nfrom keras.models import Model\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom keras import backend as K\nK.clear_session()\n\ninputs = Input(shape=(8000,1))\n\n#First Conv1D layer\nconv = Conv1D(512,240, padding='valid', activation='linear', strides=15)(inputs)\nconv = BatchNormalization()(conv)\nconv = Dropout(0.25)(conv)\n\n# Second Conv1D layer\nconv = Conv1D(512, 5, padding='valid', activation='linear', strides=1)(conv)\nconv = BatchNormalization()(conv)\nconv = Dropout(0.25)(conv)\n\n# Third Conv1D layer\nconv = Conv1D(256, 3, padding='valid', activation='relu', strides=1)(conv)\nconv = BatchNormalization()(conv)\nconv = Dropout(0.25)(conv)\n\n# Bidirectional LSTM layers\nbilstm = Bidirectional(LSTM(units=128, return_sequences=True))(conv)\nbilstm = Dropout(0.25)(bilstm)\nbilstm = BatchNormalization()(bilstm)\n\nbilstm = Bidirectional(LSTM(units=128, return_sequences=True))(bilstm)\nbilstm = Dropout(0.25)(bilstm)\nbilstm = BatchNormalization()(bilstm)\n\n# Global Average Pooling layer\npooling = GlobalAveragePooling1D()(bilstm)\n\n# Dense layers\ndense = Dense(256, activation='relu')(pooling)\ndense = Dropout(0.3)(dense)\n\ndense = Dense(256, activation='relu')(dense)\ndense = Dropout(0.3)(dense)\n\ndense = Dense(128, activation='relu')(dense)\ndense = Dropout(0.3)(dense)\n\noutputs = Dense(len(labels), activation='softmax')(dense)\n\nmodel = Model(inputs, outputs)\nmodel.summary()\n\ndense = Dense(128, activation='relu')(dense)\ndense = Dropout(0.3)(dense)\n\noutputs = Dense(len(labels), activation='softmax')(dense)\n\nmodel = Model(inputs, outputs)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:25:56.623586Z","iopub.execute_input":"2024-06-09T08:25:56.623872Z","iopub.status.idle":"2024-06-09T08:26:00.319801Z","shell.execute_reply.started":"2024-06-09T08:25:56.623847Z","shell.execute_reply":"2024-06-09T08:26:00.318880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer='adam',metrics=['accuracy'])\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=10, min_delta=0.0001) ","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:26:00.320980Z","iopub.execute_input":"2024-06-09T08:26:00.321295Z","iopub.status.idle":"2024-06-09T08:26:00.342437Z","shell.execute_reply.started":"2024-06-09T08:26:00.321263Z","shell.execute_reply":"2024-06-09T08:26:00.341575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history=model.fit(x_tr, y_tr ,epochs=200,callbacks=es, batch_size=64, validation_data=(x_val,y_val))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T08:26:00.343564Z","iopub.execute_input":"2024-06-09T08:26:00.343889Z","iopub.status.idle":"2024-06-09T11:26:10.760750Z","shell.execute_reply.started":"2024-06-09T08:26:00.343858Z","shell.execute_reply":"2024-06-09T11:26:10.759732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot \npyplot.plot(history.history['accuracy'], label='train') \npyplot.plot(history.history['val_accuracy'], label='test') \npyplot.legend()\npyplot.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:31.820056Z","iopub.execute_input":"2024-06-09T11:33:31.820842Z","iopub.status.idle":"2024-06-09T11:33:31.993284Z","shell.execute_reply.started":"2024-06-09T11:33:31.820801Z","shell.execute_reply":"2024-06-09T11:33:31.992264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\nmodel.save('best_model.hdf5')","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:38.014272Z","iopub.execute_input":"2024-06-09T11:33:38.015093Z","iopub.status.idle":"2024-06-09T11:33:38.191928Z","shell.execute_reply.started":"2024-06-09T11:33:38.015060Z","shell.execute_reply":"2024-06-09T11:33:38.191074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nnew_model = tf.keras.models.load_model('/kaggle/working/best_model.hdf5')\n\n# Check its architecture\nnew_model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:43.541056Z","iopub.execute_input":"2024-06-09T11:33:43.541783Z","iopub.status.idle":"2024-06-09T11:33:45.305179Z","shell.execute_reply.started":"2024-06-09T11:33:43.541750Z","shell.execute_reply":"2024-06-09T11:33:45.303889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(audio):\n    prob=model.predict(audio.reshape(1,8000))\n    index=np.argmax(prob[0])\n    return classes[index]","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:45.395713Z","iopub.execute_input":"2024-06-09T11:33:45.396266Z","iopub.status.idle":"2024-06-09T11:33:45.400915Z","shell.execute_reply.started":"2024-06-09T11:33:45.396215Z","shell.execute_reply":"2024-06-09T11:33:45.399968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nindex=random.randint(0,len(x_val)-1)\nsamples=x_val[index]\nprint(\"Audio:\",classes[np.argmax(y_val[index])])\n#ipd.Audio(samples, rate=8000)\nprint(\"Text:\",predict(samples))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:46.338628Z","iopub.execute_input":"2024-06-09T11:33:46.339471Z","iopub.status.idle":"2024-06-09T11:33:47.879677Z","shell.execute_reply.started":"2024-06-09T11:33:46.339440Z","shell.execute_reply":"2024-06-09T11:33:47.878844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x=x_te, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:33:51.020738Z","iopub.execute_input":"2024-06-09T11:33:51.021100Z","iopub.status.idle":"2024-06-09T11:34:02.981490Z","shell.execute_reply.started":"2024-06-09T11:33:51.021070Z","shell.execute_reply":"2024-06-09T11:34:02.980456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_te.shape,predictions.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:02.983532Z","iopub.execute_input":"2024-06-09T11:34:02.984309Z","iopub.status.idle":"2024-06-09T11:34:02.989334Z","shell.execute_reply.started":"2024-06-09T11:34:02.984271Z","shell.execute_reply":"2024-06-09T11:34:02.988413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\nconfusion_matrix = confusion_matrix(y_te.argmax(axis=1),predictions.argmax(axis=1))\nconfusion_matrix\nprint(classification_report(y_te.argmax(axis=1), predictions.argmax(axis=1)))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:02.990476Z","iopub.execute_input":"2024-06-09T11:34:02.990730Z","iopub.status.idle":"2024-06-09T11:34:03.030106Z","shell.execute_reply.started":"2024-06-09T11:34:02.990706Z","shell.execute_reply":"2024-06-09T11:34:03.029214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\n\nsn.heatmap(confusion_matrix, annot=True, cmap=plt.cm.Blues)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:03.032606Z","iopub.execute_input":"2024-06-09T11:34:03.032887Z","iopub.status.idle":"2024-06-09T11:34:04.999504Z","shell.execute_reply.started":"2024-06-09T11:34:03.032862Z","shell.execute_reply":"2024-06-09T11:34:04.998601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pred_actual_df(dataset, y_test, loaded_model, le):\n    preds = loaded_model.predict(dataset, verbose=0)\n    preds = preds.argmax(axis=1)\n    # predictions \n    preds = preds.astype(int).flatten()\n    preds = (le.inverse_transform((preds)))\n    preds = pd.DataFrame({'predictedvalues': preds})\n\n    # Actual labels\n    actual= y_test.argmax(axis=1)\n    actual = actual.astype(int).flatten()\n    actual = (le.inverse_transform((actual)))\n    actual = pd.DataFrame({'actualvalues': actual})\n    \n    finaldf = actual.join(preds)\n    return finaldf","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:05.000802Z","iopub.execute_input":"2024-06-09T11:34:05.001155Z","iopub.status.idle":"2024-06-09T11:34:05.008725Z","shell.execute_reply.started":"2024-06-09T11:34:05.001122Z","shell.execute_reply":"2024-06-09T11:34:05.007752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_compare_df = pred_actual_df(x_te, y_te, model, le)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:05.009818Z","iopub.execute_input":"2024-06-09T11:34:05.010151Z","iopub.status.idle":"2024-06-09T11:34:16.885049Z","shell.execute_reply.started":"2024-06-09T11:34:05.010125Z","shell.execute_reply":"2024-06-09T11:34:16.884022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_compare_df","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:16.886533Z","iopub.execute_input":"2024-06-09T11:34:16.886918Z","iopub.status.idle":"2024-06-09T11:34:16.901793Z","shell.execute_reply.started":"2024-06-09T11:34:16.886881Z","shell.execute_reply":"2024-06-09T11:34:16.900985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists('/kaggle/working/Predict Board'):\n    os.makedirs('/kaggle/working/Predict Board')\ntest_compare_df.to_csv('/kaggle/working/Predict Board/Predictions Test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:16.902925Z","iopub.execute_input":"2024-06-09T11:34:16.903196Z","iopub.status.idle":"2024-06-09T11:34:16.923546Z","shell.execute_reply.started":"2024-06-09T11:34:16.903172Z","shell.execute_reply":"2024-06-09T11:34:16.922836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n# the confusion matrix heat map plot\ndef print_confusion_matrix(confusion_matrix, class_names, figsize = (10,7), fontsize=14):\n    \"\"\"Prints a confusion matrix, as returned by sklearn.metrics.confusion_matrix, as a heatmap.\n    \n    Arguments\n    ---------\n    confusion_matrix: numpy.ndarray\n        The numpy.ndarray object returned from a call to sklearn.metrics.confusion_matrix. \n        Similarly constructed ndarrays can also be used.\n    class_names: list\n        An ordered list of class names, in the order they index the given confusion matrix.\n    figsize: tuple\n        A 2-long tuple, the first value determining the horizontal size of the ouputted figure,\n        the second determining the vertical size. Defaults to (10,7).\n    fontsize: int\n        Font size for axes labels. Defaults to 14.\n        \n    Returns\n    -------\n    matplotlib.figure.Figure\n        The resulting confusion matrix figure\n    \"\"\"\n    df_cm = pd.DataFrame(\n        confusion_matrix, index=class_names, columns=class_names, \n    )\n    fig = plt.figure(figsize=figsize)\n    try:\n        heatmap = sns.heatmap(df_cm, annot=True, fmt=\"d\")\n    except ValueError:\n        raise ValueError(\"Confusion matrix values must be integers.\")\n        \n    heatmap.yaxis.set_ticklabels(heatmap.yaxis.get_ticklabels(), rotation=0, ha='right', fontsize=fontsize)\n    heatmap.xaxis.set_ticklabels(heatmap.xaxis.get_ticklabels(), rotation=45, ha='right', fontsize=fontsize)\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:16.924546Z","iopub.execute_input":"2024-06-09T11:34:16.924822Z","iopub.status.idle":"2024-06-09T11:34:16.932533Z","shell.execute_reply.started":"2024-06-09T11:34:16.924798Z","shell.execute_reply":"2024-06-09T11:34:16.931647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ntest_compare_df= pd.read_csv('/kaggle/working/Predict Board/Predictions Test.csv')\nclasses = test_compare_df.actualvalues.unique()\nclasses.sort()    \npreds_value = test_compare_df['actualvalues'].values\nactual_value = test_compare_df['predictedvalues'].values\nc_test = confusion_matrix(preds_value, actual_value)\nprint_confusion_matrix(c_test, class_names = classes)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:16.935736Z","iopub.execute_input":"2024-06-09T11:34:16.936152Z","iopub.status.idle":"2024-06-09T11:34:18.886142Z","shell.execute_reply.started":"2024-06-09T11:34:16.936127Z","shell.execute_reply":"2024-06-09T11:34:18.885217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cm = pd.DataFrame(confusion_matrix(preds_value, actual_value), index=classes, columns=classes)\nfig = plt.figure(figsize=(10,7))\ntry:\n    heatmap = sns.heatmap(df_cm, annot=True, fmt=\"d\")\nexcept ValueError:\n    raise ValueError(\"Confusion matrix values must be integers.\")\n\nheatmap.yaxis.set_ticklabels(heatmap.yaxis.get_ticklabels(), rotation=0, ha='right', fontsize=14)\nheatmap.xaxis.set_ticklabels(heatmap.xaxis.get_ticklabels(), rotation=45, ha='right', fontsize=14)\nplt.ylabel('True label')\nplt.xlabel('Predicted label')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T11:34:18.887543Z","iopub.execute_input":"2024-06-09T11:34:18.887908Z","iopub.status.idle":"2024-06-09T11:34:21.447954Z","shell.execute_reply.started":"2024-06-09T11:34:18.887872Z","shell.execute_reply":"2024-06-09T11:34:21.446984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}