{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORTING THE LIBRARIES","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pickle\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.datasets import load_files\nfrom keras.utils import np_utils\nimport matplotlib.pyplot as plt\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D\nfrom keras.layers import Dropout, Flatten, Dense\nfrom keras.models import Sequential\nfrom keras.utils.vis_utils import plot_model\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.utils import to_categorical\nfrom sklearn.metrics import confusion_matrix\nfrom keras.preprocessing import image                  \nfrom tqdm import tqdm\n\nimport seaborn as sns\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:25:23.144663Z","iopub.execute_input":"2021-09-15T08:25:23.145167Z","iopub.status.idle":"2021-09-15T08:25:31.152459Z","shell.execute_reply.started":"2021-09-15T08:25:23.145129Z","shell.execute_reply":"2021-09-15T08:25:31.151449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretty display for notebooks\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:25:36.436991Z","iopub.execute_input":"2021-09-15T08:25:36.437383Z","iopub.status.idle":"2021-09-15T08:25:36.444462Z","shell.execute_reply.started":"2021-09-15T08:25:36.437346Z","shell.execute_reply":"2021-09-15T08:25:36.443365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-15T08:25:40.55395Z","iopub.execute_input":"2021-09-15T08:25:40.554328Z","iopub.status.idle":"2021-09-15T08:25:41.301043Z","shell.execute_reply.started":"2021-09-15T08:25:40.554297Z","shell.execute_reply":"2021-09-15T08:25:41.299623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the train,test and model directories\n\nWe will create the directories for train,test and model training paths if not present","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"../input/state-farm-distracted-driver-detection/imgs\"\nTEST_DIR = os.path.join(DATA_DIR,\"test\")\nTRAIN_DIR = os.path.join(DATA_DIR,\"train\")\nMODEL_PATH = os.path.join(os.getcwd(),\"model\",\"self_trained\")\nPICKLE_DIR = os.path.join(os.getcwd(),\"pickle_files\")\nCSV_DIR = os.path.join(os.getcwd(),\"csv_files\")","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:25:44.062494Z","iopub.execute_input":"2021-09-15T08:25:44.062887Z","iopub.status.idle":"2021-09-15T08:25:44.070623Z","shell.execute_reply.started":"2021-09-15T08:25:44.062843Z","shell.execute_reply":"2021-09-15T08:25:44.068831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(TEST_DIR):\n    print(\"Testing data does not exists\")\nif not os.path.exists(TRAIN_DIR):\n    print(\"Training data does not exists\")\nif not os.path.exists(MODEL_PATH):\n    print(\"Model path does not exists\")\n    os.makedirs(MODEL_PATH)\n    print(\"Model path created\")\nif not os.path.exists(PICKLE_DIR):\n    os.makedirs(PICKLE_DIR)\nif not os.path.exists(CSV_DIR):\n    os.makedirs(CSV_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:25:47.509902Z","iopub.execute_input":"2021-09-15T08:25:47.510598Z","iopub.status.idle":"2021-09-15T08:25:47.528856Z","shell.execute_reply.started":"2021-09-15T08:25:47.510562Z","shell.execute_reply":"2021-09-15T08:25:47.527115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"We will create a csv file having the location of the files present for training and test images and their associated class if present so that it is easily traceable.","metadata":{}},{"cell_type":"code","source":"def create_csv(DATA_DIR,filename):\n    class_names = os.listdir(DATA_DIR)\n    data = list()\n    if(os.path.isdir(os.path.join(DATA_DIR,class_names[0]))):\n        for class_name in class_names:\n            file_names = os.listdir(os.path.join(DATA_DIR,class_name))\n            for file in file_names:\n                data.append({\n                    \"Filename\":os.path.join(DATA_DIR,class_name,file),\n                    \"ClassName\":class_name\n                })\n    else:\n        class_name = \"test\"\n        file_names = os.listdir(DATA_DIR)\n        for file in file_names:\n            data.append(({\n                \"FileName\":os.path.join(DATA_DIR,file),\n                \"ClassName\":class_name\n            }))\n    data = pd.DataFrame(data)\n    data.to_csv(os.path.join(os.getcwd(),\"csv_files\",filename),index=False)\n\ncreate_csv(TRAIN_DIR,\"train.csv\")\ncreate_csv(TEST_DIR,\"test.csv\")\ndata_train = pd.read_csv(os.path.join(os.getcwd(),\"csv_files\",\"train.csv\"))\ndata_test = pd.read_csv(os.path.join(os.getcwd(),\"csv_files\",\"test.csv\"))\n","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:25:54.158738Z","iopub.execute_input":"2021-09-15T08:25:54.159278Z","iopub.status.idle":"2021-09-15T08:25:59.044616Z","shell.execute_reply.started":"2021-09-15T08:25:54.159244Z","shell.execute_reply":"2021-09-15T08:25:59.043672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:26:04.213847Z","iopub.execute_input":"2021-09-15T08:26:04.214361Z","iopub.status.idle":"2021-09-15T08:26:04.251241Z","shell.execute_reply.started":"2021-09-15T08:26:04.214329Z","shell.execute_reply":"2021-09-15T08:26:04.249861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['ClassName'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:26:07.739225Z","iopub.execute_input":"2021-09-15T08:26:07.739829Z","iopub.status.idle":"2021-09-15T08:26:07.756576Z","shell.execute_reply.started":"2021-09-15T08:26:07.739776Z","shell.execute_reply":"2021-09-15T08:26:07.75503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.describe()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-15T08:26:11.295413Z","iopub.execute_input":"2021-09-15T08:26:11.295878Z","iopub.status.idle":"2021-09-15T08:26:11.371932Z","shell.execute_reply.started":"2021-09-15T08:26:11.295844Z","shell.execute_reply":"2021-09-15T08:26:11.370984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nnf = data_train['ClassName'].value_counts(sort=False)\nlabels = data_train['ClassName'].value_counts(sort=False).index.tolist()\ny = np.array(nf)\nwidth = 1/1.5\nN = len(y)\nx = range(N)\n\nfig = plt.figure(figsize=(20,15))\nay = fig.add_subplot(211)\n\nplt.xticks(x, labels, size=15)\nplt.yticks(size=15)\n\nay.bar(x, y, width, color=\"blue\")\n\nplt.title('Bar Chart',size=25)\nplt.xlabel('classname',size=15)\nplt.ylabel('Count',size=15)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:26:17.196625Z","iopub.execute_input":"2021-09-15T08:26:17.197159Z","iopub.status.idle":"2021-09-15T08:26:17.441683Z","shell.execute_reply.started":"2021-09-15T08:26:17.197125Z","shell.execute_reply":"2021-09-15T08:26:17.440661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:26:32.420019Z","iopub.execute_input":"2021-09-15T08:26:32.420417Z","iopub.status.idle":"2021-09-15T08:26:32.433167Z","shell.execute_reply.started":"2021-09-15T08:26:32.420384Z","shell.execute_reply":"2021-09-15T08:26:32.432135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:26:56.161102Z","iopub.execute_input":"2021-09-15T08:26:56.161461Z","iopub.status.idle":"2021-09-15T08:26:56.167701Z","shell.execute_reply.started":"2021-09-15T08:26:56.161433Z","shell.execute_reply":"2021-09-15T08:26:56.166399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observation:\n1. There are total 22424 training samples\n2. There are total 79726 training samples\n3. The training dataset is equally balanced to a great extent and hence we need not do any downsampling of the data","metadata":{}},{"cell_type":"markdown","source":"## Converting into numerical values","metadata":{}},{"cell_type":"code","source":"labels_list = list(set(data_train['ClassName'].values.tolist()))\nlabels_id = {label_name:id for id,label_name in enumerate(labels_list)}\nprint(labels_id)\ndata_train['ClassName'].replace(labels_id,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:02.835029Z","iopub.execute_input":"2021-09-15T08:27:02.835624Z","iopub.status.idle":"2021-09-15T08:27:02.873702Z","shell.execute_reply.started":"2021-09-15T08:27:02.835591Z","shell.execute_reply":"2021-09-15T08:27:02.872943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(os.path.join(os.getcwd(),\"pickle_files\",\"labels_list.pkl\"),\"wb\") as handle:\n    pickle.dump(labels_id,handle)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:07.451105Z","iopub.execute_input":"2021-09-15T08:27:07.451477Z","iopub.status.idle":"2021-09-15T08:27:07.456517Z","shell.execute_reply.started":"2021-09-15T08:27:07.451446Z","shell.execute_reply":"2021-09-15T08:27:07.455693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = to_categorical(data_train['ClassName'])\nprint(labels.shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:10.766287Z","iopub.execute_input":"2021-09-15T08:27:10.766922Z","iopub.status.idle":"2021-09-15T08:27:10.77296Z","shell.execute_reply.started":"2021-09-15T08:27:10.766887Z","shell.execute_reply":"2021-09-15T08:27:10.77189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Splitting into Train and Test sets","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nxtrain,xtest,ytrain,ytest = train_test_split(data_train.iloc[:,0],labels,test_size = 0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:15.938008Z","iopub.execute_input":"2021-09-15T08:27:15.938356Z","iopub.status.idle":"2021-09-15T08:27:15.963193Z","shell.execute_reply.started":"2021-09-15T08:27:15.938328Z","shell.execute_reply":"2021-09-15T08:27:15.962072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Converting into 64*64 images \nYou can substitute 64,64 to 224,224 for better results only if ram is >32gb","metadata":{}},{"cell_type":"code","source":"\ndef path_to_tensor(img_path):\n    # loads RGB image as PIL.Image.Image type\n    img = image.load_img(img_path, target_size=(64, 64))\n    # convert PIL.Image.Image type to 3D tensor with shape (64, 64, 3)\n    x = image.img_to_array(img)\n    # convert 3D tensor to 4D tensor with shape (1, 64,64, 3) and return 4D tensor\n    return np.expand_dims(x, axis=0)\n\ndef paths_to_tensor(img_paths):\n    list_of_tensors = [path_to_tensor(img_path) for img_path in tqdm(img_paths)]\n    return np.vstack(list_of_tensors)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:22.537452Z","iopub.execute_input":"2021-09-15T08:27:22.537878Z","iopub.status.idle":"2021-09-15T08:27:22.545427Z","shell.execute_reply.started":"2021-09-15T08:27:22.537843Z","shell.execute_reply":"2021-09-15T08:27:22.544629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom PIL import ImageFile                            \nImageFile.LOAD_TRUNCATED_IMAGES = True                 \n\n# pre-process the data for Keras\ntrain_tensors = paths_to_tensor(xtrain).astype('float32')/255 - 0.5\n","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:27:26.030678Z","iopub.execute_input":"2021-09-15T08:27:26.031353Z","iopub.status.idle":"2021-09-15T08:31:00.86885Z","shell.execute_reply.started":"2021-09-15T08:27:26.031299Z","shell.execute_reply":"2021-09-15T08:31:00.867847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_tensors = paths_to_tensor(xtest).astype('float32')/255 - 0.5\n","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:31:06.747466Z","iopub.execute_input":"2021-09-15T08:31:06.748224Z","iopub.status.idle":"2021-09-15T08:31:58.416356Z","shell.execute_reply.started":"2021-09-15T08:31:06.748183Z","shell.execute_reply":"2021-09-15T08:31:58.414957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##takes too much ram \n## run this if your ram is greater than 16gb \n# test_tensors = paths_to_tensor(data_test.iloc[:,0]).astype('float32')/255 - 0.5 ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the Model","metadata":{}},{"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(filters=64, kernel_size=2, padding='same', activation='relu', input_shape=(64,64,3), kernel_initializer='glorot_normal'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Conv2D(filters=128, kernel_size=2, padding='same', activation='relu', kernel_initializer='glorot_normal'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Conv2D(filters=256, kernel_size=2, padding='same', activation='relu', kernel_initializer='glorot_normal'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Conv2D(filters=512, kernel_size=2, padding='same', activation='relu', kernel_initializer='glorot_normal'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.5))\nmodel.add(Flatten())\nmodel.add(Dense(500, activation='relu', kernel_initializer='glorot_normal'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(10, activation='softmax', kernel_initializer='glorot_normal'))\n\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:33:01.387084Z","iopub.execute_input":"2021-09-15T08:33:01.387493Z","iopub.status.idle":"2021-09-15T08:33:01.700621Z","shell.execute_reply.started":"2021-09-15T08:33:01.387462Z","shell.execute_reply":"2021-09-15T08:33:01.699006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model,to_file=os.path.join(MODEL_PATH,\"model_distracted_driver.png\"),show_shapes=True,show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:33:07.648445Z","iopub.execute_input":"2021-09-15T08:33:07.648915Z","iopub.status.idle":"2021-09-15T08:33:08.736132Z","shell.execute_reply.started":"2021-09-15T08:33:07.64885Z","shell.execute_reply":"2021-09-15T08:33:08.734984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='rmsprop', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:33:16.279334Z","iopub.execute_input":"2021-09-15T08:33:16.279962Z","iopub.status.idle":"2021-09-15T08:33:16.30038Z","shell.execute_reply.started":"2021-09-15T08:33:16.279908Z","shell.execute_reply":"2021-09-15T08:33:16.29948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = os.path.join(MODEL_PATH,\"distracted-{epoch:02d}-{val_accuracy:.2f}.hdf5\")\ncheckpoint = ModelCheckpoint(filepath, monitor='val_accuracy', verbose=1, save_best_only=True, mode='max',period=1)\ncallbacks_list = [checkpoint]","metadata":{"execution":{"iopub.status.busy":"2021-09-15T08:33:19.755414Z","iopub.execute_input":"2021-09-15T08:33:19.756271Z","iopub.status.idle":"2021-09-15T08:33:19.762966Z","shell.execute_reply.started":"2021-09-15T08:33:19.756202Z","shell.execute_reply":"2021-09-15T08:33:19.761964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = model.fit(train_tensors,ytrain,validation_data = (valid_tensors, ytest),epochs=25, batch_size=40, shuffle=True,callbacks=callbacks_list)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-15T08:33:23.03372Z","iopub.execute_input":"2021-09-15T08:33:23.035957Z","iopub.status.idle":"2021-09-15T09:56:27.384735Z","shell.execute_reply.started":"2021-09-15T08:33:23.035911Z","shell.execute_reply":"2021-09-15T09:56:27.38382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 12))\nax1.plot(model_history.history['loss'], color='b', label=\"Training loss\")\nax1.plot(model_history.history['val_loss'], color='r', label=\"validation loss\")\nax1.set_xticks(np.arange(1, 25, 1))\nax1.set_yticks(np.arange(0, 1, 0.1))\n\nax2.plot(model_history.history['accuracy'], color='b', label=\"Training accuracy\")\nax2.plot(model_history.history['val_accuracy'], color='r',label=\"Validation accuracy\")\nax2.set_xticks(np.arange(1, 25, 1))\n\nlegend = plt.legend(loc='best', shadow=True)\nplt.tight_layout()\nplt.show()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-15T10:31:31.252286Z","iopub.execute_input":"2021-09-15T10:31:31.252685Z","iopub.status.idle":"2021-09-15T10:31:31.277519Z","shell.execute_reply.started":"2021-09-15T10:31:31.252653Z","shell.execute_reply":"2021-09-15T10:31:31.274825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Analysis\n\nFinding the Confusion matrix,Precision,Recall and F1 score to analyse the model thus created ","metadata":{}},{"cell_type":"code","source":"\ndef print_confusion_matrix(confusion_matrix, class_names, figsize = (10,7), fontsize=14):\n    df_cm = pd.DataFrame(\n        confusion_matrix, index=class_names, columns=class_names, \n    )\n    fig = plt.figure(figsize=figsize)\n    try:\n        heatmap = sns.heatmap(df_cm, annot=True, fmt=\"d\")\n    except ValueError:\n        raise ValueError(\"Confusion matrix values must be integers.\")\n    heatmap.yaxis.set_ticklabels(heatmap.yaxis.get_ticklabels(), rotation=0, ha='right', fontsize=fontsize)\n    heatmap.xaxis.set_ticklabels(heatmap.xaxis.get_ticklabels(), rotation=45, ha='right', fontsize=fontsize)\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    fig.savefig(os.path.join(MODEL_PATH,\"confusion_matrix.png\"))\n    return fig\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_heatmap(n_labels, n_predictions, class_names):\n    labels = n_labels #sess.run(tf.argmax(n_labels, 1))\n    predictions = n_predictions #sess.run(tf.argmax(n_predictions, 1))\n\n#     confusion_matrix = sess.run(tf.contrib.metrics.confusion_matrix(labels, predictions))\n    matrix = confusion_matrix(labels.argmax(axis=1),predictions.argmax(axis=1))\n    row_sum = np.sum(matrix, axis = 1)\n    w, h = matrix.shape\n\n    c_m = np.zeros((w, h))\n\n    for i in range(h):\n        c_m[i] = matrix[i] * 100 / row_sum[i]\n\n    c = c_m.astype(dtype = np.uint8)\n\n    \n    heatmap = print_confusion_matrix(c, class_names, figsize=(18,10), fontsize=20)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = list()\nfor name,idx in labels_id.items():\n    class_names.append(name)\n# print(class_names)\nypred = model.predict(valid_tensors)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_heatmap(ytest,ypred,class_names)","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Precision Recall F1 Score","metadata":{}},{"cell_type":"code","source":"ypred_class = np.argmax(ypred,axis=1)\n# print(ypred_class[:10])\nytest = np.argmax(ytest,axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(ytest,ypred_class)\nprint('Accuracy: %f' % accuracy)\n# precision tp / (tp + fp)\nprecision = precision_score(ytest, ypred_class,average='weighted')\nprint('Precision: %f' % precision)\n# recall: tp / (tp + fn)\nrecall = recall_score(ytest,ypred_class,average='weighted')\nprint('Recall: %f' % recall)\n# f1: 2 tp / (2 tp + fp + fn)\nf1 = f1_score(ytest,ypred_class,average='weighted')\nprint('F1 score: %f' % f1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}