{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORTING THE LIBRARIES","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pickle\nimport numpy as np\nimport seaborn as sns\nimport secrets\nimport cv2\nfrom PIL import Image\nfrom PIL import ImageFile\nfrom sklearn.datasets import load_files\nfrom keras.utils import np_utils\nimport matplotlib.pyplot as plt\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D\nfrom keras.layers import Dropout, Flatten, Dense\nfrom keras.models import Sequential\nfrom keras.utils.vis_utils import plot_model\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.utils import to_categorical\nfrom sklearn.metrics import confusion_matrix\nfrom keras.preprocessing import image                  \nfrom tqdm import tqdm\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n\nfrom keras.applications.vgg16 import VGG16\n\nimport seaborn as sns\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the train,test and model directories\n\nWe will create the directories for train,test and model training paths if not present","metadata":{}},{"cell_type":"code","source":"TEST_DIR = os.path.join(os.getcwd(),\"imgs\",\"test\")\nTRAIN_DIR = os.path.join(os.getcwd(),\"imgs\",\"train\")\nMODEL_PATH = os.path.join(os.getcwd(),\"model\",\"self_trained\")\nPICKLE_DIR = os.path.join(os.getcwd(),\"pickle_files\")\nCSV_DIR = os.path.join(os.getcwd(),\"csv_files\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(TEST_DIR):\n    print(\"Testing data does not exists\")\nif not os.path.exists(TRAIN_DIR):\n    print(\"Training data does not exists\")\nif not os.path.exists(MODEL_PATH):\n    print(\"Model path does not exists\")\n    os.makedirs(MODEL_PATH)\n    print(\"Model path created\")\nif not os.path.exists(PICKLE_DIR):\n    os.makedirs(PICKLE_DIR)\nif not os.path.exists(CSV_DIR):\n    os.makedirs(CSV_DIR)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the data augmentation definition\n\ngen_per_image = 1\ngen_per_class = 200\nrotation_range = 5\nwidth_shift_range = 0.02\nheight_shift_range = 0.02\nshear_range = 0.01\nzoom_range = 0.05\nhorizontal_flip = False\nfill_mode = \"nearest\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nheight_shift_range = 0.02\nshear_range = 0.01\nzoom_range = 0.05\nhorizontal_flip = False\nfill_mode = \"nearest\"\n\ndef increase_brightness(img, value):\n    hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    h, s, v = cv2.split(hsv)\n\n    lim = 255 - value\n    v[v > lim] = 255\n    v[v <= lim] += value\n\n    final_hsv = cv2.merge((h, s, v))\n    img = cv2.cvtColor(final_hsv, cv2.COLOR_HSV2BGR)\n    return img\n\ndef change_contrast(img, level):\n    img = Image.fromarray(img.astype('uint8'))\n    factor = (259 * (level + 255)) / (255 * (259 - level))\n    def contrast(c):\n        return 128 + factor * (c - 128)\n    return np.array(img.point(contrast))\n\ndef pad_img(img):\n    h, w = img.shape[:2]\n    new_h = int((5 + secrets.randbelow(16)) * h / 100) + h\n    new_w = int((5 + secrets.randbelow(16)) * w / 100) + w\n\n    full_sheet = np.ones((new_h, new_w, 3)) * 255\n\n    p_X = secrets.randbelow(new_h - img.shape[0])\n    p_Y = secrets.randbelow(new_w - img.shape[1])\n\n    full_sheet[p_X : p_X + img.shape[0], p_Y : p_Y + img.shape[1]] = img\n\n    full_sheet = cv2.resize(full_sheet, (w, h), interpolation = cv2.INTER_AREA)\n\n    return full_sheet.astype(np.uint8)\n\ndef preprocess_img(img):\n    img = np.array(img)\n\n    x = secrets.randbelow(2)\n\n    if x == 0:\n        # img = pad_img(img)\n        img = increase_brightness(img, secrets.randbelow(26))\n        img = change_contrast(img, secrets.randbelow(51))\n    else:\n        # img = pad_img(img)\n        img = change_contrast(img, secrets.randbelow(51))\n        img = increase_brightness(img, secrets.randbelow(26))\n\n    return img\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 128\nIMAGE_SIZE = 224\nNUM_EPOCH = 400","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Initialise the parameters for Augmentation.\ndatagen = ImageDataGenerator(\n        rotation_range = rotation_range,\n        width_shift_range = width_shift_range,\n        height_shift_range = height_shift_range,\n        shear_range = shear_range,\n        zoom_range = zoom_range,\n        horizontal_flip = horizontal_flip,\n        fill_mode = fill_mode,\n        validation_split = 0.2,\n        preprocessing_function = preprocess_img)\n\n\ntrain_data = datagen.flow_from_directory(TRAIN_DIR,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=BATCH_SIZE,\n                                        subset='training',shuffle=False)\n\nvalid_data = datagen.flow_from_directory(TRAIN_DIR,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=BATCH_SIZE,\n                                        subset='validation',shuffle=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Defining the Model","metadata":{}},{"cell_type":"code","source":"model = VGG16(include_top=False)\nmodel.summary()","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_vgg16 = model.predict(train_data,verbose=1)\nvalid_vgg16 = model.predict(valid_data,verbose=1)","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.classes","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train shape\",train_vgg16.shape)\nprint(\"Validation shape\",valid_vgg16.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = train_vgg16[0]\nvalid_features = valid_vgg16[0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train features shape\",train_features.shape)\nprint(\"Validation features shape\",valid_features.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VGG16_model = Sequential()\nVGG16_model.add(GlobalAveragePooling2D(input_shape=train_features.shape))\nVGG16_model.add(Dense(10, activation='softmax', kernel_initializer='glorot_normal'))\n\nVGG16_model.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VGG16_model.compile(loss='sparse_categorical_crossentropy', optimizer='rmsprop', metrics=['accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(VGG16_model,to_file=os.path.join(os.getcwd(),\"model\",\"vgg16\",\"model_distracted_driver_vgg16.png\"),show_shapes=True,show_layer_names=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = os.path.join(MODEL_PATH,\"distracted-{epoch:02d}-{val_accuracy:.2f}.hdf5\")\ncheckpoint = ModelCheckpoint(filepath, monitor='val_accuracy', verbose=1, save_best_only=True, mode='max')\ncallbacks_list = [checkpoint]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = VGG16_model.fit(train_vgg16,train_data.classes,validation_data = (valid_vgg16,valid_data.classes),epochs=400,shuffle=True,callbacks=callbacks_list)","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 12))\nax1.plot(model_history.history['loss'], color='b', label=\"Training loss\")\nax1.plot(model_history.history['val_loss'], color='r', label=\"validation loss\")\nax1.set_xticks(np.arange(1, 25, 1))\nax1.set_yticks(np.arange(0, 1, 0.1))\n\nax2.plot(model_history.history['accuracy'], color='b', label=\"Training accuracy\")\nax2.plot(model_history.history['val_accuracy'], color='r',label=\"Validation accuracy\")\nax2.set_xticks(np.arange(1, 25, 1))\n\nlegend = plt.legend(loc='best', shadow=True)\nplt.tight_layout()\nplt.show()","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Analysis\n\nFinding the Confusion matrix,Precision,Recall and F1 score to analyse the model thus created ","metadata":{}},{"cell_type":"code","source":"\ndef print_confusion_matrix(confusion_matrix, class_names, figsize = (10,7), fontsize=14):\n    df_cm = pd.DataFrame(\n        confusion_matrix, index=class_names, columns=class_names, \n    )\n    fig = plt.figure(figsize=figsize)\n    try:\n        heatmap = sns.heatmap(df_cm, annot=True, fmt=\"d\")\n    except ValueError:\n        raise ValueError(\"Confusion matrix values must be integers.\")\n    heatmap.yaxis.set_ticklabels(heatmap.yaxis.get_ticklabels(), rotation=0, ha='right', fontsize=fontsize)\n    heatmap.xaxis.set_ticklabels(heatmap.xaxis.get_ticklabels(), rotation=45, ha='right', fontsize=fontsize)\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    fig.savefig(os.path.join(MODEL_PATH,\"confusion_matrix.png\"))\n    return fig\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_heatmap(n_labels, n_predictions, class_names):\n    labels = n_labels #sess.run(tf.argmax(n_labels, 1))\n    predictions = n_predictions #sess.run(tf.argmax(n_predictions, 1))\n\n#     confusion_matrix = sess.run(tf.contrib.metrics.confusion_matrix(labels, predictions))\n    matrix = confusion_matrix(labels,predictions.argmax(axis=1))\n    row_sum = np.sum(matrix, axis = 1)\n    w, h = matrix.shape\n\n    c_m = np.zeros((w, h))\n\n    for i in range(h):\n        c_m[i] = matrix[i] * 100 / row_sum[i]\n\n    c = c_m.astype(dtype = np.uint8)\n\n    \n    heatmap = print_confusion_matrix(c, class_names, figsize=(18,10), fontsize=20)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ypred = VGG16_model.predict(valid_vgg16)\n\nvalid_list = valid_data.classes.tolist()\n\nypred_class = np.argmax(ypred,axis=1)\nytest = valid_list","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = list()\nfor name,idx in valid_data.class_indices.items():\n    class_names.append(name)\nprint(class_names)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_heatmap(ytest,ypred,class_names)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Precision Recall F1 Score","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score(ytest,ypred_class)\nprint('Accuracy: %f' % accuracy)\n# precision tp / (tp + fp)\nprecision = precision_score(ytest, ypred_class,average='weighted')\nprint('Precision: %f' % precision)\n# recall: tp / (tp + fn)\nrecall = recall_score(ytest,ypred_class,average='weighted')\nprint('Recall: %f' % recall)\n# f1: 2 tp / (2 tp + fp + fn)\nf1 = f1_score(ytest,ypred_class,average='weighted')\nprint('F1 score: %f' % f1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}