{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-16T19:11:27.649572Z","iopub.execute_input":"2022-03-16T19:11:27.650399Z","iopub.status.idle":"2022-03-16T19:12:04.561937Z","shell.execute_reply.started":"2022-03-16T19:11:27.650262Z","shell.execute_reply":"2022-03-16T19:12:04.551179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pickle\nimport shutil\nimport numpy as np\nimport seaborn as sns\nfrom sklearn.datasets import load_files\nfrom keras.utils import np_utils\nimport matplotlib.pyplot as plt\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D\nfrom keras.layers import Dropout, Flatten, Dense\nfrom keras.models import Sequential\nfrom keras.utils.vis_utils import plot_model\nfrom keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score,f1_score\n\n\nfrom PIL import ImageFile   \nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing import image                  \nfrom tqdm import tqdm\n\nfrom keras.applications.vgg16 import VGG16","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:34.677111Z","iopub.execute_input":"2022-03-16T19:13:34.677842Z","iopub.status.idle":"2022-03-16T19:13:43.37204Z","shell.execute_reply.started":"2022-03-16T19:13:34.677795Z","shell.execute_reply":"2022-03-16T19:13:43.370921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = \"../input/state-farm-distracted-driver-detection/imgs\"\nTEST_DIR = os.path.join(DATA_DIR,\"test\")\nTRAIN_DIR = os.path.join(DATA_DIR,\"train\")\n\nCSV_DIR = os.path.join(os.getcwd(),\"csv_files\")\n\nMODEL_PATH = os.path.join(os.getcwd(),\"model\",\"vgg16\")\nPICKLE_PATH = os.path.join(os.getcwd(),\"pickle\")\nTEST_CSV = os.path.join(os.getcwd(),\"csv_files\",\"test.csv\")\nTRAIN_CSV = os.path.join(os.getcwd(),\"csv_files\",\"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:46.97399Z","iopub.execute_input":"2022-03-16T19:13:46.974661Z","iopub.status.idle":"2022-03-16T19:13:46.985392Z","shell.execute_reply.started":"2022-03-16T19:13:46.974618Z","shell.execute_reply":"2022-03-16T19:13:46.981393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists(TEST_DIR):\n    print(\"Testing data does not exists\")\nif not os.path.exists(TRAIN_DIR):\n    print(\"Training data does not exists\")\nif not os.path.exists(MODEL_PATH):\n    print(\"Model path does not exists\")\n    os.makedirs(MODEL_PATH)\n    print(\"Model path created\")\nelse:\n    shutil.rmtree(MODEL_PATH)\n    os.makedirs(MODEL_PATH)\nif not os.path.exists(PICKLE_PATH):\n    os.makedirs(PICKLE_PATH)\nif not os.path.exists(CSV_DIR):\n    os.makedirs(CSV_DIR)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:47.88641Z","iopub.execute_input":"2022-03-16T19:13:47.887508Z","iopub.status.idle":"2022-03-16T19:13:47.89994Z","shell.execute_reply.started":"2022-03-16T19:13:47.887452Z","shell.execute_reply":"2022-03-16T19:13:47.898986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(os.getcwd())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:50.494061Z","iopub.execute_input":"2022-03-16T19:13:50.495231Z","iopub.status.idle":"2022-03-16T19:13:50.505773Z","shell.execute_reply.started":"2022-03-16T19:13:50.495161Z","shell.execute_reply":"2022-03-16T19:13:50.50492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_csv(DATA_DIR,filename):\n    class_names = os.listdir(DATA_DIR)\n    data = list()\n    if(os.path.isdir(os.path.join(DATA_DIR,class_names[0]))):\n        for class_name in class_names:\n            file_names = os.listdir(os.path.join(DATA_DIR,class_name))\n            for file in file_names:\n                data.append({\n                    \"Filename\":os.path.join(DATA_DIR,class_name,file),\n                    \"ClassName\":class_name\n                })\n    else:\n        class_name = \"test\"\n        file_names = os.listdir(DATA_DIR)\n        for file in file_names:\n            data.append(({\n                \"FileName\":os.path.join(DATA_DIR,file),\n                \"ClassName\":class_name\n            }))\n    data = pd.DataFrame(data)\n    data.to_csv(os.path.join(os.getcwd(),\"csv_files\",filename),index=False)\n\ncreate_csv(TRAIN_DIR,\"train.csv\")\ncreate_csv(TEST_DIR,\"test.csv\")\ndata_train = pd.read_csv(os.path.join(os.getcwd(),\"csv_files\",\"train.csv\"))\ndata_test = pd.read_csv(os.path.join(os.getcwd(),\"csv_files\",\"test.csv\"))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:50.872856Z","iopub.execute_input":"2022-03-16T19:13:50.874339Z","iopub.status.idle":"2022-03-16T19:13:51.967847Z","shell.execute_reply.started":"2022-03-16T19:13:50.874262Z","shell.execute_reply":"2022-03-16T19:13:51.966886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv(TRAIN_CSV)\ndata_test = pd.read_csv(TEST_CSV)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:54.858555Z","iopub.execute_input":"2022-03-16T19:13:54.859755Z","iopub.status.idle":"2022-03-16T19:13:55.017607Z","shell.execute_reply.started":"2022-03-16T19:13:54.859704Z","shell.execute_reply":"2022-03-16T19:13:55.01639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_list = list(set(data_train['ClassName'].values.tolist()))\nlabels_id = {label_name:id for id,label_name in enumerate(labels_list)}\nprint(labels_id)\ndata_train['ClassName'].replace(labels_id,inplace=True)\n\nlabels = to_categorical(data_train['ClassName'])\nprint(labels.shape)\n\nwith open(os.path.join(PICKLE_PATH,\"labels_list_vgg16.pkl\"),\"wb\") as handle:\n    pickle.dump(labels_id,handle)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:55.045478Z","iopub.execute_input":"2022-03-16T19:13:55.045864Z","iopub.status.idle":"2022-03-16T19:13:55.090767Z","shell.execute_reply.started":"2022-03-16T19:13:55.045815Z","shell.execute_reply":"2022-03-16T19:13:55.089659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain,xtest,ytrain,ytest = train_test_split(data_train.iloc[:,0],labels,test_size = 0.2,random_state=42)\ndef path_to_tensor(img_path):\n    # loads RGB image as PIL.Image.Image type\n    img = image.load_img(img_path, target_size=(64, 64))\n    # convert PIL.Image.Image type to 3D tensor with shape (64, 64, 3)\n    x = image.img_to_array(img)\n    # convert 3D tensor to 4D tensor with shape (1, 64, 64, 3) and return 4D tensor\n    return np.expand_dims(x, axis=0)\n\ndef paths_to_tensor(img_paths):\n    list_of_tensors = [path_to_tensor(img_path) for img_path in tqdm(img_paths)]\n    return np.vstack(list_of_tensors)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:57.170459Z","iopub.execute_input":"2022-03-16T19:13:57.171231Z","iopub.status.idle":"2022-03-16T19:13:57.185171Z","shell.execute_reply.started":"2022-03-16T19:13:57.171192Z","shell.execute_reply":"2022-03-16T19:13:57.183803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageFile.LOAD_TRUNCATED_IMAGES = True                 \n# pre-process the data for Keras\ntrain_tensors = paths_to_tensor(xtrain).astype('float32')/255 - 0.5\nvalid_tensors = paths_to_tensor(xtest).astype('float32')/255 - 0.5","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:13:57.332531Z","iopub.execute_input":"2022-03-16T19:13:57.332935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = VGG16(include_top=False)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:56:56.549465Z","iopub.execute_input":"2022-03-16T18:56:56.549795Z","iopub.status.idle":"2022-03-16T18:56:59.625966Z","shell.execute_reply.started":"2022-03-16T18:56:56.54976Z","shell.execute_reply":"2022-03-16T18:56:59.624857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_vgg16 = model.predict(train_tensors,verbose=1)\nvalid_vgg16 = model.predict(valid_tensors,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:57:21.666695Z","iopub.execute_input":"2022-03-16T18:57:21.667076Z","iopub.status.idle":"2022-03-16T18:57:36.421356Z","shell.execute_reply.started":"2022-03-16T18:57:21.667027Z","shell.execute_reply":"2022-03-16T18:57:36.420199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train shape\",train_vgg16.shape)\nprint(\"Validation shape\",valid_vgg16.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:57:40.935665Z","iopub.execute_input":"2022-03-16T18:57:40.937509Z","iopub.status.idle":"2022-03-16T18:57:40.944485Z","shell.execute_reply.started":"2022-03-16T18:57:40.937466Z","shell.execute_reply":"2022-03-16T18:57:40.943382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features = train_vgg16[0]\nvalid_features = valid_vgg16[0]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:57:48.485616Z","iopub.execute_input":"2022-03-16T18:57:48.485937Z","iopub.status.idle":"2022-03-16T18:57:48.490677Z","shell.execute_reply.started":"2022-03-16T18:57:48.485901Z","shell.execute_reply":"2022-03-16T18:57:48.489574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train features shape\",train_features.shape)\nprint(\"Validation features shape\",valid_features.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:57:55.237571Z","iopub.execute_input":"2022-03-16T18:57:55.238505Z","iopub.status.idle":"2022-03-16T18:57:55.244884Z","shell.execute_reply.started":"2022-03-16T18:57:55.238434Z","shell.execute_reply":"2022-03-16T18:57:55.243488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VGG16_model = Sequential()\nVGG16_model.add(GlobalAveragePooling2D(input_shape=train_features.shape))\nVGG16_model.add(Dense(10, activation='softmax', kernel_initializer='glorot_normal'))\n\nVGG16_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:58:03.11252Z","iopub.execute_input":"2022-03-16T18:58:03.112877Z","iopub.status.idle":"2022-03-16T18:58:03.146624Z","shell.execute_reply.started":"2022-03-16T18:58:03.112841Z","shell.execute_reply":"2022-03-16T18:58:03.145346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VGG16_model.compile(loss='categorical_crossentropy', optimizer='rmsprop', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:58:10.300974Z","iopub.execute_input":"2022-03-16T18:58:10.301279Z","iopub.status.idle":"2022-03-16T18:58:10.320116Z","shell.execute_reply.started":"2022-03-16T18:58:10.301246Z","shell.execute_reply":"2022-03-16T18:58:10.319211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(VGG16_model,to_file=os.path.join(os.getcwd(),\"model\",\"vgg16\",\"model_distracted_driver_vgg16.png\"),show_shapes=True,show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:58:17.712777Z","iopub.execute_input":"2022-03-16T18:58:17.713103Z","iopub.status.idle":"2022-03-16T18:58:18.905058Z","shell.execute_reply.started":"2022-03-16T18:58:17.713068Z","shell.execute_reply":"2022-03-16T18:58:18.903834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = os.path.join(MODEL_PATH,\"distracted-{epoch:02d}-{val_accuracy:.2f}.hdf5\")\ncheckpoint = ModelCheckpoint(filepath, monitor='val_accuracy', verbose=1, save_best_only=True, mode='max',period=1)\ncallbacks_list = [checkpoint]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T18:58:28.0955Z","iopub.execute_input":"2022-03-16T18:58:28.095932Z","iopub.status.idle":"2022-03-16T18:58:28.103906Z","shell.execute_reply.started":"2022-03-16T18:58:28.095896Z","shell.execute_reply":"2022-03-16T18:58:28.102553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = VGG16_model.fit(train_vgg16,ytrain,validation_data = (valid_vgg16, ytest),epochs=30, batch_size=16, shuffle=True,callbacks=callbacks_list)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:00:44.183517Z","iopub.execute_input":"2022-03-16T19:00:44.183825Z","iopub.status.idle":"2022-03-16T19:02:51.711422Z","shell.execute_reply.started":"2022-03-16T19:00:44.183793Z","shell.execute_reply":"2022-03-16T19:02:51.710229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_confusion_matrix(confusion_matrix, class_names, figsize = (10,7), fontsize=14):\n    df_cm = pd.DataFrame(\n        confusion_matrix, index=class_names, columns=class_names, \n    )\n    fig = plt.figure(figsize=figsize)\n    try:\n        heatmap = sns.heatmap(df_cm, annot=True, fmt=\"d\")\n    except ValueError:\n        raise ValueError(\"Confusion matrix values must be integers.\")\n    heatmap.yaxis.set_ticklabels(heatmap.yaxis.get_ticklabels(), rotation=0, ha='right', fontsize=fontsize)\n    heatmap.xaxis.set_ticklabels(heatmap.xaxis.get_ticklabels(), rotation=45, ha='right', fontsize=fontsize)\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    fig.savefig(os.path.join(MODEL_PATH,\"confusion_matrix.png\"))\n    return fig","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:02:55.587635Z","iopub.execute_input":"2022-03-16T19:02:55.587948Z","iopub.status.idle":"2022-03-16T19:02:55.600211Z","shell.execute_reply.started":"2022-03-16T19:02:55.587914Z","shell.execute_reply":"2022-03-16T19:02:55.599195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_heatmap(n_labels, n_predictions, class_names):\n    labels = n_labels #sess.run(tf.argmax(n_labels, 1))\n    predictions = n_predictions #sess.run(tf.argmax(n_predictions, 1))\n\n#     confusion_matrix = sess.run(tf.contrib.metrics.confusion_matrix(labels, predictions))\n    matrix = confusion_matrix(labels.argmax(axis=1),predictions.argmax(axis=1))\n    row_sum = np.sum(matrix, axis = 1)\n    w, h = matrix.shape\n\n    c_m = np.zeros((w, h))\n\n    for i in range(h):\n        c_m[i] = matrix[i] * 100 / row_sum[i]\n\n    c = c_m.astype(dtype = np.uint8)\n\n    \n    heatmap = print_confusion_matrix(c, class_names, figsize=(18,10), fontsize=20)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:02:58.008114Z","iopub.execute_input":"2022-03-16T19:02:58.008957Z","iopub.status.idle":"2022-03-16T19:02:58.016167Z","shell.execute_reply.started":"2022-03-16T19:02:58.008919Z","shell.execute_reply":"2022-03-16T19:02:58.01501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = list()\nfor name,idx in labels_id.items():\n    class_names.append(name)\n# print(class_names)\nypred = VGG16_model.predict(valid_vgg16,verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:03:00.469991Z","iopub.execute_input":"2022-03-16T19:03:00.470824Z","iopub.status.idle":"2022-03-16T19:03:00.781838Z","shell.execute_reply.started":"2022-03-16T19:03:00.470786Z","shell.execute_reply":"2022-03-16T19:03:00.780729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_heatmap(ytest,ypred,class_names)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:03:00.784366Z","iopub.execute_input":"2022-03-16T19:03:00.784686Z","iopub.status.idle":"2022-03-16T19:03:01.904303Z","shell.execute_reply.started":"2022-03-16T19:03:00.784655Z","shell.execute_reply":"2022-03-16T19:03:01.90309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ypred_class = np.argmax(ypred,axis=1)\n# print(ypred_class[:10])\nytest = np.argmax(ytest,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:03:04.587146Z","iopub.execute_input":"2022-03-16T19:03:04.587952Z","iopub.status.idle":"2022-03-16T19:03:04.594983Z","shell.execute_reply.started":"2022-03-16T19:03:04.587898Z","shell.execute_reply":"2022-03-16T19:03:04.593643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = accuracy_score(ytest,ypred_class)\nprint('Accuracy: %f' % accuracy)\n# precision tp / (tp + fp)\nprecision = precision_score(ytest, ypred_class,average='weighted')\nprint('Precision: %f' % precision)\n# recall: tp / (tp + fn)\nrecall = recall_score(ytest,ypred_class,average='weighted')\nprint('Recall: %f' % recall)\n# f1: 2 tp / (2 tp + fp + fn)\nf1 = f1_score(ytest,ypred_class,average='weighted')\nprint('F1 score: %f' % f1)","metadata":{"execution":{"iopub.status.busy":"2022-03-16T19:03:04.933741Z","iopub.execute_input":"2022-03-16T19:03:04.934099Z","iopub.status.idle":"2022-03-16T19:03:04.959025Z","shell.execute_reply.started":"2022-03-16T19:03:04.934023Z","shell.execute_reply":"2022-03-16T19:03:04.958009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}