{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n    <h2><center>Distracted Driver Detection Using CNN</center></h2>\n    <center><p>Languaje used: Python</p>\n        <p>Framework used: tf.keras</p>\n    </center>\n</div>","metadata":{}},{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"#Files\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nimport glob\n\n#DATA\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import one_hot\nfrom keras.utils.np_utils import to_categorical\nfrom sklearn.model_selection import train_test_split\n\n#CNN\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom keras.models import Sequential\nfrom keras.layers import Convolution2D,MaxPooling2D,Flatten,Dense, Dropout, LSTM\nfrom keras.optimizers import Adam\nfrom keras.losses import CategoricalCrossentropy\n\n#VIS\nfrom keras.utils.vis_utils import plot_model","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing data","metadata":{}},{"cell_type":"code","source":"def _prepareData(path): \n    '''\n    parameters: path(STR) of the directory and flag(INT) to know if we prepare data of training or testing\n    return: (LIST) of images of the dataset and the (LIST) of labels\n    \n    For training:\n    -Read images of every directory and extract all images\n    -Resize to (128,128,3)\n    -Read the directory name and asign as a class\n    '''\n    imgsList = []\n    labels = []\n    for directory in sorted(glob.glob(os.path.join(path, '*')), key = lambda k: k.split(\"/\")[-1]):\n            for imgs in glob.glob(os.path.join(directory,'*.jpg')):\n                img_cv = cv2.imread(imgs)\n                img_cv_r = cv2.resize(img_cv,(128,128))\n                imgsList.append(img_cv_r)\n                labels.append(int(directory.split(\"/\")[-1].replace('c','')))\n    \n    X_Train, X_Test, Y_Train, Y_Test =  train_test_split(imgsList,labels, test_size = 0.2)\n    Y_Train = tf.keras.utils.to_categorical(Y_Train, num_classes=10)\n    Y_Test = tf.keras.utils.to_categorical(Y_Test, num_classes=10)\n\n    return np.array(X_Train), np.array(X_Test), Y_Train, Y_Test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Data","metadata":{}},{"cell_type":"code","source":"#Paths\npathTrain_Images = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train/\"\npathPropagate_Images =  \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test/\"\n\n#List of Images for Train and Test\nX_Train, X_Test, Y_Train, Y_Test = _prepareData(pathTrain_Images)\n\nprint(\"Size X_Train: {}, Size Y_Train: {}\".format(len(X_Train),len(Y_Train)))\nprint(\"Size X_Test: {}, Size Y_Test: {}\".format(len(X_Test),len(Y_Test)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check data integrity\n\n### Classes:\n* c0: safe driving\n* c1: texting - right\n* c2: talking on the phone - right\n* c3: texting - left\n* c4: talking on the phone - left\n* c5: operating the radio\n* c6: drinking\n* c7: reaching behind\n* c8: hair and makeup\n* c9: talking to passenger","metadata":{}},{"cell_type":"code","source":"print(len(X_Train))\nprint(X_Train[202].shape)\nprint(X_Train.shape)\nsave = X_Train\nz_train = X_Train\nz_new = np.squeeze((tf.image.rgb_to_grayscale(z_train)))\nprint(z_new.shape)\nprint(z_new.shape[1:])\nz_test  =  np.squeeze((tf.image.rgb_to_grayscale(X_Test)))\nim = X_Train[202]\n\nRGB_im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\nplt.imshow(RGB_im)\nplt.show()\nprint(\"Class: {}\".format(Y_Train[202]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check data distribution","metadata":{}},{"cell_type":"code","source":"data_file = pd.read_csv(\"/kaggle/input/state-farm-distracted-driver-detection/driver_imgs_list.csv\")\ndata_classes = data_file.loc[:,['classname','img']].groupby(by='classname').count().reset_index()\n\ndata_x = list(pd.unique(data_file['classname']))\ndata_y =list(data_classes['img'])\n\n# Parámetros de ploteo (Se va a generar un plot diferente para cada Clase)\nplt.rcParams.update({'font.size': 22})\nplt.figure(figsize=(30,10))\nplt.bar(data_x, data_y, color=['cornflowerblue', 'lightblue', 'steelblue'])  \nplt.ylabel('Count classes')\nplt.title('Classes')\nplt.xticks(rotation=45)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create architecture","metadata":{}},{"cell_type":"code","source":"\nmodel = keras.models.Sequential()\n\n#model.add(keras.layers.InputLayer(\n#    input_shape=(128, 3)\n#))\n\n#model.add(keras.layers.Flatten())\n#model.add(keras.layers.Dense(units=1024, activation='relu',name = 'fc_1'))\n#model.add(keras.layers.Dropout(rate=0.2))\n\nmodel.add(keras.layers.SimpleRNN(512, input_shape=(z_new.shape[1:]), activation='relu', return_sequences=True))\n#model.add(keras.layers.LSTM(128, activation='relu'))\nmodel.add(keras.layers.Dropout(0.1))\n\nmodel.add(keras.layers.SimpleRNN(512, activation='relu'))\nmodel.add(keras.layers.Dropout(0.1))\n\n\n#model.add(keras.layers.Dense(units=512, activation='relu',name = 'fc_2'))\n#model.add(keras.layers.Dense(units=256, activation='relu',name = 'fc3.5'))\nmodel.add(keras.layers.Dense(units=10,activation='softmax',name = 'fc_3'))\n\nmodel.save('/tmp/model')\n#model.compute_output_shape(input_shape=(256,8,8,1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.random.set_seed(1)\n#model.build(input_shape=(None,128,128,3))\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=1e-2, decay=1e-5), loss=tf.keras.losses.CategoricalCrossentropy(from_logits = False), metrics = ['accuracy'])\nprint(model.summary())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Train model","metadata":{}},{"cell_type":"code","source":"history = model.fit(x = z_new, y=Y_Train,epochs = 30, batch_size = 1000 , verbose = 1,validation_split=0.2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate model with test data","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(z_test, Y_Test, verbose = 1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\nnow = datetime.now()\n\ncurrent_time = now.strftime(\"%H:%M\")\nprint(current_time)\n\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\n\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\n#plt.ylim([0.9,1])\nplt.legend(['train','test'], loc='upper left')\nplt.tight_layout()\nplt.savefig(\"Accuracy\" + current_time + \".png\")\nplt.show()\n\n\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\n\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\n#plt.ylim([0,.4])\nplt.legend(['train','test'], loc='upper left')\nplt.tight_layout()\nplt.savefig(\"Loss\" + current_time + \".png\")\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save weights","metadata":{}},{"cell_type":"code","source":"model_json = model.to_json()\nmodel.save_weights('Train_weights_1.h5',overwrite = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('Train_weights_1.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show architecture distribution","metadata":{}},{"cell_type":"code","source":"keras.utils.plot_model(model,\"model.png\",show_shapes = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict Test data and create a submission file","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame({'img':[],'c0':[], 'c1':[],'c2':[], 'c3':[], 'c4':[],'c5':[], 'c6':[], 'c7':[], 'c8':[], 'c9':[]})\ndef _submission(pathPropagate_Images,df):\n    for imgs in glob.glob(os.path.join(pathPropagate_Images,'*.jpg')):\n        img_cv = cv2.imread(imgs)\n        img_cv_r = cv2.resize(img_cv,(128,128))\n        img_cv_predict = np.reshape(img_cv_r,[1,128,128,3])\n        arr_predict = model.predict(img_cv_predict,batch_size = 1)\n        #print(imgs.split('/')[-1])\n        df = df.append(\n            {\n                'img':imgs.split('/')[-1],\n                'c0':round(arr_predict[0][0],2), \n                'c1':round(arr_predict[0][1],2),\n                'c2':round(arr_predict[0][2],2),\n                'c3':round(arr_predict[0][3],2),\n                'c4':round(arr_predict[0][4],2),\n                'c5':round(arr_predict[0][5],2),\n                'c6':round(arr_predict[0][6],2),\n                'c7':round(arr_predict[0][7],2),\n                'c8':round(arr_predict[0][8],2),\n                'c9':round(arr_predict[0][9],2)\n            },\n            ignore_index=True\n        )\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_cv = cv2.imread(\"/kaggle/input/state-farm-distracted-driver-detection/imgs/test/img_41.jpg\")\nimg_cv_r = cv2.resize(img_cv,(128,128))\nimg_cv_predict = np.reshape(img_cv_r,[1,128,128,3])\narr_predict = model.predict(img_cv_predict,batch_size = 1)\n\nprint(arr_predict)\nprint(round(arr_predict[0][9],2))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pathPropagate_Images =  \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test/\"\ndf = _submission(pathPropagate_Images,df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.shape)\ndf.head(50)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save submission file","metadata":{}},{"cell_type":"code","source":"df.to_csv('submission_file.csv',index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}