{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Distracted Driver Recognition - Deep learning","metadata":{}},{"cell_type":"markdown","source":"This is the code for our Machine Learning Report","metadata":{}},{"cell_type":"markdown","source":"\n\n* [Part 1: Importing the libraries](#section-one)\n* [Part 2: Preprocessing](#section-two)\n* [Part 3: EDA](#section-three)\n    - [Part 3.1: Statistics](#threeone)\n    - [Part 3.2: Visualization](#threetwo)\n* [Part 4: ANN](#section-four)\n    - [Part 4.1: Creating the Model](#fourone)\n    - [Part 4.2: Training the Model](#train-model)\n    - [Part 4.3: Testing the Model](#test-model)\n    - [Part 4.4: Experiments](#fourfour)\n* [Part 5: RNN](#section-five)\n    - [Part 5.1: Prepare data for RNN](#fiveone)\n    - [Part 5.2: Creating the Model](#fivetwo)\n    - [Part 5.3: Training the Model](#fivethree)\n    - [Part 5.4: Testing the Model](#fivefour)","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Part 1: Importing the libraries","metadata":{}},{"cell_type":"code","source":"#Files\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport cv2\nimport glob\n\n#DATA\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.preprocessing.text import one_hot\nfrom keras.utils.np_utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.image as mpimg\n\n#CNN\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom keras.models import Sequential\nfrom keras.layers import Convolution2D,MaxPooling2D,Flatten,Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom keras.losses import CategoricalCrossentropy\n\n#VIS\nfrom keras.utils.vis_utils import plot_model","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-19T11:42:46.971589Z","iopub.execute_input":"2022-01-19T11:42:46.972061Z","iopub.status.idle":"2022-01-19T11:42:51.478528Z","shell.execute_reply.started":"2022-01-19T11:42:46.971940Z","shell.execute_reply":"2022-01-19T11:42:51.477440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# Part 2: Preprocessing","metadata":{}},{"cell_type":"code","source":"def _prepareData(path): \n    '''\n    params: path(string)\n    return: [list list] of images in dataset and the list of labels\n    '''\n    labelsList = []\n    listOfimg = []\n    for directory in sorted(glob.glob(os.path.join(path, '*')), key = lambda k: k.split(\"/\")[-1]):\n            for img in glob.glob(os.path.join(directory,'*.jpg')):\n                imgcv = cv2.imread(img)\n                imgcv_r = cv2.resize(imgcv,(128,128)) #Resize to 128,128\n                listOfimg.append(imgcv_r)\n                labelsList.append(int(directory.split(\"/\")[-1].replace('c','')))\n    \n    X_Train, X_Test, Y_Train, Y_Test =  train_test_split(listOfimg,labelsList, test_size = 0.2)\n    Y_Train = tf.keras.utils.to_categorical(Y_Train, num_classes=10)\n    Y_Test = tf.keras.utils.to_categorical(Y_Test, num_classes=10)\n\n    return np.array(X_Train), np.array(X_Test), Y_Train, Y_Test","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:42:51.480017Z","iopub.execute_input":"2022-01-19T11:42:51.480540Z","iopub.status.idle":"2022-01-19T11:42:51.493790Z","shell.execute_reply.started":"2022-01-19T11:42:51.480499Z","shell.execute_reply":"2022-01-19T11:42:51.491050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''#Paths\npathTrainImages = \"/kaggle/input/state-farm-distracted-driver-detection/img/train/\"\npathPropagateImages =  \"/kaggle/input/state-farm-distracted-driver-detection/img/test/\"\n\n#List of Images for Train and Test\nX_Train, X_Test, Y_Train, Y_Test = _prepareData(pathTrainImages)\n\nprint(\"Size X_Train: {}, Size Y_Train: {}\".format(len(X_Train),len(Y_Train)))\nprint(\"Size X_Test: {}, Size Y_Test: {}\".format(len(X_Test),len(Y_Test)))\n'''\n#Paths\npathTrain_Images = \"/kaggle/input/state-farm-distracted-driver-detection/imgs/train/\"\npathPropagate_Images =  \"/kaggle/input/state-farm-distracted-driver-detection/imgs/test/\"\n\n#List of Images for Train and Test\nX_Train, X_Test, Y_Train, Y_Test = _prepareData(pathTrain_Images)\n\nprint(\"Size X_Train: {}, Size Y_Train: {}\".format(len(X_Train),len(Y_Train)))\nprint(\"Size X_Test: {}, Size Y_Test: {}\".format(len(X_Test),len(Y_Test)))","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:42:51.495176Z","iopub.execute_input":"2022-01-19T11:42:51.496329Z","iopub.status.idle":"2022-01-19T11:46:39.147358Z","shell.execute_reply.started":"2022-01-19T11:42:51.496265Z","shell.execute_reply":"2022-01-19T11:46:39.146495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n# Part 3: EDA","metadata":{}},{"cell_type":"markdown","source":"<a id=\"threeone\"></a>\n# Part 3.1: Statistics","metadata":{}},{"cell_type":"code","source":"data_file = pd.read_csv(\"/kaggle/input/state-farm-distracted-driver-detection/driver_imgs_list.csv\")\ndata_classes = data_file.loc[:,['classname','img']].groupby(by='classname').count().reset_index()\n\ndata_x = list(pd.unique(data_file['classname']))\ndata_y =list(data_classes['img'])\n\n# Parámetros de ploteo (Se va a generar un plot diferente para cada Clase)\nplt.rcParams.update({'font.size': 22})\nplt.figure(figsize=(30,10))\nplt.bar(data_x, data_y, color=['cornflowerblue', 'lightblue', 'steelblue'])  \nplt.ylabel('Count classes')\nplt.title('Classes')\nplt.xticks(rotation=45)\ndata_file.head","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:39.163668Z","iopub.execute_input":"2022-01-19T11:46:39.164115Z","iopub.status.idle":"2022-01-19T11:46:39.689057Z","shell.execute_reply.started":"2022-01-19T11:46:39.164082Z","shell.execute_reply":"2022-01-19T11:46:39.688066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"threetwo\"></a>\n# Part 3.2: Visualization","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\npx.histogram(data_file, x=\"classname\", color=\"classname\", title=\"Number of images by categories \")","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:39.690405Z","iopub.execute_input":"2022-01-19T11:46:39.690617Z","iopub.status.idle":"2022-01-19T11:46:43.552904Z","shell.execute_reply.started":"2022-01-19T11:46:39.690590Z","shell.execute_reply":"2022-01-19T11:46:43.552004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the frequency of images per driver\ndriversID = pd.DataFrame((data_file['subject'].value_counts()).reset_index())\ndriversID.columns = ['driver_id', 'Counts']\npx.histogram(driversID, x=\"driver_id\",y=\"Counts\" ,color=\"driver_id\", title=\"Number of images by subjects \")","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:43.554278Z","iopub.execute_input":"2022-01-19T11:46:43.555299Z","iopub.status.idle":"2022-01-19T11:46:43.765352Z","shell.execute_reply.started":"2022-01-19T11:46:43.555249Z","shell.execute_reply":"2022-01-19T11:46:43.764305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories = {'c0': 'Safe driving', \n                'c1': 'Texting - right', \n                'c2': 'Talking on the phone - right', \n                'c3': 'Texting - left', \n                'c4': 'Talking on the phone - left', \n                'c5': 'Operating the radio', \n                'c6': 'Drinking', \n                'c7': 'Reaching behind', \n                'c8': 'Hair and makeup', \n                'c9': 'Talking to passenger'}\n\n\nplt.figure(figsize = (12, 20))\nimage_count = 1\nBASE_URL = '../input/state-farm-distracted-driver-detection/imgs/train/'\nfor directory in os.listdir(BASE_URL):\n    if directory[0] != '.':\n        for i, file in enumerate(os.listdir(BASE_URL + directory)):\n            if i == 1:\n                break\n            else:\n                fig = plt.subplot(5, 2, image_count)\n                image_count += 1\n                image = mpimg.imread(BASE_URL + directory + '/' + file)\n                plt.imshow(image)\n                plt.title(categories[directory])","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:43.766904Z","iopub.execute_input":"2022-01-19T11:46:43.767400Z","iopub.status.idle":"2022-01-19T11:46:46.564798Z","shell.execute_reply.started":"2022-01-19T11:46:43.767337Z","shell.execute_reply":"2022-01-19T11:46:46.563821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n# Part 4: ANN","metadata":{}},{"cell_type":"markdown","source":"<a id=\"fourone\"></a>\n## Part 4.1: Creating the Model","metadata":{}},{"cell_type":"code","source":"model = keras.models.Sequential()\n\nmodel.add(keras.layers.InputLayer(\n    input_shape=(128, 128, 3)\n))\n\n\nmodel.add(keras.layers.Flatten())\n#model.add(keras.layers.Dense(units=1024, activation='relu',name = 'fc_1'))\n#model.add(keras.layers.Dropout(rate=0.2))\nmodel.add(keras.layers.Dense(units=512, activation='relu',name = 'fc_2'))\nmodel.add(keras.layers.Dense(units=10,activation='softmax',name = 'fc_3'))\nmodel.save('/tmp/model')\n#model.compute_output_shape(input_shape=(256,8,8,1))","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:46.566345Z","iopub.execute_input":"2022-01-19T11:46:46.566726Z","iopub.status.idle":"2022-01-19T11:46:48.229585Z","shell.execute_reply.started":"2022-01-19T11:46:46.566687Z","shell.execute_reply":"2022-01-19T11:46:48.228583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.random.set_seed(1)\n#model.build(input_shape=(None,128,128,3))\nmodel.compile(optimizer=tf.keras.optimizers.Adam(), loss=tf.keras.losses.CategoricalCrossentropy(from_logits = False), metrics = ['accuracy'])\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:48.232188Z","iopub.execute_input":"2022-01-19T11:46:48.232559Z","iopub.status.idle":"2022-01-19T11:46:48.259660Z","shell.execute_reply.started":"2022-01-19T11:46:48.232510Z","shell.execute_reply":"2022-01-19T11:46:48.258461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"train-model\"></a>\n## Part 4.2: Training the Model","metadata":{}},{"cell_type":"code","source":"history = model.fit(x = X_Train, y=Y_Train,epochs = 10, batch_size = 500, verbose = 1,validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:46:48.264078Z","iopub.execute_input":"2022-01-19T11:46:48.267387Z","iopub.status.idle":"2022-01-19T11:49:50.685098Z","shell.execute_reply.started":"2022-01-19T11:46:48.267326Z","shell.execute_reply":"2022-01-19T11:49:50.684416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"test-model\"></a>\n## Part 4.3: Evaluating the Model","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(X_Test, Y_Test, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:49:50.686968Z","iopub.execute_input":"2022-01-19T11:49:50.687566Z","iopub.status.idle":"2022-01-19T11:49:54.634970Z","shell.execute_reply.started":"2022-01-19T11:49:50.687521Z","shell.execute_reply":"2022-01-19T11:49:54.634324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\n\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\n#plt.ylim([0.9,1])\nplt.legend(['train','test'], loc='upper left')\nplt.show()\n\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\n\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\n#plt.ylim([0,.4])\nplt.legend(['train','test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:49:54.637294Z","iopub.execute_input":"2022-01-19T11:49:54.637622Z","iopub.status.idle":"2022-01-19T11:49:55.029810Z","shell.execute_reply.started":"2022-01-19T11:49:54.637588Z","shell.execute_reply":"2022-01-19T11:49:55.029167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"fourfour\"></a>\n## Part 4.4: Experiments","metadata":{}},{"cell_type":"code","source":"tf.random.set_seed(1)\n#model.build(input_shape=(None,128,128,3))\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001), loss=tf.keras.losses.CategoricalCrossentropy(from_logits = False), metrics = ['accuracy'])\nprint(model.summary())\n\nhistory = model.fit(x = X_Train, y=Y_Train,epochs = 10, batch_size = 500, verbose = 1,validation_split=0.2)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:49:55.031563Z","iopub.execute_input":"2022-01-19T11:49:55.032201Z","iopub.status.idle":"2022-01-19T11:52:55.345082Z","shell.execute_reply.started":"2022-01-19T11:49:55.032151Z","shell.execute_reply":"2022-01-19T11:52:55.344019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(X_Test, Y_Test, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:52:55.347004Z","iopub.execute_input":"2022-01-19T11:52:55.347392Z","iopub.status.idle":"2022-01-19T11:52:59.285066Z","shell.execute_reply.started":"2022-01-19T11:52:55.347344Z","shell.execute_reply":"2022-01-19T11:52:59.284164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five\"></a>\n# Part 5: RNN","metadata":{}},{"cell_type":"markdown","source":"<a id=\"fiveone\"></a>\n## Part 5.1: Prepare data for RNN\n","metadata":{}},{"cell_type":"code","source":"print(len(X_Train))\nprint(X_Train[202].shape)\nprint(X_Train.shape)\nsave = X_Train\nz_train = X_Train\nz_new = np.squeeze((tf.image.rgb_to_grayscale(z_train)))\nprint(z_new.shape)\nz_test  =  np.squeeze((tf.image.rgb_to_grayscale(X_Test)))\nim = X_Train[202]\n\nRGB_im = cv2.cvtColor(im, cv2.COLOR_BGR2RGB)\nplt.imshow(RGB_im)\nplt.show()\nprint(\"Class: {}\".format(Y_Train[202]))","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:52:59.286767Z","iopub.execute_input":"2022-01-19T11:52:59.287028Z","iopub.status.idle":"2022-01-19T11:53:05.517154Z","shell.execute_reply.started":"2022-01-19T11:52:59.286995Z","shell.execute_reply":"2022-01-19T11:53:05.516331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"fivetwo\"></a>\n## Part 5.2: Creating the Model","metadata":{}},{"cell_type":"code","source":"model = keras.models.Sequential()\nmodel.add(keras.layers.SimpleRNN(128, input_shape=(z_new.shape[1:]), activation='relu', return_sequences=True))\nmodel.add(keras.layers.SimpleRNN(128, activation='relu',  return_sequences=True))\nmodel.add(keras.layers.SimpleRNN(128, activation='relu'))\n\nmodel.add(keras.layers.Dense(units=10,activation='softmax',name = 'fc_3'))\n\nmodel.save('/tmp/model')","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:53:05.518457Z","iopub.execute_input":"2022-01-19T11:53:05.518683Z","iopub.status.idle":"2022-01-19T11:53:09.919618Z","shell.execute_reply.started":"2022-01-19T11:53:05.518654Z","shell.execute_reply":"2022-01-19T11:53:09.918549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.random.set_seed(1)\n#model.build(input_shape=(None,128,128,3))\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=1e-3, decay=1e-5), loss=tf.keras.losses.CategoricalCrossentropy(from_logits = False), metrics = ['accuracy'])\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:53:09.922180Z","iopub.execute_input":"2022-01-19T11:53:09.922445Z","iopub.status.idle":"2022-01-19T11:53:09.943012Z","shell.execute_reply.started":"2022-01-19T11:53:09.922415Z","shell.execute_reply":"2022-01-19T11:53:09.941919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"fivethree\"></a>\n## Part 5.3: Training the Model","metadata":{}},{"cell_type":"code","source":"history = model.fit(x = z_new, y=Y_Train,epochs = 30, batch_size = 1000 , verbose = 1,validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-01-19T11:53:09.944357Z","iopub.execute_input":"2022-01-19T11:53:09.944696Z","iopub.status.idle":"2022-01-19T12:03:34.429608Z","shell.execute_reply.started":"2022-01-19T11:53:09.944649Z","shell.execute_reply":"2022-01-19T12:03:34.428723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"fivefour\"></a>\n## Part 5.4: Testing the Model","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = model.evaluate(z_test, Y_Test, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-01-19T12:03:34.432344Z","iopub.execute_input":"2022-01-19T12:03:34.432979Z","iopub.status.idle":"2022-01-19T12:03:46.064094Z","shell.execute_reply.started":"2022-01-19T12:03:34.432940Z","shell.execute_reply":"2022-01-19T12:03:46.063438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\n\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\n#plt.ylim([0.9,1])\nplt.legend(['train','test'], loc='upper left')\nplt.show()\n\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\n\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\n#plt.ylim([0,.4])\nplt.legend(['train','test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-19T12:03:46.065962Z","iopub.execute_input":"2022-01-19T12:03:46.066899Z","iopub.status.idle":"2022-01-19T12:03:46.468201Z","shell.execute_reply.started":"2022-01-19T12:03:46.066850Z","shell.execute_reply":"2022-01-19T12:03:46.467580Z"},"trusted":true},"execution_count":null,"outputs":[]}]}