{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Karnarjun2000@gmail.com","metadata":{}},{"cell_type":"markdown","source":"# Distracted Driver Detection - Kaggle Competition","metadata":{}},{"cell_type":"code","source":"#importing necessary library like numpy and pandas\nimport numpy as np\nimport pandas as pd","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-06-30T06:47:58.110115Z","iopub.execute_input":"2021-06-30T06:47:58.110434Z","iopub.status.idle":"2021-06-30T06:47:58.115117Z","shell.execute_reply.started":"2021-06-30T06:47:58.110401Z","shell.execute_reply":"2021-06-30T06:47:58.113861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing necessary library \nimport os    # to extract the data from system or if using kaggle then from directory\nfrom pathlib import Path      # to locate the path of file\n\nfrom keras.preprocessing import image   # to convert images to the array\n\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import train_test_split    # for creating training and test data\nimport keras","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:47:58.134873Z","iopub.execute_input":"2021-06-30T06:47:58.135398Z","iopub.status.idle":"2021-06-30T06:48:03.885223Z","shell.execute_reply.started":"2021-06-30T06:47:58.135358Z","shell.execute_reply":"2021-06-30T06:48:03.884449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This location is of kaggle dirctory where we can find image\np=Path(\"../input/state-farm-distracted-driver-detection/imgs/train\")","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:48:03.886769Z","iopub.execute_input":"2021-06-30T06:48:03.887166Z","iopub.status.idle":"2021-06-30T06:48:03.894906Z","shell.execute_reply.started":"2021-06-30T06:48:03.887126Z","shell.execute_reply":"2021-06-30T06:48:03.894054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dirs=p.glob('*')  # accessing all the content images of above path","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:48:03.896377Z","iopub.execute_input":"2021-06-30T06:48:03.896793Z","iopub.status.idle":"2021-06-30T06:48:03.905104Z","shell.execute_reply.started":"2021-06-30T06:48:03.896694Z","shell.execute_reply":"2021-06-30T06:48:03.904242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=[]   # x will containt image array of each image\ny=[]    # y will contains label of images","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:48:03.908504Z","iopub.execute_input":"2021-06-30T06:48:03.908742Z","iopub.status.idle":"2021-06-30T06:48:03.915065Z","shell.execute_reply.started":"2021-06-30T06:48:03.908719Z","shell.execute_reply":"2021-06-30T06:48:03.914108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in dirs:\n    # as if i is converted into string then last char is digit can be assigne as number labels 0 to 9 for 10 labels\n    label=str(i).split('/')[-1]  \n    for img_path in i.glob('*.jpg'):    # going into each image of each labels for converting it into array\n        image_data=image.load_img(img_path,target_size=(100,100))  # loading the image into size of 100,100\n        \n        #converting each image into corresponding array\n        image_data_array=image.img_to_array(image_data)    \n        \n        # appending the above array into x\n        x.append(image_data_array)\n        \n        # appending the corresponding label into the y\n        y.append(label[-1])\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:48:03.918816Z","iopub.execute_input":"2021-06-30T06:48:03.919152Z","iopub.status.idle":"2021-06-30T06:52:07.707563Z","shell.execute_reply.started":"2021-06-30T06:48:03.919121Z","shell.execute_reply":"2021-06-30T06:52:07.706777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting list into numpy array for fast processing\nx=np.array(x)\ny=np.array(y)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:07.709806Z","iopub.execute_input":"2021-06-30T06:52:07.710108Z","iopub.status.idle":"2021-06-30T06:52:08.525178Z","shell.execute_reply.started":"2021-06-30T06:52:07.710081Z","shell.execute_reply":"2021-06-30T06:52:08.524407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating training and testing data with test size of 30% all all data\nx_train,x_test,y_train,y_test=train_test_split(x,y,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:08.526757Z","iopub.execute_input":"2021-06-30T06:52:08.527143Z","iopub.status.idle":"2021-06-30T06:52:09.226674Z","shell.execute_reply.started":"2021-06-30T06:52:08.527105Z","shell.execute_reply":"2021-06-30T06:52:09.225747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# chekcing the shape of data\nx_train.shape,x_test.shape,y_train.shape,y_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.227976Z","iopub.execute_input":"2021-06-30T06:52:09.228433Z","iopub.status.idle":"2021-06-30T06:52:09.236416Z","shell.execute_reply.started":"2021-06-30T06:52:09.228395Z","shell.execute_reply":"2021-06-30T06:52:09.235315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encoding the y_train and y_test into one hot coding\ny_train=keras.utils.to_categorical(y_train)\ny_test=keras.utils.to_categorical(y_test)","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.237946Z","iopub.execute_input":"2021-06-30T06:52:09.238303Z","iopub.status.idle":"2021-06-30T06:52:09.259567Z","shell.execute_reply.started":"2021-06-30T06:52:09.238267Z","shell.execute_reply":"2021-06-30T06:52:09.258897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.261452Z","iopub.execute_input":"2021-06-30T06:52:09.261707Z","iopub.status.idle":"2021-06-30T06:52:09.26851Z","shell.execute_reply.started":"2021-06-30T06:52:09.261684Z","shell.execute_reply":"2021-06-30T06:52:09.267542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.270075Z","iopub.execute_input":"2021-06-30T06:52:09.270491Z","iopub.status.idle":"2021-06-30T06:52:09.282243Z","shell.execute_reply.started":"2021-06-30T06:52:09.270418Z","shell.execute_reply":"2021-06-30T06:52:09.28152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting each array's value between 0 to 1\nx_train,x_test=x_train/255,x_test/255","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.285488Z","iopub.execute_input":"2021-06-30T06:52:09.285742Z","iopub.status.idle":"2021-06-30T06:52:09.979197Z","shell.execute_reply.started":"2021-06-30T06:52:09.285716Z","shell.execute_reply":"2021-06-30T06:52:09.978029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.980712Z","iopub.execute_input":"2021-06-30T06:52:09.981097Z","iopub.status.idle":"2021-06-30T06:52:09.990765Z","shell.execute_reply.started":"2021-06-30T06:52:09.981059Z","shell.execute_reply":"2021-06-30T06:52:09.989869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Above all process is for date modification so that can be utilised for applyig machine learning algorith\n# below is for machine learning part applied through keras","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:09.992178Z","iopub.execute_input":"2021-06-30T06:52:09.992753Z","iopub.status.idle":"2021-06-30T06:52:09.998416Z","shell.execute_reply.started":"2021-06-30T06:52:09.992712Z","shell.execute_reply":"2021-06-30T06:52:09.997296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# importing models and layers from keras\nfrom keras import models,layers","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:10.000259Z","iopub.execute_input":"2021-06-30T06:52:10.000829Z","iopub.status.idle":"2021-06-30T06:52:10.008916Z","shell.execute_reply.started":"2021-06-30T06:52:10.000788Z","shell.execute_reply":"2021-06-30T06:52:10.007985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using Sequential model of keras for applying predicting\nmodel=models.Sequential()\n\n# as we are working with images so convolution layer is being applied with 32 unit with filter of size 3,3 and padding same.\n# as, it is first layer so input size should be given as image is of size 100,100 and with three channel so input_shape is (100,100,3)\nmodel.add(layers.Conv2D(32, (3, 3),padding='same',input_shape=(100,100,3),activation='relu'))\n# maxpooling is applied for reducing the actual pixel of image       \nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# dropout is applied for removing the redundancy\nmodel.add(layers.Dropout(0.5))\n\n# again convolutio layer is applied with 64 units of filte size(3,3) with padding same and activation function applied is relu\nmodel.add(layers.Conv2D(64, (3, 3), padding='same', activation='relu'))\n\n# maxpooling is applied for reducing the actual pixel of image       \nmodel.add(layers.MaxPooling2D((2, 2)))\n\n# dropout is applied for removing the redundancy\nmodel.add(layers.Dropout(0.5))\n\n# again a convolution layer , maxpooling and dropout layer are applied.\nmodel.add(layers.Conv2D(128, (3, 3), padding='same', activation='relu'))\nmodel.add(layers.MaxPooling2D((8, 8)))\nmodel.add(layers.Dropout(0.5))\n\n# as to give input to covolution layer is 2D and output is also 2D.\n# but input to dense neural network to be 1D, so converted into 1D using Flatten.\nmodel.add(layers.Flatten())\n# Flatten inpute is applied into Dense neural network of 300 units with activation function of relu\nmodel.add(layers.Dense(300,activation='relu'))\n# as final output should be one hot encodes of size 10, so final layer has 10 units and activation function applied is softmax for one hot encoding\nmodel.add(layers.Dense(10, activation='softmax'))\n\n# Neural netwrok part is complete","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:10.012163Z","iopub.execute_input":"2021-06-30T06:52:10.012555Z","iopub.status.idle":"2021-06-30T06:52:11.846113Z","shell.execute_reply.started":"2021-06-30T06:52:10.012494Z","shell.execute_reply":"2021-06-30T06:52:11.845201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# it is printing the summary of each layers and parameter in each layer.\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:11.849728Z","iopub.execute_input":"2021-06-30T06:52:11.850052Z","iopub.status.idle":"2021-06-30T06:52:11.869254Z","shell.execute_reply.started":"2021-06-30T06:52:11.850025Z","shell.execute_reply":"2021-06-30T06:52:11.868287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now compiling the model with optimizer adam and loss function is categorical_crossentropy because of one-hot-encoding,and evaluaton of model will be done on accuracy\nmodel.compile(optimizer='adam',loss='categorical_crossentropy',metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:11.870465Z","iopub.execute_input":"2021-06-30T06:52:11.870821Z","iopub.status.idle":"2021-06-30T06:52:11.888509Z","shell.execute_reply.started":"2021-06-30T06:52:11.870791Z","shell.execute_reply":"2021-06-30T06:52:11.88769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fitting the model with training data, with epochs 10 helping in increasing accuracy, usng validationg data as testing data\nmodel.fit(x_train,y_train,epochs=10,validation_data=(x_test,y_test))\n# we can see that \n# accuracy is more than 94% which is better accuracy \n#but can be increased more by changing different epochs number and modify number of units in layers etc.","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:52:11.890714Z","iopub.execute_input":"2021-06-30T06:52:11.891327Z","iopub.status.idle":"2021-06-30T06:53:10.724548Z","shell.execute_reply.started":"2021-06-30T06:52:11.89129Z","shell.execute_reply":"2021-06-30T06:53:10.723778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting the accuracy on testing data\nmodel.evaluate(x_test,y_test)\n# we can see the testing accuracy is more than 98% which is more than better accuracy.\n# moreover accuracy on testing is more than training hence this model is not overfitting.","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:53:10.727582Z","iopub.execute_input":"2021-06-30T06:53:10.727852Z","iopub.status.idle":"2021-06-30T06:53:12.315053Z","shell.execute_reply.started":"2021-06-30T06:53:10.727816Z","shell.execute_reply":"2021-06-30T06:53:12.314391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:53:12.318038Z","iopub.execute_input":"2021-06-30T06:53:12.318294Z","iopub.status.idle":"2021-06-30T06:53:12.327021Z","shell.execute_reply.started":"2021-06-30T06:53:12.318268Z","shell.execute_reply":"2021-06-30T06:53:12.326353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:53:12.330146Z","iopub.execute_input":"2021-06-30T06:53:12.330435Z","iopub.status.idle":"2021-06-30T06:53:12.341938Z","shell.execute_reply.started":"2021-06-30T06:53:12.330401Z","shell.execute_reply":"2021-06-30T06:53:12.34115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x_test[0])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:53:12.34496Z","iopub.execute_input":"2021-06-30T06:53:12.345343Z","iopub.status.idle":"2021-06-30T06:53:12.58387Z","shell.execute_reply.started":"2021-06-30T06:53:12.345314Z","shell.execute_reply":"2021-06-30T06:53:12.58289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-30T06:53:12.584964Z","iopub.execute_input":"2021-06-30T06:53:12.585291Z","iopub.status.idle":"2021-06-30T06:53:12.593256Z","shell.execute_reply.started":"2021-06-30T06:53:12.585257Z","shell.execute_reply":"2021-06-30T06:53:12.592253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}