{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Distracted Driver Detection - Kaggle Competition","metadata":{}},{"cell_type":"markdown","source":"Given the dataset consisting of driver images in car and corresponding labels for 10 nos. categories (e.g. safe driving, texting, talking etc.), the task is to build a classification model to predict the category for that image.","metadata":{}},{"cell_type":"markdown","source":"## Analyzing data","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom shutil import copyfile,rmtree\nimport random\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:07:59.205814Z","iopub.execute_input":"2023-07-18T13:07:59.206606Z","iopub.status.idle":"2023-07-18T13:07:59.211753Z","shell.execute_reply.started":"2023-07-18T13:07:59.206557Z","shell.execute_reply":"2023-07-18T13:07:59.210773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting the list of all train images\ndata = pd.read_csv(\"/kaggle/input/state-farm-distracted-driver-detection/driver_imgs_list.csv\", usecols = [1,2])\ndata.nunique(),data.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:07:59.213786Z","iopub.execute_input":"2023-07-18T13:07:59.214562Z","iopub.status.idle":"2023-07-18T13:07:59.267768Z","shell.execute_reply.started":"2023-07-18T13:07:59.214528Z","shell.execute_reply":"2023-07-18T13:07:59.266753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list of all classes\nclasses_list = data['classname'].unique()\nclasses_list","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:07:59.269836Z","iopub.execute_input":"2023-07-18T13:07:59.270179Z","iopub.status.idle":"2023-07-18T13:07:59.279247Z","shell.execute_reply.started":"2023-07-18T13:07:59.270147Z","shell.execute_reply":"2023-07-18T13:07:59.277514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dictionary containing all train data file names, class wise\ntrain_data_files={}\nfor cls, image_name in data.values:\n    key = cls\n    if key in train_data_files:\n        train_data_files[key].append(image_name)\n    else:\n        train_data_files[key] = [image_name]\n\n# printing the size of dataset for each class\nfor key in train_data_files:\n    print(key, \":\", len(train_data_files[key]))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-18T13:07:59.282262Z","iopub.execute_input":"2023-07-18T13:07:59.282860Z","iopub.status.idle":"2023-07-18T13:07:59.332213Z","shell.execute_reply.started":"2023-07-18T13:07:59.282763Z","shell.execute_reply":"2023-07-18T13:07:59.331129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splittin, transofrming and generating image data","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 128\nIMAGE_SIZE = 224","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:07:59.334979Z","iopub.execute_input":"2023-07-18T13:07:59.335534Z","iopub.status.idle":"2023-07-18T13:07:59.339946Z","shell.execute_reply.started":"2023-07-18T13:07:59.335497Z","shell.execute_reply":"2023-07-18T13:07:59.338819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR='/kaggle/input/state-farm-distracted-driver-detection/imgs/train'\ndatagen = ImageDataGenerator(\n        rescale = 1./255,\n        validation_split = 0.2\n)\n\ntraining_data = datagen.flow_from_directory(TRAIN_DIR,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=BATCH_SIZE,\n                                        subset='training',shuffle=False)\n\nevaluating_data = datagen.flow_from_directory(TRAIN_DIR,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=BATCH_SIZE,\n                                        subset='validation',shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:27:43.978049Z","iopub.execute_input":"2023-07-18T13:27:43.979102Z","iopub.status.idle":"2023-07-18T13:28:38.735880Z","shell.execute_reply.started":"2023-07-18T13:27:43.979052Z","shell.execute_reply":"2023-07-18T13:28:38.734935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating the model - CNN","metadata":{}},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dense, Flatten","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.737682Z","iopub.execute_input":"2023-07-18T13:28:38.738159Z","iopub.status.idle":"2023-07-18T13:28:38.743685Z","shell.execute_reply.started":"2023-07-18T13:28:38.738113Z","shell.execute_reply":"2023-07-18T13:28:38.742519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.Sequential([\n      Conv2D(16, (3,3), activation='relu', input_shape = (IMAGE_SIZE, IMAGE_SIZE, 3)),\n      MaxPooling2D(2, 2),\n      Conv2D(32, (3,3), activation='relu'),\n      MaxPooling2D(2, 2),\n      Conv2D(64, (3,3), activation='relu'),\n      MaxPooling2D(2, 2),\n      Flatten(),\n      Dense(1024, activation='relu'),\n      Dense(10, activation='softmax')\n])","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.745111Z","iopub.execute_input":"2023-07-18T13:28:38.745764Z","iopub.status.idle":"2023-07-18T13:28:38.823874Z","shell.execute_reply.started":"2023-07-18T13:28:38.745729Z","shell.execute_reply":"2023-07-18T13:28:38.823000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile the model\nmodel.compile(optimizer= Adam(learning_rate = 0.001), loss = 'categorical_crossentropy', metrics = ['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.826231Z","iopub.execute_input":"2023-07-18T13:28:38.826540Z","iopub.status.idle":"2023-07-18T13:28:38.861345Z","shell.execute_reply.started":"2023-07-18T13:28:38.826507Z","shell.execute_reply":"2023-07-18T13:28:38.860659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MIN_DELTA=0.005\nEPOCHS=20\nPATIENCE=2","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.862274Z","iopub.execute_input":"2023-07-18T13:28:38.862581Z","iopub.status.idle":"2023-07-18T13:28:38.866771Z","shell.execute_reply.started":"2023-07-18T13:28:38.862548Z","shell.execute_reply":"2023-07-18T13:28:38.865984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to stop training if no significant change in validation data accuracy\nes = EarlyStopping(monitor = 'val_accuracy', patience = PATIENCE, min_delta = MIN_DELTA)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.867883Z","iopub.execute_input":"2023-07-18T13:28:38.868236Z","iopub.status.idle":"2023-07-18T13:28:38.878682Z","shell.execute_reply.started":"2023-07-18T13:28:38.868203Z","shell.execute_reply":"2023-07-18T13:28:38.877860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fitting and generating the model\nmodel.fit(\n    training_data, \n    epochs = EPOCHS, \n    validation_data = evaluating_data,\n    callbacks = [es]\n         )","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:28:38.879697Z","iopub.execute_input":"2023-07-18T13:28:38.880046Z","iopub.status.idle":"2023-07-18T13:51:02.846968Z","shell.execute_reply.started":"2023-07-18T13:28:38.880013Z","shell.execute_reply":"2023-07-18T13:51:02.845950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict and load test data","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import image_dataset_from_directory","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:52:17.480684Z","iopub.execute_input":"2023-07-18T13:52:17.481087Z","iopub.status.idle":"2023-07-18T13:52:17.486580Z","shell.execute_reply.started":"2023-07-18T13:52:17.481053Z","shell.execute_reply":"2023-07-18T13:52:17.485506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_dir=\"/kaggle/input/state-farm-distracted-driver-detection/imgs\"","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:52:17.823037Z","iopub.execute_input":"2023-07-18T13:52:17.823952Z","iopub.status.idle":"2023-07-18T13:52:17.828875Z","shell.execute_reply.started":"2023-07-18T13:52:17.823905Z","shell.execute_reply":"2023-07-18T13:52:17.827876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transforming test data same as training data\ntest_data_gen = ImageDataGenerator(\n    rescale = 1./225\n)\n\ntest_data = test_data_gen.flow_from_directory(img_dir,\n                          target_size = (IMAGE_SIZE, IMAGE_SIZE),\n                          classes = ['test'],\n                          shuffle = False,\n                          batch_size = BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:52:18.249483Z","iopub.execute_input":"2023-07-18T13:52:18.249843Z","iopub.status.idle":"2023-07-18T13:53:23.177805Z","shell.execute_reply.started":"2023-07-18T13:52:18.249812Z","shell.execute_reply":"2023-07-18T13:53:23.176818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = model.predict(test_data)\npredicted.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-18T13:53:23.179641Z","iopub.execute_input":"2023-07-18T13:53:23.180024Z","iopub.status.idle":"2023-07-18T14:05:56.023729Z","shell.execute_reply.started":"2023-07-18T13:53:23.179987Z","shell.execute_reply":"2023-07-18T14:05:56.022762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Loading the prediction in required format","metadata":{}},{"cell_type":"code","source":"test_data_files = image_dataset_from_directory(\n    '/kaggle/input/state-farm-distracted-driver-detection/imgs/test',\n     labels = None,\n    label_mode=None,\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T14:05:56.026899Z","iopub.execute_input":"2023-07-18T14:05:56.027226Z","iopub.status.idle":"2023-07-18T14:06:35.550975Z","shell.execute_reply.started":"2023-07-18T14:05:56.027197Z","shell.execute_reply":"2023-07-18T14:06:35.549874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(predicted)\ndf.columns = ['c0','c1','c2','c3','c4','c5','c6','c7','c8','c9']\nfilepath = [i.split('/')[-1] for i in test_data_files.file_paths]\ndf1 = pd.DataFrame(filepath)\ndf1.columns = ['img']\ndf = df1.join(df)\ndf.to_csv('/kaggle/working/output.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-18T14:06:35.553395Z","iopub.execute_input":"2023-07-18T14:06:35.553758Z","iopub.status.idle":"2023-07-18T14:06:36.856579Z","shell.execute_reply.started":"2023-07-18T14:06:35.553722Z","shell.execute_reply":"2023-07-18T14:06:36.855578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-07-18T14:06:36.858079Z","iopub.execute_input":"2023-07-18T14:06:36.858440Z","iopub.status.idle":"2023-07-18T14:06:36.895895Z","shell.execute_reply.started":"2023-07-18T14:06:36.858405Z","shell.execute_reply":"2023-07-18T14:06:36.894674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}