{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Mobile Net using revamped dataset 😄","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport os\nimport cv2\nfrom glob import glob\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport random\nfrom pathlib import Path\nimport shutil\n\nimport os\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image, ImageDraw, ImageEnhance","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:40:03.387924Z","iopub.execute_input":"2021-10-09T12:40:03.388267Z","iopub.status.idle":"2021-10-09T12:40:08.736815Z","shell.execute_reply.started":"2021-10-09T12:40:03.388237Z","shell.execute_reply":"2021-10-09T12:40:08.735966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nprint(os.listdir(\"/kaggle/input/distracted-drivers-v2/\"))\n#print(os.listdir(\"/kaggle/working/new_images_train\"))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T11:58:57.415644Z","iopub.execute_input":"2021-10-09T11:58:57.415978Z","iopub.status.idle":"2021-10-09T11:58:57.430275Z","shell.execute_reply.started":"2021-10-09T11:58:57.415946Z","shell.execute_reply":"2021-10-09T11:58:57.429189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/distracted-drivers-v2/\"\n\ntrain_path = os.path.join(BASE_DIR,'train-revamped/train/')\nblurred_path = '/kaggle/input/blurred-drivers-test/new_images_train/'\ntest_path = os.path.join(BASE_DIR,'test-revamped/test/')\nval_path = os.path.join(BASE_DIR,'validation-revamped/validation/')\n\nprint('No. of images in training set = ',str(len(glob(train_path +'*/*'))))\nprint('No. of images in testing set = ',str(len(glob(test_path +'*/*'))))\nprint('No. of images in validation set = ',str(len(glob(val_path +'*/*'))))\nprint('No. of images in blurred set = ',str(len(glob(blurred_path +'*/*'))))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T11:59:01.486494Z","iopub.execute_input":"2021-10-09T11:59:01.486851Z","iopub.status.idle":"2021-10-09T11:59:08.446576Z","shell.execute_reply.started":"2021-10-09T11:59:01.486819Z","shell.execute_reply":"2021-10-09T11:59:08.445613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Using flow from directory to create dataframes containing test, train & validation dataset**","metadata":{}},{"cell_type":"code","source":"#Creating file dataframe to use flow from directory\ndef image_df(image_folder, extension): \n    filepaths = list(image_folder.glob(r'**/*.' + extension)) #search through all directories and search through\n    labels = list(map(lambda x : os.path.split(os.path.split(x)[0])[1], filepaths ))\n    filepaths = pd.Series(filepaths, name = 'filepaths').astype(str)\n    labels = pd.Series(labels, name = \"Label\")\n    images = pd.concat([filepaths, labels], axis = 1)\n    return images ","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:46:08.872642Z","iopub.execute_input":"2021-10-09T12:46:08.873036Z","iopub.status.idle":"2021-10-09T12:46:08.878422Z","shell.execute_reply.started":"2021-10-09T12:46:08.873003Z","shell.execute_reply":"2021-10-09T12:46:08.877579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train\nimage_folder_train = Path('/kaggle/input/blurred-drivers-test/new_images_train')\ntrain_df = image_df(image_folder_train, 'jpeg')\n#Test \nimage_folder_test = Path('/kaggle/input/distracteddriversrevampeddataset/test-revamped/test')\ntest_df = image_df(image_folder_test, 'jpg')\n#Validation\nimage_folder_val = Path('/kaggle/input/distracteddriversrevampeddataset/validation-revamped/validation')\nval_df = image_df(image_folder_val, 'jpg')","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:46:39.919786Z","iopub.execute_input":"2021-10-09T12:46:39.920132Z","iopub.status.idle":"2021-10-09T12:46:44.053978Z","shell.execute_reply.started":"2021-10-09T12:46:39.920102Z","shell.execute_reply":"2021-10-09T12:46:44.053039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:46:44.057549Z","iopub.execute_input":"2021-10-09T12:46:44.057835Z","iopub.status.idle":"2021-10-09T12:46:44.065340Z","shell.execute_reply.started":"2021-10-09T12:46:44.057808Z","shell.execute_reply":"2021-10-09T12:46:44.064050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shuffle dataframes so that algotithm will not learn from just one class at a time --> Not necessary cuz keras generator can shuffle for you\ntrain_df = train_df.sample(frac=1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:02.215806Z","iopub.execute_input":"2021-10-09T12:47:02.216296Z","iopub.status.idle":"2021-10-09T12:47:02.231207Z","shell.execute_reply.started":"2021-10-09T12:47:02.216253Z","shell.execute_reply":"2021-10-09T12:47:02.230424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **EDA**","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:10.688266Z","iopub.execute_input":"2021-10-09T12:47:10.688609Z","iopub.status.idle":"2021-10-09T12:47:10.707756Z","shell.execute_reply.started":"2021-10-09T12:47:10.688578Z","shell.execute_reply":"2021-10-09T12:47:10.706770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking shape of dataframe\nprint(f'Shape of the training dataframe is {train_df.shape}')\nprint(f'Shape of the test dataframe is {test_df.shape}')\nprint(f'Shape of the validation dataframe is {val_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:12.559549Z","iopub.execute_input":"2021-10-09T12:47:12.559908Z","iopub.status.idle":"2021-10-09T12:47:12.565269Z","shell.execute_reply.started":"2021-10-09T12:47:12.559875Z","shell.execute_reply":"2021-10-09T12:47:12.564431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", None, \"display.max_columns\", None)\ntrain_df_dist = train_df.groupby(by=['Label'], dropna=False).count()\nprint('Looking at distribution of classes for train data')\ndisplay(train_df_dist)\n#Class distribution is fairly uneven. Undersample to even it out","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:16.175990Z","iopub.execute_input":"2021-10-09T12:47:16.176314Z","iopub.status.idle":"2021-10-09T12:47:16.198043Z","shell.execute_reply.started":"2021-10-09T12:47:16.176282Z","shell.execute_reply":"2021-10-09T12:47:16.196935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", None, \"display.max_columns\", None)\ntest_df_dist = test_df.groupby(by=['Label'], dropna=False).count()\nprint('Looking at distribution of classes for test data')\ndisplay(test_df_dist)\n#Class distribution is fairly even","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:23.133336Z","iopub.execute_input":"2021-10-09T12:47:23.133662Z","iopub.status.idle":"2021-10-09T12:47:23.146178Z","shell.execute_reply.started":"2021-10-09T12:47:23.133632Z","shell.execute_reply":"2021-10-09T12:47:23.145340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", None, \"display.max_columns\", None)\nval_df_dist = val_df.groupby(by=['Label'], dropna=False).count()\nprint('Looking at distribution of classes for validation data')\ndisplay(val_df_dist)\n#Class distribution is fairly even...","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:25.225557Z","iopub.execute_input":"2021-10-09T12:47:25.225940Z","iopub.status.idle":"2021-10-09T12:47:25.241336Z","shell.execute_reply.started":"2021-10-09T12:47:25.225907Z","shell.execute_reply":"2021-10-09T12:47:25.240271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Blurring faces using pyimagesearch functions... + saving to working directory for use","metadata":{}},{"cell_type":"code","source":"# import the necessary packages\n!pip install imutils\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport argparse\nimport imutils\nimport cv2\nimport os\n\ndef plt_imshow(title, image):\n    # convert the image frame BGR to RGB color space and display it\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image)\n    plt.title(title)\n    plt.grid(False)\n    plt.show()\n    \ndef anonymize_face_simple(image, factor=3.0):\n    # automatically determine the size of the blurring kernel based\n    # on the spatial dimensions of the input image\n    (h, w) = image.shape[:2]\n    kW = int(w / factor)\n    kH = int(h / factor)\n\n    # ensure the width of the kernel is odd\n    if kW % 2 == 0:\n        kW -= 1\n\n    # ensure the height of the kernel is odd\n    if kH % 2 == 0:\n        kH -= 1\n\n    # apply a Gaussian blur to the input image using our computed\n    # kernel size\n    return cv2.GaussianBlur(image, (kW, kH), 0)\n\ndef anonymize_face_pixelate(image, blocks=3):\n    # divide the input image into NxN blocks\n    (h, w) = image.shape[:2]\n    xSteps = np.linspace(0, w, blocks + 1, dtype=\"int\")\n    ySteps = np.linspace(0, h, blocks + 1, dtype=\"int\")\n\n    # loop over the blocks in both the x and y direction\n    for i in range(1, len(ySteps)):\n        for j in range(1, len(xSteps)):\n            # compute the starting and ending (x, y)-coordinates\n            # for the current block\n            startX = xSteps[j - 1]\n            startY = ySteps[i - 1]\n            endX = xSteps[j]\n            endY = ySteps[i]\n\n            # extract the ROI using NumPy array slicing, compute the\n            # mean of the ROI, and then draw a rectangle with the\n            # mean RGB values over the ROI in the original image\n            roi = image[startY:endY, startX:endX]\n            (B, G, R) = [int(x) for x in cv2.mean(roi)[:3]]\n            cv2.rectangle(image, (startX, startY), (endX, endY),\n                (B, G, R), -1)\n\n    # return the pixelated blurred image\n    return image","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:01:23.789047Z","iopub.execute_input":"2021-10-09T10:01:23.789674Z","iopub.status.idle":"2021-10-09T10:01:33.698006Z","shell.execute_reply.started":"2021-10-09T10:01:23.789632Z","shell.execute_reply":"2021-10-09T10:01:33.697107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creating new directories to store all the images\nos.mkdir(\"/kaggle/working/new_images_test_only\")\nnames = ['c0', 'c1', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9', 'c2']\nfor i in names:\n    os.makedirs(f'/kaggle/working/new_images_train/{i}')","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:01:37.934972Z","iopub.execute_input":"2021-10-09T10:01:37.935331Z","iopub.status.idle":"2021-10-09T10:01:37.944255Z","shell.execute_reply.started":"2021-10-09T10:01:37.935297Z","shell.execute_reply":"2021-10-09T10:01:37.943128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Downloading pyimagesearch required functions\n!wget https://s3-us-west-2.amazonaws.com/static.pyimagesearch.com/opencv-face-blurring/opencv-face-blurring.zip\n!unzip -qq opencv-face-blurring.zip\n%cd opencv-face-blurring","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:01:42.382411Z","iopub.execute_input":"2021-10-09T10:01:42.382725Z","iopub.status.idle":"2021-10-09T10:01:46.469029Z","shell.execute_reply.started":"2021-10-09T10:01:42.382696Z","shell.execute_reply":"2021-10-09T10:01:46.468123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for j in range(len(train_df)):\n    img_path = train_df['filepaths'][j]\n    class_num = train_df['Label'][j]\n    #arguments for face blurring \n    args = {\n        \"image\": img_path,\n        \"face\": \"face_detector\",\n        \"method\": \"pixelated\",\n        \"blocks\": 20,\n        \"confidence\": 0.5\n\n    }\n#     try: \n    # load our serialized face detector model from disk\n    prototxtPath = os.path.sep.join([args[\"face\"], \"deploy.prototxt\"])\n    weightsPath = os.path.sep.join([args[\"face\"],\n        \"res10_300x300_ssd_iter_140000.caffemodel\"])\n    net = cv2.dnn.readNet(prototxtPath, weightsPath)\n\n    # load the input image from disk, clone it, and grab the image spatial\n    # dimensions\n    image = cv2.imread(img_path)\n    #orig = image.copy()\n    (h, w) = image.shape[:2]\n\n    # construct a blob from the image\n    blob = cv2.dnn.blobFromImage(image, 1.0, (300, 300),\n        (104.0, 177.0, 123.0))\n\n    # pass the blob through the network and obtain the face detections\n    # print(\"[INFO] computing face detections...\")\n    net.setInput(blob)\n    detections = net.forward()\n\n\n    # loop over the detections\n    for i in range(0, detections.shape[2]):\n        # extract the confidence (i.e., probability) associated with the\n        # detection\n        confidence = detections[0, 0, i, 2]\n\n        # filter out weak detections by ensuring the confidence is greater\n        # than the minimum confidence\n        if confidence > args[\"confidence\"]:\n            # compute the (x, y)-coordinates of the bounding box for the\n            # object\n            box = detections[0, 0, i, 3:7] * np.array([w, h, w, h])\n            (startX, startY, endX, endY) = box.astype(\"int\")\n\n            # extract the face ROI\n            face = image[startY:endY, startX:endX]\n\n            # check to see if we are applying the \"simple\" face blurring\n            # method\n#             if args[\"method\"] == \"simple\":\n#                 face = anonymize_face_simple(face, factor=3.0)\n\n#             # otherwise, we must be applying the \"pixelated\" face\n#             # anonymization method\n#             else:\n            face = anonymize_face_pixelate(face,\n                blocks=args[\"blocks\"])\n\n            # store the blurred face in the output image\n            image[startY:endY, startX:endX] = face\n            \n\n    im_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    Image.fromarray(im_rgb).save(f'/kaggle/working/new_images_train/{class_num}/img_{j}.jpeg')\n    \n#     im = Image.fromarray(image)\n#     im.save(f'/kaggle/working/new_images_train/{class_num}/img_{j}.jpeg')\n    print(f'img_{j}')\n#     except:\n#         pass\n","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:17:54.944165Z","iopub.execute_input":"2021-10-09T10:17:54.944511Z","iopub.status.idle":"2021-10-09T10:45:05.314086Z","shell.execute_reply.started":"2021-10-09T10:17:54.944482Z","shell.execute_reply":"2021-10-09T10:45:05.313186Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking number of images loaded\nBLURRED_DIR = \"/kaggle/working/new_images_train/\"\nprint('No. of images in blurred training set = ',str(len(glob(BLURRED_DIR +'*/*'))))","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:45:05.315838Z","iopub.execute_input":"2021-10-09T10:45:05.31631Z","iopub.status.idle":"2021-10-09T10:45:05.367955Z","shell.execute_reply.started":"2021-10-09T10:45:05.316239Z","shell.execute_reply":"2021-10-09T10:45:05.367023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:02:36.837803Z","iopub.execute_input":"2021-10-09T12:02:36.838190Z","iopub.status.idle":"2021-10-09T12:02:36.844708Z","shell.execute_reply.started":"2021-10-09T12:02:36.838157Z","shell.execute_reply":"2021-10-09T12:02:36.843656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Zipping images for download**","metadata":{}},{"cell_type":"code","source":"%%!\nzip -r new_images_train.zip /kaggle/working/new_images_train","metadata":{"execution":{"iopub.status.busy":"2021-10-09T10:51:58.443462Z","iopub.execute_input":"2021-10-09T10:51:58.443778Z","iopub.status.idle":"2021-10-09T10:52:16.575527Z","shell.execute_reply.started":"2021-10-09T10:51:58.44375Z","shell.execute_reply":"2021-10-09T10:52:16.574526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Undersampling classes from blurred images for even class distribution","metadata":{}},{"cell_type":"code","source":"#for training set \ntrain_df_undersampled_list = []\nfor category in train_df['Label'].unique(): \n    category_slice = train_df.query(f'Label ==@category')\n    train_df_undersampled_list.append(category_slice.sample(1060, random_state = 1))\ntrain_df_undersampled = pd.concat(train_df_undersampled_list, axis = 0).sample(frac = 1.0, random_state = 1).reset_index(drop=True)\ntrain_df_undersampled.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:47:59.349609Z","iopub.execute_input":"2021-10-09T12:47:59.349966Z","iopub.status.idle":"2021-10-09T12:47:59.414460Z","shell.execute_reply.started":"2021-10-09T12:47:59.349932Z","shell.execute_reply":"2021-10-09T12:47:59.413711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option(\"display.max_rows\", None, \"display.max_columns\", None)\ntrain_df_undersampled_dist = train_df_undersampled.groupby(by=['Label'], dropna=False).count()\nprint('Looking at distribution of classes for training data')\ndisplay(train_df_undersampled_dist)\n#Class distribution is now even","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:48:02.128102Z","iopub.execute_input":"2021-10-09T12:48:02.128428Z","iopub.status.idle":"2021-10-09T12:48:02.146850Z","shell.execute_reply.started":"2021-10-09T12:48:02.128398Z","shell.execute_reply":"2021-10-09T12:48:02.145812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating Generators\n- Loads in images from dataframe to suit transfer learning model requirements","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ngenerator = tf.keras.preprocessing.image.ImageDataGenerator(\n    preprocessing_function = tf.keras.applications.mobilenet_v2.preprocess_input \n)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:48:05.936499Z","iopub.execute_input":"2021-10-09T12:48:05.936860Z","iopub.status.idle":"2021-10-09T12:48:05.942216Z","shell.execute_reply.started":"2021-10-09T12:48:05.936822Z","shell.execute_reply":"2021-10-09T12:48:05.940973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = generator.flow_from_dataframe(\n    dataframe = train_df_undersampled,\n    x_col = 'filepaths', \n    y_col = 'Label',\n    target_size = (224, 224), #Target image dimensions,,,\n    color_mode = 'rgb', \n    class_mode = 'categorical',\n    batch_size = 32, \n    shuffle = True, \n    seed = 1\n)\n\ntest_images = generator.flow_from_dataframe(\n    dataframe = test_df,\n    x_col = 'filepaths', \n    y_col = 'Label',\n    target_size = (224, 224), #Default for \n    color_mode = 'rgb', \n    class_mode = 'categorical',\n    batch_size = 32, \n    shuffle = False, \n    seed = 1\n)\n\nval_images = generator.flow_from_dataframe(\n    dataframe = val_df,\n    x_col = 'filepaths', \n    y_col = 'Label',\n    target_size = (224, 224), #Default for \n    color_mode = 'rgb', \n    class_mode = 'categorical',\n    batch_size = 32, \n    shuffle = True, \n    seed = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:48:13.575724Z","iopub.execute_input":"2021-10-09T12:48:13.576136Z","iopub.status.idle":"2021-10-09T12:48:25.682168Z","shell.execute_reply.started":"2021-10-09T12:48:13.576104Z","shell.execute_reply":"2021-10-09T12:48:25.681220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Modelling using MobileNet\n- Adding Dropout layer. Used atfer conv2d and pooling layers, and dense layer","metadata":{}},{"cell_type":"code","source":"#Pre-trained model used to extract features from the images \n#Removing top lets us use it for our own classification purposes\npretrained_model = tf.keras.applications.MobileNetV2(\n    input_shape = (224,224,3), \n    include_top = False, #include top gives a classification layer for 1000 classes :O\n    weights = 'imagenet', \n    pooling = 'max' #Ensures that output of pre-trained model is 1-dimensional, output is singlet vector\n    \n)\npretrained_model.trainable = False #So weights of imagenet will not be changed ","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:49:57.013151Z","iopub.execute_input":"2021-10-09T12:49:57.013473Z","iopub.status.idle":"2021-10-09T12:50:00.205150Z","shell.execute_reply.started":"2021-10-09T12:49:57.013444Z","shell.execute_reply":"2021-10-09T12:50:00.204214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers, models\ninputs = pretrained_model.input\n\nx = layers.BatchNormalization(axis=1)\nx = layers.Dense(512, activation = 'relu')(pretrained_model.output) #128 neurons\nx = layers.Dense(256, activation = 'relu')(x)\nx = layers.Dense(128, activation = 'relu')(x)\nx = layers.Dropout(0.2, seed = 1)(x)\n\n#final layer\noutputs = layers.Dense(10, activation = 'softmax')(x)\n#softmax to make all pr values sum to 1\n\nmodel = tf.keras.Model(inputs, outputs)\nprint(model.summary())\n","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:50:03.967234Z","iopub.execute_input":"2021-10-09T12:50:03.967566Z","iopub.status.idle":"2021-10-09T12:50:04.087571Z","shell.execute_reply.started":"2021-10-09T12:50:03.967536Z","shell.execute_reply":"2021-10-09T12:50:04.085476Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Training process!","metadata":{}},{"cell_type":"code","source":"model.compile(\n    optimizer = 'adam', \n    loss = 'categorical_crossentropy', #data generator has encoded the classes already, so in vector form\n    metrics = ['accuracy']\n)\n\ncallback = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', patience = 10,\n                                           restore_best_weights = True)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:50:12.086251Z","iopub.execute_input":"2021-10-09T12:50:12.086574Z","iopub.status.idle":"2021-10-09T12:50:12.105487Z","shell.execute_reply.started":"2021-10-09T12:50:12.086543Z","shell.execute_reply":"2021-10-09T12:50:12.104528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#change was from avg to max pooling \nhistory = model.fit(\n    train_images,\n    validation_data = val_images, \n    epochs = 100,\n    callbacks = callback\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T12:50:13.621602Z","iopub.execute_input":"2021-10-09T12:50:13.621992Z","iopub.status.idle":"2021-10-09T13:11:01.748454Z","shell.execute_reply.started":"2021-10-09T12:50:13.621958Z","shell.execute_reply":"2021-10-09T13:11:01.747440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"Mobilenetv2_trial2_blurred_undersampled.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-10-08T17:24:39.418762Z","iopub.status.idle":"2021-10-08T17:24:39.419344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Results","metadata":{}},{"cell_type":"code","source":"results = model.evaluate(test_images, verbose = 0)\nprint(f\"Test Accuracy: {results[1]*100:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-09T14:00:43.228548Z","iopub.execute_input":"2021-10-09T14:00:43.228910Z","iopub.status.idle":"2021-10-09T14:01:31.449509Z","shell.execute_reply.started":"2021-10-09T14:00:43.228877Z","shell.execute_reply":"2021-10-09T14:01:31.448551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report \nfrom sklearn.metrics import confusion_matrix \n\npredictions = np.argmax(model.predict(test_images), axis =1 ) #Gets index of class with highest predicted pr\n\n#confusion matrix \ncm = confusion_matrix(test_images.labels, predictions)\nclr = classification_report(test_images.labels, predictions, target_names = test_images.class_indices)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-09T14:01:31.451249Z","iopub.execute_input":"2021-10-09T14:01:31.451938Z","iopub.status.idle":"2021-10-09T14:01:54.713136Z","shell.execute_reply.started":"2021-10-09T14:01:31.451894Z","shell.execute_reply":"2021-10-09T14:01:54.712159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting confusion matrix\nplt.figure(figsize = (14,14))\nsns.heatmap(cm, annot = True, fmt = 'g', vmin = 0 , cmap = 'Blues')\nplt.xticks(ticks = np.arange(10) + 0.5, labels  = test_images.class_indices, rotation = 90) #label names spaced in the middle \nplt.yticks(ticks = np.arange(10) + 0.5, labels  = test_images.class_indices, rotation = 0)\nplt.xlabel(\"Predicted classes\")\nplt.ylabel(\"Actual classes\")\nplt.title('Confusion Matrix for MobileNet v2')\n","metadata":{"execution":{"iopub.status.busy":"2021-10-09T14:01:54.714821Z","iopub.execute_input":"2021-10-09T14:01:54.715197Z","iopub.status.idle":"2021-10-09T14:01:55.514355Z","shell.execute_reply.started":"2021-10-09T14:01:54.715161Z","shell.execute_reply":"2021-10-09T14:01:55.513539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Classfication Report for MobileNetv2\")\nprint(clr)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T14:01:55.515731Z","iopub.execute_input":"2021-10-09T14:01:55.516109Z","iopub.status.idle":"2021-10-09T14:01:55.521489Z","shell.execute_reply.started":"2021-10-09T14:01:55.516070Z","shell.execute_reply":"2021-10-09T14:01:55.520238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(2, 1, figsize=(12, 12))\nax1.plot(history.history['loss'], color='b', label=\"Training loss\")\nax1.plot(history.history['val_loss'], color='r', label=\"validation loss\")\n#ax1.set_xticks(np.arange(1, 400, 1))\n#ax1.set_yticks(np.arange(0, 1, 0.1))\n\nax2.plot(history.history['accuracy'], color='b', label=\"Training accuracy\")\nax2.plot(history.history['val_accuracy'], color='r',label=\"Validation accuracy\")\n#ax2.set_xticks(np.arange(1, 400, 1))\n\nlegend = plt.legend(loc='best', shadow=True)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2021-10-09T14:03:57.263274Z","iopub.execute_input":"2021-10-09T14:03:57.263621Z","iopub.status.idle":"2021-10-09T14:03:57.722175Z","shell.execute_reply.started":"2021-10-09T14:03:57.263590Z","shell.execute_reply":"2021-10-09T14:03:57.721390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Looking at feature map filters","metadata":{}},{"cell_type":"code","source":"img_path = train_df['filepaths'][1]\nimg_path","metadata":{"execution":{"iopub.status.busy":"2021-10-08T17:24:39.431315Z","iopub.status.idle":"2021-10-08T17:24:39.431932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator, img_to_array, load_img\n\nimg_path = train_df['filepaths'][1]\n# Define a new Model, Input= image \n# Output= intermediate representations for all layers in the  \n# previous model after the first.\nsuccessive_outputs = [layer.output for layer in model.layers[1:]]\nvisualization_model = tf.keras.models.Model(inputs = model.input, outputs = successive_outputs)\n#Load the input image\nimg = load_img(img_path, target_size=(224, 224))\nx   = img_to_array(img)                           \nx   = x.reshape((1,) + x.shape)\n# Rescale by 1/255\nx /= 255.0\n\n# Using visualisation model to predict on image...\nsuccessive_feature_maps = visualization_model.predict(x)\n# Retrieve are the names of the layers, so can have them as part of our plot\nlayer_names2 = []\nlayer_names = [layer.name for layer in model.layers]\nprint(layer_names)\nprint(layer_names2)\nfor layer in layer_names:\n    if 'Conv' in layer or 'conv' in layer:\n        layer_names2.append(layer)\n# \nfor layer_name, feature_map in zip(layer_names2, successive_feature_maps):\n  #print(feature_map.shape)\n  if len(feature_map.shape) == 4:\n    \n    # Plot Feature maps for the conv / maxpool layers, not the fully-connected layers\n   \n    n_features = feature_map.shape[-1]  # number of features in the feature map\n    size       = feature_map.shape[ 1]  # feature map shape (1, size, size, n_features)\n    \n    # We will tile our images in this matrix\n    display_grid = np.zeros((int(size), size * n_features))\n    \n    # Postprocess the feature to be visually palatable\n    for i in range(n_features):\n      x  = feature_map[0, :, :, i]\n      x -= x.mean()\n      x /= x.std ()\n      x *=  64\n      x += 128\n      x  = np.clip(x, 0, 255).astype('uint8')\n      # Tile each filter into a horizontal grid\n      display_grid[:, i * size : (i + 1) * size] = x\n# Display the grid\n    scale = 20. / n_features\n    plt.figure( figsize=(scale * n_features, scale) )\n    plt.title ( layer_name )\n    plt.grid  ( False )\n    plt.imshow( display_grid, aspect='auto', cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2021-10-08T17:24:39.433334Z","iopub.status.idle":"2021-10-08T17:24:39.433944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.mobilenet_v2.MobileNetV2 import preprocess_input\nfrom tensorflow.keras.preprocessing.image import load_img,img_to_array\nfrom keras.models import Model\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nimg_path = train_df['filepaths'][1]\n\nlayer_dict = dict([(layer.name, layer) for layer in model.layers])\n\nlayer_name = 'block1_conv2'\n\nmodel = Model(inputs=model.inputs, outputs=layer_dict[layer_name].output)\n\n# Perpare the image\nimage = load_img(img_path, target_size=(224, 224))\nimage = img_to_array(image)\nimage = np.expand_dims(image, axis=0)\nimage = preprocess_input(image)\n\n# Apply the model to the image\nfeature_maps = model.predict(image)\n\nsquare = 4\nindex = 1\nfor i in range(square):\n\tfor i in range(square):\n        \n\t\tax = plt.subplot(square, square, index)\n\t\tax.set_xticks([])\n\t\tax.set_yticks([])\n\n\t\tplt.imshow(feature_maps[0, :, :, index-1], cmap='viridis')\n\t\tindex += 1\n        \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-08T17:24:39.437092Z","iopub.status.idle":"2021-10-08T17:24:39.437698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Grad Cam - checking class activation maps for each class...","metadata":{}},{"cell_type":"code","source":"\n#%cd /kaggle","metadata":{"execution":{"iopub.status.busy":"2021-10-09T13:19:46.109097Z","iopub.execute_input":"2021-10-09T13:19:46.109479Z","iopub.status.idle":"2021-10-09T13:19:55.361796Z","shell.execute_reply.started":"2021-10-09T13:19:46.109443Z","shell.execute_reply":"2021-10-09T13:19:55.360811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = ['c0', 'c1', 'c2', 'c3', 'c4', 'c5', 'c6', 'c7', 'c8', 'c9']\nlist_of_img_paths = []\nfor name in names: \n    temp_df = train_df[train_df['Label'] ==  name ]\n    img_path = temp_df['filepaths'].iloc[0] \n    list_of_img_paths.append(img_path) #getting an img_path from each class for grad-cam\nprint(list_of_img_paths)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T13:17:43.668510Z","iopub.execute_input":"2021-10-09T13:17:43.668903Z","iopub.status.idle":"2021-10-09T13:17:43.702073Z","shell.execute_reply.started":"2021-10-09T13:17:43.668869Z","shell.execute_reply":"2021-10-09T13:17:43.701093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget https://s3-us-west-2.amazonaws.com/static.pyimagesearch.com/keras-gradcam/keras-grad-cam.zip\n!unzip -qq keras-grad-cam.zip\n%cd keras-grad-cam","metadata":{"execution":{"iopub.status.busy":"2021-10-09T13:19:33.002325Z","iopub.execute_input":"2021-10-09T13:19:33.002664Z","iopub.status.idle":"2021-10-09T13:19:35.262681Z","shell.execute_reply.started":"2021-10-09T13:19:33.002632Z","shell.execute_reply":"2021-10-09T13:19:35.261798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import the necessary packages\n# !pip install imutils\nfrom pyimagesearch.gradcam import GradCAM\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.applications import mobilenet_v2\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.preprocessing.image import load_img\nfrom tensorflow.keras.applications import imagenet_utils\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport argparse\nimport imutils\nimport cv2\n\ncount = 0\n\n# for img_path in list_of_img_paths: \n#     # load the original image from disk (in OpenCV format) and then\n#     # resize the image to its target dimensions\\\ndef cam(path, label, model): \n    orig = cv2.imread(path)\n    image = cv2.imread(path)\n    image = cv2.resize(orig, (224, 224))\n    image = image.astype('float32') / 255\n    image = np.expand_dims(image, axis=0)\n    \n\n    # use the network to make predictions on the input image and find - delete\n    # the class label index with the largest corresponding probability\n    preds = model.predict(image)\n    i = np.argmax(preds[0])\n\n    # initialize our gradient class activation map and build the heatmap\n    cam = GradCAM(model, i)\n    heatmap = cam.compute_heatmap(image)\n\n    # resize the resulting heatmap to the original input image dimensions\n    # and then overlay heatmap on top of the image\n    heatmap = cv2.resize(heatmap, (orig.shape[1], orig.shape[0]))\n    image = cv2.imread(path)\n    image = cv2.resize(image, (224, 224))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    (heatmap, output) = cam.overlay_heatmap(heatmap, orig, alpha=0.5)\n    \n    # Plotting figure\n    fig, ax = plt.subplots(1, 3, constrained_layout=True)\n#     ax[0].set_title(f'GRADCAM for class {label}')\n    fig.suptitle(f'GRADCAM for class {label}', fontsize=12)\n    ax[0].imshow(heatmap)\n    ax[1].imshow(image)\n    ax[2].imshow(output)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T13:59:13.308781Z","iopub.execute_input":"2021-10-09T13:59:13.309241Z","iopub.status.idle":"2021-10-09T13:59:13.321094Z","shell.execute_reply.started":"2021-10-09T13:59:13.309199Z","shell.execute_reply":"2021-10-09T13:59:13.319941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,10):\n    cam(list_of_img_paths[i], f'c{i}', model)\n\n#cam(list_of_img_paths[0], 'c0', model)","metadata":{"execution":{"iopub.status.busy":"2021-10-09T13:59:14.884240Z","iopub.execute_input":"2021-10-09T13:59:14.884612Z","iopub.status.idle":"2021-10-09T13:59:22.918515Z","shell.execute_reply.started":"2021-10-09T13:59:14.884581Z","shell.execute_reply":"2021-10-09T13:59:22.917526Z"},"trusted":true},"execution_count":null,"outputs":[]}]}