{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # \nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-05T10:48:11.125303Z","iopub.execute_input":"2023-02-05T10:48:11.126088Z","iopub.status.idle":"2023-02-05T10:48:11.140344Z","shell.execute_reply.started":"2023-02-05T10:48:11.126002Z","shell.execute_reply":"2023-02-05T10:48:11.139171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport tensorflow as tf\n\nfrom sklearn.model_selection import train_test_split\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split\nfrom collections import Counter\n\nimport cv2\nfrom concurrent import futures\nimport threading\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nfrom sklearn.preprocessing import LabelEncoder\n\nimport datetime","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:11.146873Z","iopub.execute_input":"2023-02-05T10:48:11.147184Z","iopub.status.idle":"2023-02-05T10:48:15.044226Z","shell.execute_reply.started":"2023-02-05T10:48:11.147151Z","shell.execute_reply":"2023-02-05T10:48:15.043106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data import","metadata":{}},{"cell_type":"code","source":"#getting the total number of images in the training set\n\nbase_dir = '../input'\n\ntrain_dir = os.path.join(base_dir,'train', 'train')\n\ntype1_dir = os.path.join(base_dir,'Type_1')\ntype2_dir = os.path.join(base_dir,'Type_2')\ntype3_dir = os.path.join(base_dir,'Type_3')\n\ntype1_files = glob.glob(type1_dir+'/*.jpg')\ntype2_files = glob.glob(type2_dir+'/*.jpg')\ntype3_files = glob.glob(type3_dir+'/*.jpg')\n\nadded_type1_files  =  glob.glob(os.path.join(base_dir, \"additional_Type_1_v2\", \"Type_1\")+'/*.jpg')\nadded_type2_files  =  glob.glob(os.path.join(base_dir, \"additional_Type_2_v2\", \"Type_2\")+'/*.jpg')\nadded_type3_files  =  glob.glob(os.path.join(base_dir, \"additional_Type_3_v2\", \"Type_3\")+'/*.jpg')\n\ntype1_files = type1_files + added_type1_files\ntype2_files = type2_files + added_type2_files\ntype3_files = type3_files + added_type3_files\n\n\nprint('Number of images in a train set of type 1: ', len(type1_files))\nprint('Number of images in a train set of type 2: ', len(type2_files))\nprint('Number of images in a train set of type 3: ', len(type3_files))\nprint('Total number of images in a train set: ', sum([len(type1_files), len(type2_files), len(type3_files)]))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.047603Z","iopub.execute_input":"2023-02-05T10:48:15.048469Z","iopub.status.idle":"2023-02-05T10:48:15.081158Z","shell.execute_reply.started":"2023-02-05T10:48:15.048437Z","shell.execute_reply":"2023-02-05T10:48:15.080155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Building a dataframe mapping images and Cancer type\n\nfiles_df = pd.DataFrame({\n    'filename': type1_files + type2_files + type3_files,\n    'label': ['Type_1'] * len(type1_files) + ['Type_2'] * len(type2_files) + ['Type_3'] * len(type3_files)\n})\n\nfiles_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.084193Z","iopub.execute_input":"2023-02-05T10:48:15.084525Z","iopub.status.idle":"2023-02-05T10:48:15.106838Z","shell.execute_reply.started":"2023-02-05T10:48:15.084496Z","shell.execute_reply":"2023-02-05T10:48:15.105657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shuffle data\n\nrandom_state = 42\n\nfiles_df = files_df.sample(frac=1, random_state=random_state)\n\nfiles_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.108775Z","iopub.execute_input":"2023-02-05T10:48:15.109191Z","iopub.status.idle":"2023-02-05T10:48:15.124760Z","shell.execute_reply.started":"2023-02-05T10:48:15.109152Z","shell.execute_reply":"2023-02-05T10:48:15.123655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data exploration","metadata":{}},{"cell_type":"code","source":"files_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.126617Z","iopub.execute_input":"2023-02-05T10:48:15.127006Z","iopub.status.idle":"2023-02-05T10:48:15.148228Z","shell.execute_reply.started":"2023-02-05T10:48:15.126959Z","shell.execute_reply":"2023-02-05T10:48:15.147256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check for duplicates\nlen(files_df[files_df.duplicated()])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.149903Z","iopub.execute_input":"2023-02-05T10:48:15.150277Z","iopub.status.idle":"2023-02-05T10:48:15.161316Z","shell.execute_reply.started":"2023-02-05T10:48:15.150239Z","shell.execute_reply":"2023-02-05T10:48:15.160256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get count of each type \ntype_count = pd.DataFrame(files_df['label'].value_counts()).rename(columns= {'label': 'Num_Values'})\ntype_count","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.163037Z","iopub.execute_input":"2023-02-05T10:48:15.163696Z","iopub.status.idle":"2023-02-05T10:48:15.177097Z","shell.execute_reply.started":"2023-02-05T10:48:15.163660Z","shell.execute_reply":"2023-02-05T10:48:15.176201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Display barplot of type count\nplt.figure(figsize = (15, 6))\nsns.barplot(x= type_count['Num_Values'], y= type_count.index.to_list())\nplt.title('Cervical Cancer Type Distribution')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.178458Z","iopub.execute_input":"2023-02-05T10:48:15.178754Z","iopub.status.idle":"2023-02-05T10:48:15.397561Z","shell.execute_reply.started":"2023-02-05T10:48:15.178722Z","shell.execute_reply":"2023-02-05T10:48:15.396533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Display sample images of types\nfor label in ('Type_1', 'Type_2', 'Type_3'):\n    filepaths = files_df[files_df['label']==label]['filename'].values[:5]\n    fig = plt.figure(figsize= (15, 6))\n    for i, path in enumerate(filepaths):\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        img = cv2.resize(img, (224, 224))\n        fig.add_subplot(1, 5, i+1)\n        plt.imshow(img)\n        plt.subplots_adjust(hspace=0.5)\n        plt.axis(False)\n        plt.title(label)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:15.399218Z","iopub.execute_input":"2023-02-05T10:48:15.399575Z","iopub.status.idle":"2023-02-05T10:48:19.691573Z","shell.execute_reply.started":"2023-02-05T10:48:15.399537Z","shell.execute_reply":"2023-02-05T10:48:19.690522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data propocessing","metadata":{}},{"cell_type":"code","source":"#Split training,val and test set : 70:15:15\n\ntrain_files, test_files, train_labels, test_labels = train_test_split(files_df['filename'].values,\n                                                                      files_df['label'].values, \n                                                                      test_size=0.3, \n                                                                      random_state=random_state)\n\ntest_files, val_files, test_labels, val_labels = train_test_split(test_files,\n                                                                  test_labels, \n                                                                  test_size=0.5, \n                                                                  random_state=random_state)\n\n\nprint('Number of images in train set: ', train_files.shape)\nprint('Number of images in validation set: ', val_files.shape)\nprint('Number of images in test set: ', test_files.shape, '\\n')\n\nprint('Train:', Counter(train_labels), '\\nVal:', Counter(val_labels), '\\nTest:', Counter(test_labels))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:19.694038Z","iopub.execute_input":"2023-02-05T10:48:19.694704Z","iopub.status.idle":"2023-02-05T10:48:19.707815Z","shell.execute_reply.started":"2023-02-05T10:48:19.694665Z","shell.execute_reply":"2023-02-05T10:48:19.706835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_images(files, labels):\n    features = []\n    correct_labels = []\n    bad_images = 0\n    \n    for i in range(len(files)):\n        try:\n            img = cv2.imread(files[i])\n            resized_img = cv2.resize(img, (160, 160))\n            \n            features.append(np.array(resized_img))\n            correct_labels.append(labels[i])\n                   \n        except Exception as e:\n            bad_images+=1\n            print('Encoutered bad image')\n    print('Bad images ecountered:', bad_images)\n    return np.array(features), np.array(correct_labels)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:19.709660Z","iopub.execute_input":"2023-02-05T10:48:19.710057Z","iopub.status.idle":"2023-02-05T10:48:19.717810Z","shell.execute_reply.started":"2023-02-05T10:48:19.710021Z","shell.execute_reply":"2023-02-05T10:48:19.716520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load training and evaluation data\ntrain_features, train_labels = load_images(train_files, train_labels)\nprint('Train images loaded')\nval_features, val_labels = load_images(val_files, val_labels)\nprint('Validation images loaded')\ntest_features, test_labels = load_images(test_files, test_labels)\nprint('test images loaded')","metadata":{"execution":{"iopub.status.busy":"2023-02-05T10:48:19.723609Z","iopub.execute_input":"2023-02-05T10:48:19.723951Z","iopub.status.idle":"2023-02-05T11:11:50.219834Z","shell.execute_reply.started":"2023-02-05T10:48:19.723921Z","shell.execute_reply":"2023-02-05T11:11:50.218665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check lengths of training and evaluation  sets\nlen(train_features), len(train_labels), len(val_features), len(val_labels), len(test_features), len(test_labels) ","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:50.221431Z","iopub.execute_input":"2023-02-05T11:11:50.221761Z","iopub.status.idle":"2023-02-05T11:11:50.233657Z","shell.execute_reply.started":"2023-02-05T11:11:50.221732Z","shell.execute_reply":"2023-02-05T11:11:50.232637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 32\nNUM_CLASSES = 3\nEPOCHS = 20\nINPUT_SHAPE = (160, 160, 3)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:50.235400Z","iopub.execute_input":"2023-02-05T11:11:50.236036Z","iopub.status.idle":"2023-02-05T11:11:50.242068Z","shell.execute_reply.started":"2023-02-05T11:11:50.235997Z","shell.execute_reply":"2023-02-05T11:11:50.241050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#encode train+val sets text categories with labels\nle = LabelEncoder()\nle.fit(train_labels)\n\ntrain_labels_enc = le.transform(train_labels)\nval_labels_enc = le.transform(val_labels)\n\ntrain_labels_1hotenc = tf.keras.utils.to_categorical(train_labels_enc, num_classes=NUM_CLASSES)\nval_labels_1hotenc = tf.keras.utils.to_categorical(val_labels_enc, num_classes=NUM_CLASSES)\n\nprint(train_labels[:6], train_labels_enc[:6])\nprint(train_labels[:6], train_labels_1hotenc[:6])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:50.243549Z","iopub.execute_input":"2023-02-05T11:11:50.244177Z","iopub.status.idle":"2023-02-05T11:11:50.503258Z","shell.execute_reply.started":"2023-02-05T11:11:50.244140Z","shell.execute_reply":"2023-02-05T11:11:50.501925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nle = LabelEncoder()\nle.fit(test_labels)\n\ntest_labels_enc = le.transform(test_labels)\n\ntest_labels_1hotenc = tf.keras.utils.to_categorical(test_labels_enc, num_classes=NUM_CLASSES)\n\n\nprint(test_labels[:6], test_labels_enc[:6])\nprint(test_labels[:6], test_labels_1hotenc[:6])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:50.507493Z","iopub.execute_input":"2023-02-05T11:11:50.507885Z","iopub.status.idle":"2023-02-05T11:11:50.520206Z","shell.execute_reply.started":"2023-02-05T11:11:50.507849Z","shell.execute_reply":"2023-02-05T11:11:50.519067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data augmentation","metadata":{}},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n  tf.keras.layers.RandomFlip('horizontal'),\n  tf.keras.layers.RandomRotation(0.2),\n])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:50.523103Z","iopub.execute_input":"2023-02-05T11:11:50.523636Z","iopub.status.idle":"2023-02-05T11:11:52.199532Z","shell.execute_reply.started":"2023-02-05T11:11:50.523572Z","shell.execute_reply":"2023-02-05T11:11:52.197513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nfirst_image = train_features[0]\nfor i in range(9):\n    ax = plt.subplot(3, 3, i + 1)\n    augmented_image = data_augmentation(tf.expand_dims(first_image, 0))\n    plt.imshow(augmented_image[0] / 255)\n    plt.axis('off')\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:52.203144Z","iopub.execute_input":"2023-02-05T11:11:52.203458Z","iopub.status.idle":"2023-02-05T11:11:52.994942Z","shell.execute_reply.started":"2023-02-05T11:11:52.203428Z","shell.execute_reply":"2023-02-05T11:11:52.993683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MobileNet V2 pre-trained","metadata":{}},{"cell_type":"code","source":"base_model = tf.keras.applications.MobileNetV2(include_top=False, \n                                               weights='imagenet', \n                                               input_shape=INPUT_SHAPE)\n\nbase_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:52.996752Z","iopub.execute_input":"2023-02-05T11:11:52.997304Z","iopub.status.idle":"2023-02-05T11:11:54.230942Z","shell.execute_reply.started":"2023-02-05T11:11:52.997235Z","shell.execute_reply":"2023-02-05T11:11:54.229864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(160, 160, 3))\nx = data_augmentation(inputs)\n\n#rescale pixel values\npreprocess_input = tf.keras.applications.mobilenet_v2.preprocess_input\nx = preprocess_input(x)\n\nx = base_model(x, training=False)\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\nx = tf.keras.layers.Dropout(0.2)(x)\noutputs = tf.keras.layers.Dense(3, activation='softmax')(x)\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:54.232492Z","iopub.execute_input":"2023-02-05T11:11:54.232874Z","iopub.status.idle":"2023-02-05T11:11:54.726142Z","shell.execute_reply.started":"2023-02-05T11:11:54.232836Z","shell.execute_reply":"2023-02-05T11:11:54.725123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#compile the model\nlearning_rate=1e-4\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate),\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:54.727822Z","iopub.execute_input":"2023-02-05T11:11:54.728423Z","iopub.status.idle":"2023-02-05T11:11:54.743785Z","shell.execute_reply.started":"2023-02-05T11:11:54.728383Z","shell.execute_reply":"2023-02-05T11:11:54.742716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:54.745652Z","iopub.execute_input":"2023-02-05T11:11:54.746039Z","iopub.status.idle":"2023-02-05T11:11:54.762190Z","shell.execute_reply.started":"2023-02-05T11:11:54.745999Z","shell.execute_reply":"2023-02-05T11:11:54.761271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x=train_features, \n                    y=train_labels_1hotenc, \n                    batch_size=BATCH_SIZE,\n                    epochs=EPOCHS, \n                    validation_data=(val_features, val_labels_1hotenc),\n                    verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:11:54.763743Z","iopub.execute_input":"2023-02-05T11:11:54.764126Z","iopub.status.idle":"2023-02-05T11:13:10.303074Z","shell.execute_reply.started":"2023-02-05T11:11:54.764089Z","shell.execute_reply":"2023-02-05T11:13:10.302025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def learning_performance_chart(title='Learning Perfomance', history=history):\n    #plots a chart showing the change in accuracy and loss function over epochs\n    f, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 4))\n    t = f.suptitle(title, fontsize=12)\n    f.subplots_adjust(top=0.85, wspace=0.3)\n\n    max_epoch = len(history.history['accuracy'])+1\n    epoch_list = list(range(1,max_epoch))\n    ax1.plot(epoch_list, history.history['accuracy'], label='Train Accuracy')\n    ax1.plot(epoch_list, history.history['val_accuracy'], label='Validation Accuracy')\n    ax1.set_xticks(np.arange(1, max_epoch, 5))\n    ax1.set_ylabel('Accuracy Value')\n    ax1.set_xlabel('Epoch')\n    ax1.set_title('Accuracy')\n    l1 = ax1.legend(loc=\"best\")\n\n    ax2.plot(epoch_list, history.history['loss'], label='Train Loss')\n    ax2.plot(epoch_list, history.history['val_loss'], label='Validation Loss')\n    ax2.set_xticks(np.arange(1, max_epoch, 5))\n    ax2.set_ylabel('Loss Value')\n    ax2.set_xlabel('Epoch')\n    ax2.set_title('Loss')\n    l2 = ax2.legend(loc=\"best\")\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:10.305201Z","iopub.execute_input":"2023-02-05T11:13:10.305578Z","iopub.status.idle":"2023-02-05T11:13:10.316018Z","shell.execute_reply.started":"2023-02-05T11:13:10.305540Z","shell.execute_reply":"2023-02-05T11:13:10.314139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_performance_chart(title='MobileNetV2 baseline performance', history=history)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:10.317512Z","iopub.execute_input":"2023-02-05T11:13:10.318539Z","iopub.status.idle":"2023-02-05T11:13:10.630573Z","shell.execute_reply.started":"2023-02-05T11:13:10.318501Z","shell.execute_reply":"2023-02-05T11:13:10.629527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('MobileNetV2 performance on the test set:')\nresults = model.evaluate(test_features,test_labels_1hotenc, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:10.632142Z","iopub.execute_input":"2023-02-05T11:13:10.633241Z","iopub.status.idle":"2023-02-05T11:13:12.125698Z","shell.execute_reply.started":"2023-02-05T11:13:10.633198Z","shell.execute_reply":"2023-02-05T11:13:12.124650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#keep track of the models performance\nperformance_df = pd.DataFrame(columns=['model','test set accuracy'])\n\ndef record_performance(df, model_name, test_accuracy):\n    return df.append(\n        {'model':model_name, 'test set accuracy':test_accuracy},\n        ignore_index=True)\n\nperformance_df = record_performance(performance_df, 'MobileNetV2 base', results[1])\nperformance_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.129649Z","iopub.execute_input":"2023-02-05T11:13:12.130551Z","iopub.status.idle":"2023-02-05T11:13:12.149711Z","shell.execute_reply.started":"2023-02-05T11:13:12.130508Z","shell.execute_reply":"2023-02-05T11:13:12.148601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fine tuning","metadata":{}},{"cell_type":"code","source":"tuning_model_1 = base_model\ntuning_model_1.trainable = True","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.151340Z","iopub.execute_input":"2023-02-05T11:13:12.152263Z","iopub.status.idle":"2023-02-05T11:13:12.164028Z","shell.execute_reply.started":"2023-02-05T11:13:12.152220Z","shell.execute_reply":"2023-02-05T11:13:12.162908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's take a look to see how many layers are in the base model\nprint(\"Number of layers in the base model: \", len(tuning_model_1.layers))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.165908Z","iopub.execute_input":"2023-02-05T11:13:12.166350Z","iopub.status.idle":"2023-02-05T11:13:12.175952Z","shell.execute_reply.started":"2023-02-05T11:13:12.166312Z","shell.execute_reply":"2023-02-05T11:13:12.174735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine tune at 120 layer","metadata":{}},{"cell_type":"code","source":"# Fine-tune from this layer onwards\nfine_tune_at = 120\n\n# Freeze all the layers before the `fine_tune_at` layer\nfor layer in tuning_model_1.layers[:fine_tune_at]:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.177354Z","iopub.execute_input":"2023-02-05T11:13:12.178474Z","iopub.status.idle":"2023-02-05T11:13:12.188979Z","shell.execute_reply.started":"2023-02-05T11:13:12.178435Z","shell.execute_reply":"2023-02-05T11:13:12.187967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\n# Use smaller learning rate\nmodel.compile(loss='categorical_crossentropy',\n              optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate/10),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.190659Z","iopub.execute_input":"2023-02-05T11:13:12.191150Z","iopub.status.idle":"2023-02-05T11:13:12.213021Z","shell.execute_reply.started":"2023-02-05T11:13:12.191112Z","shell.execute_reply":"2023-02-05T11:13:12.211880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.trainable_variables)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.214746Z","iopub.execute_input":"2023-02-05T11:13:12.215255Z","iopub.status.idle":"2023-02-05T11:13:12.223278Z","shell.execute_reply.started":"2023-02-05T11:13:12.215215Z","shell.execute_reply":"2023-02-05T11:13:12.222082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model some more\n\nfine_tune_epochs = 20\ntotal_epochs =  EPOCHS + fine_tune_epochs\n\nhistory_fine = model.fit(x= train_features,\n                         y=train_labels_1hotenc, \n                         batch_size=BATCH_SIZE,\n                         epochs=total_epochs,\n                         initial_epoch=history.epoch[-1],\n                         validation_data=(val_features, val_labels_1hotenc),\n                         verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:13:12.224686Z","iopub.execute_input":"2023-02-05T11:13:12.225688Z","iopub.status.idle":"2023-02-05T11:14:41.876605Z","shell.execute_reply.started":"2023-02-05T11:13:12.225648Z","shell.execute_reply":"2023-02-05T11:14:41.875642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_performance_chart(title='MobileNetV2 fine tuned baseline performance', history=history_fine)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:41.878429Z","iopub.execute_input":"2023-02-05T11:14:41.879096Z","iopub.status.idle":"2023-02-05T11:14:42.209473Z","shell.execute_reply.started":"2023-02-05T11:14:41.879052Z","shell.execute_reply":"2023-02-05T11:14:42.208456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('fine tuned MobileNetV2 performance on the test set:')\nresults = model.evaluate(test_features,test_labels_1hotenc, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:42.211200Z","iopub.execute_input":"2023-02-05T11:14:42.211602Z","iopub.status.idle":"2023-02-05T11:14:43.043415Z","shell.execute_reply.started":"2023-02-05T11:14:42.211549Z","shell.execute_reply":"2023-02-05T11:14:43.042266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"performance_df = record_performance(performance_df, 'MobileNetV2 fine tune top 38 layers', results[1])\nperformance_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.045379Z","iopub.execute_input":"2023-02-05T11:14:43.045751Z","iopub.status.idle":"2023-02-05T11:14:43.058662Z","shell.execute_reply.started":"2023-02-05T11:14:43.045721Z","shell.execute_reply":"2023-02-05T11:14:43.057653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine tune at 100 layer","metadata":{}},{"cell_type":"code","source":"tuning_model_2 = base_model\ntuning_model_2.trainable = True","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.060282Z","iopub.execute_input":"2023-02-05T11:14:43.060968Z","iopub.status.idle":"2023-02-05T11:14:43.076800Z","shell.execute_reply.started":"2023-02-05T11:14:43.060927Z","shell.execute_reply":"2023-02-05T11:14:43.075679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fine-tune from this layer onwards\nfine_tune_at = 100\n\n# Freeze all the layers before the `fine_tune_at` layer\nfor layer in tuning_model_2.layers[:fine_tune_at]:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.080293Z","iopub.execute_input":"2023-02-05T11:14:43.081302Z","iopub.status.idle":"2023-02-05T11:14:43.090642Z","shell.execute_reply.started":"2023-02-05T11:14:43.081258Z","shell.execute_reply":"2023-02-05T11:14:43.089415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model\n# Use smaller learning rate\nmodel.compile(loss='categorical_crossentropy',\n              optimizer = tf.keras.optimizers.Adam(learning_rate=learning_rate/10),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.099024Z","iopub.execute_input":"2023-02-05T11:14:43.099318Z","iopub.status.idle":"2023-02-05T11:14:43.118711Z","shell.execute_reply.started":"2023-02-05T11:14:43.099289Z","shell.execute_reply":"2023-02-05T11:14:43.117777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.trainable_variables)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.120294Z","iopub.execute_input":"2023-02-05T11:14:43.121857Z","iopub.status.idle":"2023-02-05T11:14:43.129402Z","shell.execute_reply.started":"2023-02-05T11:14:43.121816Z","shell.execute_reply":"2023-02-05T11:14:43.128136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model some more\n\nfine_tune_epochs = 20\ntotal_epochs =  EPOCHS + fine_tune_epochs\n\nhistory_fine = model.fit(x= train_features,\n                         y=train_labels_1hotenc, \n                         batch_size=BATCH_SIZE,\n                         epochs=total_epochs,\n                         initial_epoch=history.epoch[-1],\n                         validation_data=(val_features, val_labels_1hotenc),\n                         verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:14:43.131084Z","iopub.execute_input":"2023-02-05T11:14:43.131575Z","iopub.status.idle":"2023-02-05T11:16:24.935032Z","shell.execute_reply.started":"2023-02-05T11:14:43.131535Z","shell.execute_reply":"2023-02-05T11:16:24.933892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_performance_chart(title='MobileNetV2 fine tuned baseline performance', history=history_fine)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:24.939306Z","iopub.execute_input":"2023-02-05T11:16:24.939687Z","iopub.status.idle":"2023-02-05T11:16:25.268746Z","shell.execute_reply.started":"2023-02-05T11:16:24.939653Z","shell.execute_reply":"2023-02-05T11:16:25.267639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('fine tuned MobileNetV2 performance on the test set:')\nresults = model.evaluate(test_features,test_labels_1hotenc, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:25.271432Z","iopub.execute_input":"2023-02-05T11:16:25.272554Z","iopub.status.idle":"2023-02-05T11:16:26.075559Z","shell.execute_reply.started":"2023-02-05T11:16:25.272512Z","shell.execute_reply":"2023-02-05T11:16:26.074538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"performance_df = record_performance(performance_df, 'MobileNetV2 fine tune top 56 layers', results[1])\nperformance_df","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:26.077289Z","iopub.execute_input":"2023-02-05T11:16:26.078879Z","iopub.status.idle":"2023-02-05T11:16:26.093294Z","shell.execute_reply.started":"2023-02-05T11:16:26.078833Z","shell.execute_reply":"2023-02-05T11:16:26.091972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The last model showed the best results on the test set with accuracy at ~65%","metadata":{}},{"cell_type":"markdown","source":"# Prepare the submission","metadata":{}},{"cell_type":"markdown","source":"## Stage 1 test ","metadata":{}},{"cell_type":"code","source":"test_dir = os.path.join('../input/test/test')\n\ntest_files = glob.glob(test_dir+'/*.jpg')\n\nprint('Number of images in a test set:', len(test_files))","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:26.095228Z","iopub.execute_input":"2023-02-05T11:16:26.095911Z","iopub.status.idle":"2023-02-05T11:16:26.108351Z","shell.execute_reply.started":"2023-02-05T11:16:26.095866Z","shell.execute_reply":"2023-02-05T11:16:26.106927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sort test files in correct order for submission\nfrom tkinter import Tcl\ntest_files = Tcl().call('lsort', '-dict', test_files)\ntest_files[:5]","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:26.110525Z","iopub.execute_input":"2023-02-05T11:16:26.110966Z","iopub.status.idle":"2023-02-05T11:16:26.148419Z","shell.execute_reply.started":"2023-02-05T11:16:26.110925Z","shell.execute_reply":"2023-02-05T11:16:26.147260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load images\n\nfeatures = []\nbad_images = 0\n    \nfor i in range(len(test_files)):\n    try:\n        img = cv2.imread(test_files[i])\n        resized_img = cv2.resize(img, (160, 160))\n            \n        features.append(np.array(resized_img))\n                   \n    except Exception as e:\n        bad_images+=1\n        print('Encoutered bad image')\n        \nprint('Bad images ecountered:', bad_images)\ntest_images = np.array(features)","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:16:26.150170Z","iopub.execute_input":"2023-02-05T11:16:26.150580Z","iopub.status.idle":"2023-02-05T11:18:12.424646Z","shell.execute_reply.started":"2023-02-05T11:16:26.150537Z","shell.execute_reply":"2023-02-05T11:18:12.423408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:18:12.426082Z","iopub.execute_input":"2023-02-05T11:18:12.426796Z","iopub.status.idle":"2023-02-05T11:18:12.434354Z","shell.execute_reply.started":"2023-02-05T11:18:12.426751Z","shell.execute_reply":"2023-02-05T11:18:12.433244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_images)\npredictions[:5]","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:18:12.435778Z","iopub.execute_input":"2023-02-05T11:18:12.436795Z","iopub.status.idle":"2023-02-05T11:18:13.818160Z","shell.execute_reply.started":"2023-02-05T11:18:12.436756Z","shell.execute_reply":"2023-02-05T11:18:13.817023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submissions = pd.DataFrame(\n    {'image_name':test_files,\n     'Type_1':predictions[:,0],\n     'Type_2':predictions[:,1],\n     'Type_3':predictions[:,2]})\n\n\ntest_submissions.to_csv('submission.csv', index = False)\ntest_submissions.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-05T11:18:13.822428Z","iopub.execute_input":"2023-02-05T11:18:13.824687Z","iopub.status.idle":"2023-02-05T11:18:13.853494Z","shell.execute_reply.started":"2023-02-05T11:18:13.824645Z","shell.execute_reply":"2023-02-05T11:18:13.852649Z"},"trusted":true},"execution_count":null,"outputs":[]}]}