{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6243,"databundleVersionId":868544,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-06T16:19:05.162496Z","iopub.execute_input":"2021-09-06T16:19:05.163031Z","iopub.status.idle":"2021-09-06T16:19:05.167924Z","shell.execute_reply.started":"2021-09-06T16:19:05.162922Z","shell.execute_reply":"2021-09-06T16:19:05.167217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Cervical Cancer Screening Project \n# Notebook by:\n# Samuel Ozechi, Adebowale Akande\n# \n# \n# Task:\n# Develop an algorithm which accurately identifies and classifies a woman’s cervix type based on images.\n# \n# Dataset: \n# intel-mobileodt-cervical-cancer-screening\n# \n# The major objectives of this notebook includes:\n# \n# Rid the dataset of bad images.  \n# Explore the dataset for useful insights in model building. \n# Apply useful image data processes to develop an efficient classifier.\n# Build, train and evaluate a classifier using transfer learning.\n# Highlight useful steps that can improve the data.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"# import required libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport glob\nimport plotly.graph_objects as go\nimport cv2\nfrom PIL import Image\nfrom PIL import ImageFile\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport tensorflow as tf\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.utils.np_utils import to_categorical\nfrom keras.applications.vgg16 import VGG16\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Flatten , Dropout\nfrom keras.optimizers import Adam\nfrom keras.callbacks import EarlyStopping, ReduceLROnPlateau,ModelCheckpoint \nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nImageFile.LOAD_TRUNCATED_IMAGES = True","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:05.271944Z","iopub.execute_input":"2021-09-06T16:19:05.272289Z","iopub.status.idle":"2021-09-06T16:19:10.557809Z","shell.execute_reply.started":"2021-09-06T16:19:05.272255Z","shell.execute_reply":"2021-09-06T16:19:10.556955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Dataset ","metadata":{}},{"cell_type":"code","source":"# get the data for training\n\nroot_dir = '../input/intel-mobileodt-cervical-cancer-screening'\ntrain_dir = os.path.join(root_dir,'train', 'train')\n\ntype1_dir = os.path.join(train_dir, 'Type_1')\ntype2_dir = os.path.join(train_dir, 'Type_2')\ntype3_dir = os.path.join(train_dir, 'Type_3')\n\ntrain_type1_files = glob.glob(type1_dir+'/*.jpg')\ntrain_type2_files = glob.glob(type2_dir+'/*.jpg')\ntrain_type3_files = glob.glob(type3_dir+'/*.jpg')\n\nadded_type1_files  =  glob.glob(os.path.join(root_dir, \"additional_Type_1_v2\", \"Type_1\")+'/*.jpg')\nadded_type2_files  =  glob.glob(os.path.join(root_dir, \"additional_Type_2_v2\", \"Type_2\")+'/*.jpg')\nadded_type3_files  =  glob.glob(os.path.join(root_dir, \"additional_Type_3_v2\", \"Type_3\")+'/*.jpg')\n\n\ntype1_files = train_type1_files + added_type1_files\ntype2_files = train_type2_files + added_type2_files\ntype3_files = train_type3_files + added_type3_files\n\nprint(f'''Type 1 files for training: {len(type1_files)} \nType 2 files for training: {len(type2_files)}\nType 3 files for training: {len(type3_files)}''' )","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:10.560221Z","iopub.execute_input":"2021-09-06T16:19:10.560521Z","iopub.status.idle":"2021-09-06T16:19:11.597265Z","shell.execute_reply.started":"2021-09-06T16:19:10.560495Z","shell.execute_reply":"2021-09-06T16:19:11.596451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # get data for testing\n\n# test_dir = os.path.join(root_dir,'test', 'test')\n\n# test_files = glob.glob(test_dir+'/*.jpg')\n\n# print(f'''Test files for training: {len(test_files)}''' )","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:11.599258Z","iopub.execute_input":"2021-09-06T16:19:11.599654Z","iopub.status.idle":"2021-09-06T16:19:11.603058Z","shell.execute_reply.started":"2021-09-06T16:19:11.599615Z","shell.execute_reply":"2021-09-06T16:19:11.602189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataframe of file and labels\nfiles = {'filepath': type1_files + type2_files + type3_files,\n          'label': ['Type 1']* len(type1_files) + ['Type 2']* len(type2_files) + ['Type 3']* len(type3_files)}\n\nfiles_df = pd.DataFrame(files).sample(frac=1, random_state= 1).reset_index(drop=True)\nfiles_df","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:11.604607Z","iopub.execute_input":"2021-09-06T16:19:11.60516Z","iopub.status.idle":"2021-09-06T16:19:11.640562Z","shell.execute_reply.started":"2021-09-06T16:19:11.605122Z","shell.execute_reply":"2021-09-06T16:19:11.639652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning and Exploration","metadata":{}},{"cell_type":"code","source":"# describe the dataframe\nfiles_df.describe()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:11.64183Z","iopub.execute_input":"2021-09-06T16:19:11.642181Z","iopub.status.idle":"2021-09-06T16:19:11.678167Z","shell.execute_reply.started":"2021-09-06T16:19:11.642147Z","shell.execute_reply":"2021-09-06T16:19:11.677468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for duplicates\nlen(files_df[files_df.duplicated(subset=['filepath'])])","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:11.679488Z","iopub.execute_input":"2021-09-06T16:19:11.679799Z","iopub.status.idle":"2021-09-06T16:19:11.692594Z","shell.execute_reply.started":"2021-09-06T16:19:11.679767Z","shell.execute_reply":"2021-09-06T16:19:11.691888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for damaged files\nbad_files = []\nfor path in (files_df['filepath'].values):\n    try:\n        img = Image.open(path)\n    except:\n        index = files_df[files_df['filepath']==path].index.values[0]\n        bad_files.append(index)\nprint(len(bad_files))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:19:11.693725Z","iopub.execute_input":"2021-09-06T16:19:11.694048Z","iopub.status.idle":"2021-09-06T16:21:11.769227Z","shell.execute_reply.started":"2021-09-06T16:19:11.694014Z","shell.execute_reply":"2021-09-06T16:21:11.768398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # show the bad files\n# print(bad_files)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.772522Z","iopub.execute_input":"2021-09-06T16:21:11.772781Z","iopub.status.idle":"2021-09-06T16:21:11.777415Z","shell.execute_reply.started":"2021-09-06T16:21:11.772753Z","shell.execute_reply":"2021-09-06T16:21:11.776521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop the damaged files\nfiles_df.drop(bad_files, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.779575Z","iopub.execute_input":"2021-09-06T16:21:11.780042Z","iopub.status.idle":"2021-09-06T16:21:11.791227Z","shell.execute_reply.started":"2021-09-06T16:21:11.780004Z","shell.execute_reply":"2021-09-06T16:21:11.790394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check length of files in dataframe\nlen(files_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.792491Z","iopub.execute_input":"2021-09-06T16:21:11.792878Z","iopub.status.idle":"2021-09-06T16:21:11.808363Z","shell.execute_reply.started":"2021-09-06T16:21:11.792841Z","shell.execute_reply":"2021-09-06T16:21:11.80718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check unique labels\nfiles_df['label'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.809851Z","iopub.execute_input":"2021-09-06T16:21:11.810282Z","iopub.status.idle":"2021-09-06T16:21:11.82129Z","shell.execute_reply.started":"2021-09-06T16:21:11.810246Z","shell.execute_reply":"2021-09-06T16:21:11.82047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get count of each type \ntype_count = pd.DataFrame(files_df['label'].value_counts()).rename(columns= {'label': 'Num_Values'})\ntype_count","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.82252Z","iopub.execute_input":"2021-09-06T16:21:11.822863Z","iopub.status.idle":"2021-09-06T16:21:11.838115Z","shell.execute_reply.started":"2021-09-06T16:21:11.822829Z","shell.execute_reply":"2021-09-06T16:21:11.837349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display barplot of type count\nplt.figure(figsize = (15, 6))\nsns.barplot(x= type_count['Num_Values'], y= type_count.index.to_list())\nplt.title('Cervical Cancer Type Distribution')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.841252Z","iopub.execute_input":"2021-09-06T16:21:11.841539Z","iopub.status.idle":"2021-09-06T16:21:11.980189Z","shell.execute_reply.started":"2021-09-06T16:21:11.841489Z","shell.execute_reply":"2021-09-06T16:21:11.979402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The data distribution plot shows that type 3 class has more datapoints than other types, with type 1  having the least datapoints.\n\n### A pie plot is useful in visualizing the percentage of data distribution.\n","metadata":{}},{"cell_type":"code","source":"# display pieplot of label distribution\npie_plot = go.Pie(labels= type_count.index.to_list(), values= type_count.values.flatten(),\n                 hole= 0.2, text= type_count.index.to_list(), textposition='auto')\nfig = go.Figure([pie_plot])\nfig.update_layout(title_text='Pie Plot of Type Distribution')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:11.981968Z","iopub.execute_input":"2021-09-06T16:21:11.982225Z","iopub.status.idle":"2021-09-06T16:21:12.108607Z","shell.execute_reply.started":"2021-09-06T16:21:11.982194Z","shell.execute_reply":"2021-09-06T16:21:12.107879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The pie plot shows that more than half of the dataset belong to the type two class.\n### The model built with this dataset is expected to be bias towards the the type 2 class and have less precision for identifying the type 1 class. ","metadata":{}},{"cell_type":"code","source":"# display sample images of types\nfor label in ('Type 1', 'Type 2', 'Type 3'):\n    filepaths = files_df[files_df['label']==label]['filepath'].values[:5]\n    fig = plt.figure(figsize= (15, 6))\n    for i, path in enumerate(filepaths):\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n        img = cv2.resize(img, (224, 224))\n        fig.add_subplot(1, 5, i+1)\n        plt.imshow(img)\n        plt.subplots_adjust(hspace=0.5)\n        plt.axis(False)\n        plt.title(label)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:12.109725Z","iopub.execute_input":"2021-09-06T16:21:12.11005Z","iopub.status.idle":"2021-09-06T16:21:16.388436Z","shell.execute_reply.started":"2021-09-06T16:21:12.110015Z","shell.execute_reply":"2021-09-06T16:21:16.387476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Processing","metadata":{}},{"cell_type":"code","source":"#  split the data into train  and validation set\ntrain_df, eval_df = train_test_split(files_df, test_size= 0.2, stratify= files_df['label'], random_state= 1)\nval_df, test_df = train_test_split(eval_df, test_size= 0.5, stratify= eval_df['label'], random_state= 1)\nprint(len(train_df), len(val_df), len(test_df))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:16.389804Z","iopub.execute_input":"2021-09-06T16:21:16.390138Z","iopub.status.idle":"2021-09-06T16:21:16.417019Z","shell.execute_reply.started":"2021-09-06T16:21:16.390104Z","shell.execute_reply":"2021-09-06T16:21:16.416209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loads images from dataframe\ndef load_images(dataframe):\n    features = []\n    filepaths = dataframe['filepath'].values\n    labels = dataframe['label'].values\n    \n    for path in filepaths:\n        img = cv2.imread(path)\n        resized_img = cv2.resize(img, (180, 180))\n        features.append(np.array(resized_img))\n    return np.array(features), np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:16.418336Z","iopub.execute_input":"2021-09-06T16:21:16.418746Z","iopub.status.idle":"2021-09-06T16:21:16.425407Z","shell.execute_reply.started":"2021-09-06T16:21:16.418706Z","shell.execute_reply":"2021-09-06T16:21:16.424465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load training and evaluation data\ntrain_features, train_labels = load_images(train_df)\nval_features, val_labels = load_images(val_df)\ntest_features, test_labels = load_images(test_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:21:16.426703Z","iopub.execute_input":"2021-09-06T16:21:16.427045Z","iopub.status.idle":"2021-09-06T16:46:56.498719Z","shell.execute_reply.started":"2021-09-06T16:21:16.427009Z","shell.execute_reply":"2021-09-06T16:46:56.493943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check lengths of training and evaluation  sets\nlen(train_features), len(train_labels), len(test_features), len(test_labels), len(test_features), len(test_labels) ","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:46:56.507748Z","iopub.execute_input":"2021-09-06T16:46:56.508224Z","iopub.status.idle":"2021-09-06T16:46:56.518667Z","shell.execute_reply.started":"2021-09-06T16:46:56.508193Z","shell.execute_reply":"2021-09-06T16:46:56.51758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get image shape\nInputShape = train_features[0].shape\nprint(InputShape)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:46:56.520109Z","iopub.execute_input":"2021-09-06T16:46:56.520483Z","iopub.status.idle":"2021-09-06T16:46:56.535186Z","shell.execute_reply.started":"2021-09-06T16:46:56.520448Z","shell.execute_reply":"2021-09-06T16:46:56.534442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize the features\nX_train = train_features/255\nX_val  = val_features/255\nX_test  = test_features/255","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:46:56.536409Z","iopub.execute_input":"2021-09-06T16:46:56.536765Z","iopub.status.idle":"2021-09-06T16:47:01.882803Z","shell.execute_reply.started":"2021-09-06T16:46:56.536729Z","shell.execute_reply":"2021-09-06T16:47:01.881908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode the labels\nle = LabelEncoder().fit(['Type 1', 'Type 2', 'Type 3'])\ny_train = le.transform(train_labels)\ny_val = le.transform(val_labels)\ny_test = le.transform(test_labels)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:01.886172Z","iopub.execute_input":"2021-09-06T16:47:01.886454Z","iopub.status.idle":"2021-09-06T16:47:01.897054Z","shell.execute_reply.started":"2021-09-06T16:47:01.886428Z","shell.execute_reply":"2021-09-06T16:47:01.896325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check unique labels\nnp.unique(y_train)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:01.902212Z","iopub.execute_input":"2021-09-06T16:47:01.902497Z","iopub.status.idle":"2021-09-06T16:47:01.911764Z","shell.execute_reply.started":"2021-09-06T16:47:01.902469Z","shell.execute_reply":"2021-09-06T16:47:01.910718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize image data generator for training and evaluation sets\ntrain_datagen = ImageDataGenerator(\n                                rotation_range = 40,\n                                zoom_range = 0.2,\n                                width_shift_range=0.2,\n                                height_shift_range=0.2,\n                                shear_range=0.2,\n                                horizontal_flip=True,\n                                vertical_flip = True)\n\neval_datagen = ImageDataGenerator()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:01.914113Z","iopub.execute_input":"2021-09-06T16:47:01.914436Z","iopub.status.idle":"2021-09-06T16:47:01.93702Z","shell.execute_reply.started":"2021-09-06T16:47:01.914403Z","shell.execute_reply":"2021-09-06T16:47:01.935248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply data augmentation to features\nBATCH_SIZE= 32\ntrain_gen = train_datagen.flow(X_train, y_train, batch_size= BATCH_SIZE)\nval_gen = eval_datagen.flow(X_val, y_val, batch_size= BATCH_SIZE)\ntest_gen = eval_datagen.flow(X_test, y_test, batch_size= BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:01.939375Z","iopub.execute_input":"2021-09-06T16:47:01.939797Z","iopub.status.idle":"2021-09-06T16:47:05.069557Z","shell.execute_reply.started":"2021-09-06T16:47:01.939759Z","shell.execute_reply":"2021-09-06T16:47:05.068592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show shape of each  batch\nfor data_batch, labels_batch in train_gen:\n    print('data batch shape: {} \\n labels batch shape: {}'.format(data_batch.shape, labels_batch.shape))\n    break","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:05.072101Z","iopub.execute_input":"2021-09-06T16:47:05.072661Z","iopub.status.idle":"2021-09-06T16:47:05.284721Z","shell.execute_reply.started":"2021-09-06T16:47:05.072618Z","shell.execute_reply":"2021-09-06T16:47:05.283705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building","metadata":{}},{"cell_type":"code","source":"# initialize pretrained vgg model base\nconv_base = VGG16(weights= 'imagenet', include_top= False, input_shape= (180, 180, 3))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:05.28616Z","iopub.execute_input":"2021-09-06T16:47:05.286539Z","iopub.status.idle":"2021-09-06T16:47:08.444148Z","shell.execute_reply.started":"2021-09-06T16:47:05.286502Z","shell.execute_reply":"2021-09-06T16:47:08.443311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show trainable layers before freezing\nprint('This is the number of trainable weights '\n'before freezing layers in the conv base:', len(conv_base.trainable_weights))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.44544Z","iopub.execute_input":"2021-09-06T16:47:08.445756Z","iopub.status.idle":"2021-09-06T16:47:08.455052Z","shell.execute_reply.started":"2021-09-06T16:47:08.44572Z","shell.execute_reply":"2021-09-06T16:47:08.454217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# freeze few layers of pretrained model\nfor layer in conv_base.layers[:-5]:\n    layer.trainable= False","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.456135Z","iopub.execute_input":"2021-09-06T16:47:08.456604Z","iopub.status.idle":"2021-09-06T16:47:08.462424Z","shell.execute_reply.started":"2021-09-06T16:47:08.456566Z","shell.execute_reply":"2021-09-06T16:47:08.461242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show trainable layers after freezing\nprint('This is the number of trainable weights '\n'after freezing layers in the conv base:', len(conv_base.trainable_weights))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.463767Z","iopub.execute_input":"2021-09-06T16:47:08.464229Z","iopub.status.idle":"2021-09-06T16:47:08.472844Z","shell.execute_reply.started":"2021-09-06T16:47:08.464193Z","shell.execute_reply":"2021-09-06T16:47:08.471814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build model \nmodel = Sequential([conv_base, \n                    Flatten(),\n                   Dropout(0.5),\n                   Dense(3, activation='softmax')])","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.474727Z","iopub.execute_input":"2021-09-06T16:47:08.475228Z","iopub.status.idle":"2021-09-06T16:47:08.555546Z","shell.execute_reply.started":"2021-09-06T16:47:08.475189Z","shell.execute_reply":"2021-09-06T16:47:08.554625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile model\nmodel.compile(optimizer= Adam(0.0001), loss= 'sparse_categorical_crossentropy', metrics= ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.556889Z","iopub.execute_input":"2021-09-06T16:47:08.557275Z","iopub.status.idle":"2021-09-06T16:47:08.575273Z","shell.execute_reply.started":"2021-09-06T16:47:08.557217Z","shell.execute_reply":"2021-09-06T16:47:08.574259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show model summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.576526Z","iopub.execute_input":"2021-09-06T16:47:08.576911Z","iopub.status.idle":"2021-09-06T16:47:08.587946Z","shell.execute_reply.started":"2021-09-06T16:47:08.576872Z","shell.execute_reply":"2021-09-06T16:47:08.586925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define training steps\nTRAIN_STEPS = len(train_df)//BATCH_SIZE\nVAL_STEPS = len(val_df)//BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.589332Z","iopub.execute_input":"2021-09-06T16:47:08.589994Z","iopub.status.idle":"2021-09-06T16:47:08.595407Z","shell.execute_reply.started":"2021-09-06T16:47:08.589945Z","shell.execute_reply":"2021-09-06T16:47:08.594312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize callbacks\nreduceLR = ReduceLROnPlateau(monitor='val_loss', patience=10, verbose= 1, mode='min', factor=  0.2, min_lr = 1e-5)\n\nearly_stopping = EarlyStopping(monitor='val_loss', patience = 20, verbose=1, mode='min', restore_best_weights= True)\n\ncheckpoint = ModelCheckpoint('cervicalModel.weights.hdf5', monitor='val_loss', verbose=1,save_best_only=True, mode= 'min')","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.596828Z","iopub.execute_input":"2021-09-06T16:47:08.597511Z","iopub.status.idle":"2021-09-06T16:47:08.605324Z","shell.execute_reply.started":"2021-09-06T16:47:08.59747Z","shell.execute_reply":"2021-09-06T16:47:08.604255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train model\nhistory = model.fit(train_gen, steps_per_epoch= TRAIN_STEPS, validation_data=val_gen, validation_steps=VAL_STEPS, epochs= 100,\n                   callbacks= [reduceLR, early_stopping, checkpoint])","metadata":{"execution":{"iopub.status.busy":"2021-09-06T16:47:08.60675Z","iopub.execute_input":"2021-09-06T16:47:08.607484Z","iopub.status.idle":"2021-09-06T17:27:30.727069Z","shell.execute_reply.started":"2021-09-06T16:47:08.607433Z","shell.execute_reply":"2021-09-06T17:27:30.724528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read training history into dataframe\nhistory_df = pd.DataFrame(history.history)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:30.729995Z","iopub.execute_input":"2021-09-06T17:27:30.730391Z","iopub.status.idle":"2021-09-06T17:27:30.747191Z","shell.execute_reply.started":"2021-09-06T17:27:30.730349Z","shell.execute_reply":"2021-09-06T17:27:30.746481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display training and validation history\n\n# display history of accurracy\nplt.figure(figsize= (15,6))\nplt.subplot(1,2,1)\nplt.plot(history_df['accuracy'], label= 'accuracy' )\nplt.plot(history_df['val_accuracy'], label= 'val_accuracy')\n# history_df[['acc', 'val_acc']]\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Training and Validation Accuracy History')\nplt.legend()\n\n# display history of loss\nplt.subplot(1,2,2)\nplt.plot(history_df['loss'], label= 'loss')\nplt.plot(history_df['val_loss'], label= 'val_loss')\n# history_df[['loss', 'val_loss']].plot()\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Training and Validation Loss History')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:30.748255Z","iopub.execute_input":"2021-09-06T17:27:30.748574Z","iopub.status.idle":"2021-09-06T17:27:31.136724Z","shell.execute_reply.started":"2021-09-06T17:27:30.74854Z","shell.execute_reply":"2021-09-06T17:27:31.13593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The training and validation histories show model overfitting.\n### This could be as a result of the number of training samples, the quality of the image data or the data distribution ","metadata":{}},{"cell_type":"markdown","source":"#### Possible steps of improving the model\n1. Acquiring more quality data\n2. Utilizing regularization methods\n3. Using a model with less capacity","metadata":{}},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"# load best weights into model\nmodel.load_weights('cervicalModel.weights.hdf5')","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:31.137919Z","iopub.execute_input":"2021-09-06T17:27:31.138255Z","iopub.status.idle":"2021-09-06T17:27:31.211027Z","shell.execute_reply.started":"2021-09-06T17:27:31.138215Z","shell.execute_reply":"2021-09-06T17:27:31.210137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model\nmodel.save('cancer_screen_model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:31.212331Z","iopub.execute_input":"2021-09-06T17:27:31.212722Z","iopub.status.idle":"2021-09-06T17:27:31.412458Z","shell.execute_reply.started":"2021-09-06T17:27:31.212679Z","shell.execute_reply":"2021-09-06T17:27:31.411589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# evaluate model on test set\nmodel.evaluate(test_gen)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:31.413821Z","iopub.execute_input":"2021-09-06T17:27:31.414166Z","iopub.status.idle":"2021-09-06T17:27:34.076401Z","shell.execute_reply.started":"2021-09-06T17:27:31.414129Z","shell.execute_reply":"2021-09-06T17:27:34.075592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open('cancer.pickle', 'wb') as f:\n#     pickle.dump(model, f)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:34.077722Z","iopub.execute_input":"2021-09-06T17:27:34.078063Z","iopub.status.idle":"2021-09-06T17:27:34.08201Z","shell.execute_reply.started":"2021-09-06T17:27:34.078027Z","shell.execute_reply":"2021-09-06T17:27:34.080815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get test data directory\ntest_dir = os.path.join(root_dir,'test', 'test')","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:34.0837Z","iopub.execute_input":"2021-09-06T17:27:34.084185Z","iopub.status.idle":"2021-09-06T17:27:34.09186Z","shell.execute_reply.started":"2021-09-06T17:27:34.084037Z","shell.execute_reply":"2021-09-06T17:27:34.090993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load test features and labels\ntest_filenames = []\ntest_features = []\nfor filename in os.listdir(test_dir):\n    test_filenames.append(filename)\n    filepath = os.path.join(test_dir, filename)\n    img = cv2.imread(filepath)\n    resized_img = cv2.resize(img, (180, 180))\n    test_features.append(np.array(resized_img)) ","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:27:34.093318Z","iopub.execute_input":"2021-09-06T17:27:34.093733Z","iopub.status.idle":"2021-09-06T17:29:11.800414Z","shell.execute_reply.started":"2021-09-06T17:27:34.093695Z","shell.execute_reply":"2021-09-06T17:29:11.799556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show length of test features and labels\nprint(len(test_filenames), len(test_features))","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:11.802933Z","iopub.execute_input":"2021-09-06T17:29:11.803223Z","iopub.status.idle":"2021-09-06T17:29:11.808811Z","shell.execute_reply.started":"2021-09-06T17:29:11.803197Z","shell.execute_reply":"2021-09-06T17:29:11.807953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# normalize test features\ntest_X = np.array(test_features)\ntest_X = test_X/255","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:11.810285Z","iopub.execute_input":"2021-09-06T17:29:11.810901Z","iopub.status.idle":"2021-09-06T17:29:12.23311Z","shell.execute_reply.started":"2021-09-06T17:29:11.810861Z","shell.execute_reply":"2021-09-06T17:29:12.232179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get test predictions\ntest_predict = model.predict(test_X)\ntest_predict[0]","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:12.234585Z","iopub.execute_input":"2021-09-06T17:29:12.235167Z","iopub.status.idle":"2021-09-06T17:29:13.675973Z","shell.execute_reply.started":"2021-09-06T17:29:12.235128Z","shell.execute_reply":"2021-09-06T17:29:13.675135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show encoded classes\nle.classes_","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:13.677953Z","iopub.execute_input":"2021-09-06T17:29:13.678363Z","iopub.status.idle":"2021-09-06T17:29:13.685407Z","shell.execute_reply.started":"2021-09-06T17:29:13.678322Z","shell.execute_reply":"2021-09-06T17:29:13.684306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataframe of test predictions\nsub_df = pd.DataFrame(test_predict, columns= ['Type 1', 'Type 2', 'Type 3' ])\nsub_df['image_name'] = test_filenames\nsub_df = sub_df[['image_name', 'Type 1', 'Type 2', 'Type 3']]\nsub_df = sub_df.sort_values(['image_name']).reset_index(drop=True)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:13.686969Z","iopub.execute_input":"2021-09-06T17:29:13.687689Z","iopub.status.idle":"2021-09-06T17:29:13.733126Z","shell.execute_reply.started":"2021-09-06T17:29:13.687648Z","shell.execute_reply":"2021-09-06T17:29:13.732025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create csv file of test predictions\nsub_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-06T17:29:13.734688Z","iopub.execute_input":"2021-09-06T17:29:13.735086Z","iopub.status.idle":"2021-09-06T17:29:13.754115Z","shell.execute_reply.started":"2021-09-06T17:29:13.735046Z","shell.execute_reply":"2021-09-06T17:29:13.753317Z"},"trusted":true},"execution_count":null,"outputs":[]}]}