{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1 class=\"list-group-item list-group-item-success\">HISTOPATHOLOGIC CANCER DETECTION</h1></center><br>\n\n<img src = \"https://storage.googleapis.com/kaggle-competitions/kaggle/11848/logos/header.png?t=2018-11-15-01-52-19\">\n","metadata":{"_uuid":"5e3e84b04b0843f2d577775ff4495206b10acdd7"}},{"cell_type":"code","source":"# Importing Packages\n\n# Setting Random seed\nfrom numpy.random import seed\nseed(101)\nfrom tensorflow import set_random_seed\nset_random_seed(101)\n\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, Activation\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam\nimport os\nimport cv2\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nimport itertools\nimport shutil\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2022-02-01T11:51:47.628492Z","iopub.execute_input":"2022-02-01T11:51:47.628859Z","iopub.status.idle":"2022-02-01T11:51:48.433585Z","shell.execute_reply.started":"2022-02-01T11:51:47.628786Z","shell.execute_reply":"2022-02-01T11:51:48.432682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting Parameters before loading data\nIMAGE_SIZE = 96\nIMAGE_CHANNELS = 3\nSAMPLE_SIZE = 80000 # the number of images we use from each of the two classes\n","metadata":{"_uuid":"b7bbdd52c81188b8e9c528b88d9fd0da176bf4bc","execution":{"iopub.status.busy":"2022-02-01T11:51:48.436152Z","iopub.execute_input":"2022-02-01T11:51:48.436785Z","iopub.status.idle":"2022-02-01T11:51:48.442744Z","shell.execute_reply.started":"2022-02-01T11:51:48.436483Z","shell.execute_reply":"2022-02-01T11:51:48.441340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Labels as per csv file\n\n0 = no tumor tissue<br>\n1 =   has tumor tissue. <br>\n","metadata":{"_uuid":"c24d1662f8bd5fc4d8c57c29449990a3520cd96b"}},{"cell_type":"markdown","source":"### Checking for  Count of Train & Test Images","metadata":{"_uuid":"5a285343c286be191aa827ff9b836b67821176a5"}},{"cell_type":"code","source":"print(len(os.listdir('../input/train')))\nprint(len(os.listdir('../input/test')))","metadata":{"_uuid":"54461212efed65ac377369a468c80e7d708010f4","execution":{"iopub.status.busy":"2022-02-01T11:51:48.466844Z","iopub.execute_input":"2022-02-01T11:51:48.467142Z","iopub.status.idle":"2022-02-01T11:51:53.557036Z","shell.execute_reply.started":"2022-02-01T11:51:48.467078Z","shell.execute_reply":"2022-02-01T11:51:53.555945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a Dataframe containing all images","metadata":{"_uuid":"b90854e07d495d9f945a0e40189fd32a0c32bff5"}},{"cell_type":"code","source":"df_data = pd.read_csv('../input/train_labels.csv')\n\n# removing this image because it caused a training error previously\ndf_data[df_data['id'] != 'dd6dfed324f9fcb6f93f46f32fc800f2ec196be2']\n\n# removing this image because it's black\ndf_data[df_data['id'] != '9369c7278ec8bcc6c880d99194de09fc2bd4efbe']\n\n\nprint(df_data.shape)","metadata":{"_uuid":"e9c9f40ffab35044641b0dc7d9b18609af1aa25e","execution":{"iopub.status.busy":"2022-02-01T11:51:53.558507Z","iopub.execute_input":"2022-02-01T11:51:53.559148Z","iopub.status.idle":"2022-02-01T11:51:54.397399Z","shell.execute_reply.started":"2022-02-01T11:51:53.559080Z","shell.execute_reply":"2022-02-01T11:51:54.395002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check the class distribution","metadata":{"_uuid":"cfbd53b7f8ea1929952ffed6221b380012618e32"}},{"cell_type":"code","source":"df_data['label'].value_counts()","metadata":{"_uuid":"e18560bf69d3dfc0c4772e7c79bb119fd2eb634b","execution":{"iopub.status.busy":"2022-02-01T11:51:54.398465Z","iopub.execute_input":"2022-02-01T11:51:54.398771Z","iopub.status.idle":"2022-02-01T11:51:54.415297Z","shell.execute_reply.started":"2022-02-01T11:51:54.398703Z","shell.execute_reply":"2022-02-01T11:51:54.414163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display a random sample of train images  by class","metadata":{"_uuid":"6efe2e5de99c4bf92079b1a7d0b892d30fc9d518"}},{"cell_type":"code","source":"# Function for Showing images with both classes randomly\ndef draw_category_images(col_name,figure_cols, df, IMAGE_PATH):\n    \n    \"\"\"\n    Give a column in a dataframe,\n    this function takes a sample of each class and displays that\n    sample on one row. The sample size is the same as figure_cols which\n    is the number of columns in the figure.\n    Because this function takes a random sample, each time the function is run it\n    displays different images.\n    \"\"\"\n    \n\n    categories = (df.groupby([col_name])[col_name].nunique()).index\n    f, ax = plt.subplots(nrows=len(categories),ncols=figure_cols, \n                         figsize=(4*figure_cols,4*len(categories))) # adjust size here\n    # draw a number of images for each location\n    for i, cat in enumerate(categories):\n        sample = df[df[col_name]==cat].sample(figure_cols) # figure_cols is also the sample size\n        for j in range(0,figure_cols):\n            file=IMAGE_PATH + sample.iloc[j]['id'] + '.tif'\n            im=cv2.imread(file)\n            ax[i, j].imshow(im, resample=True, cmap='gray')\n            ax[i, j].set_title(cat, fontsize=16)  \n    plt.tight_layout()\n    plt.show()\n    ","metadata":{"_kg_hide-input":true,"_uuid":"1c5143f227da4262eafce8cf0210a02c8072fb8e","execution":{"iopub.status.busy":"2022-02-01T11:51:54.416569Z","iopub.execute_input":"2022-02-01T11:51:54.416876Z","iopub.status.idle":"2022-02-01T11:51:54.428519Z","shell.execute_reply.started":"2022-02-01T11:51:54.416809Z","shell.execute_reply":"2022-02-01T11:51:54.427029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_PATH = '../input/train/' \n# Displaying 4 images in each class \ndraw_category_images('label',4, df_data, IMAGE_PATH)","metadata":{"_uuid":"bd38bcfb5839975e4fee9e70b93d42c29c1b5d2e","execution":{"iopub.status.busy":"2022-02-01T11:51:54.429993Z","iopub.execute_input":"2022-02-01T11:51:54.430626Z","iopub.status.idle":"2022-02-01T11:51:56.296689Z","shell.execute_reply.started":"2022-02-01T11:51:54.430556Z","shell.execute_reply":"2022-02-01T11:51:56.295508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Balance the target distribution\nWe will reduce the number of samples in class 0.","metadata":{"_uuid":"c1150500d2772b7f36cdaa5aa5fd7f0fb4a72628"}},{"cell_type":"code","source":"# take a random sample of class 0 with size equal to num samples in class 1\ndf_0 = df_data[df_data['label'] == 0].sample(SAMPLE_SIZE, random_state = 101)\n# filter out class 1\ndf_1 = df_data[df_data['label'] == 1].sample(SAMPLE_SIZE, random_state = 101)\n\n# concat the dataframes\ndf_data = pd.concat([df_0, df_1], axis=0).reset_index(drop=True)\n# shuffle\ndf_data = shuffle(df_data)\n\ndf_data['label'].value_counts()","metadata":{"_uuid":"270fc18640b552ecc3cb0e1dd3036441db7a4a2b","execution":{"iopub.status.busy":"2022-02-01T11:51:56.334387Z","iopub.execute_input":"2022-02-01T11:51:56.337049Z","iopub.status.idle":"2022-02-01T11:51:56.452639Z","shell.execute_reply.started":"2022-02-01T11:51:56.336963Z","shell.execute_reply":"2022-02-01T11:51:56.451603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DF Overview\ndf_data.head()","metadata":{"_uuid":"a166dec3ef84c66ad9cd815b63fc1a753df2eb76","execution":{"iopub.status.busy":"2022-02-01T11:51:56.465932Z","iopub.execute_input":"2022-02-01T11:51:56.470723Z","iopub.status.idle":"2022-02-01T11:51:56.497258Z","shell.execute_reply.started":"2022-02-01T11:51:56.470656Z","shell.execute_reply":"2022-02-01T11:51:56.496194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_test_split\n\n# stratify=y creates a balanced validation set.\ny = df_data['label']\n\ndf_train, df_val = train_test_split(df_data, test_size=0.10, random_state=101, stratify=y)\n\nprint(df_train.shape)\nprint(df_val.shape)","metadata":{"_uuid":"15ba9792e6a370b7560330af15b3cfe21185c1cb","execution":{"iopub.status.busy":"2022-02-01T11:51:56.502420Z","iopub.execute_input":"2022-02-01T11:51:56.505034Z","iopub.status.idle":"2022-02-01T11:51:56.601682Z","shell.execute_reply.started":"2022-02-01T11:51:56.504948Z","shell.execute_reply":"2022-02-01T11:51:56.600828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Both classes have equal weightage in train subset\ndf_train['label'].value_counts()","metadata":{"_uuid":"7de70d915a5f1d2599725e00bdb3b9103d947883","execution":{"iopub.status.busy":"2022-02-01T11:51:56.602672Z","iopub.execute_input":"2022-02-01T11:51:56.602968Z","iopub.status.idle":"2022-02-01T11:51:56.616293Z","shell.execute_reply.started":"2022-02-01T11:51:56.602908Z","shell.execute_reply":"2022-02-01T11:51:56.614838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Both classes have equal weightage in Val subset\ndf_val['label'].value_counts()","metadata":{"_uuid":"392c0eea00be8e43a6e55438d1458650e842030b","execution":{"iopub.status.busy":"2022-02-01T11:51:56.617765Z","iopub.execute_input":"2022-02-01T11:51:56.618475Z","iopub.status.idle":"2022-02-01T11:51:56.628533Z","shell.execute_reply.started":"2022-02-01T11:51:56.618410Z","shell.execute_reply":"2022-02-01T11:51:56.627416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a Directory Structure","metadata":{"_uuid":"ba17dd34b75367fc61df6634d51dac94c3ab4951"}},{"cell_type":"code","source":"# Create a new directory\nbase_dir = 'base_dir'\nos.mkdir(base_dir)\n\n# create a path to 'base_dir' to which we will join the names of the new folders\n# train_dir\ntrain_dir = os.path.join(base_dir, 'train_dir')\nos.mkdir(train_dir)\n\n# val_dir\nval_dir = os.path.join(base_dir, 'val_dir')\nos.mkdir(val_dir)\n\n# Inside each folder we create seperate folders for each class\n# create new folders inside train_dir\nno_tumor_tissue = os.path.join(train_dir, 'a_no_tumor_tissue')\nos.mkdir(no_tumor_tissue)\nhas_tumor_tissue = os.path.join(train_dir, 'b_has_tumor_tissue')\nos.mkdir(has_tumor_tissue)\n\n\n# create new folders inside val_dir\nno_tumor_tissue = os.path.join(val_dir, 'a_no_tumor_tissue')\nos.mkdir(no_tumor_tissue)\nhas_tumor_tissue = os.path.join(val_dir, 'b_has_tumor_tissue')\nos.mkdir(has_tumor_tissue)\n","metadata":{"_uuid":"ff8acc2e92a1b1b5002d6e1bf9a1180c3256f19d","execution":{"iopub.status.busy":"2022-02-01T11:51:56.630205Z","iopub.execute_input":"2022-02-01T11:51:56.631071Z","iopub.status.idle":"2022-02-01T11:51:56.642216Z","shell.execute_reply.started":"2022-02-01T11:51:56.630977Z","shell.execute_reply":"2022-02-01T11:51:56.641201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that the folders have been created\nos.listdir('base_dir/train_dir')","metadata":{"_uuid":"03ca5d4b8b027c2712d7096314d3a79ef829b23c","execution":{"iopub.status.busy":"2022-02-01T11:51:56.644033Z","iopub.execute_input":"2022-02-01T11:51:56.644632Z","iopub.status.idle":"2022-02-01T11:51:56.658851Z","shell.execute_reply.started":"2022-02-01T11:51:56.644544Z","shell.execute_reply":"2022-02-01T11:51:56.657555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transfer the images into the folders","metadata":{"_uuid":"6a2e56340ba18f3b63c1b129fd995fecfadaa21d"}},{"cell_type":"code","source":"# Set the id as the index in df_data\ndf_data.set_index('id', inplace=True)","metadata":{"_uuid":"e84c8a9642b030094b1888af3299063f883112a6","execution":{"iopub.status.busy":"2022-02-01T11:51:56.659879Z","iopub.execute_input":"2022-02-01T11:51:56.660411Z","iopub.status.idle":"2022-02-01T11:51:56.679529Z","shell.execute_reply.started":"2022-02-01T11:51:56.660312Z","shell.execute_reply":"2022-02-01T11:51:56.678567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get a list of train and val images\ntrain_list = list(df_train['id'])\nval_list = list(df_val['id'])\n\n\n# Transfer the train images\nfor image in train_list:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = df_data.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = 'a_no_tumor_tissue'\n    if target == 1:\n        label = 'b_has_tumor_tissue'\n    \n    # source path to image\n    src = os.path.join('../input/train', fname)\n    # destination path to image\n    dst = os.path.join(train_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n\n\n# Transfer the val images\n\nfor image in val_list:\n    \n    # the id in the csv file does not have the .tif extension therefore we add it here\n    fname = image + '.tif'\n    # get the label for a certain image\n    target = df_data.loc[image,'label']\n    \n    # these must match the folder names\n    if target == 0:\n        label = 'a_no_tumor_tissue'\n    if target == 1:\n        label = 'b_has_tumor_tissue'\n    \n\n    # source path to image\n    src = os.path.join('../input/train', fname)\n    # destination path to image\n    dst = os.path.join(val_dir, label, fname)\n    # copy the image from the source to the destination\n    shutil.copyfile(src, dst)\n    \n\n\n   ","metadata":{"_uuid":"afb8969a9ee75c13bddc808a4bcc326611baaaaf","execution":{"iopub.status.busy":"2022-02-01T11:51:56.680514Z","iopub.execute_input":"2022-02-01T11:51:56.680897Z","iopub.status.idle":"2022-02-01T12:07:00.647827Z","shell.execute_reply.started":"2022-02-01T11:51:56.680831Z","shell.execute_reply":"2022-02-01T12:07:00.646690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking for how many train images we have in each folder\n\nprint(len(os.listdir('base_dir/train_dir/a_no_tumor_tissue')))\nprint(len(os.listdir('base_dir/train_dir/b_has_tumor_tissue')))","metadata":{"_uuid":"71532bfc32608289b1f773ffdbc8a7cea1bfb94c","execution":{"iopub.status.busy":"2022-02-01T12:07:00.648837Z","iopub.execute_input":"2022-02-01T12:07:00.649137Z","iopub.status.idle":"2022-02-01T12:07:00.781506Z","shell.execute_reply.started":"2022-02-01T12:07:00.649076Z","shell.execute_reply":"2022-02-01T12:07:00.779096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking how many val images we have in each folder\n\nprint(len(os.listdir('base_dir/val_dir/a_no_tumor_tissue')))\nprint(len(os.listdir('base_dir/val_dir/b_has_tumor_tissue')))\n","metadata":{"_uuid":"897e9df543bb65b47bb00019dc681125ca08ee5d","execution":{"iopub.status.busy":"2022-02-01T12:07:00.787125Z","iopub.execute_input":"2022-02-01T12:07:00.787679Z","iopub.status.idle":"2022-02-01T12:07:00.812338Z","shell.execute_reply.started":"2022-02-01T12:07:00.787437Z","shell.execute_reply":"2022-02-01T12:07:00.811162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Set Up the Generators","metadata":{"_uuid":"f8dce940ee8a7a42aacb062e4c6b5a4a54dba58f"}},{"cell_type":"code","source":"# Setting up new created directories as path\ntrain_path = 'base_dir/train_dir'\nvalid_path = 'base_dir/val_dir'\ntest_path = '../input/test'\n\nnum_train_samples = len(df_train)\nnum_val_samples = len(df_val)\ntrain_batch_size = 10\nval_batch_size = 10\n\n# Defining steps with batch size and samples\ntrain_steps = np.ceil(num_train_samples / train_batch_size)\nval_steps = np.ceil(num_val_samples / val_batch_size)","metadata":{"_uuid":"ef4fe7be09f11ff4badfd22d5fd5e03f8521ed58","execution":{"iopub.status.busy":"2022-02-01T12:07:00.826668Z","iopub.execute_input":"2022-02-01T12:07:00.827460Z","iopub.status.idle":"2022-02-01T12:07:00.837162Z","shell.execute_reply.started":"2022-02-01T12:07:00.827394Z","shell.execute_reply":"2022-02-01T12:07:00.835944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using ImageDataGenerator for loading images\ndatagen = ImageDataGenerator(rescale=1.0/255)\n\ntrain_gen = datagen.flow_from_directory(train_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=train_batch_size,\n                                        class_mode='categorical')\n\nval_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=val_batch_size,\n                                        class_mode='categorical')\n\n# Note: shuffle=False causes the test dataset to not be shuffled\ntest_gen = datagen.flow_from_directory(valid_path,\n                                        target_size=(IMAGE_SIZE,IMAGE_SIZE),\n                                        batch_size=1,\n                                        class_mode='categorical',\n                                        shuffle=False)","metadata":{"_uuid":"68fbd9d5fbb80859a82f94a12e335ce05a93bd51","execution":{"iopub.status.busy":"2022-02-01T12:07:00.838158Z","iopub.execute_input":"2022-02-01T12:07:00.838490Z","iopub.status.idle":"2022-02-01T12:07:15.816113Z","shell.execute_reply.started":"2022-02-01T12:07:00.838416Z","shell.execute_reply":"2022-02-01T12:07:15.815234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model Architecture¶","metadata":{"_uuid":"79da4d0a66a90cffe40580a596dd4d0e2bc45a9b"}},{"cell_type":"code","source":"# Model Params\nkernel_size = (3,3)\npool_size= (2,2)\nfirst_filters = 32\nsecond_filters = 64\nthird_filters = 128\ndropout_conv = 0.3\ndropout_dense = 0.3\n\n# Model Structure\nmodel = Sequential()\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu', input_shape = (96, 96, 3)))\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\nmodel.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size)) \nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(second_filters, kernel_size, activation ='relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(Conv2D(third_filters, kernel_size, activation ='relu'))\nmodel.add(MaxPooling2D(pool_size = pool_size))\nmodel.add(Dropout(dropout_conv))\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = \"relu\"))\nmodel.add(Dropout(dropout_dense))\nmodel.add(Dense(2, activation = \"softmax\"))\n\nmodel.summary()\n","metadata":{"_uuid":"b9835ea0fd0bca54138904895c39d38227a70c22","execution":{"iopub.status.busy":"2022-02-01T12:07:15.817161Z","iopub.execute_input":"2022-02-01T12:07:15.817481Z","iopub.status.idle":"2022-02-01T12:07:16.210505Z","shell.execute_reply.started":"2022-02-01T12:07:15.817421Z","shell.execute_reply":"2022-02-01T12:07:16.207814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train the Model","metadata":{"_uuid":"75cfc4fcb8dd3408d1c4fcf8cd85e0e2f5b611d7"}},{"cell_type":"code","source":"# Compliling the model\nmodel.compile(Adam(lr=0.0001), loss='binary_crossentropy', \n              metrics=['accuracy'])","metadata":{"_uuid":"9de9715f49a63b55775b10abd2f461b395e23b5d","execution":{"iopub.status.busy":"2022-02-01T12:07:16.211551Z","iopub.execute_input":"2022-02-01T12:07:16.211835Z","iopub.status.idle":"2022-02-01T12:07:16.369454Z","shell.execute_reply.started":"2022-02-01T12:07:16.211774Z","shell.execute_reply":"2022-02-01T12:07:16.368436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the labels that are associated with each index\nprint(val_gen.class_indices)","metadata":{"_uuid":"227d84a4f44c0b7256855c06ba04dabd58d89d84","execution":{"iopub.status.busy":"2022-02-01T12:07:16.370548Z","iopub.execute_input":"2022-02-01T12:07:16.370858Z","iopub.status.idle":"2022-02-01T12:07:16.378505Z","shell.execute_reply.started":"2022-02-01T12:07:16.370793Z","shell.execute_reply":"2022-02-01T12:07:16.377263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = \"model.h5\"\n\n# Checkpoint to save weights after every epoch\ncheckpoint = ModelCheckpoint(filepath, monitor='val_acc', verbose=1, \n                             save_best_only=True, mode='max')\n\n# It alters the learning rate based on metrics in each epoch\nreduce_lr = ReduceLROnPlateau(monitor='val_acc', factor=0.5, patience=2, \n                                   verbose=1, mode='max', min_lr=0.00001)\n                                                  \ncallbacks_list = [checkpoint, reduce_lr]\n\n# Fitting the model\nhistory = model.fit_generator(train_gen, steps_per_epoch=train_steps, \n                    validation_data=val_gen,\n                    validation_steps=val_steps,\n                    epochs=20, verbose=1,\n                   callbacks=callbacks_list)","metadata":{"_uuid":"a746769db61563f226288eba9aa8a6584b9e8e0b","execution":{"iopub.status.busy":"2022-02-01T12:34:04.463544Z","iopub.execute_input":"2022-02-01T12:34:04.463904Z","iopub.status.idle":"2022-02-01T14:01:24.216175Z","shell.execute_reply.started":"2022-02-01T12:34:04.463836Z","shell.execute_reply":"2022-02-01T14:01:24.214830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate the model using the val set","metadata":{"_uuid":"fa15a8afda3593973726e9087cbd98073041c908"}},{"cell_type":"code","source":"# Here the best epoch will be used.\n\n# Loading Weights file\nmodel.load_weights('model.h5')\n\nval_loss, val_acc = \\\nmodel.evaluate_generator(test_gen, \n                        steps=len(df_val))\n\nprint('val_loss:', val_loss)\nprint('val_acc:', val_acc)","metadata":{"_uuid":"428bdf5b24ff8cef35012205c3f2eb37006fc9e9","execution":{"iopub.status.busy":"2022-02-01T14:07:25.663054Z","iopub.execute_input":"2022-02-01T14:07:25.663404Z","iopub.status.idle":"2022-02-01T14:08:23.052645Z","shell.execute_reply.started":"2022-02-01T14:07:25.663336Z","shell.execute_reply":"2022-02-01T14:08:23.051469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot the Training Curves","metadata":{"_uuid":"93556c9e4b6a188cf9cb67a6519c9bc365c60caf"}},{"cell_type":"code","source":"# Display the loss and accuracy curves\n\nimport matplotlib.pyplot as plt\n\nacc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, loss, 'bo', label='Training loss')\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.legend()\nplt.figure()\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.legend()\nplt.figure()","metadata":{"_uuid":"385da8ba94a1079d17909790716b295fc2737584","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-01T14:09:59.372548Z","iopub.execute_input":"2022-02-01T14:09:59.372943Z","iopub.status.idle":"2022-02-01T14:10:00.123360Z","shell.execute_reply.started":"2022-02-01T14:09:59.372841Z","shell.execute_reply":"2022-02-01T14:10:00.121891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predictions","metadata":{"_uuid":"5636e76f23202dd1f2a27ace25e15e09619a5e4e"}},{"cell_type":"code","source":"predictions = model.predict_generator(test_gen, steps=len(df_val), verbose=1)","metadata":{"_uuid":"652d9d6aa51dc1818d1c5171212d10e141ad7de9","execution":{"iopub.status.busy":"2022-02-01T14:10:05.010048Z","iopub.execute_input":"2022-02-01T14:10:05.010430Z","iopub.status.idle":"2022-02-01T14:10:46.945410Z","shell.execute_reply.started":"2022-02-01T14:10:05.010362Z","shell.execute_reply":"2022-02-01T14:10:46.944491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To check what index keras has internally assigned to each class. \ntest_gen.class_indices","metadata":{"_uuid":"dc71f69944e7db83329417c5265a5bc31f9c4fc3","execution":{"iopub.status.busy":"2022-02-01T14:10:46.948130Z","iopub.execute_input":"2022-02-01T14:10:46.948880Z","iopub.status.idle":"2022-02-01T14:10:46.956633Z","shell.execute_reply.started":"2022-02-01T14:10:46.948809Z","shell.execute_reply":"2022-02-01T14:10:46.955381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The columns need to be ordered to match the output of the previous cell\n# Appending predictions to  a dataframe object\ndf_preds = pd.DataFrame(predictions, columns=['no_tumor_tissue', 'has_tumor_tissue'])\ndf_preds.head()","metadata":{"_uuid":"4a6709d73969f7fd597128223b110be077f84edb","execution":{"iopub.status.busy":"2022-02-01T14:12:34.552978Z","iopub.execute_input":"2022-02-01T14:12:34.553408Z","iopub.status.idle":"2022-02-01T14:12:34.574301Z","shell.execute_reply.started":"2022-02-01T14:12:34.553342Z","shell.execute_reply":"2022-02-01T14:12:34.572886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the true labels\ny_true = test_gen.classes\n\n# Get the predicted labels as probabilities\ny_pred = df_preds['has_tumor_tissue']","metadata":{"_uuid":"0954d61a4ef8bc056452b3bbad9456d45c00bed1","execution":{"iopub.status.busy":"2022-02-01T14:12:35.984922Z","iopub.execute_input":"2022-02-01T14:12:35.985325Z","iopub.status.idle":"2022-02-01T14:12:35.991078Z","shell.execute_reply.started":"2022-02-01T14:12:35.985257Z","shell.execute_reply":"2022-02-01T14:12:35.989650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Metrics","metadata":{"_uuid":"f07c814d213e814cdac4e3d05d0a8db847fbfe28"}},{"cell_type":"code","source":"# AUC Score\nfrom sklearn.metrics import roc_auc_score\nroc_auc_score(y_true, y_pred)","metadata":{"_uuid":"0b7b7a56c6fa47cc40764d0c06d64860580cbea1","execution":{"iopub.status.busy":"2022-02-01T14:12:37.525490Z","iopub.execute_input":"2022-02-01T14:12:37.525874Z","iopub.status.idle":"2022-02-01T14:12:37.539329Z","shell.execute_reply.started":"2022-02-01T14:12:37.525808Z","shell.execute_reply":"2022-02-01T14:12:37.538152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion Matrix\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","metadata":{"_uuid":"91f570e8e5f07126e5361bbf92929d786e853a09","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-02-01T14:12:38.405873Z","iopub.execute_input":"2022-02-01T14:12:38.406229Z","iopub.status.idle":"2022-02-01T14:12:38.419267Z","shell.execute_reply.started":"2022-02-01T14:12:38.406163Z","shell.execute_reply":"2022-02-01T14:12:38.417670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels of the test images.\ntest_labels = test_gen.classes","metadata":{"_uuid":"c20bc358dc753020d5a560dc56f2350c7f40a4f9","execution":{"iopub.status.busy":"2022-02-01T14:12:39.257936Z","iopub.execute_input":"2022-02-01T14:12:39.258340Z","iopub.status.idle":"2022-02-01T14:12:39.263588Z","shell.execute_reply.started":"2022-02-01T14:12:39.258266Z","shell.execute_reply":"2022-02-01T14:12:39.262446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# argmax returns the index of the max value in a row\ncm = confusion_matrix(test_labels, predictions.argmax(axis=1))","metadata":{"_uuid":"0bb4323e931ff53081994bbf58e82b1ec93ab327","execution":{"iopub.status.busy":"2022-02-01T14:12:40.085786Z","iopub.execute_input":"2022-02-01T14:12:40.086217Z","iopub.status.idle":"2022-02-01T14:12:40.168389Z","shell.execute_reply.started":"2022-02-01T14:12:40.086146Z","shell.execute_reply":"2022-02-01T14:12:40.167427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the labels of the class indices. These need to match the order shown above.\ncm_plot_labels = ['no_tumor_tissue', 'has_tumor_tissue']\nplot_confusion_matrix(cm, cm_plot_labels, title='Confusion Matrix')","metadata":{"_uuid":"ba26c7e718df937a18aa2035c4ba883252e44c79","execution":{"iopub.status.busy":"2022-02-01T14:12:41.062964Z","iopub.execute_input":"2022-02-01T14:12:41.063374Z","iopub.status.idle":"2022-02-01T14:12:41.404115Z","shell.execute_reply.started":"2022-02-01T14:12:41.063308Z","shell.execute_reply":"2022-02-01T14:12:41.402922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Classification Report\nfrom sklearn.metrics import classification_report\n\n# Generate a classification report\n\n# For this to work we need y_pred as binary labels not as probabilities\ny_pred_binary = predictions.argmax(axis=1)\n\nreport = classification_report(y_true, y_pred_binary, target_names=cm_plot_labels)\n\nprint(report)\n","metadata":{"_uuid":"91dcace7eb99aca310774b7a3a55535c9127ce55","execution":{"iopub.status.busy":"2022-02-01T14:12:41.710588Z","iopub.execute_input":"2022-02-01T14:12:41.710974Z","iopub.status.idle":"2022-02-01T14:12:41.724264Z","shell.execute_reply.started":"2022-02-01T14:12:41.710907Z","shell.execute_reply":"2022-02-01T14:12:41.722881Z"},"trusted":true},"execution_count":null,"outputs":[]}]}