{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport cv2 as cv\nfrom numpy.random import seed\nseed(45)\nimport pickle\n\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom glob import glob \nfrom tensorflow.keras import layers\nfrom tensorflow.keras import backend as K\n\n%matplotlib inline","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-11-30T06:42:51.373309Z","iopub.execute_input":"2021-11-30T06:42:51.373699Z","iopub.status.idle":"2021-11-30T06:42:51.386667Z","shell.execute_reply.started":"2021-11-30T06:42:51.373657Z","shell.execute_reply":"2021-11-30T06:42:51.385509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transfer learning with fine-tuning -- week7\nversion7","metadata":{}},{"cell_type":"markdown","source":"## Helper Functions\nThe functions below are used for merging the contents of Keras history objects, and for displaying training curves. We will use these after each training run.","metadata":{}},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.388881Z","iopub.execute_input":"2021-11-30T06:42:51.389305Z","iopub.status.idle":"2021-11-30T06:42:51.40407Z","shell.execute_reply.started":"2021-11-30T06:42:51.389263Z","shell.execute_reply":"2021-11-30T06:42:51.403133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dirname = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.405867Z","iopub.execute_input":"2021-11-30T06:42:51.406267Z","iopub.status.idle":"2021-11-30T06:42:51.420161Z","shell.execute_reply.started":"2021-11-30T06:42:51.406232Z","shell.execute_reply":"2021-11-30T06:42:51.419181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset exploration","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.42102Z","iopub.execute_input":"2021-11-30T06:42:51.421186Z","iopub.status.idle":"2021-11-30T06:42:51.650931Z","shell.execute_reply.started":"2021-11-30T06:42:51.421167Z","shell.execute_reply":"2021-11-30T06:42:51.650303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label distribution","metadata":{}},{"cell_type":"code","source":"train_labels['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.654424Z","iopub.execute_input":"2021-11-30T06:42:51.654886Z","iopub.status.idle":"2021-11-30T06:42:51.667632Z","shell.execute_reply.started":"2021-11-30T06:42:51.654848Z","shell.execute_reply":"2021-11-30T06:42:51.666789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display a DataFrame showing the proportion of observations with each \n# possible of the target variable (which is label). \n(train_labels.label.value_counts() / len(train_labels)).to_frame()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.671033Z","iopub.execute_input":"2021-11-30T06:42:51.671311Z","iopub.status.idle":"2021-11-30T06:42:51.689742Z","shell.execute_reply.started":"2021-11-30T06:42:51.671276Z","shell.execute_reply":"2021-11-30T06:42:51.689053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.info()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.693473Z","iopub.execute_input":"2021-11-30T06:42:51.693934Z","iopub.status.idle":"2021-11-30T06:42:51.7241Z","shell.execute_reply.started":"2021-11-30T06:42:51.693894Z","shell.execute_reply":"2021-11-30T06:42:51.722351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data is not entirely balanced, there is more negative samples than positive, by about 30 percent","metadata":{}},{"cell_type":"markdown","source":"# View Sample Images","metadata":{}},{"cell_type":"code","source":"positive_samples = train_labels.loc[train_labels['label'] == 1].sample(4)\nnegative_samples = train_labels.loc[train_labels['label'] == 0].sample(4)\npositive_images = []\nnegative_images = []\nfor sample in positive_samples['id']:\n    path = os.path.join(train_dirname, sample+'.tif')\n    img = cv.imread(path)\n    positive_images.append(img)\n        \nfor sample in negative_samples['id']:\n    path = os.path.join(train_dirname, sample+'.tif')\n    img = cv.imread(path)\n    negative_images.append(img)\n\nfig,axis = plt.subplots(2,4,figsize=(20,8))\nfig.suptitle('Dataset samples presentation plot',fontsize=20)\nfor i,img in enumerate(positive_images):\n    axis[0,i].imshow(img)\n    rect = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='g',facecolor='none', linestyle=':', capstyle='round')\n    axis[0,i].add_patch(rect)\naxis[0,0].set_ylabel('Positive samples', size='large')\nfor i,img in enumerate(negative_images):\n    axis[1,i].imshow(img)\n    rect = patches.Rectangle((32,32),32,32,linewidth=4,edgecolor='r',facecolor='none', linestyle=':', capstyle='round')\n    axis[1,i].add_patch(rect)\naxis[1,0].set_ylabel('Negative samples', size='large')\n    ","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:51.726329Z","iopub.execute_input":"2021-11-30T06:42:51.72663Z","iopub.status.idle":"2021-11-30T06:42:52.790173Z","shell.execute_reply.started":"2021-11-30T06:42:51.726579Z","shell.execute_reply":"2021-11-30T06:42:52.789441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting dataset","metadata":{}},{"cell_type":"markdown","source":"# Setting up learning constants","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 96\nIMG_CHANNELS = 3\n#TRAIN_SIZE=80000\nTRAIN_SIZE = 10000\nBATCH_SIZE = 64\nEPOCHS = 30","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.791381Z","iopub.execute_input":"2021-11-30T06:42:52.791783Z","iopub.status.idle":"2021-11-30T06:42:52.79664Z","shell.execute_reply.started":"2021-11-30T06:42:52.791748Z","shell.execute_reply":"2021-11-30T06:42:52.795705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Balancing the dataset","metadata":{}},{"cell_type":"code","source":"train_neg = train_labels[train_labels['label']==0].sample(TRAIN_SIZE,random_state=45)\ntrain_pos = train_labels[train_labels['label']==1].sample(TRAIN_SIZE,random_state=45)\n\ntrain_data = pd.concat([train_neg, train_pos], axis=0).reset_index(drop=True)\n\ntrain_data = shuffle(train_data)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.798617Z","iopub.execute_input":"2021-11-30T06:42:52.799096Z","iopub.status.idle":"2021-11-30T06:42:52.844687Z","shell.execute_reply.started":"2021-11-30T06:42:52.799052Z","shell.execute_reply":"2021-11-30T06:42:52.843903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.846683Z","iopub.execute_input":"2021-11-30T06:42:52.846965Z","iopub.status.idle":"2021-11-30T06:42:52.857017Z","shell.execute_reply.started":"2021-11-30T06:42:52.846932Z","shell.execute_reply":"2021-11-30T06:42:52.856123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def append_ext(fn):\n    return fn+\".tif\"","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.859158Z","iopub.execute_input":"2021-11-30T06:42:52.859529Z","iopub.status.idle":"2021-11-30T06:42:52.865188Z","shell.execute_reply.started":"2021-11-30T06:42:52.859495Z","shell.execute_reply":"2021-11-30T06:42:52.864282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Splitting the dataset","metadata":{}},{"cell_type":"code","source":"#y = train_data['label']\n#train_df, val_df = train_test_split(train_data, test_size=0.3, random_state=45, stratify=y)\n#y = val_df['label']\n#val_df, test_df = train_test_split(val_df, test_size=0.5, random_state=45, stratify=y)\n#print(train_df.shape)\n#print(val_df.shape)\n#print(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.867157Z","iopub.execute_input":"2021-11-30T06:42:52.867724Z","iopub.status.idle":"2021-11-30T06:42:52.873838Z","shell.execute_reply.started":"2021-11-30T06:42:52.867688Z","shell.execute_reply":"2021-11-30T06:42:52.872888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_data['label']\ntrain_df, valid_df = train_test_split(train_data, test_size=0.2, random_state=45, stratify=y)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.875401Z","iopub.execute_input":"2021-11-30T06:42:52.875756Z","iopub.status.idle":"2021-11-30T06:42:52.906332Z","shell.execute_reply.started":"2021-11-30T06:42:52.875723Z","shell.execute_reply":"2021-11-30T06:42:52.905653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['id'] = train_df['id'].apply(append_ext)\nvalid_df['id'] = valid_df['id'].apply(append_ext)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.908029Z","iopub.execute_input":"2021-11-30T06:42:52.908358Z","iopub.status.idle":"2021-11-30T06:42:52.936872Z","shell.execute_reply.started":"2021-11-30T06:42:52.90832Z","shell.execute_reply":"2021-11-30T06:42:52.936128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image generators for the simple CNN model","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.937982Z","iopub.execute_input":"2021-11-30T06:42:52.938402Z","iopub.status.idle":"2021-11-30T06:42:52.943363Z","shell.execute_reply.started":"2021-11-30T06:42:52.938353Z","shell.execute_reply":"2021-11-30T06:42:52.94255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\ntrain_path = '../input/histopathologic-cancer-detection/train'\ntrain_df['label'] = train_df['label'].astype(str)\nvalid_df['label'] = valid_df['label'].astype(str)\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (64,64)\n)\n\nvalid_loader = valid_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (64,64)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:42:52.945146Z","iopub.execute_input":"2021-11-30T06:42:52.945425Z","iopub.status.idle":"2021-11-30T06:43:00.558852Z","shell.execute_reply.started":"2021-11-30T06:42:52.945392Z","shell.execute_reply":"2021-11-30T06:43:00.557977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:00.560427Z","iopub.execute_input":"2021-11-30T06:43:00.561038Z","iopub.status.idle":"2021-11-30T06:43:00.568098Z","shell.execute_reply.started":"2021-11-30T06:43:00.560998Z","shell.execute_reply":"2021-11-30T06:43:00.567078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build Network","metadata":{}},{"cell_type":"markdown","source":"In this section, we will construct our neural network. For feature extraction, we will use the VGG16 model, as trained on the ImageNet dataset.\n\nIn the cell below, we will load the pretrained VGG16 model into a variable named base_model. We will set include_top=False to indicate that we only wish to use the convolutional blocks that appear before the Flatten() layer. We will not include the dense layers composing the classifier at the top of the network. Instead, we will design and train our own classifier.\n\nWe set the input_shape parameter to indicate the shape of the images that we will be feeding into the network.\n\nFinally, we set the trainable parameter of the model to False. This tells Keras that we do not wish to update the weights in the base layer during training. We only wish to train the new classifier that we will design.","metadata":{}},{"cell_type":"code","source":"base_model = tf.keras.applications.VGG16(input_shape=(64,64,3),\n                                         include_top=False,\n                                         weights='imagenet')\n\nbase_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:00.570001Z","iopub.execute_input":"2021-11-30T06:43:00.570788Z","iopub.status.idle":"2021-11-30T06:43:00.865498Z","shell.execute_reply.started":"2021-11-30T06:43:00.570752Z","shell.execute_reply":"2021-11-30T06:43:00.864712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before moving forward, let's take a look at the structure of our base model. Notice that it consists of 5 convolutional blocks, some of which contain 2 convolutional layers, and some of which contain 3. Also note that none of the weights in the model are trainable (since we have set them to not be).","metadata":{}},{"cell_type":"code","source":"base_model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:00.867046Z","iopub.execute_input":"2021-11-30T06:43:00.867318Z","iopub.status.idle":"2021-11-30T06:43:00.882831Z","shell.execute_reply.started":"2021-11-30T06:43:00.867285Z","shell.execute_reply":"2021-11-30T06:43:00.876481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VGG16 is one of many pretrained models that we could have used. Common choices include VGG16, VGG19, ResNet50, and InceptionV3. A full list of the pretrained models provided by Keras can be found here: Keras Applications\n\nWe are now ready to build a classifier for our neural network. In the cell below, we include base_model in the network as if were a single layer.","metadata":{}},{"cell_type":"code","source":"cnn = Sequential([\n    base_model,\n    \n    Flatten(),\n    \n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(16, activation='relu'),\n    Dropout(0.25),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:00.884068Z","iopub.execute_input":"2021-11-30T06:43:00.884333Z","iopub.status.idle":"2021-11-30T06:43:01.049504Z","shell.execute_reply.started":"2021-11-30T06:43:00.884301Z","shell.execute_reply":"2021-11-30T06:43:01.048697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Network","metadata":{}},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:01.050723Z","iopub.execute_input":"2021-11-30T06:43:01.05098Z","iopub.status.idle":"2021-11-30T06:43:01.15166Z","shell.execute_reply.started":"2021-11-30T06:43:01.050949Z","shell.execute_reply":"2021-11-30T06:43:01.15106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Run 1","metadata":{}},{"cell_type":"code","source":"%%time \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 15,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:01.155441Z","iopub.execute_input":"2021-11-30T06:43:01.15571Z","iopub.status.idle":"2021-11-30T06:43:49.205183Z","shell.execute_reply.started":"2021-11-30T06:43:01.155685Z","shell.execute_reply":"2021-11-30T06:43:49.204542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = h1.history\nprint(history.keys())","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.207096Z","iopub.execute_input":"2021-11-30T06:43:49.207401Z","iopub.status.idle":"2021-11-30T06:43:49.216186Z","shell.execute_reply.started":"2021-11-30T06:43:49.207366Z","shell.execute_reply":"2021-11-30T06:43:49.215208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.217722Z","iopub.execute_input":"2021-11-30T06:43:49.219006Z","iopub.status.idle":"2021-11-30T06:43:49.66514Z","shell.execute_reply.started":"2021-11-30T06:43:49.21897Z","shell.execute_reply":"2021-11-30T06:43:49.661951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Training Run 2\ntf.keras.backend.set_value(cnn.optimizer.learning_rate, 0.0001)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.666356Z","iopub.status.idle":"2021-11-30T06:43:49.667062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh2 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 15,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.668344Z","iopub.status.idle":"2021-11-30T06:43:49.669026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in history.keys():\n    history[k] += h2.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.670247Z","iopub.status.idle":"2021-11-30T06:43:49.670933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine-Tuning the Model","metadata":{}},{"cell_type":"markdown","source":"### Fine-tuning is a technique that allows us to adjust the layers in a pretrained convolutional base to let it adapt to the new dataset it is being applied to. This is accomplished by unfreezing the layers in the base, setting a low learning rate, and then training for some number of additional epochs.","metadata":{}},{"cell_type":"code","source":"base_model.trainable = True\nK.set_value(cnn.optimizer.learning_rate, 0.00001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.672142Z","iopub.status.idle":"2021-11-30T06:43:49.672815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Notice that when we view a summary of the model, we see that the number of trainable parameters has increased significantly. This is the result of unfreezing the convolutional base.","metadata":{}},{"cell_type":"code","source":"cnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.674028Z","iopub.status.idle":"2021-11-30T06:43:49.674704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will now train all layers of the model with a low learning rate.","metadata":{}},{"cell_type":"code","source":"h3 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    epochs = 10,\n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.675924Z","iopub.status.idle":"2021-11-30T06:43:49.676578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(history.keys())\nprint(h3.history.keys())","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.677814Z","iopub.status.idle":"2021-11-30T06:43:49.678498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(h3.history)\n\nh3.history['auc'] = h3.history['auc_1']\ndel h3.history['auc_1']\n\nh3.history['val_auc'] = h3.history['val_auc_1']\ndel h3.history['val_auc_1']\nprint(h3.history)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.679743Z","iopub.status.idle":"2021-11-30T06:43:49.680426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in history.keys():\n    history[k] += h3.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.68167Z","iopub.status.idle":"2021-11-30T06:43:49.682361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save Model and History","metadata":{}},{"cell_type":"code","source":"cnn.save('cancer_detection_model_v16.h5')\npickle.dump(history, open(f'cancer_detection_history_v16.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-11-30T06:43:49.683614Z","iopub.status.idle":"2021-11-30T06:43:49.684294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## ","metadata":{}}]}