{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Histopathologic Cancer Detection","metadata":{}},{"cell_type":"markdown","source":"# Import packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras import backend as K\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:32.993725Z","iopub.execute_input":"2021-12-13T12:07:32.994032Z","iopub.status.idle":"2021-12-13T12:07:38.624700Z","shell.execute_reply.started":"2021-12-13T12:07:32.993950Z","shell.execute_reply":"2021-12-13T12:07:38.623747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 64\nIMG_SIZE = 96\nRANDOM_SEED = 1982","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:38.626276Z","iopub.execute_input":"2021-12-13T12:07:38.626552Z","iopub.status.idle":"2021-12-13T12:07:38.635289Z","shell.execute_reply.started":"2021-12-13T12:07:38.626516Z","shell.execute_reply":"2021-12-13T12:07:38.632107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"dataset = '/kaggle/input/histopathologic-cancer-detection/'\ntrain_path = dataset+'train/'\ntest_path = dataset+'test/'","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:38.636640Z","iopub.execute_input":"2021-12-13T12:07:38.637296Z","iopub.status.idle":"2021-12-13T12:07:38.646850Z","shell.execute_reply.started":"2021-12-13T12:07:38.637255Z","shell.execute_reply":"2021-12-13T12:07:38.646105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(dataset+'train_labels.csv', dtype=str)\nprint('Training Set Size:', data.shape)\ndata['path'] = data.id + '.tif'\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:38.649550Z","iopub.execute_input":"2021-12-13T12:07:38.650219Z","iopub.status.idle":"2021-12-13T12:07:39.172041Z","shell.execute_reply.started":"2021-12-13T12:07:38.650177Z","shell.execute_reply":"2021-12-13T12:07:39.171361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Split","metadata":{}},{"cell_type":"code","source":"train, valid = train_test_split(data, test_size=0.2, random_state=RANDOM_SEED, stratify=data.label)\n\nprint(train.shape)\nprint(valid.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:39.173315Z","iopub.execute_input":"2021-12-13T12:07:39.174007Z","iopub.status.idle":"2021-12-13T12:07:39.565625Z","shell.execute_reply.started":"2021-12-13T12:07:39.173966Z","shell.execute_reply":"2021-12-13T12:07:39.564721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generators\nIn this section, we will split the labeled observations into training and validation sets. We will then create data loaders to feed the images into our neural network during training.","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:39.567053Z","iopub.execute_input":"2021-12-13T12:07:39.567451Z","iopub.status.idle":"2021-12-13T12:07:39.572479Z","shell.execute_reply.started":"2021-12-13T12:07:39.567391Z","shell.execute_reply":"2021-12-13T12:07:39.571586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = train_datagen.flow_from_dataframe(\n    dataframe = train,\n    directory = train_path,\n    x_col = 'path',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = RANDOM_SEED,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (IMG_SIZE,IMG_SIZE)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid,\n    directory = train_path,\n    x_col = 'path',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = RANDOM_SEED,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (IMG_SIZE,IMG_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:07:39.574151Z","iopub.execute_input":"2021-12-13T12:07:39.574448Z","iopub.status.idle":"2021-12-13T12:11:38.893920Z","shell.execute_reply.started":"2021-12-13T12:07:39.574413Z","shell.execute_reply":"2021-12-13T12:11:38.893013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:38.895290Z","iopub.execute_input":"2021-12-13T12:11:38.895559Z","iopub.status.idle":"2021-12-13T12:11:38.903214Z","shell.execute_reply.started":"2021-12-13T12:11:38.895523Z","shell.execute_reply":"2021-12-13T12:11:38.902287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN Model using VGG16","metadata":{}},{"cell_type":"code","source":"np.random.seed(RANDOM_SEED)\ntf.random.set_seed(RANDOM_SEED)\n\ncnn = tf.keras.models.load_model('../input/cancerdetection-tl-v02/cancer_model_v01.h5')","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:38.904405Z","iopub.execute_input":"2021-12-13T12:11:38.904735Z","iopub.status.idle":"2021-12-13T12:11:43.749515Z","shell.execute_reply.started":"2021-12-13T12:11:38.904649Z","shell.execute_reply":"2021-12-13T12:11:43.748737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = tf.keras.applications.VGG16(input_shape=(96,96,3),\n                                         include_top=False,\n                                         weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:43.752195Z","iopub.execute_input":"2021-12-13T12:11:43.752469Z","iopub.status.idle":"2021-12-13T12:11:44.885896Z","shell.execute_reply.started":"2021-12-13T12:11:43.752434Z","shell.execute_reply":"2021-12-13T12:11:44.885120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable = True\nfor layer in base_model.layers[:-8]:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:44.890226Z","iopub.execute_input":"2021-12-13T12:11:44.892740Z","iopub.status.idle":"2021-12-13T12:11:44.900943Z","shell.execute_reply.started":"2021-12-13T12:11:44.892108Z","shell.execute_reply":"2021-12-13T12:11:44.899964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(0.0001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:44.902011Z","iopub.execute_input":"2021-12-13T12:11:44.902441Z","iopub.status.idle":"2021-12-13T12:11:44.928494Z","shell.execute_reply.started":"2021-12-13T12:11:44.902406Z","shell.execute_reply":"2021-12-13T12:11:44.927847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:44.930900Z","iopub.execute_input":"2021-12-13T12:11:44.931082Z","iopub.status.idle":"2021-12-13T12:11:44.937200Z","shell.execute_reply.started":"2021-12-13T12:11:44.931059Z","shell.execute_reply":"2021-12-13T12:11:44.936224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"%%time \n\n# Complete one or more training runs. \n# Display training curves after each run. \n\nh2 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 20,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1,\n    use_multiprocessing=True, \n    workers=8\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-13T12:11:44.938643Z","iopub.execute_input":"2021-12-13T12:11:44.938909Z","iopub.status.idle":"2021-12-13T13:33:03.806975Z","shell.execute_reply.started":"2021-12-13T12:11:44.938875Z","shell.execute_reply":"2021-12-13T13:33:03.805911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickle_file = open(\"../input/cancerdetection-tl-v02/cancer_history_v00.pkl\", \"rb\")\nhistory = pickle.load(pickle_file)\npickle_file.close()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T13:33:03.809764Z","iopub.execute_input":"2021-12-13T13:33:03.810545Z","iopub.status.idle":"2021-12-13T13:33:03.824828Z","shell.execute_reply.started":"2021-12-13T13:33:03.810499Z","shell.execute_reply":"2021-12-13T13:33:03.824049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in h2.history.keys():\n    history[k] += h2.history[k]\n    \nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T13:33:03.826255Z","iopub.execute_input":"2021-12-13T13:33:03.826993Z","iopub.status.idle":"2021-12-13T13:33:04.980829Z","shell.execute_reply.started":"2021-12-13T13:33:03.826954Z","shell.execute_reply":"2021-12-13T13:33:04.980163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Model","metadata":{}},{"cell_type":"code","source":"cnn.save('LK_HCD_CNN_TLV3_Model.h5')\npickle.dump(history, open(f'LP_HCD_CNN_Model_V3_History.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-12-13T13:33:04.982208Z","iopub.execute_input":"2021-12-13T13:33:04.982684Z","iopub.status.idle":"2021-12-13T13:33:05.148913Z","shell.execute_reply.started":"2021-12-13T13:33:04.982645Z","shell.execute_reply":"2021-12-13T13:33:05.148185Z"},"trusted":true},"execution_count":null,"outputs":[]}]}