{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.utils import shuffle\n\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\n\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntrain_path = '/kaggle/input/histopathologic-cancer-detection/train/'\ntest_path = '/kaggle/input/histopathologic-cancer-detection/test/'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-11T15:03:39.126487Z","iopub.execute_input":"2021-11-11T15:03:39.127039Z","iopub.status.idle":"2021-11-11T15:03:44.719509Z","shell.execute_reply.started":"2021-11-11T15:03:39.126958Z","shell.execute_reply":"2021-11-11T15:03:44.717974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(f'../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\nprint('Training Set Size:', data.shape)\n\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:03:47.491585Z","iopub.execute_input":"2021-11-11T15:03:47.492343Z","iopub.status.idle":"2021-11-11T15:03:48.054692Z","shell.execute_reply.started":"2021-11-11T15:03:47.492294Z","shell.execute_reply":"2021-11-11T15:03:48.05399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['path'] = data.id + '.tif'","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:03:50.555505Z","iopub.execute_input":"2021-11-11T15:03:50.556067Z","iopub.status.idle":"2021-11-11T15:03:50.593326Z","shell.execute_reply.started":"2021-11-11T15:03:50.556027Z","shell.execute_reply":"2021-11-11T15:03:50.592472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:03:52.568561Z","iopub.execute_input":"2021-11-11T15:03:52.569104Z","iopub.status.idle":"2021-11-11T15:03:52.579898Z","shell.execute_reply.started":"2021-11-11T15:03:52.569067Z","shell.execute_reply":"2021-11-11T15:03:52.579161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(data.label.value_counts() / len(data)).to_frame().sort_index().T","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:03:55.020622Z","iopub.execute_input":"2021-11-11T15:03:55.021123Z","iopub.status.idle":"2021-11-11T15:03:55.060535Z","shell.execute_reply.started":"2021-11-11T15:03:55.021084Z","shell.execute_reply":"2021-11-11T15:03:55.059744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = data.sample(n=16).reset_index()\n\nplt.figure(figsize=(8,8))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(train_path+row.path)    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:03:57.967986Z","iopub.execute_input":"2021-11-11T15:03:57.968773Z","iopub.status.idle":"2021-11-11T15:03:59.243762Z","shell.execute_reply.started":"2021-11-11T15:03:57.968732Z","shell.execute_reply":"2021-11-11T15:03:59.242673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:04:02.596085Z","iopub.execute_input":"2021-11-11T15:04:02.596769Z","iopub.status.idle":"2021-11-11T15:04:02.601401Z","shell.execute_reply.started":"2021-11-11T15:04:02.59673Z","shell.execute_reply":"2021-11-11T15:04:02.600674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Use 50% of data - use more once model is working correctly\ntrain, valid = train_test_split(data, test_size=0.5, random_state=1, stratify=data.label)\n\nprint(train.shape)\nprint(valid.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:04:05.655423Z","iopub.execute_input":"2021-11-11T15:04:05.656309Z","iopub.status.idle":"2021-11-11T15:04:06.022673Z","shell.execute_reply.started":"2021-11-11T15:04:05.656263Z","shell.execute_reply":"2021-11-11T15:04:06.021882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create image data generators for both the training set and the validation set. \n# Use the data generators to scale the pixel values by a factor of 1/255. \n\ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train,\n    directory = train_path,\n    x_col = 'path',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    #Rescales images from 96x96\n    target_size = (96,96),\n#    interpolation = 'lanczos:center'\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid,\n    directory = train_path,\n    x_col = 'path',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    #Rescales images from 96x96\n    target_size = (96,96)    \n)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:15:14.023413Z","iopub.execute_input":"2021-11-11T15:15:14.023961Z","iopub.status.idle":"2021-11-11T15:25:26.537755Z","shell.execute_reply.started":"2021-11-11T15:15:14.023922Z","shell.execute_reply":"2021-11-11T15:25:26.536961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:28:43.571462Z","iopub.execute_input":"2021-11-11T15:28:43.571773Z","iopub.status.idle":"2021-11-11T15:28:43.578789Z","shell.execute_reply.started":"2021-11-11T15:28:43.571739Z","shell.execute_reply":"2021-11-11T15:28:43.577942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\ncnn = Sequential([\n    \n    Conv2D(32, (3,3), activation = 'relu', padding = 'same', input_shape=(96,96,3)),\n    Conv2D(32, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.5),\n    BatchNormalization(),\n\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.5),\n    BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(64, activation='relu'),\n    Dropout(0.5),\n    Dense(32, activation='relu'),\n    Dropout(0.5),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:28:47.329985Z","iopub.execute_input":"2021-11-11T15:28:47.330265Z","iopub.status.idle":"2021-11-11T15:28:49.950175Z","shell.execute_reply.started":"2021-11-11T15:28:47.330232Z","shell.execute_reply":"2021-11-11T15:28:49.949539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define an optimizer and select a learning rate. \n# Then compile the model. \n#Lowered learning rate from 0.01 to 0.001\n#Changed metric to AUC\n\nopt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:29:01.224986Z","iopub.execute_input":"2021-11-11T15:29:01.225264Z","iopub.status.idle":"2021-11-11T15:29:01.245552Z","shell.execute_reply.started":"2021-11-11T15:29:01.22523Z","shell.execute_reply":"2021-11-11T15:29:01.244879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n# Complete one or more training runs. \n# Display training curves after each run. \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 10,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-11T15:29:07.697308Z","iopub.execute_input":"2021-11-11T15:29:07.698011Z","iopub.status.idle":"2021-11-11T15:29:16.536798Z","shell.execute_reply.started":"2021-11-11T15:29:07.697974Z","shell.execute_reply":"2021-11-11T15:29:16.536044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = h1.history\nprint(history.keys())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-12T02:20:43.565169Z","iopub.execute_input":"2021-11-12T02:20:43.565695Z","iopub.status.idle":"2021-11-12T02:20:43.650832Z","shell.execute_reply.started":"2021-11-12T02:20:43.565572Z","shell.execute_reply":"2021-11-12T02:20:43.649775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('JW_Cancer_Model_v01.h5')\npickle.dump(history, open(f'JW_Cancer_History_v01.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-11-08T00:02:48.794735Z","iopub.execute_input":"2021-11-08T00:02:48.795232Z","iopub.status.idle":"2021-11-08T00:02:48.86667Z","shell.execute_reply.started":"2021-11-08T00:02:48.79519Z","shell.execute_reply":"2021-11-08T00:02:48.865925Z"},"trusted":true},"execution_count":null,"outputs":[]}]}