{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pandas as pd\nimport pickle\nimport gc\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import shuffle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport zipfile ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-10T03:20:15.957659Z","iopub.execute_input":"2021-12-10T03:20:15.958011Z","iopub.status.idle":"2021-12-10T03:20:21.443999Z","shell.execute_reply.started":"2021-12-10T03:20:15.957906Z","shell.execute_reply":"2021-12-10T03:20:21.443234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:21.445727Z","iopub.execute_input":"2021-12-10T03:20:21.446040Z","iopub.status.idle":"2021-12-10T03:20:21.592312Z","shell.execute_reply.started":"2021-12-10T03:20:21.445950Z","shell.execute_reply":"2021-12-10T03:20:21.591709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing Training Labels","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\", dtype=str)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:21.593282Z","iopub.execute_input":"2021-12-10T03:20:21.593656Z","iopub.status.idle":"2021-12-10T03:20:22.073903Z","shell.execute_reply.started":"2021-12-10T03:20:21.593622Z","shell.execute_reply":"2021-12-10T03:20:22.073141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:22.075783Z","iopub.execute_input":"2021-12-10T03:20:22.076483Z","iopub.status.idle":"2021-12-10T03:20:22.091263Z","shell.execute_reply.started":"2021-12-10T03:20:22.076444Z","shell.execute_reply":"2021-12-10T03:20:22.090509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Seeing the Distribution of Labels","metadata":{}},{"cell_type":"code","source":"y_train = train.label\n\n(train.label.value_counts() / len(train)).to_frame().T","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:22.092507Z","iopub.execute_input":"2021-12-10T03:20:22.092762Z","iopub.status.idle":"2021-12-10T03:20:22.128887Z","shell.execute_reply.started":"2021-12-10T03:20:22.092730Z","shell.execute_reply":"2021-12-10T03:20:22.128135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sampling a Few Images","metadata":{}},{"cell_type":"code","source":"# Sample 16 images from the training set and display these along with their labels.\n\nplt.figure(figsize=(10,10)) # specifying the overall grid size\n\nfor i in range(16):\n    plt.subplot(4,4,i+1)    # the number of images in the grid is 6*6 (16)\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{train[\"id\"][i]}.tif')\n    plt.imshow(img)\n    plt.text(0, -5, f'Label {train[\"label\"][i]}')\n    plt.axis('off')\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:22.129949Z","iopub.execute_input":"2021-12-10T03:20:22.130636Z","iopub.status.idle":"2021-12-10T03:20:23.248278Z","shell.execute_reply.started":"2021-12-10T03:20:22.130598Z","shell.execute_reply":"2021-12-10T03:20:23.247504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Taking Even Amount of Neg and Pos Labels","metadata":{}},{"cell_type":"code","source":"train_neg = train[train['label']=='0'].sample(10000,random_state=1)\ntrain_pos = train[train['label']=='1'].sample(10000,random_state=1)\n\ntrain_data = pd.concat([train_neg, train_pos], axis=0).reset_index(drop=True)\n\ntrain = shuffle(train_data)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.249419Z","iopub.execute_input":"2021-12-10T03:20:23.249672Z","iopub.status.idle":"2021-12-10T03:20:23.348276Z","shell.execute_reply.started":"2021-12-10T03:20:23.249638Z","shell.execute_reply":"2021-12-10T03:20:23.347539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.349684Z","iopub.execute_input":"2021-12-10T03:20:23.349991Z","iopub.status.idle":"2021-12-10T03:20:23.363375Z","shell.execute_reply.started":"2021-12-10T03:20:23.349955Z","shell.execute_reply":"2021-12-10T03:20:23.362643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to apply the .tif extension\ndef append_ext(fn):\n    return fn+\".tif\"\n\n\ntrain['id'] = train['id'].apply(append_ext)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.365026Z","iopub.execute_input":"2021-12-10T03:20:23.365525Z","iopub.status.idle":"2021-12-10T03:20:23.383964Z","shell.execute_reply.started":"2021-12-10T03:20:23.365488Z","shell.execute_reply":"2021-12-10T03:20:23.383185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Splitting the Data","metadata":{}},{"cell_type":"code","source":"# Split the dataframe train into two DataFrames named train_df and valid_df. \n# Use 20% of the data for the validation set. \n# Use stratified sampling so that the label proportions are preserved.\n# Set a random seed for the split. \n\ntrain_df, valid_df = train_test_split(train, test_size=0.2, random_state=1, stratify=train.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.387673Z","iopub.execute_input":"2021-12-10T03:20:23.388209Z","iopub.status.idle":"2021-12-10T03:20:23.426234Z","shell.execute_reply.started":"2021-12-10T03:20:23.388145Z","shell.execute_reply":"2021-12-10T03:20:23.425435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Datagenerators","metadata":{}},{"cell_type":"code","source":"# Create image data generators for both the training set and the validation set. \n# Use the data generators to scale the pixel values by a factor of 1/255. \ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.427399Z","iopub.execute_input":"2021-12-10T03:20:23.427722Z","iopub.status.idle":"2021-12-10T03:20:23.432327Z","shell.execute_reply.started":"2021-12-10T03:20:23.427685Z","shell.execute_reply":"2021-12-10T03:20:23.431584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Complete the code for the data loaders below. \n\nBATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = '../input/histopathologic-cancer-detection/train/',\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (32,32)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = '../input/histopathologic-cancer-detection/train/',\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (32,32)\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:23.433751Z","iopub.execute_input":"2021-12-10T03:20:23.434658Z","iopub.status.idle":"2021-12-10T03:20:44.657995Z","shell.execute_reply.started":"2021-12-10T03:20:23.434620Z","shell.execute_reply":"2021-12-10T03:20:44.656138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run this cell to determine the number of training and validation batches. \n\nTR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:44.659394Z","iopub.execute_input":"2021-12-10T03:20:44.659659Z","iopub.status.idle":"2021-12-10T03:20:44.665274Z","shell.execute_reply.started":"2021-12-10T03:20:44.659623Z","shell.execute_reply":"2021-12-10T03:20:44.664302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building the CNN","metadata":{}},{"cell_type":"code","source":"# Use this cell to construct a convolutional neural network model. \n# Your model should make use of each of the following layer types:\n#    Conv2D, MaxPooling2D, Dropout, BatchNormalization, Flatten, Dense\n# You can start by mimicking the architecture used in the \n# Aerial Cactus competetition, but you should explore different architectures\n# by adding more layers and/or adding more nodes in individual layers\n\nnp.random.seed(1)\ntf.random.set_seed(1)\n\ncnn1 = Sequential([\n    Conv2D(32, (3,3), activation = 'relu', padding = 'same', input_shape=(32,32,3)),\n    BatchNormalization(),\n    Conv2D(32, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.2),\n    BatchNormalization(),\n\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    BatchNormalization(),\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.4),\n    BatchNormalization(),\n    \n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    BatchNormalization(),\n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.5),\n    BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(16, activation='relu'),\n    Dropout(0.2),\n    BatchNormalization(),\n    # we have 2 here because we have 2 classes\n    Dense(2, activation='softmax')\n])\n\ncnn1.summary()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:44.666875Z","iopub.execute_input":"2021-12-10T03:20:44.667396Z","iopub.status.idle":"2021-12-10T03:20:47.415715Z","shell.execute_reply.started":"2021-12-10T03:20:44.667359Z","shell.execute_reply":"2021-12-10T03:20:47.415030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(0.001)\ncnn1.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:47.416925Z","iopub.execute_input":"2021-12-10T03:20:47.417175Z","iopub.status.idle":"2021-12-10T03:20:47.436595Z","shell.execute_reply.started":"2021-12-10T03:20:47.417140Z","shell.execute_reply":"2021-12-10T03:20:47.435965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fitting the CNN","metadata":{}},{"cell_type":"code","source":"%%time \n\nh1 = cnn1.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 20,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:20:47.437829Z","iopub.execute_input":"2021-12-10T03:20:47.438092Z","iopub.status.idle":"2021-12-10T03:29:15.463475Z","shell.execute_reply.started":"2021-12-10T03:20:47.438060Z","shell.execute_reply":"2021-12-10T03:29:15.462656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = h1.history\nprint(history.keys())","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:29:15.466336Z","iopub.execute_input":"2021-12-10T03:29:15.466617Z","iopub.status.idle":"2021-12-10T03:29:15.471260Z","shell.execute_reply.started":"2021-12-10T03:29:15.466581Z","shell.execute_reply":"2021-12-10T03:29:15.470425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Graphing the Results","metadata":{}},{"cell_type":"code","source":"# Graph the result\n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\n\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\n\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\n\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:29:15.472853Z","iopub.execute_input":"2021-12-10T03:29:15.473454Z","iopub.status.idle":"2021-12-10T03:29:16.377407Z","shell.execute_reply.started":"2021-12-10T03:29:15.473413Z","shell.execute_reply":"2021-12-10T03:29:16.376749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Another Training Run to Smooth out Validation Graph","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.set_value(cnn1.optimizer.learning_rate, 0.0001)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:29:16.380432Z","iopub.execute_input":"2021-12-10T03:29:16.382499Z","iopub.status.idle":"2021-12-10T03:29:16.389495Z","shell.execute_reply.started":"2021-12-10T03:29:16.382459Z","shell.execute_reply":"2021-12-10T03:29:16.388764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh2 = cnn1.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 20,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:29:16.394311Z","iopub.execute_input":"2021-12-10T03:29:16.396565Z","iopub.status.idle":"2021-12-10T03:37:16.437433Z","shell.execute_reply.started":"2021-12-10T03:29:16.396523Z","shell.execute_reply":"2021-12-10T03:37:16.436732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Graph the result\n\nfor k in history.keys():\n    history[k] += h2.history[k]\n    \n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\n\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\n\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\n\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:37:16.439196Z","iopub.execute_input":"2021-12-10T03:37:16.439614Z","iopub.status.idle":"2021-12-10T03:37:16.985510Z","shell.execute_reply.started":"2021-12-10T03:37:16.439574Z","shell.execute_reply":"2021-12-10T03:37:16.984829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One Last Training Run","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.set_value(cnn1.optimizer.learning_rate, 0.0001)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:37:16.986874Z","iopub.execute_input":"2021-12-10T03:37:16.987344Z","iopub.status.idle":"2021-12-10T03:37:16.992816Z","shell.execute_reply.started":"2021-12-10T03:37:16.987305Z","shell.execute_reply":"2021-12-10T03:37:16.991702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh3 = cnn1.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 20,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:37:16.994020Z","iopub.execute_input":"2021-12-10T03:37:16.994257Z","iopub.status.idle":"2021-12-10T03:44:54.817507Z","shell.execute_reply.started":"2021-12-10T03:37:16.994222Z","shell.execute_reply":"2021-12-10T03:44:54.816050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Graph the result\nfor k in history.keys():\n    history[k] += h2.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\n\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\n\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\n\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:44:54.819196Z","iopub.execute_input":"2021-12-10T03:44:54.819445Z","iopub.status.idle":"2021-12-10T03:44:55.350468Z","shell.execute_reply.started":"2021-12-10T03:44:54.819408Z","shell.execute_reply":"2021-12-10T03:44:55.349781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Saving the Model","metadata":{}},{"cell_type":"code","source":"cnn1.save('cancer_model15.h5')\npickle.dump(history, open(f'cancer_history15.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-12-10T03:44:55.351506Z","iopub.execute_input":"2021-12-10T03:44:55.351853Z","iopub.status.idle":"2021-12-10T03:44:55.452401Z","shell.execute_reply.started":"2021-12-10T03:44:55.351821Z","shell.execute_reply":"2021-12-10T03:44:55.451686Z"},"trusted":true},"execution_count":null,"outputs":[]}]}