{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import namespaces","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\n\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport os\nfrom tensorflow.keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-20T23:06:07.265979Z","iopub.execute_input":"2021-11-20T23:06:07.266626Z","iopub.status.idle":"2021-11-20T23:06:12.016591Z","shell.execute_reply.started":"2021-11-20T23:06:07.266536Z","shell.execute_reply":"2021-11-20T23:06:12.015817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"# Load the training data into a DataFrame named 'train'. \n# Print the shape of the resulting DataFrame. \n# You do not need the test data in this notebook. \n\ntrain = pd.read_csv(f'../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\nprint('Training Set Size:', train.shape)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:12.02012Z","iopub.execute_input":"2021-11-20T23:06:12.020694Z","iopub.status.idle":"2021-11-20T23:06:12.684163Z","shell.execute_reply.started":"2021-11-20T23:06:12.020663Z","shell.execute_reply":"2021-11-20T23:06:12.68346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Lets play with 1% data to check if all code works\n# # Comment this when running the entire code\n# ignore, train = train_test_split(train, test_size=0.01, random_state=1, stratify=train.label)\n# print('Training Set Size:', train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:12.685613Z","iopub.execute_input":"2021-11-20T23:06:12.686093Z","iopub.status.idle":"2021-11-20T23:06:13.062046Z","shell.execute_reply.started":"2021-11-20T23:06:12.686045Z","shell.execute_reply":"2021-11-20T23:06:13.061267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets update the dataset to include filename extensions","metadata":{}},{"cell_type":"code","source":"train['id'] = train['id'].apply(lambda x: f'{x}.tif')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:13.064082Z","iopub.execute_input":"2021-11-20T23:06:13.064818Z","iopub.status.idle":"2021-11-20T23:06:13.075468Z","shell.execute_reply.started":"2021-11-20T23:06:13.064776Z","shell.execute_reply":"2021-11-20T23:06:13.074799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"(train.label.value_counts() / len(train)).to_frame().sort_index().T","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:13.076702Z","iopub.execute_input":"2021-11-20T23:06:13.077454Z","iopub.status.idle":"2021-11-20T23:06:13.095584Z","shell.execute_reply.started":"2021-11-20T23:06:13.07742Z","shell.execute_reply":"2021-11-20T23:06:13.094859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View Sample of Images","metadata":{}},{"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train\"\nprint('Training Images:', len(os.listdir(train_path)))\n\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize=(8,8))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{row.id}')    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:13.096733Z","iopub.execute_input":"2021-11-20T23:06:13.097034Z","iopub.status.idle":"2021-11-20T23:06:17.646397Z","shell.execute_reply.started":"2021-11-20T23:06:13.097Z","shell.execute_reply":"2021-11-20T23:06:17.645649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generators","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, random_state=1, stratify=train.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:17.647482Z","iopub.execute_input":"2021-11-20T23:06:17.647751Z","iopub.status.idle":"2021-11-20T23:06:17.661313Z","shell.execute_reply.started":"2021-11-20T23:06:17.647712Z","shell.execute_reply":"2021-11-20T23:06:17.660624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create image data generators for both the training set and the validation set. \n# Use the data generators to scale the pixel values by a factor of 1/255. \n\ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:17.662688Z","iopub.execute_input":"2021-11-20T23:06:17.663155Z","iopub.status.idle":"2021-11-20T23:06:17.668101Z","shell.execute_reply.started":"2021-11-20T23:06:17.663117Z","shell.execute_reply":"2021-11-20T23:06:17.667475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Complete the code for the data loaders below. \n\nBATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:17.669281Z","iopub.execute_input":"2021-11-20T23:06:17.670278Z","iopub.status.idle":"2021-11-20T23:06:22.801578Z","shell.execute_reply.started":"2021-11-20T23:06:17.670235Z","shell.execute_reply":"2021-11-20T23:06:22.800814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.804486Z","iopub.execute_input":"2021-11-20T23:06:22.804714Z","iopub.status.idle":"2021-11-20T23:06:22.810078Z","shell.execute_reply.started":"2021-11-20T23:06:22.804682Z","shell.execute_reply":"2021-11-20T23:06:22.809246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Network","metadata":{}},{"cell_type":"code","source":"SEED = 1\n\ndata_augmentation = tf.keras.Sequential([\n    layers.RandomFlip(\"horizontal_and_vertical\", seed=SEED, input_shape=(96,96,3)),\n    layers.RandomRotation(0.5, seed=SEED),\n    layers.RandomZoom(0.3, 0.3, seed=SEED),\n    layers.RandomContrast(0.3, seed=SEED),\n    layers.RandomTranslation(0.3, 0.3, seed=SEED)\n])\n\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\ncnn = Sequential([\n    \n    # Cropping2D(cropping=((32, 32), (32, 32)), input_shape=(96,96,3)),\n    \n    data_augmentation,\n    \n    Conv2D(32, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(32, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.3),\n    BatchNormalization(),\n\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.4),\n    BatchNormalization(),\n\n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.5),\n    BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(64, activation='relu'),\n    Dropout(0.5),\n    Dense(32, activation='relu'),\n    Dropout(0.4),\n    Dense(16, activation='relu'),\n    Dropout(0.3),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:07:23.686939Z","iopub.execute_input":"2021-11-20T23:07:23.687194Z","iopub.status.idle":"2021-11-20T23:07:26.516686Z","shell.execute_reply.started":"2021-11-20T23:07:23.687165Z","shell.execute_reply":"2021-11-20T23:07:26.515879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Network","metadata":{}},{"cell_type":"code","source":"# Define an optimizer and select a learning rate. \n# Then compile the model. \n\nopt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:07:26.518072Z","iopub.execute_input":"2021-11-20T23:07:26.518325Z","iopub.status.idle":"2021-11-20T23:07:26.53745Z","shell.execute_reply.started":"2021-11-20T23:07:26.518293Z","shell.execute_reply":"2021-11-20T23:07:26.536762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n# Complete one or more training runs. \n# Display training curves after each run. \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 35,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1,\n    use_multiprocessing=True, \n    workers=8\n)\n\nhistory = h1.history\nprint(history.keys())","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:07:30.432045Z","iopub.execute_input":"2021-11-20T23:07:30.432583Z","iopub.status.idle":"2021-11-20T23:09:14.697094Z","shell.execute_reply.started":"2021-11-20T23:07:30.432548Z","shell.execute_reply":"2021-11-20T23:09:14.695964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.829805Z","iopub.status.idle":"2021-11-20T23:06:22.830466Z","shell.execute_reply.started":"2021-11-20T23:06:22.830232Z","shell.execute_reply":"2021-11-20T23:06:22.830255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Run 2","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.set_value(cnn.optimizer.learning_rate, 0.0001)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.831667Z","iopub.status.idle":"2021-11-20T23:06:22.832311Z","shell.execute_reply.started":"2021-11-20T23:06:22.83206Z","shell.execute_reply":"2021-11-20T23:06:22.832084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh2 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 35,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1,\n    use_multiprocessing=True, \n    workers=8\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.833521Z","iopub.status.idle":"2021-11-20T23:06:22.834169Z","shell.execute_reply.started":"2021-11-20T23:06:22.833924Z","shell.execute_reply":"2021-11-20T23:06:22.833947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in history.keys():\n    history[k] += h2.history[k]\n\nepoch_range = range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,3,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,3,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.subplot(1,3,3)\nplt.plot(epoch_range, history['auc'], label='Training')\nplt.plot(epoch_range, history['val_auc'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('AUC'); plt.title('AUC')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.83539Z","iopub.status.idle":"2021-11-20T23:06:22.836Z","shell.execute_reply.started":"2021-11-20T23:06:22.835768Z","shell.execute_reply":"2021-11-20T23:06:22.835791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Model and History","metadata":{}},{"cell_type":"code","source":"cnn.save('cancer_model_v01.h5')\npickle.dump(history, open(f'cancer_history_v01.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-11-20T23:06:22.837185Z","iopub.status.idle":"2021-11-20T23:06:22.837819Z","shell.execute_reply.started":"2021-11-20T23:06:22.837588Z","shell.execute_reply":"2021-11-20T23:06:22.837611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}