{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import namespaces","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\n\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport os\nfrom tensorflow.keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-27T14:26:25.524830Z","iopub.execute_input":"2021-11-27T14:26:25.525173Z","iopub.status.idle":"2021-11-27T14:26:32.003530Z","shell.execute_reply.started":"2021-11-27T14:26:25.525103Z","shell.execute_reply":"2021-11-27T14:26:32.002754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:32.005418Z","iopub.execute_input":"2021-11-27T14:26:32.005814Z","iopub.status.idle":"2021-11-27T14:26:32.015323Z","shell.execute_reply.started":"2021-11-27T14:26:32.005778Z","shell.execute_reply":"2021-11-27T14:26:32.014132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"# Load the training data into a DataFrame named 'train'. \n# Print the shape of the resulting DataFrame. \n# You do not need the test data in this notebook. \n\ntrain = pd.read_csv(f'../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\nprint('Training Set Size:', train.shape)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:32.017705Z","iopub.execute_input":"2021-11-27T14:26:32.018123Z","iopub.status.idle":"2021-11-27T14:26:32.649167Z","shell.execute_reply.started":"2021-11-27T14:26:32.018067Z","shell.execute_reply":"2021-11-27T14:26:32.648393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Lets play with 1% data to check if all code works\n# # Comment this when running the entire code\n# ignore, train = train_test_split(train, test_size=0.01, random_state=1, stratify=train.label)\n# print('Training Set Size:', train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:32.651428Z","iopub.execute_input":"2021-11-27T14:26:32.651888Z","iopub.status.idle":"2021-11-27T14:26:33.005438Z","shell.execute_reply.started":"2021-11-27T14:26:32.651849Z","shell.execute_reply":"2021-11-27T14:26:33.004652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets update the dataset to include filename extensions","metadata":{}},{"cell_type":"code","source":"train['id'] = train['id'].apply(lambda x: f'{x}.tif')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:33.006550Z","iopub.execute_input":"2021-11-27T14:26:33.007271Z","iopub.status.idle":"2021-11-27T14:26:33.020940Z","shell.execute_reply.started":"2021-11-27T14:26:33.007234Z","shell.execute_reply":"2021-11-27T14:26:33.020290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"(train.label.value_counts() / len(train)).to_frame().sort_index().T","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:33.022946Z","iopub.execute_input":"2021-11-27T14:26:33.023145Z","iopub.status.idle":"2021-11-27T14:26:33.040040Z","shell.execute_reply.started":"2021-11-27T14:26:33.023122Z","shell.execute_reply":"2021-11-27T14:26:33.039036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View Sample of Images","metadata":{}},{"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train\"\nprint('Training Images:', len(os.listdir(train_path)))\n\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize=(8,8))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{row.id}')    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:33.041715Z","iopub.execute_input":"2021-11-27T14:26:33.042197Z","iopub.status.idle":"2021-11-27T14:26:55.518288Z","shell.execute_reply.started":"2021-11-27T14:26:33.042057Z","shell.execute_reply":"2021-11-27T14:26:55.517530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generators","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, random_state=1, stratify=train.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:55.519275Z","iopub.execute_input":"2021-11-27T14:26:55.519491Z","iopub.status.idle":"2021-11-27T14:26:55.532347Z","shell.execute_reply.started":"2021-11-27T14:26:55.519463Z","shell.execute_reply":"2021-11-27T14:26:55.531408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create image data generators for both the training set and the validation set. \n# Use the data generators to scale the pixel values by a factor of 1/255. \n\ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:55.533734Z","iopub.execute_input":"2021-11-27T14:26:55.534190Z","iopub.status.idle":"2021-11-27T14:26:55.538597Z","shell.execute_reply.started":"2021-11-27T14:26:55.534155Z","shell.execute_reply":"2021-11-27T14:26:55.537948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Complete the code for the data loaders below. \n\nBATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = train_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:26:55.541568Z","iopub.execute_input":"2021-11-27T14:26:55.542301Z","iopub.status.idle":"2021-11-27T14:27:02.418941Z","shell.execute_reply.started":"2021-11-27T14:26:55.542244Z","shell.execute_reply":"2021-11-27T14:27:02.418128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:27:02.420593Z","iopub.execute_input":"2021-11-27T14:27:02.421176Z","iopub.status.idle":"2021-11-27T14:27:02.427439Z","shell.execute_reply.started":"2021-11-27T14:27:02.421134Z","shell.execute_reply":"2021-11-27T14:27:02.426405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Network","metadata":{}},{"cell_type":"code","source":"base_model = tf.keras.applications.InceptionResNetV2(include_top=False,\n                                         weights='imagenet')\n\nbase_model.trainable = False\nbase_model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:27:02.428566Z","iopub.execute_input":"2021-11-27T14:27:02.429331Z","iopub.status.idle":"2021-11-27T14:27:11.387553Z","shell.execute_reply.started":"2021-11-27T14:27:02.429296Z","shell.execute_reply":"2021-11-27T14:27:11.386802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 1\n\ndata_augmentation = tf.keras.Sequential([\n    layers.RandomFlip(\"horizontal_and_vertical\", seed=SEED, input_shape=(96,96,3)),\n    layers.RandomRotation(0.5, seed=SEED),\n    layers.RandomZoom(0.3, 0.3, seed=SEED),\n    layers.RandomContrast(0.3, seed=SEED),\n    layers.RandomTranslation(0.3, 0.3, seed=SEED)\n])\n\n\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\ncnn = Sequential([\n    \n    data_augmentation,\n    base_model,\n\n    Flatten(),\n    \n    Dense(32, activation='relu'),\n    Dropout(0.5),\n    Dense(16, activation='relu'),\n    Dropout(0.25),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:27:11.388708Z","iopub.execute_input":"2021-11-27T14:27:11.389463Z","iopub.status.idle":"2021-11-27T14:27:13.237485Z","shell.execute_reply.started":"2021-11-27T14:27:11.389430Z","shell.execute_reply":"2021-11-27T14:27:13.236763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Network","metadata":{}},{"cell_type":"code","source":"# Define an optimizer and select a learning rate. \n# Then compile the model. \n\nopt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:27:13.238705Z","iopub.execute_input":"2021-11-27T14:27:13.238942Z","iopub.status.idle":"2021-11-27T14:27:13.266503Z","shell.execute_reply.started":"2021-11-27T14:27:13.238906Z","shell.execute_reply":"2021-11-27T14:27:13.265874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n# Complete one or more training runs. \n# Display training curves after each run. \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 25,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1,\n    use_multiprocessing=True, \n    workers=8\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:27:13.267650Z","iopub.execute_input":"2021-11-27T14:27:13.267889Z","iopub.status.idle":"2021-11-27T14:30:54.071561Z","shell.execute_reply.started":"2021-11-27T14:27:13.267856Z","shell.execute_reply":"2021-11-27T14:30:54.069798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:30:54.074069Z","iopub.execute_input":"2021-11-27T14:30:54.074418Z","iopub.status.idle":"2021-11-27T14:30:54.625813Z","shell.execute_reply.started":"2021-11-27T14:30:54.074376Z","shell.execute_reply":"2021-11-27T14:30:54.625144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Run 2","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.set_value(cnn.optimizer.learning_rate, 0.0001)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:30:54.627045Z","iopub.execute_input":"2021-11-27T14:30:54.627441Z","iopub.status.idle":"2021-11-27T14:30:54.633528Z","shell.execute_reply.started":"2021-11-27T14:30:54.627399Z","shell.execute_reply":"2021-11-27T14:30:54.632909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh2 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 25,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1,\n    use_multiprocessing=True, \n    workers=8\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:30:54.634893Z","iopub.execute_input":"2021-11-27T14:30:54.635300Z","iopub.status.idle":"2021-11-27T14:34:02.854074Z","shell.execute_reply.started":"2021-11-27T14:30:54.635262Z","shell.execute_reply":"2021-11-27T14:34:02.850727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1, h2])\nvis_training(history, start=15)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:34:02.884880Z","iopub.execute_input":"2021-11-27T14:34:02.885249Z","iopub.status.idle":"2021-11-27T14:34:04.066538Z","shell.execute_reply.started":"2021-11-27T14:34:02.885203Z","shell.execute_reply":"2021-11-27T14:34:04.065867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Model and History","metadata":{}},{"cell_type":"code","source":"cnn.save('cancer_model_v02.h5')\npickle.dump(history, open(f'cancer_history_v02.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2021-11-27T14:34:04.067636Z","iopub.execute_input":"2021-11-27T14:34:04.068016Z","iopub.status.idle":"2021-11-27T14:34:05.317678Z","shell.execute_reply.started":"2021-11-27T14:34:04.067972Z","shell.execute_reply":"2021-11-27T14:34:05.316924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}