{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Histopathic Cancer Detection (HCD)\n### Taylor Kern","metadata":{}},{"cell_type":"markdown","source":"# Prepare Enviornment","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pickle\nimport os\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import * \n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import backend as k\n\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' ","metadata":{"execution":{"iopub.status.busy":"2022-04-16T16:23:52.098652Z","iopub.execute_input":"2022-04-16T16:23:52.099382Z","iopub.status.idle":"2022-04-16T16:23:58.068127Z","shell.execute_reply.started":"2022-04-16T16:23:52.099289Z","shell.execute_reply":"2022-04-16T16:23:58.067358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-16T16:23:58.072813Z","iopub.execute_input":"2022-04-16T16:23:58.07478Z","iopub.status.idle":"2022-04-16T16:23:58.086968Z","shell.execute_reply.started":"2022-04-16T16:23:58.074739Z","shell.execute_reply":"2022-04-16T16:23:58.086401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Training DataFrame","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-04-16T16:23:58.090665Z","iopub.execute_input":"2022-04-16T16:23:58.092595Z","iopub.status.idle":"2022-04-16T16:23:58.526642Z","shell.execute_reply.started":"2022-04-16T16:23:58.092546Z","shell.execute_reply":"2022-04-16T16:23:58.525959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:12:10.00883Z","iopub.execute_input":"2022-04-05T19:12:10.009447Z","iopub.status.idle":"2022-04-05T19:12:10.030883Z","shell.execute_reply.started":"2022-04-05T19:12:10.009414Z","shell.execute_reply":"2022-04-05T19:12:10.029449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.id = train.id + '.tif'\n","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:12:10.257154Z","iopub.execute_input":"2022-04-05T19:12:10.257683Z","iopub.status.idle":"2022-04-05T19:12:10.277143Z","shell.execute_reply.started":"2022-04-05T19:12:10.257652Z","shell.execute_reply":"2022-04-05T19:12:10.276001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.125648Z","iopub.status.idle":"2022-04-05T19:11:58.126102Z","shell.execute_reply.started":"2022-04-05T19:11:58.125873Z","shell.execute_reply":"2022-04-05T19:11:58.125897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"(train.label.value_counts() / len(train)).to_frame().sort_index().T\n","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.127315Z","iopub.status.idle":"2022-04-05T19:11:58.127749Z","shell.execute_reply.started":"2022-04-05T19:11:58.127517Z","shell.execute_reply":"2022-04-05T19:11:58.127541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract Images","metadata":{}},{"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train\"\n\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize=(6,6))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{row.id}')    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.129697Z","iopub.status.idle":"2022-04-05T19:11:58.130188Z","shell.execute_reply.started":"2022-04-05T19:11:58.129907Z","shell.execute_reply":"2022-04-05T19:11:58.129933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and Validation Sets","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, random_state=1, stratify=train.label)","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.131473Z","iopub.status.idle":"2022-04-05T19:11:58.131869Z","shell.execute_reply.started":"2022-04-05T19:11:58.131653Z","shell.execute_reply":"2022-04-05T19:11:58.131675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generators","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale=1/255)\nvalidation_datagen = ImageDataGenerator(rescale=1/255)","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.133134Z","iopub.status.idle":"2022-04-05T19:11:58.133542Z","shell.execute_reply.started":"2022-04-05T19:11:58.13333Z","shell.execute_reply":"2022-04-05T19:11:58.133352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntrain_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)\n\nvalid_loader = train_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2022-04-05T19:11:58.134805Z","iopub.status.idle":"2022-04-05T19:11:58.135226Z","shell.execute_reply.started":"2022-04-05T19:11:58.134985Z","shell.execute_reply":"2022-04-05T19:11:58.135007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_loader)\nVA_STEPS = len(valid_loader)\n\nprint(TR_STEPS)\nprint(VA_STEPS)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Base Model","metadata":{}},{"cell_type":"code","source":"base_model = tf.keras.applications.VGG19(\n    input_shape=(96,96,3), \n    include_top=False, \n    weights='imagenet'\n)\n\nbase_model.trainable = False\n\nbase_model.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build and Train","metadata":{}},{"cell_type":"code","source":"np.random.seed(1)\ntf.random.set_seed(1)\n\ncnn = Sequential([\n    base_model,\n    BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(16, activation='relu'),\n    Dropout(0.5),\n    Dense(8, activation='relu'),\n    Dropout(0.5),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])\n\ncnn.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opt = tf.keras.optimizers.Adam(0.001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nh1 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 40,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fine Tuning","metadata":{}},{"cell_type":"code","source":"base_model.trainable = True\nk.set_value(cnn.optimizer.learning_rate, 0.00001)\ncnn.compile(loss='categorical_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.summary()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train 2","metadata":{}},{"cell_type":"code","source":"%%time \n\nh2 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 30,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h2.history['auc'] = h2.history['auc_1']\nh2.history['val_auc'] = h2.history['val_auc_1']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1, h2])\nvis_training(history, start=10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training 3","metadata":{}},{"cell_type":"code","source":"%%time \n\nh3 = cnn.fit(\n    x = train_loader, \n    steps_per_epoch = TR_STEPS, \n    epochs = 20,\n    validation_data = valid_loader, \n    validation_steps = VA_STEPS, \n    verbose = 1\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h3.history['auc'] = h3.history['auc_1'] \nh3.history['val_auc'] = h3.history['val_auc_1'] ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1, h2, h3])\nvis_training(history, start=10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('HCDv01.h5')\npickle.dump(history, open(f'HCDv01.pkl', 'wb'))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/histopathologic-cancer-detection/sample_submission.csv')\n\nprint('Test Set Size:', test.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['filename'] = test.id + '.tif'\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = \"../input/histopathologic-cancer-detection/test\"\nprint('Test Images:', len(os.listdir(test_path)))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\ntest_datagen = ImageDataGenerator(rescale=1/255)\n\ntest_loader = test_datagen.flow_from_dataframe(\n    dataframe = test,\n    directory = test_path,\n    x_col = 'filename',\n    batch_size = BATCH_SIZE,\n    shuffle = False,\n    class_mode = None,\n    target_size = (96,96)\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_probs = cnn.predict(test_loader)\nprint(test_probs.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test_loader))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_probs[:10,].round(2))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = np.argmax(test_probs, axis=1)\nprint(test_pred[:10])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/histopathologic-cancer-detection/sample_submission.csv')\nsubmission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.label = test_probs[:,1]\nsubmission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', header=True, index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}