{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import namespaces","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport matplotlib.image as mpimg\n\nfrom sklearn.model_selection import train_test_split\n\nimport pickle\n\nimport tensorflow as tf\nfrom keras.utils.vis_utils import plot_model\nfrom tensorflow import keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport os\nfrom tensorflow.keras import layers\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, roc_auc_score, plot_confusion_matrix\nimport itertools","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-06T21:56:17.307295Z","iopub.execute_input":"2021-12-06T21:56:17.307572Z","iopub.status.idle":"2021-12-06T21:56:17.328216Z","shell.execute_reply.started":"2021-12-06T21:56:17.307539Z","shell.execute_reply":"2021-12-06T21:56:17.327546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# List of Models we created","metadata":{}},{"cell_type":"markdown","source":"### [Cancer Detection CNN Model with Dense 256 Layer](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-cnn-v1-model)  \t\n### [Cancer Detection CNN Model with Dense 512 Layer](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-cnn-v2-model)   \t\n### [Cancer Detection ResNet50 Model](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-resnet50-model)   \t\n### [Cancer Detection VGG16 Model](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-vgg16-model) \t\n### [Cancer Detection Ensemble Model](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-ensemble-model) \t\n### [Cancer Detection CNN Model with Dense 512 Layer, built using PyTorch](https://www.kaggle.com/lokanathpatro/lp-cancer-detection-cnn-model-with-pytorch) ","metadata":{}},{"cell_type":"code","source":"cnn_models = {\n    'Dense256': {'model' : '../input/lp-cancer-detection-cnnv1-model-after-40epochs/LP_Cancer_Detection_CNN_Model.h5',\n                 'history' : '../input/lp-cancer-detection-cnnv1-model-after-40epochs/LP_Cancer_Detection_CNN_Model_History.pkl',\n                 'color': 'green'},\n    'Dense512': {'model' : '../input/lp-cancer-detection-cnnv2-model-after-40epochs/LP_Cancer_Detection_CNN_Model.h5',\n                 'history' : '../input/lp-cancer-detection-cnnv2-model-after-40epochs/LP_Cancer_Detection_CNN_Model_History.pkl',\n                 'color': 'blue'},\n    'ResNet50' : {'model' : '../input/cancer-models/cancer_model_v01_version10.h5',\n                  'history' : '../input/cancer-models/cancer_history_v02_ResNet50V2.pkl',\n                  'color': 'indigo'},\n    'VGG16' : {'model' : '../input/lp-cancer-detection-vgg16-model-after-40epochs/LP_HCD_VGG16_Model.h5',\n               'history' : '../input/lp-cancer-detection-vgg16-model-after-40epochs/LP_HCD_VGG16_Model_History.pkl',\n               'color': 'black'}\n}","metadata":{"execution":{"iopub.status.busy":"2021-12-06T22:39:45.401802Z","iopub.execute_input":"2021-12-06T22:39:45.402458Z","iopub.status.idle":"2021-12-06T22:39:45.407866Z","shell.execute_reply.started":"2021-12-06T22:39:45.402423Z","shell.execute_reply":"2021-12-06T22:39:45.407089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model's Summary","metadata":{}},{"cell_type":"code","source":"for model in cnn_models.keys():\n    print(' ')\n    print('=========================================================================')\n    print('  '+model)\n    print('=========================================================================')\n    print(' ')\n    cnn = keras.models.load_model(cnn_models[model]['model'])\n    cnn.summary()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model's Performance ","metadata":{}},{"cell_type":"code","source":"columns = ['model', 'epoch', 'accuracy', 'auc', 'loss', 'val_accuracy', 'val_auc', 'val_loss']\nperformance = pd.DataFrame(columns = columns)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for model in cnn_models.keys():\n    pickle_file = open(cnn_models[model]['history'], \"rb\")\n    history = pickle.load(pickle_file)\n    pickle_file.close()\n    epoch_range = range(1, len(history['loss'])+1)\n    performance = performance.append(\n        pd.DataFrame(\n            list(zip([model for i in epoch_range], epoch_range, \n                 history['accuracy'], history['auc'], history['loss'],\n                 history['val_accuracy'], history['val_auc'], history['val_loss'])), \n            columns = columns))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[20,10])\nplt.subplot(1,2,1)\nfor model in cnn_models.keys():\n    modelPerformance = performance[performance['model']==model]\n    plt.plot(modelPerformance.epoch, modelPerformance.accuracy, label=model, color=cnn_models[model]['color'])\n    plt.plot(modelPerformance.epoch, modelPerformance.val_accuracy, '--', color=cnn_models[model]['color'])\nplt.xlabel('Epoch'); \nplt.ylabel('Accuracy'); \nplt.subplot(1,2,2)\nfor model in cnn_models.keys():\n    modelPerformance = performance[performance['model']==model]\n    plt.plot(modelPerformance.epoch, modelPerformance.auc, label=model, color=cnn_models[model]['color'])\n    plt.plot(modelPerformance.epoch, modelPerformance.val_auc, '--', color=cnn_models[model]['color'])\nplt.xlabel('Epoch');\nplt.ylabel('AUC'); \nplt.legend(loc = 4)\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper Functions","metadata":{}},{"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n\n    print(cm)\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    fmt = '.2f' if normalize else 'd'\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, format(cm[i, j], fmt),\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2021-12-06T21:49:56.170577Z","iopub.execute_input":"2021-12-06T21:49:56.170827Z","iopub.status.idle":"2021-12-06T21:49:56.180096Z","shell.execute_reply.started":"2021-12-06T21:49:56.170799Z","shell.execute_reply":"2021-12-06T21:49:56.179177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Dataset","metadata":{}},{"cell_type":"code","source":"# Load the training data into a DataFrame named 'train'. \n# We do not need the test data in this notebook. \ntrain = pd.read_csv(f'../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\n\n# Print the shape of the resulting DataFrame. \nprint('Training Set Size:', train.shape)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-06T21:50:02.107365Z","iopub.execute_input":"2021-12-06T21:50:02.107620Z","iopub.status.idle":"2021-12-06T21:50:02.608325Z","shell.execute_reply.started":"2021-12-06T21:50:02.107591Z","shell.execute_reply":"2021-12-06T21:50:02.607624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets update the dataset to include filename extensions","metadata":{}},{"cell_type":"code","source":"train['path'] = train['id'].apply(lambda x: f'{x}.tif')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-06T21:50:04.797350Z","iopub.execute_input":"2021-12-06T21:50:04.797918Z","iopub.status.idle":"2021-12-06T21:50:04.883260Z","shell.execute_reply.started":"2021-12-06T21:50:04.797881Z","shell.execute_reply":"2021-12-06T21:50:04.882587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we don't have access to the test labels, we split the dataset and use the test split for this purpose.","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.01, random_state=1, stratify=train.label)\n\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-06T21:50:52.186247Z","iopub.execute_input":"2021-12-06T21:50:52.186499Z","iopub.status.idle":"2021-12-06T21:50:52.552250Z","shell.execute_reply.started":"2021-12-06T21:50:52.186470Z","shell.execute_reply":"2021-12-06T21:50:52.551512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 64\ntrain_path = \"../input/histopathologic-cancer-detection/train\"\nprint('Training Images:', len(os.listdir(train_path)))\n\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n\nvalid_loader = valid_datagen.flow_from_dataframe(\n    dataframe = valid_df,\n    directory = train_path,\n    x_col = 'id',\n    y_col = 'label',\n    batch_size = BATCH_SIZE,\n    seed = 1,\n    shuffle = True,\n    class_mode = 'categorical',\n    target_size = (96,96)\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-05T23:29:13.030208Z","iopub.execute_input":"2021-12-05T23:29:13.030894Z","iopub.status.idle":"2021-12-05T23:29:13.956972Z","shell.execute_reply.started":"2021-12-05T23:29:13.030858Z","shell.execute_reply":"2021-12-05T23:29:13.956066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Predictions","metadata":{}},{"cell_type":"code","source":"#change to validation dataset\nvalid_probs = cnn.predict(valid_loader)\nprint(valid_probs.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-05T23:29:16.428731Z","iopub.execute_input":"2021-12-05T23:29:16.429272Z","iopub.status.idle":"2021-12-05T23:29:34.421573Z","shell.execute_reply.started":"2021-12-05T23:29:16.429234Z","shell.execute_reply":"2021-12-05T23:29:34.420771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate the model","metadata":{}},{"cell_type":"code","source":"y_val = valid_loader.classes\ny_pred = np.argmax(valid_probs, axis=1)\ntarget_names = ['Benign (No Cancer)', 'Malignant (Has Cancer)']","metadata":{"execution":{"iopub.status.busy":"2021-12-05T23:30:38.97077Z","iopub.execute_input":"2021-12-05T23:30:38.971051Z","iopub.status.idle":"2021-12-05T23:30:38.976319Z","shell.execute_reply.started":"2021-12-05T23:30:38.971021Z","shell.execute_reply":"2021-12-05T23:30:38.975624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Accuracy\n\nAccuracy is one metric for evaluating classification models. Informally, accuracy is the fraction of predictions our model got right. Formally, accuracy has the following definition. \n\n$\nAccuracy=\\frac{Number of correct predictions}{Number of total predictons}\n$","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score(y_val, y_pred)\nprint('For our model, the accuracy is %f' % accuracy)","metadata":{"execution":{"iopub.status.busy":"2021-12-05T23:40:28.808225Z","iopub.execute_input":"2021-12-05T23:40:28.808925Z","iopub.status.idle":"2021-12-05T23:40:28.817446Z","shell.execute_reply.started":"2021-12-05T23:40:28.808888Z","shell.execute_reply":"2021-12-05T23:40:28.815674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion matrix\n\nA confusion matrix tell us the percentage of examples from each class in our test set that our model predicted correctly. In the case of an imbalanced dataset like the one we're dealing with, this is a better measure of our model's performance than overall accuracy.","metadata":{}},{"cell_type":"code","source":"# Compute confusion matrix\ncm=confusion_matrix(y_val, y_pred)\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\nplt.figure()\nplot_confusion_matrix(cm, target_names, title='Confusion matrix')","metadata":{"execution":{"iopub.status.busy":"2021-12-06T00:43:27.584738Z","iopub.execute_input":"2021-12-06T00:43:27.585473Z","iopub.status.idle":"2021-12-06T00:43:27.821725Z","shell.execute_reply.started":"2021-12-06T00:43:27.585437Z","shell.execute_reply":"2021-12-06T00:43:27.820989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tp = cm[0][0] # actual Benign and predicted Benign\nfn = cm[0][1] # actual Benign and predicted Malignant\ntn = cm[1][0] # actual Malignant and predicted Benign\nfp = cm[1][1] # actual Malignant and predicted Malignant\nprint(\"The preceding confusion matrix shows that of the\", tp + fn, \"samples that were Benign,\",\n      \"the model correctly classified \", tp, \"as Benign (\", tp, \"true positives),\",\n      \"and incorrectly classified \", fn, \"as Malignant (\", fn, \"false negative).\")\nprint(\"Similarly, of\", tn + fp, \"samples that actually were Malignant, \", tn, \" were correctly classified (\", tn, \"true negatives)\", \n      \"and\", fp, \"were incorrectly classified (\", fp, \"false positives).\")","metadata":{"execution":{"iopub.status.busy":"2021-12-06T00:47:25.257106Z","iopub.execute_input":"2021-12-06T00:47:25.257579Z","iopub.status.idle":"2021-12-06T00:47:25.26951Z","shell.execute_reply.started":"2021-12-06T00:47:25.257539Z","shell.execute_reply":"2021-12-06T00:47:25.26853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Classification report\n\nClassification report allows us to look at Precision and Recall.\n\nPrecision is defined as follows\n\n$\nPrecision\\ =\\ \\frac{TP}{TP+FP}\n$\n\nPrecision helps us answer the question _\"What proportion of positive identifications was actually correct?\"_\n\nRecall is is defined as follows\n\n$\nRecall\\ =\\ \\frac{TP}{TP+FN}\n$\n\nRecall helps us answers the question _\"What proportion of actual positives was identified correctly?\"_\n","metadata":{}},{"cell_type":"code","source":"print(classification_report(y_val, y_pred, target_names=target_names))","metadata":{"execution":{"iopub.status.busy":"2021-12-06T00:50:16.45645Z","iopub.execute_input":"2021-12-06T00:50:16.456906Z","iopub.status.idle":"2021-12-06T00:50:16.472382Z","shell.execute_reply.started":"2021-12-06T00:50:16.456867Z","shell.execute_reply":"2021-12-06T00:50:16.471671Z"},"trusted":true},"execution_count":null,"outputs":[]}]}