{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"e53e1831-98b8-42b6-abab-a7e8a381d8b1","_cell_guid":"2fe5de17-8604-4c11-9fce-06d18348c496","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:58:20.659087Z","iopub.execute_input":"2021-09-04T08:58:20.659571Z","iopub.status.idle":"2021-09-04T08:58:59.405955Z","shell.execute_reply.started":"2021-09-04T08:58:20.659527Z","shell.execute_reply":"2021-09-04T08:58:59.395251Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import libraries\n\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\n\nfrom glob import glob\n\nfrom sklearn.metrics import confusion_matrix\nfrom mlxtend.plotting import plot_confusion_matrix\n\nimport tensorflow as tf\n\nfrom tensorflow.keras import utils\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, Flatten, ZeroPadding2D, Conv2D, MaxPooling2D, BatchNormalization\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, DirectoryIterator\nfrom tensorflow.keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\n\nfrom keras.applications.vgg16 import VGG16\nfrom keras.applications.vgg16 import preprocess_input\nfrom keras.applications.vgg16 import decode_predictions\n\n# For reproducibility\nnp.random.seed(42)","metadata":{"_uuid":"d1e440c2-276a-4187-8598-b723402b63e1","_cell_guid":"14ac95d4-4086-4716-8589-14aaadc12277","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:59:20.019119Z","iopub.execute_input":"2021-09-04T08:59:20.019546Z","iopub.status.idle":"2021-09-04T08:59:27.14246Z","shell.execute_reply.started":"2021-09-04T08:59:20.019511Z","shell.execute_reply":"2021-09-04T08:59:27.141327Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Modelling & Predicting Pneumonia w/ Neural Networks***","metadata":{"_uuid":"059ae6ac-735a-41af-86ae-e4b89db4b7e0","_cell_guid":"a0d7c8a6-7c39-43f3-94bc-3e105f15c599","trusted":true}},{"cell_type":"code","source":"# Imports\nimport os\nimport cv2\nimport glob\nimport time\nimport pydicom\nimport skimage\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom skimage import feature, filters\n%matplotlib inline\n\nfrom functools import partial\nfrom collections import defaultdict\nfrom joblib import Parallel, delayed\nfrom lightgbm import LGBMClassifier\nfrom tqdm import tqdm\n\n# Tensorflow / Keras\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom keras import models\nfrom keras import layers\n\n# sklearn\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import RandomizedSearchCV\n\nsns.set_style('whitegrid')\nnp.warnings.filterwarnings('ignore')","metadata":{"_uuid":"92a3d7e0-f44e-4bc4-a628-4da08e08b9a7","_cell_guid":"153b3002-3374-406f-b36d-1a318b52c4ce","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:59:27.143854Z","iopub.execute_input":"2021-09-04T08:59:27.14419Z","iopub.status.idle":"2021-09-04T08:59:29.668009Z","shell.execute_reply.started":"2021-09-04T08:59:27.14416Z","shell.execute_reply":"2021-09-04T08:59:29.667098Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List our paths\ntrainImagesPath = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\ntestImagesPath = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images\"\n\nlabelsPath = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nclassInfoPath = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv\"\n\n# Read the labels and classinfo\nlabels = pd.read_csv(labelsPath)\ndetails = pd.read_csv(classInfoPath)","metadata":{"_uuid":"f6f0a293-51ea-4400-a89d-99fcfa8e4b18","_cell_guid":"2fed0ddc-f8d7-4984-971d-60fc39ae430c","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:59:29.669614Z","iopub.execute_input":"2021-09-04T08:59:29.669883Z","iopub.status.idle":"2021-09-04T08:59:29.822371Z","shell.execute_reply.started":"2021-09-04T08:59:29.669857Z","shell.execute_reply":"2021-09-04T08:59:29.821156Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Attaining our Training & Testing Data in Proper Format**","metadata":{"_uuid":"8eda0f23-26ba-465d-a7ee-969f54834b99","_cell_guid":"769ba2bf-07a5-470d-bb8f-983b72c6d526","trusted":true}},{"cell_type":"code","source":"\"\"\"\n@Description: Reads an array of dicom image paths, and returns an array of the images after they have been read\n\n@Inputs: An array of filepaths for the images\n\n@Output: Returns an array of the images after they have been read\n\"\"\"\ndef readDicomData(data):\n    \n    res = []\n    \n    for filePath in tqdm(data): # Loop over data\n        \n        # We use stop_before_pixels to avoid reading the image (Saves on speed/memory)\n        f = pydicom.read_file(filePath, stop_before_pixels=True)\n        res.append(f)\n    \n    return res","metadata":{"_uuid":"81e34154-07bb-443a-9c9a-1e9dea7b4004","_cell_guid":"79d23083-5b88-457f-9378-490b5781064d","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:59:29.823967Z","iopub.execute_input":"2021-09-04T08:59:29.824357Z","iopub.status.idle":"2021-09-04T08:59:29.831805Z","shell.execute_reply.started":"2021-09-04T08:59:29.824314Z","shell.execute_reply":"2021-09-04T08:59:29.830455Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get an array of the test & training file paths\ntrainFilepaths = glob.glob(f\"{trainImagesPath}/*.dcm\")\ntestFilepaths = glob.glob(f\"{testImagesPath}/*.dcm\")\n\n# Read data into an array\ntrainImages = readDicomData(trainFilepaths[:5000])\ntestImages = readDicomData(testFilepaths)","metadata":{"_uuid":"8378ddcc-9283-4f3c-99a7-4bc6a7c7b454","_cell_guid":"cd80c202-de06-497a-953d-db5fb79b72f3","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T08:59:31.697479Z","iopub.execute_input":"2021-09-04T08:59:31.697835Z","iopub.status.idle":"2021-09-04T09:00:26.374408Z","shell.execute_reply.started":"2021-09-04T08:59:31.697807Z","shell.execute_reply":"2021-09-04T09:00:26.373292Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Balancing our Data:\n\nWe balance our data as CNNs work best on evenly balanced data**","metadata":{"_uuid":"8849b44f-31ae-4757-a36c-894e19832295","_cell_guid":"741a4ea5-623d-4a31-ae81-6bced3191c44","trusted":true}},{"cell_type":"code","source":"COUNT_NORMAL = len(labels.loc[labels['Target'] == 0]) # Number of patients with no pneumonia\nCOUNT_PNE = len(labels.loc[labels['Target'] == 1]) # Number of patients with pneumonia\nTRAIN_IMG_COUNT = len(trainFilepaths) # Total patients\n\n# We calculate the weight of each\nweight_for_0 = (1 / COUNT_NORMAL)*(TRAIN_IMG_COUNT)/2.0 \nweight_for_1 = (1 / COUNT_PNE)*(TRAIN_IMG_COUNT)/2.0\n\nclassWeight = {0: weight_for_0, \n               1: weight_for_1}\n\nprint(f\"Weights: {classWeight}\")","metadata":{"_uuid":"e7314694-2154-461e-be1b-d917b1eeabfd","_cell_guid":"21b2cdfa-834d-4a45-87b7-0e9f17d5ae7c","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:00:34.618678Z","iopub.execute_input":"2021-09-04T09:00:34.619207Z","iopub.status.idle":"2021-09-04T09:00:34.640213Z","shell.execute_reply.started":"2021-09-04T09:00:34.619174Z","shell.execute_reply":"2021-09-04T09:00:34.638877Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Get Train_Y & Test_Y**","metadata":{"_uuid":"dc092c05-0636-4157-ba81-131c42a45ede","_cell_guid":"19ee6c23-830e-4793-ac8a-901df94b3dd4","trusted":true}},{"cell_type":"code","source":"\"\"\"\n@Description: This function parses the medical images meta-data contained\n\n@Inputs: Takes in the dicom image after it has been read\n\n@Output: Returns the unpacked data and the group elements keywords\n\"\"\"\ndef parseMetadata(dcm):\n    \n    unpackedData = {}\n    groupElemToKeywords = {}\n    \n    for d in dcm: # Iterate here to force conversion from lazy RawDataElement to DataElement\n        pass\n    \n    # Un-pack Data\n    for tag, elem in dcm.items():\n        tagGroup = tag.group\n        tagElem = tag.elem\n        keyword = elem.keyword\n        groupElemToKeywords[(tagGroup, tagElem)] = keyword\n        value = elem.value\n        unpackedData[keyword] = value\n        \n    return unpackedData, groupElemToKeywords","metadata":{"_uuid":"94e97fd7-3898-4ca7-a5fc-e8e076bcb8fa","_cell_guid":"7eb5a6d2-40c0-4441-b9fd-53e8225ec5ed","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:00:37.097741Z","iopub.execute_input":"2021-09-04T09:00:37.098129Z","iopub.status.idle":"2021-09-04T09:00:37.105352Z","shell.execute_reply.started":"2021-09-04T09:00:37.098098Z","shell.execute_reply":"2021-09-04T09:00:37.103988Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# These parse the metadata into dictionaries\ntrainMetaDicts, trainKeyword = zip(*[parseMetadata(x) for x in tqdm(trainImages)])\ntestMetaDicts, testKeyword = zip(*[parseMetadata(x) for x in tqdm(testImages)])","metadata":{"_uuid":"6f2c5ebe-5715-44d8-ab55-bb1d9c67fda6","_cell_guid":"be9b679a-3b21-4503-bfe5-51513a10d1f7","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:00:39.137802Z","iopub.execute_input":"2021-09-04T09:00:39.138179Z","iopub.status.idle":"2021-09-04T09:00:45.902088Z","shell.execute_reply.started":"2021-09-04T09:00:39.138149Z","shell.execute_reply":"2021-09-04T09:00:45.900986Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n@Description: This function goes through the dicom image information and returns 1 or 0\n              depending on whether the image contains Pneumonia or not\n\n@Inputs: A dataframe containing the metadata\n\n@Output: Returns the Y result (i.e: our train and test y)\n\"\"\"\ndef createY(df):\n    y = (df['SeriesDescription'] == 'view: PA')\n    Y = np.zeros(len(y)) # Initialise Y\n    \n    for i in range(len(y)):\n        if(y[i] == True):\n            Y[i] = 1\n    \n    return Y\n\n\ntrain_df = pd.DataFrame.from_dict(data=trainMetaDicts)\ntest_df = pd.DataFrame.from_dict(data=testMetaDicts)\n\ntrain_df['dataset'] = 'train'\ntest_df['dataset'] = 'test'\n\ndf = train_df\ndf2 = test_df\n\ntrain_Y = createY(df) # Create training Y \ntest_Y = createY(df2) # Create testing Y","metadata":{"_uuid":"befea754-7acd-449c-a167-91ec514d783a","_cell_guid":"7af8bec1-f077-49d6-9e27-2e996ae2367f","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:01:21.858678Z","iopub.execute_input":"2021-09-04T09:01:21.859066Z","iopub.status.idle":"2021-09-04T09:01:22.063475Z","shell.execute_reply.started":"2021-09-04T09:01:21.859029Z","shell.execute_reply":"2021-09-04T09:01:22.062469Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Get Train_X & Test_X**","metadata":{"_uuid":"e0fca9a4-b749-41bc-b0e4-02320fd8d3a7","_cell_guid":"ad697c5e-20b8-4aed-bd87-db7a2bcf2a5e","trusted":true}},{"cell_type":"code","source":"\"\"\"\n@Description: This decodes an image by reading the pixel array, resizing it into the correct format and\n              normalising the pixels\n\n@Inputs:\n    - filePath: This is the filepath of the image that we want to decode\n\n@Output:\n    - img: This is the image after it has been decoded\n\"\"\"\ndef decodeImage(filePath):\n    image = pydicom.read_file(filePath).pixel_array\n    image = cv2.resize(image, (128, 128))\n    return (image/255)","metadata":{"_uuid":"6814413b-c152-4d04-a71c-511a7bf94ce0","_cell_guid":"9a9190cd-ad27-4c3a-b270-4e7972359459","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:01:25.017728Z","iopub.execute_input":"2021-09-04T09:01:25.018106Z","iopub.status.idle":"2021-09-04T09:01:25.023729Z","shell.execute_reply.started":"2021-09-04T09:01:25.018076Z","shell.execute_reply":"2021-09-04T09:01:25.022584Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get our train x in the correct shape\ntrain_X = []\n\nfor filePath in tqdm(trainFilepaths[:5000]):\n    \n    img = decodeImage(filePath)\n    train_X.append(img)\n\ntrain_X = np.array(train_X) # Convert to np.array\ntrain_X_rgb = np.repeat(train_X[..., np.newaxis], 3, -1) # Reshape into rgb format","metadata":{"_uuid":"57bb376d-a0c0-4ecc-8681-c81b2f410b17","_cell_guid":"bb4f6ce8-882a-45e3-ae81-ecc1ec1462de","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:01:27.017804Z","iopub.execute_input":"2021-09-04T09:01:27.018177Z","iopub.status.idle":"2021-09-04T09:02:21.15214Z","shell.execute_reply.started":"2021-09-04T09:01:27.018147Z","shell.execute_reply":"2021-09-04T09:02:21.151069Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get our test x in the correct shape for NN\ntest_X = []\n\nfor filePath in tqdm(testFilepaths):\n    img_test = decodeImage(filePath) # Decode & Resize\n    test_X.append(img_test)\n\ntest_X = np.array(test_X) # Convert to np array\ntest_X_rgb = np.repeat(test_X[..., np.newaxis], 3, -1) # Reshape into rgb format","metadata":{"_uuid":"21b31027-dfec-4460-8583-4165272983f8","_cell_guid":"89f5daba-54f1-4b46-8366-7efb85f211c2","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:02:40.178491Z","iopub.execute_input":"2021-09-04T09:02:40.178863Z","iopub.status.idle":"2021-09-04T09:03:12.533984Z","shell.execute_reply.started":"2021-09-04T09:02:40.178833Z","shell.execute_reply":"2021-09-04T09:03:12.533Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n@Description: This function plots our metrics for our models across epochs\n\n@Inputs: The history of the fitted model\n\n@Output: N/A\n\"\"\"\ndef plottingScores(hist):\n    fig, ax = plt.subplots(1, 5, figsize=(20, 3))\n    ax = ax.ravel()\n\n    for i, met in enumerate(['accuracy', 'precision', 'recall', 'AUC', 'loss']):\n        ax[i].plot(hist.history[met])\n        ax[i].plot(hist.history['val_' + met])\n        ax[i].set_title('Model {}'.format(met))\n        ax[i].set_xlabel('epochs')\n        ax[i].set_ylabel(met)\n        ax[i].legend(['train', 'val'])","metadata":{"_uuid":"e39460e7-f638-483d-8ea9-654a61dabad5","_cell_guid":"825519c7-e526-4fee-afb9-8edee5eb7042","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:13.457449Z","iopub.execute_input":"2021-09-04T09:03:13.457785Z","iopub.status.idle":"2021-09-04T09:03:13.465381Z","shell.execute_reply.started":"2021-09-04T09:03:13.457758Z","shell.execute_reply":"2021-09-04T09:03:13.464272Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Metrics Evaluation:\n\nFor our metrics, we want to include precision and recall as they will provide use with more info on how good our model is\n\n     Accuracy: This tells us what fraction of the labels are correct.\n        Since our data is not balanced, accuracy might give a skewed sense of a good model\n     Precision: This tells us the number of true positives (TP) over the sum of TP and    \n     false positives (FP).\n       It shows what fraction of labeled positives are actually correct.\n    Recall: The number of TP over the sum of TP and false negatves (FN).\n      It shows what fraction of actual positives are correct.","metadata":{"_uuid":"8971c4bd-9552-4e64-9cc6-1cd7d16f169f","_cell_guid":"77ed52fc-ed96-4489-adea-d77cf23dca10","trusted":true}},{"cell_type":"markdown","source":"","metadata":{"_uuid":"018d1878-df7c-4270-84f7-3ed5a06b1b48","_cell_guid":"694e204e-6609-48b0-ad5a-b0c52ced6d6f","trusted":true}},{"cell_type":"code","source":"# These our our scoring metrics that are going to be used to evaluate our models\nMETRICS = ['accuracy', \n           tf.keras.metrics.Precision(name='precision'), \n           tf.keras.metrics.Recall(name='recall'), \n           tf.keras.metrics.AUC(name='AUC')]","metadata":{"_uuid":"592e072f-479b-4583-9484-121c63456b0c","_cell_guid":"c132bdb8-6586-4de7-ae6e-575bdaa7b8c3","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:16.697459Z","iopub.execute_input":"2021-09-04T09:03:16.697826Z","iopub.status.idle":"2021-09-04T09:03:16.766521Z","shell.execute_reply.started":"2021-09-04T09:03:16.69779Z","shell.execute_reply":"2021-09-04T09:03:16.765523Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tuning our Models with Callbacks:\n\n   We'll use Keras callbacks to further finetune our model.\n   The checkpoint callback saves the best weights of the model, so next time we want to use    the model, we do not have to spend time training it.\n   The early stopping callback stops the training process when the model starts becoming      stagnant, or even worse, when the model starts overfitting.\n   Since we set restore_best_weights to True, the returned model at the end of the training    process will be the model with the best weights (i.e. low loss and high accuracy).","metadata":{"_uuid":"f91568ce-bc30-4d9d-97d8-340d5f2c7cb6","_cell_guid":"2dc50eab-f668-43e0-9fc8-1a32f144f0ec","trusted":true}},{"cell_type":"code","source":"# Define our callback functions to pass when fitting our NNs\ndef exponential_decay(lr0, s):\n    def exponential_decay_fn(epoch):\n        return lr0 * 0.1 **(epoch / s)\n    return exponential_decay_fn\n\nexponential_decay_fn = exponential_decay(0.01, 20)\n\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(exponential_decay_fn)\n\ncheckpoint_cb = tf.keras.callbacks.ModelCheckpoint(\"xray_model.h5\", save_best_only=True)\n\nearly_stopping_cb = tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True)","metadata":{"_uuid":"1771d2c8-e823-4e34-b2c5-15442b3602fb","_cell_guid":"b7d5de5a-c5c0-482a-87b6-9025538754ef","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:18.698029Z","iopub.execute_input":"2021-09-04T09:03:18.698384Z","iopub.status.idle":"2021-09-04T09:03:18.705121Z","shell.execute_reply.started":"2021-09-04T09:03:18.698354Z","shell.execute_reply":"2021-09-04T09:03:18.703866Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building Model #1 - Fully Connected Model**","metadata":{"_uuid":"037b4fad-69f8-449f-91c5-8af4da70ef3f","_cell_guid":"9cac5e8f-4b0d-4ab4-b538-a02527183447","trusted":true}},{"cell_type":"code","source":"\"\"\"\n@Description: This function builds our simple Fully-connected NN\n\n@Inputs: N/A\n\n@Output: Returns the FCNN Model\n\"\"\"\ndef build_fcnn_model():\n    \n    # Basic model with a flattening layer followng by 2 dense layers\n    # The first dense layer is using relu and the 2nd one is using sigmoid\n    model = tf.keras.models.Sequential([\n                tf.keras.layers.Flatten(input_shape = (128, 128, 3)), \n                tf.keras.layers.Dense(128, activation = \"relu\"), \n                tf.keras.layers.Dense(1, activation = \"sigmoid\")\n                ])\n    \n    return model","metadata":{"_uuid":"ca729f1f-b97d-42ba-92b7-519f9fd49274","_cell_guid":"6a9db92f-7b39-45ea-b641-8db8b19d2173","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:20.69763Z","iopub.execute_input":"2021-09-04T09:03:20.698043Z","iopub.status.idle":"2021-09-04T09:03:20.704341Z","shell.execute_reply.started":"2021-09-04T09:03:20.698001Z","shell.execute_reply":"2021-09-04T09:03:20.703248Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build our FCNN model and compile\nmodel_fcnn = build_fcnn_model()\nmodel_fcnn.summary()\nmodel_fcnn.compile(optimizer=\"adam\", loss=\"binary_crossentropy\", metrics=METRICS) # Compile","metadata":{"_uuid":"121061e0-f007-4f8c-900b-fc366930119d","_cell_guid":"dc8e043e-ef50-427f-b470-1d42e9787850","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:23.137575Z","iopub.execute_input":"2021-09-04T09:03:23.138153Z","iopub.status.idle":"2021-09-04T09:03:23.240529Z","shell.execute_reply.started":"2021-09-04T09:03:23.138116Z","shell.execute_reply":"2021-09-04T09:03:23.238867Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Fitting Model to Training Data**","metadata":{"_uuid":"784d4e95-854d-4430-8af7-0db6e8a77fda","_cell_guid":"9c71338f-ecd8-48bc-a6fa-e3840852dba0","trusted":true}},{"cell_type":"code","source":"history_fcnn = model_fcnn.fit(train_X_rgb, \n                          train_Y,  \n                          epochs = 30,\n                          batch_size = 128,\n                          validation_split = 0.2, \n                          class_weight = classWeight, \n                          verbose = 1,\n                          callbacks = [checkpoint_cb, early_stopping_cb, lr_scheduler]) # Fit the model","metadata":{"_uuid":"dd7e4638-58db-40fc-8280-7036b933a3bb","_cell_guid":"a6e71047-79d7-4946-b540-f59e3a99c4a9","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:03:27.178062Z","iopub.execute_input":"2021-09-04T09:03:27.178474Z","iopub.status.idle":"2021-09-04T09:04:29.101861Z","shell.execute_reply.started":"2021-09-04T09:03:27.178443Z","shell.execute_reply":"2021-09-04T09:04:29.100816Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate and display results\nresults = model_fcnn.evaluate(test_X_rgb, test_Y) # Evaluate the model on test data\nresults = dict(zip(model_fcnn.metrics_names,results))\n\nprint(results)\nplottingScores(history_fcnn) # Visualise scores","metadata":{"_uuid":"a319a58f-9e26-40e8-b2a3-1d376457ba20","_cell_guid":"559d7e02-40fd-4642-bcb7-3db11f18218c","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:09:33.117212Z","iopub.execute_input":"2021-09-04T09:09:33.11763Z","iopub.status.idle":"2021-09-04T09:09:36.10412Z","shell.execute_reply.started":"2021-09-04T09:09:33.117598Z","shell.execute_reply":"2021-09-04T09:09:36.1029Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Building Model #2 - CNN\n\nIn our CNN model, fewer parameters are needed because every convolutional layer reduces the dimensions of the input through the convolution operation.","metadata":{"_uuid":"e8d82012-c7a8-4b4d-9627-d17f9dae4b49","_cell_guid":"64db7418-fba3-4741-8914-d5b19f2bd042","trusted":true}},{"cell_type":"code","source":"\"\"\"\n@Description: This function builds our custom CNN Model\n\n@Inputs: N/A\n\n@Output: Returns the CNN model\n\"\"\"\ndef build_cnn_model():\n    model = tf.keras.Sequential([\n        tf.keras.layers.Conv2D(32, kernel_size=(3,3), strides=(1,1), padding = 'valid', activation = 'relu', input_shape=(128, 128, 3)), #  convolutional layer\n        tf.keras.layers.MaxPool2D(pool_size=(2,2)), # flatten output of conv\n        \n        tf.keras.layers.Conv2D(32, kernel_size=(3,3), strides=(1,1), padding = 'valid', activation = 'relu'), #  convolutional layer\n        tf.keras.layers.MaxPool2D(pool_size=(2,2)), # flatten output of conv\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Conv2D(64, 3, activation = 'relu', padding = 'valid'),\n        tf.keras.layers.Conv2D(128, 3, activation = 'relu', padding = 'valid'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.MaxPool2D(),\n        tf.keras.layers.Dropout(0.4),\n        \n        tf.keras.layers.Flatten(), # flatten output of conv\n        tf.keras.layers.Dense(512, activation = \"relu\"), # hidden layer\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Dense(128, activation = \"relu\"), #  output layer\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Dense(1, activation = \"sigmoid\")])\n    \n    return model","metadata":{"_uuid":"c0b2b4b8-a939-4b86-bcfd-f96ce2fe8da3","_cell_guid":"04d59491-9ba3-4043-9d9f-c49c88a6ea48","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:09:40.133084Z","iopub.execute_input":"2021-09-04T09:09:40.133447Z","iopub.status.idle":"2021-09-04T09:09:40.145262Z","shell.execute_reply.started":"2021-09-04T09:09:40.133418Z","shell.execute_reply":"2021-09-04T09:09:40.14364Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and compile model\nmodel_cnn = build_cnn_model()\nmodel_cnn.summary()\nmodel_cnn.compile(optimizer='adam', loss='binary_crossentropy', metrics=METRICS)","metadata":{"_uuid":"0f7959f1-1975-4007-a514-fb8d81f26aad","_cell_guid":"347a3f6e-3b94-4a04-96ff-6ae93b3289c6","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:09:41.592345Z","iopub.execute_input":"2021-09-04T09:09:41.592694Z","iopub.status.idle":"2021-09-04T09:09:41.793926Z","shell.execute_reply.started":"2021-09-04T09:09:41.592663Z","shell.execute_reply":"2021-09-04T09:09:41.792983Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit model\nhistory_cnn = model_cnn.fit(train_X_rgb, \n                      train_Y,  \n                      epochs=30, \n                      validation_split = 0.15, \n                      batch_size=128,\n                      class_weight=classWeight,\n                      callbacks=[checkpoint_cb, early_stopping_cb, lr_scheduler],\n                      verbose=1) # Fit the model","metadata":{"_uuid":"b52b489c-d2b3-43cb-920d-42392fbb5f2d","_cell_guid":"93180ce2-ee65-4a03-acfc-d935da8d1702","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:09:42.942599Z","iopub.execute_input":"2021-09-04T09:09:42.942975Z","iopub.status.idle":"2021-09-04T09:26:14.474313Z","shell.execute_reply.started":"2021-09-04T09:09:42.94294Z","shell.execute_reply":"2021-09-04T09:26:14.473423Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evalute the models results and put into a dict\nresults = model_cnn.evaluate(test_X_rgb, test_Y)\nresults = dict(zip(model_cnn.metrics_names,results))\n\nprint(results)\nplottingScores(history_cnn) # Visualise scores","metadata":{"_uuid":"9209632e-e06f-44e8-8335-e4d06417944f","_cell_guid":"c0d8a247-e1fd-47e0-84d2-ab67e1a2e691","collapsed":false,"execution":{"iopub.status.busy":"2021-09-04T09:26:14.475675Z","iopub.execute_input":"2021-09-04T09:26:14.476224Z","iopub.status.idle":"2021-09-04T09:26:29.563829Z","shell.execute_reply.started":"2021-09-04T09:26:14.476189Z","shell.execute_reply":"2021-09-04T09:26:29.563061Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"1d93e5b5-f1c8-4a7f-b36e-c5246c17f834","_cell_guid":"dd7c3f21-8580-4f83-94bf-e61a53f6622a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}