{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-30T00:41:57.420977Z","iopub.execute_input":"2021-11-30T00:41:57.421210Z","iopub.status.idle":"2021-11-30T00:41:57.442405Z","shell.execute_reply.started":"2021-11-30T00:41:57.421141Z","shell.execute_reply":"2021-11-30T00:41:57.441713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Other necessary imports\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image\nimport keras\nfrom keras import models\nfrom keras import layers\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:41:57.444851Z","iopub.execute_input":"2021-11-30T00:41:57.445286Z","iopub.status.idle":"2021-11-30T00:42:01.937596Z","shell.execute_reply.started":"2021-11-30T00:41:57.445250Z","shell.execute_reply":"2021-11-30T00:42:01.936872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining a results visualization function\ndef visualize_results(history):\n    '''\n    From https://machinelearningmastery.com/display-deep-learning-model-training-history-in-keras/\n    \n    Input: keras history object (output from trained model)\n    '''\n    #Instantiate values\n    train_loss = history.history['loss']\n    train_acc = history.history['accuracy']\n    train_recall = history.history['recall']\n    train_aucroc = history.history['auc']\n    val_loss = history.history['val_loss']\n    val_acc = history.history['val_accuracy']\n    val_recall = history.history['val_recall']\n    val_aucroc = history.history['val_auc']\n    \n    #Create figure for plotting\n    fig, [(ax1, ax2), (ax3, ax4)] = plt.subplots(2, 2, figsize=(15, 10))\n    fig.suptitle('Model Results')\n    #plt.xlabel('Epoch')\n    \n    #Plot Loss\n    ax1.plot(train_loss)\n    ax1.plot(val_loss)\n    ax1.set_ylabel('Loss')\n    ax1.set_xlabel('Epochs')\n    ax1.legend(['train', 'val'])\n    \n    #Plot Accuracy\n    ax2.plot(train_acc)\n    ax2.plot(val_acc)\n    ax2.set_ylabel('Accuracy')\n    ax2.set_xlabel('Epochs')\n    ax2.legend(['train', 'val'])\n    \n    #Plot Recall\n    ax3.plot(train_recall)\n    ax3.plot(val_recall)\n    ax3.set_ylabel('Recall')\n    ax3.set_xlabel('Epochs')\n    ax3.legend(['train', 'val'])\n    \n    #Plot AUC-ROC\n    ax4.plot(train_aucroc)\n    ax4.plot(val_aucroc)\n    ax4.set_ylabel('AUC-ROC')\n    ax4.set_xlabel('Epochs')\n    ax4.legend(['train', 'val'])\n    \n    plt.show();","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:01.939495Z","iopub.execute_input":"2021-11-30T00:42:01.940250Z","iopub.status.idle":"2021-11-30T00:42:01.956885Z","shell.execute_reply.started":"2021-11-30T00:42:01.940209Z","shell.execute_reply":"2021-11-30T00:42:01.956226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Preprocessed Data","metadata":{}},{"cell_type":"code","source":"#Load data\ntrain_images = np.load('../input/eda-and-data-preprocessing/train_images.npy')\ntrain_labels = np.load('../input/eda-and-data-preprocessing/train_labels.npy')\n\ntrain_images_third = np.load('../input/eda-and-data-preprocessing/train_images_third.npy')\ntrain_labels_third = np.load('../input/eda-and-data-preprocessing/train_labels_third.npy')\n\nval_images = np.load('../input/eda-and-data-preprocessing/val_images.npy')\nval_labels = np.load('../input/eda-and-data-preprocessing/val_labels.npy')\n\ntest_images = np.load('../input/eda-and-data-preprocessing/test_images.npy')\ntest_labels = np.load('../input/eda-and-data-preprocessing/test_labels.npy')","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:01.959193Z","iopub.execute_input":"2021-11-30T00:42:01.959785Z","iopub.status.idle":"2021-11-30T00:42:34.805314Z","shell.execute_reply.started":"2021-11-30T00:42:01.959745Z","shell.execute_reply":"2021-11-30T00:42:34.804567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Explore the dataset again\nprint (\"Number of training samples: \" + str(train_images.shape[0]))\nprint (\"A third of training samples: \" + str(train_images_third.shape[0]))\nprint (\"Number of validation samples: \" + str(val_images.shape[0]))\nprint (\"Number of testing samples: \" + str(test_images.shape[0]))\nprint (\"===\")\nprint (\"train_images shape: \" + str(train_images.shape))\nprint (\"train_labels shape: \" + str(train_labels.shape))\nprint (\"A third of train_images shape: \" + str(train_images_third.shape))\nprint (\"A third of train_labels shape: \" + str(train_labels_third.shape))\nprint (\"val_images shape: \" + str(val_images.shape))\nprint (\"val_labels shape: \" + str(val_labels.shape))\nprint (\"test_images shape: \" + str(test_images.shape))\nprint (\"test_labels shape: \" + str(test_labels.shape))","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:34.807799Z","iopub.execute_input":"2021-11-30T00:42:34.808314Z","iopub.status.idle":"2021-11-30T00:42:34.818769Z","shell.execute_reply.started":"2021-11-30T00:42:34.808274Z","shell.execute_reply":"2021-11-30T00:42:34.818082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"#Build my first CNN model\nfirst_cnn_model = models.Sequential()\nfirst_cnn_model.add(layers.Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128,  3)))\nfirst_cnn_model.add(layers.MaxPooling2D((2, 2)))\n\nfirst_cnn_model.add(layers.Conv2D(32, (4, 4), activation='relu'))\nfirst_cnn_model.add(layers.MaxPooling2D((2, 2)))\n\nfirst_cnn_model.add(layers.Conv2D(64, (3, 3), activation='relu'))\nfirst_cnn_model.add(layers.MaxPooling2D((2, 2)))\n\nfirst_cnn_model.add(layers.Flatten())\nfirst_cnn_model.add(layers.Dense(64, activation='relu'))\nfirst_cnn_model.add(layers.Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:34.820322Z","iopub.execute_input":"2021-11-30T00:42:34.820658Z","iopub.status.idle":"2021-11-30T00:42:37.414770Z","shell.execute_reply.started":"2021-11-30T00:42:34.820621Z","shell.execute_reply":"2021-11-30T00:42:37.414075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get the summary of the model\nfirst_cnn_model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:37.416135Z","iopub.execute_input":"2021-11-30T00:42:37.416373Z","iopub.status.idle":"2021-11-30T00:42:37.427952Z","shell.execute_reply.started":"2021-11-30T00:42:37.416341Z","shell.execute_reply":"2021-11-30T00:42:37.427064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Compile first cnn model\nfirst_cnn_model.compile(optimizer='adam',\n                       loss='binary_crossentropy',\n                       metrics=['accuracy', 'Recall', 'AUC'])\n\n#Instantiate an EarlyStopping Object\nes = EarlyStopping(monitor='val_loss', mode='min', patience=5)\nmc = ModelCheckpoint('best_cnn_model.h5', monitor='val_recall', mode='max', verbose=1, save_best_only=True)\n\n#And fit the cnn model to the training images, validating on the val images\nresults = first_cnn_model.fit(train_images_third,\n                            train_labels_third,\n                            epochs=50,\n                            batch_size=32,\n                            callbacks = [es,mc],\n                            validation_data=(val_images, val_labels))","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:42:37.429599Z","iopub.execute_input":"2021-11-30T00:42:37.429846Z","iopub.status.idle":"2021-11-30T00:44:09.914207Z","shell.execute_reply.started":"2021-11-30T00:42:37.429795Z","shell.execute_reply":"2021-11-30T00:44:09.913483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize Results \nvisualize_results(results)","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:44:09.915537Z","iopub.execute_input":"2021-11-30T00:44:09.916960Z","iopub.status.idle":"2021-11-30T00:45:06.668739Z","shell.execute_reply.started":"2021-11-30T00:44:09.916919Z","shell.execute_reply":"2021-11-30T00:45:06.668073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save Weights\nfirst_cnn_model.save_weights('cnn_model_weights.h5')","metadata":{"execution":{"iopub.status.busy":"2021-11-30T00:45:06.672016Z","iopub.execute_input":"2021-11-30T00:45:06.672487Z","iopub.status.idle":"2021-11-30T00:45:06.696167Z","shell.execute_reply.started":"2021-11-30T00:45:06.672447Z","shell.execute_reply":"2021-11-30T00:45:06.695535Z"},"trusted":true},"execution_count":null,"outputs":[]}]}