{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Histopathologic Cancer Detection : Image classification\n\n**Description:** In this competition, you must create an algorithm to identify metastatic cancer in small image patches taken from larger digital pathology scans. The data for this competition is a slightly modified version of the PatchCamelyon (PCam) benchmark dataset (the original PCam dataset contains duplicate images due to its probabilistic sampling, however, the version presented on Kaggle does not contain duplicates).","metadata":{}},{"cell_type":"markdown","source":"## Setting up the Environment","metadata":{}},{"cell_type":"code","source":"# Libraries\nimport pandas as pd                     # data processing\nimport numpy as np                      # linear algebra; asarray, save, load\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nfrom matplotlib.image import imread\nfrom tqdm import tqdm_notebook\n\nimport os, warnings, random, time, multiprocessing, pickle\n","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:38.946983Z","iopub.execute_input":"2022-05-27T00:07:38.947311Z","iopub.status.idle":"2022-05-27T00:07:38.959576Z","shell.execute_reply.started":"2022-05-27T00:07:38.947255Z","shell.execute_reply":"2022-05-27T00:07:38.958809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import BatchNormalization\n\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0\"         # Set for GPU use\n# os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"          # Set for CPU use\n# os.environ[\"CUDA_DEVICE_ORDER\"] = \"PCI_BUS_ID\"     # Set for CPU use\n\ndevice_name = tf.test.gpu_device_name()\n\nif device_name != '/device:GPU:0':\n    print('GPU device not found')\n    workers = multiprocessing.cpu_count()\n    print('You have %d Cores' % workers)\nelse:\n    print('Found GPU at: {}'.format(device_name))\n    physical_devices = tf.config.list_physical_devices('GPU')\n    print(\"Num GPUs Available: \", len(physical_devices))\n#    tf.config.experimental.set_memory_growth(physical_devices[0], True)\n#    tf.debugging.set_log_device_placement(False)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:38.964536Z","iopub.execute_input":"2022-05-27T00:07:38.965268Z","iopub.status.idle":"2022-05-27T00:07:41.581013Z","shell.execute_reply.started":"2022-05-27T00:07:38.965231Z","shell.execute_reply":"2022-05-27T00:07:41.580073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting Variables\nwarnings.filterwarnings('ignore')\npd.options.display.max_rows = 30\npd.options.display.float_format = \"{:.2f}\".format\n%matplotlib inline\nmpl.style.use('ggplot')","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:41.582302Z","iopub.execute_input":"2022-05-27T00:07:41.582820Z","iopub.status.idle":"2022-05-27T00:07:41.594159Z","shell.execute_reply.started":"2022-05-27T00:07:41.582755Z","shell.execute_reply":"2022-05-27T00:07:41.593486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DB_PATH   = r'../input/histopathologic-cancer-detection/'\nTRAIN_DIR = r'../input/histopathologic-cancer-detection/train/'\nTEST_DIR  = r'../input/histopathologic-cancer-detection/test/'\nDIR       = ['train/', 'test/']","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:41.597548Z","iopub.execute_input":"2022-05-27T00:07:41.598924Z","iopub.status.idle":"2022-05-27T00:07:41.606569Z","shell.execute_reply.started":"2022-05-27T00:07:41.598884Z","shell.execute_reply":"2022-05-27T00:07:41.605876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Dataframes","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(DB_PATH + 'train_labels.csv',dtype=str)\ntrain_df.id = train_df.id + '.tif'","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:41.607733Z","iopub.execute_input":"2022-05-27T00:07:41.608138Z","iopub.status.idle":"2022-05-27T00:07:41.951613Z","shell.execute_reply.started":"2022-05-27T00:07:41.608105Z","shell.execute_reply":"2022-05-27T00:07:41.950799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train Shape: ' , train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:41.955778Z","iopub.execute_input":"2022-05-27T00:07:41.957952Z","iopub.status.idle":"2022-05-27T00:07:41.977514Z","shell.execute_reply.started":"2022-05-27T00:07:41.957902Z","shell.execute_reply":"2022-05-27T00:07:41.976786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(train_df.label.value_counts() / len(train_df)).to_frame().sort_index().T","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:41.978608Z","iopub.execute_input":"2022-05-27T00:07:41.978947Z","iopub.status.idle":"2022-05-27T00:07:42.027088Z","shell.execute_reply.started":"2022-05-27T00:07:41.978916Z","shell.execute_reply":"2022-05-27T00:07:42.026385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.028401Z","iopub.execute_input":"2022-05-27T00:07:42.028668Z","iopub.status.idle":"2022-05-27T00:07:42.039148Z","shell.execute_reply.started":"2022-05-27T00:07:42.028632Z","shell.execute_reply":"2022-05-27T00:07:42.038027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing and Generate the Dataset","metadata":{}},{"cell_type":"markdown","source":"### Preprocessing","metadata":{}},{"cell_type":"code","source":"# Check for image string\n\nimg = TRAIN_DIR + train_df.id[5]\nfobj = open(img, \"rb\")\nfobj.peek(10)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-05-27T00:07:42.048876Z","iopub.execute_input":"2022-05-27T00:07:42.049355Z","iopub.status.idle":"2022-05-27T00:07:42.072229Z","shell.execute_reply.started":"2022-05-27T00:07:42.049318Z","shell.execute_reply":"2022-05-27T00:07:42.066233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ## Checking for Corrupted images\n\n# num_skipped = 0\n# for folder_name in DIR:\n#     folder_path = os.path.join(DB_PATH, folder_name)\n#     for fname in os.listdir(folder_path):\n#         fpath = os.path.join(folder_path, fname)\n#         try:\n#             fobj = open(fpath, \"rb\")\n#             is_jfif = tf.compat.as_bytes('tif') in fobj.peek(10)   # check for 'tif' string\n#         finally:\n#             fobj.close()\n\n#         if not is_jfif:\n#             num_skipped += 1\n#             # Delete corrupted image\n#             print (fpath)\n#             # os.remove(fpath)\n\n# print(\"%d Corrupted images\" % num_skipped)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.073448Z","iopub.execute_input":"2022-05-27T00:07:42.073739Z","iopub.status.idle":"2022-05-27T00:07:42.082431Z","shell.execute_reply.started":"2022-05-27T00:07:42.073700Z","shell.execute_reply":"2022-05-27T00:07:42.080678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Check for any completely black or white images\n\n# dark_th = 10 / 255                             # If no pixel reaches this threshold, image is considered too dark\n# bright_th = 245 / 255                          # If no pixel is under this threshold, image is considerd too bright\n# too_dark_idx = []\n# too_bright_idx = []\n\n# x_tot = np.zeros(3)\n# x2_tot = np.zeros(3)\n# counted_ones = 0\n\n# for i, idx in tqdm_notebook(enumerate(train_df['id']), 'Computing...(220.025 total files)'):\n#     path = os.path.join(TRAIN_DIR, idx)\n#     imagearray = imread(path).reshape(-1,3)\n    \n#     if((imagearray.max() / 255) < dark_th):            # is this too dark\n#         too_dark_idx.append(idx)\n#         continue                                       # do not include in statistics\n    \n#     if((imagearray.min() / 255) > bright_th):          # is this too bright\n#         too_bright_idx.append(idx)\n#         continue                                       # do not include in statistics\n\n# print('There was {0} extremely dark image'.format(len(too_dark_idx)))\n# print('and {0} extremely bright images'.format(len(too_bright_idx)))\n# print('Dark one:')\n# print(too_dark_idx)\n# print('Bright ones:')\n# print(too_bright_idx)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.083911Z","iopub.execute_input":"2022-05-27T00:07:42.084138Z","iopub.status.idle":"2022-05-27T00:07:42.100126Z","shell.execute_reply.started":"2022-05-27T00:07:42.084109Z","shell.execute_reply":"2022-05-27T00:07:42.097140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unusable = too_dark_idx + too_bright_idx\n\nunusable = ['9369c7278ec8bcc6c880d99194de09fc2bd4efbe.tif', '9071b424ec2e84deeb59b54d2450a6d0172cf701.tif',\n            'f6f1d771d14f7129a6c3ac2c220d90992c30c10b.tif', '5f30d325d895d873d3e72a82ffc0101c45cba4a8.tif',\n            '54df3640d17119486e5c5f98019d2a92736feabc.tif', '5a268c0241b8510465cb002c4452d63fec71028a.tif',\n            'c448cd6574108cf14514ad5bc27c0b2c97fc1a83.tif']\n\nplt.figure(figsize=(10,10))\ni = 0\nfor n in unusable:\n    img = imread(TRAIN_DIR + n)\n    plt.subplot(6,6,i+1)\n    plt.imshow(img) \n    plt.axis('off')\n    i = i+1\n    plt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.101228Z","iopub.execute_input":"2022-05-27T00:07:42.101448Z","iopub.status.idle":"2022-05-27T00:07:42.580341Z","shell.execute_reply.started":"2022-05-27T00:07:42.101421Z","shell.execute_reply":"2022-05-27T00:07:42.579727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Remove corrupted or unusable\n\nfor n in unusable:\n    idx = train_df[train_df['id'] == n].index\n    train_df.drop(idx, inplace=True)\n    print ('Deleting image ', n)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.583808Z","iopub.execute_input":"2022-05-27T00:07:42.587641Z","iopub.status.idle":"2022-05-27T00:07:42.965256Z","shell.execute_reply.started":"2022-05-27T00:07:42.587602Z","shell.execute_reply":"2022-05-27T00:07:42.964462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot few Train images\n\nsample = train_df.sample(n=18).reset_index()\nplt.figure(figsize=(10,10))\nplt.suptitle('Histopathologic scans of lymph node sections',fontsize=16)\nfor i, row in sample.iterrows():\n    img = imread(TRAIN_DIR + f'{row.id}')    \n    label = row.label\n\n    plt.subplot(6,6,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:42.966527Z","iopub.execute_input":"2022-05-27T00:07:42.967286Z","iopub.status.idle":"2022-05-27T00:07:43.694747Z","shell.execute_reply.started":"2022-05-27T00:07:42.967246Z","shell.execute_reply":"2022-05-27T00:07:43.694056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,5, figsize=(20,8))\nfig.suptitle('Histopathologic scans of lymph node sections',fontsize=20)\n\n\n# Negatives\nfor i, idx in enumerate(train_df[train_df['label'] == '0']['id'][:5]):\n    path = os.path.join(TRAIN_DIR, idx)\n    img = imread(path)\n    ax[0,i].imshow(img)\nax[0,0].set_ylabel('Negative samples', size='large')\n\n\n# # Positives\nfor i, idx in enumerate(train_df[train_df['label'] == '1']['id'][:5]):\n    path = os.path.join(TRAIN_DIR, idx)\n    img = imread(path)\n    ax[1,i].imshow(img)\nax[1,0].set_ylabel('Positive samples', size='large');","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:43.696160Z","iopub.execute_input":"2022-05-27T00:07:43.696767Z","iopub.status.idle":"2022-05-27T00:07:45.191312Z","shell.execute_reply.started":"2022-05-27T00:07:43.696717Z","shell.execute_reply":"2022-05-27T00:07:45.190539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:45.192444Z","iopub.execute_input":"2022-05-27T00:07:45.192794Z","iopub.status.idle":"2022-05-27T00:07:45.205304Z","shell.execute_reply.started":"2022-05-27T00:07:45.192764Z","shell.execute_reply":"2022-05-27T00:07:45.204669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Generators","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nSEED           = 1337\nSPLIT_SIZE     = 0.2\nBATCH_SIZE     = 32\nIMAGE_SIZE     = (96,96)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:45.206699Z","iopub.execute_input":"2022-05-27T00:07:45.207714Z","iopub.status.idle":"2022-05-27T00:07:45.212950Z","shell.execute_reply.started":"2022-05-27T00:07:45.207674Z","shell.execute_reply":"2022-05-27T00:07:45.212378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(rescale=1./255., validation_split=SPLIT_SIZE)\n\ntrain_gen = datagen.flow_from_dataframe(\n    dataframe  = train_df,\n    directory  = TRAIN_DIR,\n    color_mode = 'rgb',\n    x_col      = 'id',\n    y_col      = 'label',\n    subset     = 'training',\n    batch_size = BATCH_SIZE,\n    seed       = SEED,\n    shuffle    = True,\n    class_mode = 'binary',\n    target_size = IMAGE_SIZE)\n\nvalid_gen = datagen.flow_from_dataframe(\n    dataframe  = train_df,\n    directory  = TRAIN_DIR,\n    color_mode = 'rgb',\n    x_col      = 'id',\n    y_col      = 'label',\n    subset     = 'validation',\n    batch_size = BATCH_SIZE,\n    seed       = SEED,\n    shuffle    = True,\n    class_mode = 'binary',\n    target_size = IMAGE_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:07:45.214073Z","iopub.execute_input":"2022-05-27T00:07:45.214889Z","iopub.status.idle":"2022-05-27T00:10:17.247498Z","shell.execute_reply.started":"2022-05-27T00:07:45.214854Z","shell.execute_reply":"2022-05-27T00:10:17.246729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_steps = np.ceil(len(train_gen) / BATCH_SIZE)\nval_steps = np.ceil(len(valid_gen) / BATCH_SIZE)\n\nprint('Steps:')\nprint('Train: %d | Validation: %d ' %(train_steps, val_steps))","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.248890Z","iopub.execute_input":"2022-05-27T00:10:17.249320Z","iopub.status.idle":"2022-05-27T00:10:17.256501Z","shell.execute_reply.started":"2022-05-27T00:10:17.249281Z","shell.execute_reply":"2022-05-27T00:10:17.255504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building a Model","metadata":{}},{"cell_type":"code","source":"# Model Libraries\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Input, Dropout, Flatten, Conv2D, MaxPooling2D, Dense, Activation\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.258041Z","iopub.execute_input":"2022-05-27T00:10:17.258355Z","iopub.status.idle":"2022-05-27T00:10:17.266913Z","shell.execute_reply.started":"2022-05-27T00:10:17.258275Z","shell.execute_reply":"2022-05-27T00:10:17.266216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variables\n\nkernel_size     = (3,3)\npool_size       = (2,2)\ninput_shape     = (96,96,3)\n\nfirst_filters   = 32\nsecond_filters  = 64\nthird_filters   = 128\nfouth_filters   = 256\n\ndropout_conv = 0.3\ndropout_dense = 0.3\n\nearly_stopping  = EarlyStopping(\n        monitor = 'val_acc',\n      min_delta = 0.001,\n       patience = 5,\n        verbose = 1,\n           mode = 'auto')\n\nreduce_lr    = ReduceLROnPlateau(\n    monitor  ='val_acc',\n    factor   = 0.5,\n    patience = 2,\n    verbose  = 1,\n    mode     = 'max',\n    min_lr   = 0.00001)\n\ncallbacks       = [early_stopping, reduce_lr]\n\noptimizer       = Adam(learning_rate=0.0001)                 # SGD(lr=0.001, momentum=0.9), Adam, RMSprop\nloss            = 'binary_crossentropy'                      # 'categorical_crossentropy', 'binary_crossentropy'\nmetric          = 'accuracy'\nactivation      = 'sigmoid'                                  # 'sigmoid'; 'softmax'\nepochs          = 10\nval_split       = 0.2","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.268327Z","iopub.execute_input":"2022-05-27T00:10:17.268590Z","iopub.status.idle":"2022-05-27T00:10:17.294371Z","shell.execute_reply.started":"2022-05-27T00:10:17.268535Z","shell.execute_reply":"2022-05-27T00:10:17.293627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####################  Plot Loss Curves  #####################\n\ndef Plot_Train(hlist, start=1):\n\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n  \n    epoch_range = range(start, len(history['loss']) +1)\n    s           = slice(start-1, None)\n    n           = int(len(history.keys()) / 2)\n    \n    plt.figure(figsize=[14,4])\n    for i in range(n):\n        k = list(history.keys())[i]\n        plt.subplot(1, n, i+1)\n        plt.plot(epoch_range, history[k][s], label='Training')\n        plt.plot(epoch_range, history['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.297364Z","iopub.execute_input":"2022-05-27T00:10:17.297903Z","iopub.status.idle":"2022-05-27T00:10:17.306574Z","shell.execute_reply.started":"2022-05-27T00:10:17.297865Z","shell.execute_reply":"2022-05-27T00:10:17.305706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Building the Base Model\n\ndef model_vgg16(input_shape, activation):\n    \n    model = tf.keras.applications.vgg16.VGG16(\n        input_shape   = input_shape,\n        include_top   = False,\n        weights = 'imagenet')\n    \n    x = model.layers[-1].output\n    # model.layers.pop()\n    x = layers.GlobalAveragePooling2D()(x)\n    output = layers.Dense(1, activation=activation)(x)\n    model.trainable = False    \n    model = keras.Model(inputs=model.input, outputs=output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.307781Z","iopub.execute_input":"2022-05-27T00:10:17.308613Z","iopub.status.idle":"2022-05-27T00:10:17.316231Z","shell.execute_reply.started":"2022-05-27T00:10:17.308577Z","shell.execute_reply":"2022-05-27T00:10:17.315459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_mobile(input_shape, activation):\n\n    model = tf.keras.applications.MobileNetV3Large(\n                             input_shape = input_shape,\n                             # input_tensor=input_shape,\n                             alpha       = 1.0,\n                             minimalistic= False,\n                             include_top = False,\n                             weights     = 'imagenet',\n                             classes     = 1000,\n                             pooling     = None,\n                             dropout_rate= 0.2,\n                             classifier_activation='softmax',\n                             include_preprocessing=True)\n      \n    x = model.layers[-1].output\n    x = layers.GlobalAveragePooling2D()(x)\n    output = layers.Dense(1, activation=activation)(x)\n    model.trainable = False \n    model = keras.Model(inputs=model.input, outputs=output)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.317565Z","iopub.execute_input":"2022-05-27T00:10:17.317822Z","iopub.status.idle":"2022-05-27T00:10:17.329820Z","shell.execute_reply.started":"2022-05-27T00:10:17.317788Z","shell.execute_reply":"2022-05-27T00:10:17.329051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_custom(input_shape, activation):\n    \n    model = Sequential()\n    model.add(Conv2D(first_filters, kernel_size, activation = 'relu', input_shape = input_shape))\n    model.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\n    model.add(Conv2D(first_filters, kernel_size, activation = 'relu'))\n    model.add(MaxPooling2D(pool_size = pool_size)) \n    model.add(Dropout(dropout_conv))\n    \n    model.add(Conv2D(second_filters, kernel_size, activation ='relu'))\n    model.add(Conv2D(second_filters, kernel_size, activation ='relu'))\n    model.add(Conv2D(second_filters, kernel_size, activation ='relu'))\n    model.add(MaxPooling2D(pool_size = pool_size))\n    model.add(Dropout(dropout_conv))\n    \n    model.add(Conv2D(third_filters, kernel_size, activation ='relu'))\n    model.add(Conv2D(third_filters, kernel_size, activation ='relu'))\n    model.add(Conv2D(third_filters, kernel_size, activation ='relu'))\n    model.add(MaxPooling2D(pool_size = pool_size))\n    model.add(Dropout(dropout_conv))\n    \n    model.add(Flatten())\n    model.add(Dense(256, activation = \"relu\"))\n    model.add(Dropout(dropout_dense))\n    model.add(Dense(1, activation = activation))\n    \n    return model    ","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.330918Z","iopub.execute_input":"2022-05-27T00:10:17.331258Z","iopub.status.idle":"2022-05-27T00:10:17.343374Z","shell.execute_reply.started":"2022-05-27T00:10:17.331221Z","shell.execute_reply":"2022-05-27T00:10:17.342541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_custom2(input_shape, activation):\n    \n    model = Sequential()\n    model.add(Conv2D(first_filters, kernel_size, activation = 'relu', input_shape = input_shape))\n    model.add(Conv2D(first_filters, kernel_size, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(MaxPooling2D(pool_size = pool_size)) \n    model.add(Dropout(dropout_conv))\n    \n    model.add(Conv2D(second_filters, kernel_size, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(Conv2D(second_filters, kernel_size, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(MaxPooling2D(pool_size = pool_size))\n    model.add(Dropout(dropout_conv))\n    \n    model.add(Conv2D(third_filters, kernel_size, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(Conv2D(third_filters, kernel_size, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(MaxPooling2D(pool_size = pool_size))\n    model.add(Dropout(dropout_conv))\n    \n    model.add(Flatten())\n    model.add(Dense(256, use_bias=False))\n    model.add(BatchNormalization())\n    model.add(Activation(\"relu\"))\n    model.add(Dropout(dropout_dense))\n    model.add(Dense(1, activation = \"sigmoid\"))\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:17.344676Z","iopub.execute_input":"2022-05-27T00:10:17.345069Z","iopub.status.idle":"2022-05-27T00:10:17.356984Z","shell.execute_reply.started":"2022-05-27T00:10:17.345036Z","shell.execute_reply":"2022-05-27T00:10:17.356102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_vgg = model_vgg16(input_shape, activation)\nmodel_vgg.compile(loss=loss, optimizer=optimizer, metrics=[metric, tf.keras.metrics.AUC()])\n\n# model.save(DB_PATH + 'HCDmVGG16.h5',\n#         overwrite=True,\n#         include_optimizer=True,\n#         save_format=None,\n#         signatures=None,\n#         options=None,\n#         save_traces=True)\n\nmodel_vgg.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-27T00:10:17.358287Z","iopub.execute_input":"2022-05-27T00:10:17.359430Z","iopub.status.idle":"2022-05-27T00:10:19.153056Z","shell.execute_reply.started":"2022-05-27T00:10:17.359394Z","shell.execute_reply":"2022-05-27T00:10:19.152341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Trainning the Model","metadata":{}},{"cell_type":"code","source":"# model.fit(\n#     x                     = None,               \n#     y                     = None,               \n#     batch_size            = None,                # Number of samples per gradient update\n#     epochs                = 1,                   # Number of iterations over the entire x and y\n#     verbose               = 'auto',              # Verbosity mode. 0 = silent, 1 = progress bar, 2 = one line per epoch.\n#     callbacks             = None,                # List of keras.callbacks.Callback instances\n#     validation_split      = 0.0,                 # Fraction of the training data to be used as validation data\n#     validation_data       = None,                # Data on which to evaluate the loss and any model metrics\n#     shuffle               = True,                # Boolean (to shuffle training data before each epoch) or str (for 'batch')\n#     class_weight          = None,                # \n#     sample_weight         = None,                # \n#     initial_epoch         = 0,                   # \n#     steps_per_epoch       = None,                # \n#     validation_steps      = None,                # \n#     validation_batch_size = None,                # \n#     validation_freq       = 1,                   # \n#     max_queue_size        = 10,                  # \n#     workers               = 4,                   # \n#     use_multiprocessing   = False)               # ","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:10:19.154260Z","iopub.execute_input":"2022-05-27T00:10:19.154500Z","iopub.status.idle":"2022-05-27T00:10:19.159524Z","shell.execute_reply.started":"2022-05-27T00:10:19.154466Z","shell.execute_reply":"2022-05-27T00:10:19.158510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = time.time()\n# with tf.device('/CPU:0'):              # Running on CPU\nh1 = model_vgg.fit(x                = train_gen,\n                   steps_per_epoch  = train_steps,\n                   validation_data  = valid_gen,\n                   validation_steps = val_steps,\n                   epochs           = epochs)\n\nprint('Fitting model in %.2f secs' % (time.time()-s))\n\n# pickle.dump(h3, open(f'HCDmVGG16.pkl', 'wb'))\n\n# with open(DB_PATH 'HCDmVGG16.pkl', 'wb') as file_pkl:\n#     pickle.dump(history.history, file_pkl)\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-27T00:10:19.161162Z","iopub.execute_input":"2022-05-27T00:10:19.161468Z","iopub.status.idle":"2022-05-27T00:16:55.415640Z","shell.execute_reply.started":"2022-05-27T00:10:19.161432Z","shell.execute_reply":"2022-05-27T00:16:55.414868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mod1 = tf.keras.models.load_model(\n#     DB_PATH + 'HCDmVGG16.h5',\n#     custom_objects=None,\n#     compile=True,\n#     options=None)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:16:55.417181Z","iopub.execute_input":"2022-05-27T00:16:55.417488Z","iopub.status.idle":"2022-05-27T00:16:55.422949Z","shell.execute_reply.started":"2022-05-27T00:16:55.417450Z","shell.execute_reply":"2022-05-27T00:16:55.422268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Plot_Train([h1])","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:16:55.423974Z","iopub.execute_input":"2022-05-27T00:16:55.424604Z","iopub.status.idle":"2022-05-27T00:16:55.849693Z","shell.execute_reply.started":"2022-05-27T00:16:55.424564Z","shell.execute_reply":"2022-05-27T00:16:55.849017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model 2","metadata":{}},{"cell_type":"code","source":"model_mob = model_mobile(input_shape, activation)\nmodel_mob.compile(loss=loss, optimizer=optimizer, metrics=[metric, tf.keras.metrics.AUC()])\n\n# model.save(DB_PATH + 'HCDmMNv3LG.h5',\n#            overwrite=True,\n#            include_optimizer=True,\n#            save_format=None,\n#            signatures=None,\n#            options=None,\n#            save_traces=True)\n\nmodel_mob.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:16:55.850871Z","iopub.execute_input":"2022-05-27T00:16:55.851270Z","iopub.status.idle":"2022-05-27T00:16:58.723435Z","shell.execute_reply.started":"2022-05-27T00:16:55.851234Z","shell.execute_reply":"2022-05-27T00:16:58.722684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = time.time()\n# with tf.device('/CPU:0'):              # Running on CPU\nh2 = model_mob.fit(x                = train_gen,\n                   steps_per_epoch  = train_steps,\n                   validation_data  = valid_gen,\n                   validation_steps = val_steps,\n                   epochs           = epochs)\n\nprint('Fitting model in %.2f secs' % (time.time()-s))\n\n# pickle.dump(h3, open(f'HCDmMNv3LG.pkl', 'wb'))\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-27T00:16:58.724766Z","iopub.execute_input":"2022-05-27T00:16:58.725005Z","iopub.status.idle":"2022-05-27T00:23:07.112511Z","shell.execute_reply.started":"2022-05-27T00:16:58.724972Z","shell.execute_reply":"2022-05-27T00:23:07.111263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mod2 = tf.keras.models.load_model(\n#     DB_PATH + 'HCDmMNv3LG.h5',\n#     custom_objects=None,\n#     compile=True,\n#     options=None)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:23:07.114004Z","iopub.execute_input":"2022-05-27T00:23:07.114289Z","iopub.status.idle":"2022-05-27T00:23:07.119082Z","shell.execute_reply.started":"2022-05-27T00:23:07.114253Z","shell.execute_reply":"2022-05-27T00:23:07.118224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Plot_Train([h2])","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:23:07.120659Z","iopub.execute_input":"2022-05-27T00:23:07.120969Z","iopub.status.idle":"2022-05-27T00:23:07.573685Z","shell.execute_reply.started":"2022-05-27T00:23:07.120932Z","shell.execute_reply":"2022-05-27T00:23:07.573006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model 3","metadata":{}},{"cell_type":"code","source":"model_c1 = model_custom(input_shape, activation)\nmodel_c1.compile(loss=loss, optimizer=optimizer, metrics=[metric, tf.keras.metrics.AUC()])\n\n# model.save(DB_PATH + 'HCDmCustom.h5',\n#         overwrite=True,\n#         include_optimizer=True,\n#         save_format=None,\n#         signatures=None,\n#         options=None,\n#         save_traces=True)\n\nmodel_c1.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-27T00:23:07.574995Z","iopub.execute_input":"2022-05-27T00:23:07.575461Z","iopub.status.idle":"2022-05-27T00:23:07.697938Z","shell.execute_reply.started":"2022-05-27T00:23:07.575424Z","shell.execute_reply":"2022-05-27T00:23:07.697224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = time.time()\n# with tf.device('/CPU:0'):              # Running on CPU\nh3 = model_c1.fit(x                = train_gen,\n                  steps_per_epoch  = train_steps,\n                  validation_data  = valid_gen,\n                  validation_steps = val_steps,\n                  epochs           = epochs)\n\nprint('Fitting model in %.2f secs' % (time.time()-s))\n\n# pickle.dump(h3, open(f'HCDmCustom.pkl', 'wb'))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-27T00:23:07.700129Z","iopub.execute_input":"2022-05-27T00:23:07.700391Z","iopub.status.idle":"2022-05-27T00:29:39.921296Z","shell.execute_reply.started":"2022-05-27T00:23:07.700357Z","shell.execute_reply":"2022-05-27T00:29:39.920343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mod3 = tf.keras.models.load_model(\n#     DB_PATH + 'HCDmCustom.h5',\n#     custom_objects=None,\n#     compile=True,\n#     options=None)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:29:39.922684Z","iopub.execute_input":"2022-05-27T00:29:39.923020Z","iopub.status.idle":"2022-05-27T00:29:39.927319Z","shell.execute_reply.started":"2022-05-27T00:29:39.922979Z","shell.execute_reply":"2022-05-27T00:29:39.926470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Plot_Train([h3])","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:29:39.930863Z","iopub.execute_input":"2022-05-27T00:29:39.931498Z","iopub.status.idle":"2022-05-27T00:29:40.575572Z","shell.execute_reply.started":"2022-05-27T00:29:39.931464Z","shell.execute_reply":"2022-05-27T00:29:40.574882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model 4","metadata":{}},{"cell_type":"code","source":"model_c2 = model_custom2(input_shape, activation)\nmodel_c2.compile(loss=loss, optimizer=optimizer, metrics=[metric, tf.keras.metrics.AUC()])\n\n# model.save(DB_PATH + 'HCDmCustom2.h5',\n#         overwrite=True,\n#         include_optimizer=True,\n#         save_format=None,\n#         signatures=None,\n#         options=None,\n#         save_traces=True)\n\nmodel_c2.summary()","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:29:40.582884Z","iopub.execute_input":"2022-05-27T00:29:40.585040Z","iopub.status.idle":"2022-05-27T00:29:40.778218Z","shell.execute_reply.started":"2022-05-27T00:29:40.585000Z","shell.execute_reply":"2022-05-27T00:29:40.777531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = time.time()\nh4 = model_c2.fit(x                = train_gen,\n                  steps_per_epoch  = train_steps,\n                  validation_data  = valid_gen,\n                  validation_steps = val_steps,\n                  epochs           = epochs)\n\nprint('Fitting model in %.2f secs' % (time.time()-s))","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:29:40.779440Z","iopub.execute_input":"2022-05-27T00:29:40.779675Z","iopub.status.idle":"2022-05-27T00:32:59.610881Z","shell.execute_reply.started":"2022-05-27T00:29:40.779643Z","shell.execute_reply":"2022-05-27T00:32:59.610094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Plot_Train([h4])","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:32:59.612965Z","iopub.execute_input":"2022-05-27T00:32:59.613791Z","iopub.status.idle":"2022-05-27T00:33:00.060988Z","shell.execute_reply.started":"2022-05-27T00:32:59.613752Z","shell.execute_reply":"2022-05-27T00:33:00.060157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Validating the Model","metadata":{}},{"cell_type":"code","source":"### Evaluate the model\n### Returns the loss value & metrics values for the model in test mode.\n\n# model.evaluate(x              = valid_gen,\n#                y              = None,\n#                steps          = test_steps\n#                batch_size     = None,\n#                verbose        = 'auto',\n#                sample_weight  = None,\n#                callbacks      = None,\n#                max_queue_size = 10,\n#                workers        = 1,\n#                use_multiprocessing = False,\n#                return_dict    = False,","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:33:00.062496Z","iopub.execute_input":"2022-05-27T00:33:00.062763Z","iopub.status.idle":"2022-05-27T00:33:00.066916Z","shell.execute_reply.started":"2022-05-27T00:33:00.062727Z","shell.execute_reply":"2022-05-27T00:33:00.066000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predicting the Model","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(DB_PATH + 'sample_submission.csv')\ntest_df['filename'] = test_df.id + '.tif'","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:33:00.068312Z","iopub.execute_input":"2022-05-27T00:33:00.068558Z","iopub.status.idle":"2022-05-27T00:33:00.152877Z","shell.execute_reply.started":"2022-05-27T00:33:00.068525Z","shell.execute_reply":"2022-05-27T00:33:00.152203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Test Images:', len(os.listdir(TEST_DIR)))\n\ndatagen_test = ImageDataGenerator(rescale=1./255.)\n\ntest_gen = datagen_test.flow_from_dataframe(\n    dataframe  = test_df,\n    directory  = TEST_DIR,\n    color_mode = 'rgb',\n    x_col      = 'filename',\n    batch_size = 32,\n    seed       = SEED,\n    shuffle    = False,\n    class_mode = None,\n    target_size = IMAGE_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:33:00.154021Z","iopub.execute_input":"2022-05-27T00:33:00.154336Z","iopub.status.idle":"2022-05-27T00:34:46.468468Z","shell.execute_reply.started":"2022-05-27T00:33:00.154292Z","shell.execute_reply":"2022-05-27T00:34:46.467652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_steps = np.ceil(len(test_gen) / BATCH_SIZE)\ntest_images_path = len(os.listdir(TEST_DIR))\n\nprint('Test Images in path:', test_images_path)\nprint('Test Dataframe Size:', len(test_df))\nprint('Steps: ', test_steps)","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:34:46.469688Z","iopub.execute_input":"2022-05-27T00:34:46.470584Z","iopub.status.idle":"2022-05-27T00:34:46.506460Z","shell.execute_reply.started":"2022-05-27T00:34:46.470545Z","shell.execute_reply":"2022-05-27T00:34:46.505617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_c2.predict(\n    test_gen,\n#    steps=test_steps,\n    verbose=1)\n\npredictions.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:36:53.199494Z","iopub.execute_input":"2022-05-27T00:36:53.199838Z","iopub.status.idle":"2022-05-27T00:41:50.134879Z","shell.execute_reply.started":"2022-05-27T00:36:53.199798Z","shell.execute_reply":"2022-05-27T00:41:50.134061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(DB_PATH + 'sample_submission.csv', index_col='id')\nsubmission.label = predictions","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:41:50.136622Z","iopub.execute_input":"2022-05-27T00:41:50.137396Z","iopub.status.idle":"2022-05-27T00:41:50.195862Z","shell.execute_reply.started":"2022-05-27T00:41:50.137357Z","shell.execute_reply":"2022-05-27T00:41:50.195157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-27T00:43:25.834586Z","iopub.execute_input":"2022-05-27T00:43:25.834871Z","iopub.status.idle":"2022-05-27T00:43:26.002996Z","shell.execute_reply.started":"2022-05-27T00:43:25.834840Z","shell.execute_reply":"2022-05-27T00:43:26.002231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}