{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T02:22:11.744364Z","iopub.execute_input":"2022-07-27T02:22:11.745024Z","iopub.status.idle":"2022-07-27T02:22:11.748959Z","shell.execute_reply.started":"2022-07-27T02:22:11.744989Z","shell.execute_reply":"2022-07-27T02:22:11.748244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# directories\n\nmain_dir = '/kaggle/input/uw-madison-gi-tract-image-segmentation'\ntrain_folder = os.path.join(main_dir, 'train')\ntest_folder = os.path.join(main_dir, 'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:21.522811Z","iopub.execute_input":"2022-07-27T02:22:21.523157Z","iopub.status.idle":"2022-07-27T02:22:21.528057Z","shell.execute_reply.started":"2022-07-27T02:22:21.523128Z","shell.execute_reply":"2022-07-27T02:22:21.526962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Download train dataset","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(main_dir, 'train.csv'))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:21.609237Z","iopub.execute_input":"2022-07-27T02:22:21.609536Z","iopub.status.idle":"2022-07-27T02:22:22.233335Z","shell.execute_reply.started":"2022-07-27T02:22:21.609510Z","shell.execute_reply":"2022-07-27T02:22:22.232462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Investigate train dataset and preprocess it","metadata":{}},{"cell_type":"code","source":"print(len(train_df))\nprint('class null values: {:.2f}%'.format(train_df['class'].isna().sum() / len(train_df) * 100))\nprint('segmentation null values: {:.2f}%'.format(train_df['segmentation'].isna().sum() / len(train_df) * 100))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.234977Z","iopub.execute_input":"2022-07-27T02:22:22.235321Z","iopub.status.idle":"2022-07-27T02:22:22.263895Z","shell.execute_reply.started":"2022-07-27T02:22:22.235287Z","shell.execute_reply":"2022-07-27T02:22:22.263136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how many mask-classes are presented\nclasses = train_df.loc[:, 'class'].unique().tolist()\nclasses","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.265035Z","iopub.execute_input":"2022-07-27T02:22:22.265335Z","iopub.status.idle":"2022-07-27T02:22:22.281884Z","shell.execute_reply.started":"2022-07-27T02:22:22.265305Z","shell.execute_reply":"2022-07-27T02:22:22.281026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# does each image have three masks (by number of classes)?\nfor cl in classes:\n    print('{}: {}'.format(cl, len(train_df[train_df['class'] == cl])))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.284079Z","iopub.execute_input":"2022-07-27T02:22:22.284417Z","iopub.status.idle":"2022-07-27T02:22:22.340529Z","shell.execute_reply.started":"2022-07-27T02:22:22.284383Z","shell.execute_reply":"2022-07-27T02:22:22.339769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group train dataset by id, add 3 new columns: 'large_bowel', 'small_bowel', 'stomach'\n#--------------------------------------------------------------------------------------\ntrain_df_grouped = train_df.copy()\ntrain_df_grouped.set_index('id', inplace = True)\n\nseg_list = []\nfor cl in classes:\n    seg = train_df_grouped[train_df_grouped['class'] == cl]['segmentation']\n    seg.name = cl\n    seg_list.append(seg)\n    \ntrain_df_grouped = pd.concat(seg_list, axis=1).reset_index()\ntrain_df_grouped.fillna('', inplace = True)\ntrain_df_grouped.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.341535Z","iopub.execute_input":"2022-07-27T02:22:22.341936Z","iopub.status.idle":"2022-07-27T02:22:22.439911Z","shell.execute_reply.started":"2022-07-27T02:22:22.341903Z","shell.execute_reply":"2022-07-27T02:22:22.439005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_case_day_slice(x):\n    \n    #--------------------------------------------------------------------------------------\n    # function that parses a string (full_path or image_id)\n    # and returns case, day, slice_ \n    #--------------------------------------------------------------------------------------\n    \n    case = re.search('case[0-9]+', x).group()[len('case'):]\n    day = re.search('day[0-9]+', x).group()[len('day'):]\n    slice_ = re.search('slice_[0-9]+', x).group()[len('slice_'):]\n    return case, day, slice_","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.441221Z","iopub.execute_input":"2022-07-27T02:22:22.441964Z","iopub.status.idle":"2022-07-27T02:22:22.449275Z","shell.execute_reply.started":"2022-07-27T02:22:22.441926Z","shell.execute_reply":"2022-07-27T02:22:22.448521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from glob import glob\nimport re\n\ncase_day_slice = ['case', 'day', 'slice_']\n\ndef process_df(df, path):\n    \n    #--------------------------------------------------------------------------------------\n    # add new columns:\n    # ['case', 'day', 'slice_']\n    # and\n    # 'full_path'\n    #--------------------------------------------------------------------------------------\n    \n    df = df.copy()\n    df.loc[:, case_day_slice] = df.id.apply(get_case_day_slice).to_list()\n    \n    # get list of all images \n    all_images = glob(os.path.join(path, \"**\", \"*.png\"), recursive = True)\n    img_df = pd.DataFrame(all_images, columns = ['full_path'])\n    img_df.loc[:, case_day_slice] = img_df.full_path.apply(get_case_day_slice).to_list()\n    \n    return df.merge(img_df, on = case_day_slice, how = 'left')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.450790Z","iopub.execute_input":"2022-07-27T02:22:22.451349Z","iopub.status.idle":"2022-07-27T02:22:22.459937Z","shell.execute_reply.started":"2022-07-27T02:22:22.451306Z","shell.execute_reply":"2022-07-27T02:22:22.459122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_grouped = process_df(train_df_grouped, train_folder)\ntrain_df_grouped.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:22.461365Z","iopub.execute_input":"2022-07-27T02:22:22.462374Z","iopub.status.idle":"2022-07-27T02:22:34.174548Z","shell.execute_reply.started":"2022-07-27T02:22:22.462337Z","shell.execute_reply":"2022-07-27T02:22:34.173636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove mislabeled training data\n\n# thank you:\n# https://www.kaggle.com/code/caomaobin/uw-madison-data-incorrect-in-case7day0\n# https://www.kaggle.com/code/dschettler8845/uwm-gi-tract-image-segmentation-eda\n\ntrain_df_grouped = train_df_grouped[(train_df_grouped['case'] != 7) | (train_df_grouped['day'] != 0)].reset_index(drop = True)\ntrain_df_grouped = train_df_grouped[(train_df_grouped['case'] != 81) | (train_df_grouped['day'] != 30)].reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:34.176091Z","iopub.execute_input":"2022-07-27T02:22:34.176484Z","iopub.status.idle":"2022-07-27T02:22:34.225608Z","shell.execute_reply.started":"2022-07-27T02:22:34.176430Z","shell.execute_reply":"2022-07-27T02:22:34.224721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Encode/decode RLE mask functions","metadata":{}},{"cell_type":"code","source":"# see: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\n\n# from competition overview:\n# \"pairs of values that contain a start position and a run length. E.g. '1 3' implies starting at pixel 1 and running a total of 3 pixels (1,2,3).\"\"\n\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[::2], s[1::2])]\n    starts -= 1\n    ends = starts + lengths\n    \n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:34.228729Z","iopub.execute_input":"2022-07-27T02:22:34.229016Z","iopub.status.idle":"2022-07-27T02:22:34.238065Z","shell.execute_reply.started":"2022-07-27T02:22:34.228991Z","shell.execute_reply":"2022-07-27T02:22:34.236816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Display several examples","metadata":{}},{"cell_type":"markdown","source":"### 4.1 Investigate different ways to open/represent image","metadata":{}},{"cell_type":"markdown","source":"This part of notebook appeared when I was wondering what approach to open/decode images would be better to use for exact this project. You can safely skip this part of the notebook ","metadata":{}},{"cell_type":"code","source":"# get an example of image having non-null 'large_bowel' mask\nexample_row = train_df_grouped[train_df_grouped['large_bowel'] != ''].iloc[0, :]\nexample_row","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.239669Z","iopub.execute_input":"2022-07-27T02:22:34.240447Z","iopub.status.idle":"2022-07-27T02:22:34.259865Z","shell.execute_reply.started":"2022-07-27T02:22:34.240410Z","shell.execute_reply":"2022-07-27T02:22:34.258986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = example_row.full_path\nimg_path","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.261588Z","iopub.execute_input":"2022-07-27T02:22:34.262119Z","iopub.status.idle":"2022-07-27T02:22:34.267851Z","shell.execute_reply.started":"2022-07-27T02:22:34.262085Z","shell.execute_reply":"2022-07-27T02:22:34.267008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display an image\n# + image size\n# + values of 100th row\nimport matplotlib.pyplot as plt\n\ndef print_img_info(img):\n    img_arr = np.asarray(img)\n    print(img_arr.shape)\n    print(img_arr[100])\n    plt.imshow(img)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.269393Z","iopub.execute_input":"2022-07-27T02:22:34.270045Z","iopub.status.idle":"2022-07-27T02:22:34.275522Z","shell.execute_reply.started":"2022-07-27T02:22:34.270009Z","shell.execute_reply":"2022-07-27T02:22:34.274753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# matplotlib\n\n# from documentation:\n# PNG images are returned as float arrays (0-1). \n# All other formats are returned as int arrays, with a bit depth determined by the file's contents.\n\n# all values divided by 255 * 255\n\nimport matplotlib.image as mpimg\nprint_img_info(mpimg.imread(img_path, format = 'png'))","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.276782Z","iopub.execute_input":"2022-07-27T02:22:34.277605Z","iopub.status.idle":"2022-07-27T02:22:34.511892Z","shell.execute_reply.started":"2022-07-27T02:22:34.277572Z","shell.execute_reply":"2022-07-27T02:22:34.511037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cv2\n# each value - [0, 255]\n\nimport cv2\nprint_img_info(cv2.imread(img_path, cv2.IMREAD_GRAYSCALE))","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.512833Z","iopub.execute_input":"2022-07-27T02:22:34.513145Z","iopub.status.idle":"2022-07-27T02:22:34.869073Z","shell.execute_reply.started":"2022-07-27T02:22:34.513113Z","shell.execute_reply":"2022-07-27T02:22:34.868339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\nprint_img_info(PIL.Image.open(img_path, formats = ['PNG']))","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:34.870305Z","iopub.execute_input":"2022-07-27T02:22:34.871152Z","iopub.status.idle":"2022-07-27T02:22:35.058699Z","shell.execute_reply.started":"2022-07-27T02:22:34.871114Z","shell.execute_reply":"2022-07-27T02:22:35.057950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# keras\nimport tensorflow as tf\nimg = tf.keras.preprocessing.image.load_img(img_path, color_mode = \"grayscale\")\nimg = tf.keras.preprocessing.image.img_to_array(img)\nprint_img_info(img[:, :, 0])","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:35.059745Z","iopub.execute_input":"2022-07-27T02:22:35.060435Z","iopub.status.idle":"2022-07-27T02:22:40.135292Z","shell.execute_reply.started":"2022-07-27T02:22:35.060396Z","shell.execute_reply":"2022-07-27T02:22:40.134489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_seq = example_row.large_bowel","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:40.136760Z","iopub.execute_input":"2022-07-27T02:22:40.137418Z","iopub.status.idle":"2022-07-27T02:22:40.145067Z","shell.execute_reply.started":"2022-07-27T02:22:40.137378Z","shell.execute_reply":"2022-07-27T02:22:40.144323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = rle_decode(mask_seq, img.shape)\nprint(mask.shape)\nprint(mask[90, :, 0])\nplt.imshow(mask)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:40.146561Z","iopub.execute_input":"2022-07-27T02:22:40.147409Z","iopub.status.idle":"2022-07-27T02:22:40.328931Z","shell.execute_reply.started":"2022-07-27T02:22:40.147377Z","shell.execute_reply":"2022-07-27T02:22:40.328117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_mask = tf.image.resize(mask, (128,128), method = 'nearest').numpy()\nprint(small_mask.shape)\nprint(small_mask[40, :, 0])\nplt.imshow(small_mask)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:40.330179Z","iopub.execute_input":"2022-07-27T02:22:40.330672Z","iopub.status.idle":"2022-07-27T02:22:42.928292Z","shell.execute_reply.started":"2022-07-27T02:22:40.330636Z","shell.execute_reply":"2022-07-27T02:22:42.927364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_mask = cv2.resize(mask, (128,128))\nprint(small_mask.shape)\nprint(small_mask[40, :])\nplt.imshow(small_mask)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-27T02:22:42.930141Z","iopub.execute_input":"2022-07-27T02:22:42.930829Z","iopub.status.idle":"2022-07-27T02:22:43.118102Z","shell.execute_reply.started":"2022-07-27T02:22:42.930779Z","shell.execute_reply":"2022-07-27T02:22:43.117357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.2 Display several examples","metadata":{}},{"cell_type":"code","source":"num = 5\nsegmentation_df_example = train_df_grouped[train_df_grouped.large_bowel != ''].sample(num)\n\nfig, ax = plt.subplots(num, 3, figsize=(18, 8*num))\nfor i in range(num):\n    record = segmentation_df_example.iloc[i, :]\n    \n    img = mpimg.imread(record.full_path, format = 'png')\n    ax[i, 0].imshow(img)\n    ax[i, 0].set_title(record.id)\n    \n    mask = np.zeros(img.shape)\n    for j, cl in enumerate(classes):\n        mask += rle_decode(record[cl], img.shape)*(j + 1) / 4 * np.max(img)\n    ax[i, 1].imshow(mask)\n    \n    ax[i, 2].imshow(img + mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:43.119290Z","iopub.execute_input":"2022-07-27T02:22:43.119735Z","iopub.status.idle":"2022-07-27T02:22:45.310744Z","shell.execute_reply.started":"2022-07-27T02:22:43.119697Z","shell.execute_reply":"2022-07-27T02:22:45.309518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Data Generator","metadata":{}},{"cell_type":"code","source":"from enum import Enum, auto\n\nclass GeneratorMode(Enum):\n    TRAIN = auto()\n    TEST = auto()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:45.312062Z","iopub.execute_input":"2022-07-27T02:22:45.312633Z","iopub.status.idle":"2022-07-27T02:22:45.317133Z","shell.execute_reply.started":"2022-07-27T02:22:45.312597Z","shell.execute_reply":"2022-07-27T02:22:45.316245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimg_size = 160\n\nclass ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size,\n                 mode = GeneratorMode.TRAIN):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.image_size = img_size\n        self.mode = mode\n        \n        self.len = len(df)\n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        images = np.zeros((self.batch_size, self.image_size, self.image_size, 1))\n        masks = np.zeros((self.batch_size, self.image_size, self.image_size, len(classes)))\n        for i , pos in enumerate(range(start, end)):\n            row = self.df.iloc[pos, :]\n            \n            #image\n            image = tf.keras.preprocessing.image.load_img(row['full_path'], color_mode = \"grayscale\")\n            image = tf.keras.preprocessing.image.img_to_array(image)\n            image_shape = image.shape\n            \n            image = tf.image.resize(image, (self.image_size, self.image_size), method = 'nearest').numpy()\n            image = image.reshape((self.image_size, self.image_size))\n            images[i, :, :, 0] = image / 255.\n            \n            # masks (only train/val mode)\n            if self.mode == GeneratorMode.TRAIN:\n                for j, cl in enumerate(classes):\n                    feat = row[cl]\n                    mask = rle_decode(feat, image_shape)\n                    mask = tf.image.resize(mask, (self.image_size, self.image_size), method = 'nearest').numpy()\n                    mask = mask.reshape((self.image_size, self.image_size))\n                    masks[i, :, :, j] = mask\n              \n        return (images, masks) if self.mode == GeneratorMode.TRAIN else images\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:25.975774Z","iopub.execute_input":"2022-07-27T02:23:25.976615Z","iopub.status.idle":"2022-07-27T02:23:25.991259Z","shell.execute_reply.started":"2022-07-27T02:23:25.976573Z","shell.execute_reply":"2022-07-27T02:23:25.990462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split train dataset into train and validation parts\n\nfrom sklearn.model_selection import train_test_split\ntrain_set, val_set = train_test_split(train_df_grouped, test_size = 0.1, shuffle = True, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:29.659593Z","iopub.execute_input":"2022-07-27T02:23:29.660295Z","iopub.status.idle":"2022-07-27T02:23:29.681387Z","shell.execute_reply.started":"2022-07-27T02:23:29.660258Z","shell.execute_reply":"2022-07-27T02:23:29.680416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_gen = ImageDataGen(train_set, 64)\nval_data_gen = ImageDataGen(val_set, 64)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:29.792012Z","iopub.execute_input":"2022-07-27T02:23:29.792846Z","iopub.status.idle":"2022-07-27T02:23:29.796884Z","shell.execute_reply.started":"2022-07-27T02:23:29.792803Z","shell.execute_reply":"2022-07-27T02:23:29.795988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Unet Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.208430Z","iopub.execute_input":"2022-07-27T02:23:38.209354Z","iopub.status.idle":"2022-07-27T02:23:38.217430Z","shell.execute_reply.started":"2022-07-27T02:23:38.209299Z","shell.execute_reply":"2022-07-27T02:23:38.214780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://pyimagesearch.com/2022/02/21/u-net-image-segmentation-in-keras/\ndef double_conv_block(x, n_filters):\n    x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n    x = layers.Conv2D(n_filters, 3, padding = \"same\", activation = \"relu\", kernel_initializer = \"he_normal\")(x)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.235164Z","iopub.execute_input":"2022-07-27T02:23:38.235533Z","iopub.status.idle":"2022-07-27T02:23:38.242075Z","shell.execute_reply.started":"2022-07-27T02:23:38.235470Z","shell.execute_reply":"2022-07-27T02:23:38.241024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def downsample_block(x, n_filters):\n    f = double_conv_block(x, n_filters)\n    p = layers.MaxPool2D(2)(f)\n    p = layers.Dropout(0.3)(p)\n    return f, p","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.285201Z","iopub.execute_input":"2022-07-27T02:23:38.285797Z","iopub.status.idle":"2022-07-27T02:23:38.294074Z","shell.execute_reply.started":"2022-07-27T02:23:38.285760Z","shell.execute_reply":"2022-07-27T02:23:38.293353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def upsample_block(x, conv_features, n_filters):\n    x = layers.Conv2DTranspose(n_filters, 3, 2, padding=\"same\")(x)\n    x = layers.concatenate([x, conv_features])\n    x = layers.Dropout(0.3)(x)\n    x = double_conv_block(x, n_filters)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.297115Z","iopub.execute_input":"2022-07-27T02:23:38.297651Z","iopub.status.idle":"2022-07-27T02:23:38.307529Z","shell.execute_reply.started":"2022-07-27T02:23:38.297618Z","shell.execute_reply":"2022-07-27T02:23:38.306724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_unet_model():\n    inputs = layers.Input(shape=(img_size, img_size, 1))\n    \n    f1, p1 = downsample_block(inputs, 64)\n    f2, p2 = downsample_block(p1, 128)\n    f3, p3 = downsample_block(p2, 256)\n    f4, p4 = downsample_block(p3, 512)\n    \n    bottleneck = double_conv_block(p4, 1024)\n    \n    u6 = upsample_block(bottleneck, f4, 512)\n    u7 = upsample_block(u6, f3, 256)\n    u8 = upsample_block(u7, f2, 128)\n    u9 = upsample_block(u8, f1, 64)\n    \n    outputs = layers.Conv2D(len(classes), 1, padding=\"same\", activation = \"sigmoid\")(u9)\n    \n    unet_model = tf.keras.Model(inputs, outputs, name=\"U-Net\")\n    \n    return unet_model","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.315387Z","iopub.execute_input":"2022-07-27T02:23:38.316056Z","iopub.status.idle":"2022-07-27T02:23:38.329203Z","shell.execute_reply.started":"2022-07-27T02:23:38.316019Z","shell.execute_reply":"2022-07-27T02:23:38.328368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metrics\n\n# https://www.kaggle.com/code/samuelcortinhas/uwmgi-segmentation-unet-keras-inference\n\nfrom keras import backend as K\n\n# addind a 'smooth' value which equals to 1 just to avoid dividing by 0 in case y_true and y_pred consist of 0\ndef dice_coef(y_true, y_pred, smooth = 1):\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = K.sum(y_true_f * y_pred_f)\n    return (2. * intersection + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n\n#intersection over union coefficient\ndef iou_coef(y_true, y_pred, smooth = 1):\n    intersection = K.sum(K.abs(y_true * y_pred), axis=[1,2,3])\n    union = K.sum(y_true,[1,2,3])+K.sum(y_pred,[1,2,3])-intersection\n    iou = K.mean((intersection + smooth) / (union + smooth), axis=0)\n    return iou","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.339580Z","iopub.execute_input":"2022-07-27T02:23:38.340126Z","iopub.status.idle":"2022-07-27T02:23:38.353001Z","shell.execute_reply.started":"2022-07-27T02:23:38.340092Z","shell.execute_reply":"2022-07-27T02:23:38.352200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOSS func\n\nfrom tensorflow.keras.losses import BinaryCrossentropy\n\ndef loss_f(y_true, y_pred):\n    bce = BinaryCrossentropy()\n    return dice_loss(y_true, y_pred) + bce(y_true, y_pred)\n    \ndef dice_loss(y_true, y_pred):\n    return 1. - dice_coef(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.365533Z","iopub.execute_input":"2022-07-27T02:23:38.368000Z","iopub.status.idle":"2022-07-27T02:23:38.376095Z","shell.execute_reply.started":"2022-07-27T02:23:38.367966Z","shell.execute_reply":"2022-07-27T02:23:38.375330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unet_model = build_unet_model()\nunet_model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate = 1e-4),\n                   loss = loss_f, metrics = [dice_coef, iou_coef])\n\nunet_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.393263Z","iopub.execute_input":"2022-07-27T02:23:38.395376Z","iopub.status.idle":"2022-07-27T02:23:38.747747Z","shell.execute_reply.started":"2022-07-27T02:23:38.395342Z","shell.execute_reply":"2022-07-27T02:23:38.746031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# callbacks\nfrom tensorflow.keras.callbacks import EarlyStopping, LearningRateScheduler, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(monitor = 'val_loss', patience = 3, verbose = 1, restore_best_weights = True)\nreduce_lr = ReduceLROnPlateau(monitor = \"val_loss\", factor = 0.1, patience = 2, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.749531Z","iopub.execute_input":"2022-07-27T02:23:38.749924Z","iopub.status.idle":"2022-07-27T02:23:38.757261Z","shell.execute_reply.started":"2022-07-27T02:23:38.749887Z","shell.execute_reply":"2022-07-27T02:23:38.756515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 18\nhistory = unet_model.fit(train_data_gen,\n                         validation_data = val_data_gen, \n                         epochs = EPOCHS, callbacks = [early_stop, reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:38.758585Z","iopub.execute_input":"2022-07-27T02:23:38.759161Z","iopub.status.idle":"2022-07-27T02:24:03.458142Z","shell.execute_reply.started":"2022-07-27T02:23:38.758998Z","shell.execute_reply":"2022-07-27T02:24:03.455534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Print acc/loss ","metadata":{}},{"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\n\niou_coef = history.history['iou_coef']\nval_iou_coef = history.history['val_iou_coef']\n\ndice_coef = history.history['dice_coef']\nval_dice_coef = history.history['val_dice_coef']\n\n\nepochs = range(1, len(loss) + 1)\n\nplt.figure(figsize=(20, 5))\n\n# loss\nplt.subplot(1,3,1)\nplt.plot(epochs, loss, 'bo', label = 'Trainig loss')\nplt.plot(epochs, val_loss, 'r', label = 'Validation loss')\nplt.legend()\n\n# iou\nplt.subplot(1,3,2)\nplt.plot(epochs, iou_coef, 'bo', label = 'Training iou accuracy')\nplt.plot(epochs, val_iou_coef, 'r', label = 'Validation iou accuracy')\nplt.legend()\n\n# dice\nplt.subplot(1,3,3)\nplt.plot(epochs, dice_coef, 'bo', label = 'Training dice accuracy')\nplt.plot(epochs, val_dice_coef, 'r', label = 'Validation dice accuracy')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Predict result on validation set. Display result and compare it with initial masks","metadata":{}},{"cell_type":"code","source":"X, y = val_data_gen[2]\npred = unet_model.predict(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num = 5\nfig, ax = plt.subplots(num, 3, figsize=(18, 8 * num))\n\nfor i in range(num):\n    img = X[i, :, :, 0]\n    masks = y[i]\n    pred_masks = pred[i]\n    ax[i, 0].imshow(img)\n    \n    mask = np.zeros(masks.shape[:-1])\n    for j in range(masks.shape[-1]):\n        m = masks[:, :, j]\n        mask += m*(j+1)/4*np.max(img)\n    ax[i, 1].imshow(mask)\n    \n    pred_mask = np.zeros(masks.shape[:-1])\n    for j, cl in enumerate(classes):\n        m = (pred_masks[:, :, j] > 0.5).astype(np.float32)\n        pred_mask += m*(j+1)/4*np.max(img)\n    ax[i, 2].imshow(pred_mask)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Test dataset","metadata":{}},{"cell_type":"code","source":"# Test set\ntest_df = pd.read_csv('../input/uw-madison-gi-tract-image-segmentation/sample_submission.csv')\n\nif len(test_df) == 0:\n    DEBUG = True\n    test_df = train_df.iloc[:10*16*3,:].copy()\n    test_df.loc[:, \"segmentation\"] = ''\n    test_df = test_df.rename(columns = {\"segmentation\":\"predicted\"})\nelse:\n    DEBUG = False\n\nsubmission = test_df.copy()\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9.1 Download all .png files to DataFrame","metadata":{}},{"cell_type":"code","source":"path = test_folder if not DEBUG else train_folder\npath","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataset structured like 'train_df_grouped' \ntest_df = process_df(test_df['id'].drop_duplicates().to_frame(), path)\ntest_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9.2 DataGenerator","metadata":{}},{"cell_type":"code","source":"test_datagen = ImageDataGen(test_df, batch_size = 1, mode = GeneratorMode.TEST)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9.3 Predict","metadata":{}},{"cell_type":"code","source":"for i in range(len(test_df)):\n    pred = unet_model.predict(test_datagen[i])[0]\n    pred = (pred > 0.5).astype(np.float32)\n    \n    # remember that shape is (160x160x3)!\n    # so we need to resize predicted mask back to image size\n    img = mpimg.imread(test_df.iloc[i, :].full_path, format = 'png')\n    pred = tf.image.resize(pred, img.shape, method = 'nearest').numpy()\n\n    for j in range(pred.shape[-1]):\n        id = test_df.iloc[i, :].id\n        cls = classes[j]\n        mask = pred[:, :, j]\n        submission.loc[(submission['id'] == id) & (submission['class'] == cls), 'predicted'] = rle_encode(mask)\n        \nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display predicted and resized masks\n\nimport random\n\nif DEBUG:\n    num = 5\n    ids = random.choices(test_df.id.unique().tolist(), k = num)\n    fig, ax = plt.subplots(num, 3, figsize=(18, 8 * num))\n\n    for i in range(num):\n        original = train_df_grouped[train_df_grouped.id == ids[i]]\n        predicted = submission[submission.id == ids[i]]\n\n        img = mpimg.imread(original.full_path.values[0], format = 'png')\n        ax[i, 0].imshow(img)\n        ax[i, 0].set_title(ids[i])\n        \n        mask = np.zeros(img.shape)\n        for j, cl in enumerate(classes):\n            mask += rle_decode(original[cl].values[0], img.shape)*(j + 1) / 4 * np.max(img)\n        ax[i, 1].imshow(mask)\n\n        \n        pred_mask = np.zeros(img.shape)\n        for j, cl in enumerate(classes):\n            pred_mask += rle_decode(predicted[predicted['class'] == cl]['predicted'].values[0], img.shape)*(j + 1) / 4 * np.max(img)\n        ax[i, 2].imshow(pred_mask)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9.4 Submit","metadata":{}},{"cell_type":"code","source":"# from overview:\n# Submission file must be named submission.csv\nsubmission.to_csv('submission.csv',index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}