{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-competitions/kaggle/27923/logos/header.png?t=2021-06-02-20-30-25)","metadata":{}},{"cell_type":"markdown","source":"## NoteBook\n\n## Training\n* Training: [Unet 2.5D Augmentation](https://www.kaggle.com/code/juhjoo/uw-madison-gi-tract-training)\n\n## Inference\n* Inference: [Unet 2.5D Augmentation](https://www.kaggle.com/juhjoo/uw-madison-gi-tract-inference)\n\n### Data and Dataset\n* Data: [2.5D TFRecord Data](https://www.kaggle.com/code/juhjoo/uw-madison-gi-tract-dataset-tfrecords)\n* Dataset: [2.5D TFRecord Dataset](https://www.kaggle.com/datasets/juhjoo/uwmadisondataset)\n\n### Summary of Notebook\n\n* Model: UNET\n* Backbone: EfficientNet B4\n* Image Size: 320\n* Learning Rate: maximum 3e-3, cosine decay with warmup\n* Epochs: 30 (2 for warmup and the rests are cosine decay)\n* Batch Size: 64\n* Folds: 5 (trains 1 fold only)\n* Loss: 0.5 BCE + 0.5 Dice\n* Data Augmentations: Albumentations like\n* 2.5D stride 2","metadata":{}},{"cell_type":"markdown","source":"# Import Librarie","metadata":{}},{"cell_type":"code","source":"# Core\nihttps://storage.googleapis.com/kaggle-competitions/kaggle/27923/logos/header.png?t=2021-06-02-20-30-25mport pandas as pd\nimport numpy as np\nimport os\nimport cv2\nimport gc\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm\nfrom datetime import datetime\nimport json,itertools\nfrom typing import Optional\nfrom glob import glob\nimport time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom sklearn.model_selection import StratifiedKFold, KFold, StratifiedGroupKFold\nimport matplotlib.gridspec as gridspec\nimport matplotlib.patches as mpatches\nimport matplotlib as mpl\n\n# keras\nfrom tensorflow import keras\nimport tensorflow as tf\nimport keras\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.layers import Input\nfrom keras.layers.convolutional import Conv2D, Conv2DTranspose\nfrom keras.layers.pooling import MaxPooling2D\nfrom keras.layers.merge import concatenate\nfrom keras.losses import binary_crossentropy\nfrom keras.callbacks import Callback, ModelCheckpoint, EarlyStopping\nfrom keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:46.680983Z","iopub.execute_input":"2022-08-09T16:19:46.681362Z","iopub.status.idle":"2022-08-09T16:19:46.692457Z","shell.execute_reply.started":"2022-08-09T16:19:46.681330Z","shell.execute_reply":"2022-08-09T16:19:46.691268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    wandb = True\n    competition = \"UW-Madison-GI\"\n    _wandb_kernel = \"juhjoo\"\n    debug = False\n    exp_name = \"v1\"\n    comment = \"UNet-EfficientNetB4-160x160-aug-2.5D\"\n\n    # Use verbose=0 for silent, 1 for interactive\n    verbose = 0\n    display_plot = True\n\n    # Device for training\n    device = None  # device is automatically selected\n\n    # Model & Backbone\n    model_name = \"UNet\"\n    backbone = \"efficientnetb4\"\n\n    # Seeding for reproducibility\n    seed = 101\n\n    # Number of folds\n    folds = 5\n\n    # Which Folds to train\n    selected_folds = [0, 1, 2, 3, 4]\n\n    # Image Size\n    img_size = [160, 160]\n    IMAGE_SIZE = 160\n\n    # Batch Size & Epochs\n    batch_size = 8\n    drop_remainder = False\n    epochs = 20\n    steps_per_execution = None\n\n    # Loss & Optimizer\n    optimizer = \"Adam\"\n    lr = 3e-4\n    patience = 5\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    \n    n_splits = 5\n    \n    clip = False\n    \n    augment = True\n    \n    # Horizontal & Vertical Flip\n    hflip = 0.5\n    vflip = 0.5\n\n    # Random Bright\n    brightness_limit = 0.2\n    contrast_limit = 0.2\n    Brightp = 0.75\n\n    # Clip values to [0, 1]\n    clip = False\n\n    # ShiftScaleRotate\n    shift_limit=0.125\n    scale_limit=0.1\n    rotate_limit=20\n    rotatep=0.075\n\n    # CutOut\n    drop_prob = 0.5\n    drop_cnt = 10\n    drop_size = 0.05","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:46.741427Z","iopub.execute_input":"2022-08-09T16:19:46.742125Z","iopub.status.idle":"2022-08-09T16:19:46.753539Z","shell.execute_reply.started":"2022-08-09T16:19:46.742087Z","shell.execute_reply":"2022-08-09T16:19:46.752101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16\nim_width = 320\nim_height = 320","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:46.788483Z","iopub.execute_input":"2022-08-09T16:19:46.788978Z","iopub.status.idle":"2022-08-09T16:19:46.794819Z","shell.execute_reply.started":"2022-08-09T16:19:46.788937Z","shell.execute_reply":"2022-08-09T16:19:46.793620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read and Pre-Process DataSet\n\n* Dataset: [2.5D TFRecord Dataset](https://www.kaggle.com/datasets/juhjoo/uwmadisondataset)","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/uw-madison-gi-tract-image-segmentation/train.csv')\nprint(train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:46.819150Z","iopub.execute_input":"2022-08-09T16:19:46.819902Z","iopub.status.idle":"2022-08-09T16:19:47.189393Z","shell.execute_reply.started":"2022-08-09T16:19:46.819851Z","shell.execute_reply":"2022-08-09T16:19:47.188113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/uw-madison-gi-tract-image-segmentation/sample_submission.csv')\n\nif len(test_df)==0:\n    DEBUG=True\n    test_df = train_df.iloc[:10*16*3,:]\n    test_df[\"segmentation\"]=''\n    test_df=test_df.rename(columns={\"segmentation\":\"predicted\"})\nelse:\n    DEBUG=False\n\nsubmission=test_df.copy()\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:47.192086Z","iopub.execute_input":"2022-08-09T16:19:47.192911Z","iopub.status.idle":"2022-08-09T16:19:47.212791Z","shell.execute_reply.started":"2022-08-09T16:19:47.192861Z","shell.execute_reply":"2022-08-09T16:19:47.211146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(df, subset=\"train\"):\n    #--------------------------------------------------------------------------\n    df[\"case\"] = df[\"id\"].apply(lambda x: int(x.split(\"_\")[0].replace(\"case\", \"\")))\n    df[\"day\"] = df[\"id\"].apply(lambda x: int(x.split(\"_\")[1].replace(\"day\", \"\")))\n    df[\"slice\"] = df[\"id\"].apply(lambda x: x.split(\"_\")[3])\n    #--------------------------------------------------------------------------\n    if (subset==\"train\") or (DEBUG):\n        DIR=\"../input/uw-madison-gi-tract-image-segmentation/train\"\n    else:\n        DIR=\"../input/uw-madison-gi-tract-image-segmentation/test\"\n    \n    all_images = glob(os.path.join(DIR, \"**\", \"*.png\"), recursive=True)\n    x = all_images[0].rsplit(\"/\", 4)[0] ## ../input/uw-madison-gi-tract-image-segmentation/train\n\n    path_partial_list = []\n    for i in range(0, df.shape[0]):\n        path_partial_list.append(os.path.join(x,\n                              \"case\"+str(df[\"case\"].values[i]),\n                              \"case\"+str(df[\"case\"].values[i])+\"_\"+ \"day\"+str(df[\"day\"].values[i]),\n                              \"scans\",\n                              \"slice_\"+str(df[\"slice\"].values[i])))\n    df[\"path_partial\"] = path_partial_list\n    #--------------------------------------------------------------------------\n    path_partial_list = []\n    for i in range(0, len(all_images)):\n        path_partial_list.append(str(all_images[i].rsplit(\"_\",4)[0]))\n\n    tmp_df = pd.DataFrame()\n    tmp_df['path_partial'] = path_partial_list\n    tmp_df['path'] = all_images\n\n    #--------------------------------------------------------------------------\n    df = df.merge(tmp_df, on=\"path_partial\").drop(columns=[\"path_partial\"])\n    #--------------------------------------------------------------------------\n    df[\"width\"] = df[\"path\"].apply(lambda x: int(x[:-4].rsplit(\"_\",4)[1]))\n    df[\"height\"] = df[\"path\"].apply(lambda x: int(x[:-4].rsplit(\"_\",4)[2]))\n    #--------------------------------------------------------------------------\n    del x, path_partial_list, tmp_df\n    #--------------------------------------------------------------------------\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:47.214932Z","iopub.execute_input":"2022-08-09T16:19:47.215401Z","iopub.status.idle":"2022-08-09T16:19:47.231738Z","shell.execute_reply.started":"2022-08-09T16:19:47.215356Z","shell.execute_reply":"2022-08-09T16:19:47.230335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = preprocessing(train_df, subset=\"train\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:47.236414Z","iopub.execute_input":"2022-08-09T16:19:47.238037Z","iopub.status.idle":"2022-08-09T16:19:51.453422Z","shell.execute_reply.started":"2022-08-09T16:19:47.237980Z","shell.execute_reply":"2022-08-09T16:19:51.452295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=preprocessing(test_df, subset=\"test\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:51.454526Z","iopub.execute_input":"2022-08-09T16:19:51.454846Z","iopub.status.idle":"2022-08-09T16:19:52.434947Z","shell.execute_reply.started":"2022-08-09T16:19:51.454806Z","shell.execute_reply":"2022-08-09T16:19:52.433787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def restructure(df, subset=\"train\"):\n    # RESTRUCTURE  DATAFRAME\n    df_out = pd.DataFrame({'id': df['id'][::3]})\n\n    if subset==\"train\":\n        df_out['large_bowel'] = df['segmentation'][::3].values\n        df_out['small_bowel'] = df['segmentation'][1::3].values\n        df_out['stomach'] = df['segmentation'][2::3].values\n\n    df_out['path'] = df['path'][::3].values\n    df_out['case'] = df['case'][::3].values\n    df_out['day'] = df['day'][::3].values\n    df_out['slice'] = df['slice'][::3].values\n    df_out['width'] = df['width'][::3].values\n    df_out['height'] = df['height'][::3].values\n\n    df_out=df_out.reset_index(drop=True)\n    df_out=df_out.fillna('')\n    if subset==\"train\":\n        df_out['count'] = np.sum(df_out.iloc[:,1:4]!='',axis=1).values\n    \n    return df_out","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.436569Z","iopub.execute_input":"2022-08-09T16:19:52.437256Z","iopub.status.idle":"2022-08-09T16:19:52.448601Z","shell.execute_reply.started":"2022-08-09T16:19:52.437212Z","shell.execute_reply":"2022-08-09T16:19:52.447319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=restructure(train_df, subset=\"train\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.450354Z","iopub.execute_input":"2022-08-09T16:19:52.450715Z","iopub.status.idle":"2022-08-09T16:19:52.530478Z","shell.execute_reply.started":"2022-08-09T16:19:52.450682Z","shell.execute_reply":"2022-08-09T16:19:52.529102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=restructure(test_df, subset=\"test\")\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.534343Z","iopub.execute_input":"2022-08-09T16:19:52.534684Z","iopub.status.idle":"2022-08-09T16:19:52.555644Z","shell.execute_reply.started":"2022-08-09T16:19:52.534653Z","shell.execute_reply":"2022-08-09T16:19:52.554439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[(train_df['case']!=7)|(train_df['day']!=0)]\ntrain_df = train_df[(train_df['case']!=81)|(train_df['day']!=30)]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.557138Z","iopub.execute_input":"2022-08-09T16:19:52.557442Z","iopub.status.idle":"2022-08-09T16:19:52.588141Z","shell.execute_reply.started":"2022-08-09T16:19:52.557413Z","shell.execute_reply":"2022-08-09T16:19:52.586881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.591647Z","iopub.execute_input":"2022-08-09T16:19:52.592022Z","iopub.status.idle":"2022-08-09T16:19:52.815067Z","shell.execute_reply.started":"2022-08-09T16:19:52.591987Z","shell.execute_reply":"2022-08-09T16:19:52.813817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run-length encoding\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    \n    return ' '.join(str(x) for x in runs)\n\ndef rle_decode(mask_rle, shape, color=1):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return\n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros((shape[0] * shape[1], shape[2]), dtype=np.float32)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = color\n    return img.reshape(shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.816683Z","iopub.execute_input":"2022-08-09T16:19:52.817090Z","iopub.status.idle":"2022-08-09T16:19:52.827972Z","shell.execute_reply.started":"2022-08-09T16:19:52.817057Z","shell.execute_reply":"2022-08-09T16:19:52.827023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics\ndef dice_coef(y_true, y_pred, smooth=1):\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = K.sum(y_true_f * y_pred_f)\n    return (2. * intersection + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n\ndef iou_coef(y_true, y_pred, smooth=1):\n    intersection = K.sum(K.abs(y_true * y_pred), axis=[1,2,3])\n    union = K.sum(y_true,[1,2,3])+K.sum(y_pred,[1,2,3])-intersection\n    iou = K.mean((intersection + smooth) / (union + smooth), axis=0)\n    return iou\n\ndef dice_loss(y_true, y_pred):\n    smooth = 1.\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = y_true_f * y_pred_f\n    score = (2. * K.sum(intersection) + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n    return 1. - score\n\ndef bce_dice_loss(y_true, y_pred):\n    return binary_crossentropy(tf.cast(y_true, tf.float32), y_pred) + dice_loss(tf.cast(y_true, tf.float32), y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.829498Z","iopub.execute_input":"2022-08-09T16:19:52.831019Z","iopub.status.idle":"2022-08-09T16:19:52.845600Z","shell.execute_reply.started":"2022-08-09T16:19:52.830970Z","shell.execute_reply":"2022-08-09T16:19:52.844348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Images reshaped to (im_height,im_width)\nclass DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, batch_size = BATCH_SIZE, subset=\"train\", shuffle=False):\n        super().__init__()\n        self.df = df\n        self.shuffle = shuffle\n        self.subset = subset\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(df))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.floor(len(self.df) / self.batch_size))\n    \n    def on_epoch_end(self):\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        X = np.empty((self.batch_size,im_height,im_width,3))\n        y = np.empty((self.batch_size,im_height,im_width,3))\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        for i, img_path in enumerate(self.df['path'].iloc[indexes]):\n            w=self.df['width'].iloc[indexes[i]]\n            h=self.df['height'].iloc[indexes[i]]\n            img = self.__load_grayscale(img_path)  # shape: (im_height,im_width,1)\n            X[i,] = img   # broadcast to shape: (im_height,im_width,3)\n            if self.subset == 'train':\n                for k,j in enumerate([\"large_bowel\",\"small_bowel\",\"stomach\"]):\n                    rles = self.df[j].iloc[indexes[i]]\n                    mask = rle_decode(rles, shape=(h, w, 1))\n                    mask = cv2.resize(mask, (im_height,im_width))\n                    y[i,:,:,k] = mask\n        if self.subset == 'train':\n            return X, y\n        else: \n            return X\n        \n        # To do: add data augmentation\n        \n    def __load_grayscale(self, img_path):\n        img = cv2.imread(img_path, cv2.IMREAD_ANYDEPTH)\n        dsize = (im_height,im_width)\n        img = cv2.resize(img, dsize)\n        img = img.astype(np.float32) / 255.\n        img = np.expand_dims(img, axis=-1)\n        return img","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:19:52.847274Z","iopub.execute_input":"2022-08-09T16:19:52.848140Z","iopub.status.idle":"2022-08-09T16:19:52.865117Z","shell.execute_reply.started":"2022-08-09T16:19:52.848096Z","shell.execute_reply":"2022-08-09T16:19:52.863912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the Model","metadata":{}},{"cell_type":"code","source":"# Load trained model\ncustom_objects = custom_objects={\n    'dice_coef': dice_coef,\n    'iou_coef': iou_coef,\n    'bce_dice_loss': bce_dice_loss\n}\n\nmodel = tf.keras.models.load_model('../input/uwmodelver2/', custom_objects=custom_objects)\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:24:33.473856Z","iopub.execute_input":"2022-08-09T16:24:33.474313Z","iopub.status.idle":"2022-08-09T16:24:44.832894Z","shell.execute_reply.started":"2022-08-09T16:24:33.474275Z","shell.execute_reply":"2022-08-09T16:24:44.831587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_batches = DataGenerator(test_df, batch_size = BATCH_SIZE, subset=\"test\", shuffle=False)\nnum_batches = int(len(test_df)/BATCH_SIZE)\n\nfor i in range(num_batches):\n    # Predict\n    preds = model.predict(pred_batches[i],verbose=0)     # shape: (16,im_height,im_width,3)\n    \n    # Rle encode\n    for j in range(BATCH_SIZE):\n        for k in range(3):\n            pred_img = cv2.resize(preds[j,:,:,k], (test_df.loc[i*BATCH_SIZE+j,\"width\"], test_df.loc[i*BATCH_SIZE+j,\"height\"]), interpolation=cv2.INTER_NEAREST) # resize probabilities to original shape\n            pred_img = (pred_img>0.5).astype(dtype='uint8')    # classify\n            submission.loc[3*(i*BATCH_SIZE+j)+k,'predicted'] = rle_encode(pred_img)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:20:03.586417Z","iopub.status.idle":"2022-08-09T16:20:03.586831Z","shell.execute_reply.started":"2022-08-09T16:20:03.586633Z","shell.execute_reply":"2022-08-09T16:20:03.586653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)\nsubmission.tail(70)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T16:20:03.588070Z","iopub.status.idle":"2022-08-09T16:20:03.588472Z","shell.execute_reply.started":"2022-08-09T16:20:03.588281Z","shell.execute_reply":"2022-08-09T16:20:03.588300Z"},"trusted":true},"execution_count":null,"outputs":[]}]}