{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## UW-Madison GI Tract Image Segmentation","metadata":{}},{"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-competitions/kaggle/27923/logos/header.png?t=2021-06-02-20-30-25)","metadata":{}},{"cell_type":"markdown","source":"## Training\n* Data: [Unet 2.5D Augmentation](https://www.kaggle.com/code/juhjoo/uw-madison-gi-tract-training)\n\n### Data and Dataset\n* Data: [2.5D TFRecord Data](https://www.kaggle.com/code/juhjoo/uw-madison-gi-tract-dataset-tfrecords)\n* Dataset: [2.5D TFRecord Dataset](https://www.kaggle.com/datasets/juhjoo/uwmadisondataset)","metadata":{}},{"cell_type":"markdown","source":"## Reference\n\nI appriciate to the author for sharing valuable notebook\n\n[UWMGI: 2.5D TFRecord Data](https://www.kaggle.com/code/awsaf49/uwmgi-2-5d-tfrecord-data)\n\n[UWMGI Image Segmentation Make TFRecords](https://www.kaggle.com/code/tt195361/uwmgi-image-segmentation-make-tfrecords)","metadata":{}},{"cell_type":"markdown","source":"## Data Preparation","metadata":{}},{"cell_type":"code","source":"# Core\nimport pandas as pd\nimport numpy as np\nimport os\nimport cv2\nimport gc\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm\nfrom datetime import datetime\nimport json,itertools\nfrom typing import Optional\nfrom glob import glob\nimport warnings\nfrom IPython import display as ipd\nwarnings.filterwarnings(\"ignore\")\nimport matplotlib.gridspec as gridspec\nimport matplotlib.patches as mpatches\nimport matplotlib as mpl\nfrom matplotlib.patches import Rectangle\nfrom sklearn.model_selection import StratifiedKFold, KFold, StratifiedGroupKFold\nimport random\nfrom joblib import Parallel, delayed\nimport os, shutil\n\n# Keras\nfrom tensorflow import keras\nimport tensorflow as tf\nimport keras\nfrom keras import backend as K\nfrom keras.models import Model\nfrom keras.layers import Input\nfrom keras.layers.convolutional import Conv2D, Conv2DTranspose\nfrom keras.layers.pooling import MaxPooling2D\nfrom keras.layers.merge import concatenate\nfrom keras.losses import binary_crossentropy\nfrom keras.callbacks import Callback, ModelCheckpoint, EarlyStopping\nfrom keras.models import load_model, save_model","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:35.873162Z","iopub.execute_input":"2022-07-22T09:16:35.873637Z","iopub.status.idle":"2022-07-22T09:16:35.886840Z","shell.execute_reply.started":"2022-07-22T09:16:35.873601Z","shell.execute_reply":"2022-07-22T09:16:35.885927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=0) :\n    np.random.seed(seed)\n    random.seed(seed)\n    tf.random.set_seed(seed)\nset_seed()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.326795Z","iopub.execute_input":"2022-07-22T09:16:44.327231Z","iopub.status.idle":"2022-07-22T09:16:44.334089Z","shell.execute_reply.started":"2022-07-22T09:16:44.327196Z","shell.execute_reply":"2022-07-22T09:16:44.333000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 32\nEPOCHS = 40\nn_splits = 5\nfold_selected = 2   # 1,...,5\nim_width = 320\nim_height = 320\nNO_EMPTY = True  # set False to include images with empty mask in WandB\nNUM_LOG = 1000 # for WandB interactive Visualiztion\nCHANNELS = 3\nSTRIDE = 2\nIMAGE_SIZE = None\nFOLDS = 5\nSEED = 101\nAUTO = tf.data.experimental.AUTOTUNE\n\nTRAIN_ROOT_DIR = \"../input/uw-madison-gi-tract-image-segmentation/\"\nTEST_ROOT_DIR = \"../input/uw-madison-gi-tract-image-segmentation/test/\"","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.335782Z","iopub.execute_input":"2022-07-22T09:16:44.336341Z","iopub.status.idle":"2022-07-22T09:16:44.347006Z","shell.execute_reply.started":"2022-07-22T09:16:44.336304Z","shell.execute_reply":"2022-07-22T09:16:44.346090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Meta Data","metadata":{}},{"cell_type":"code","source":"train_df_original = pd.read_csv(TRAIN_ROOT_DIR + 'train.csv')\n\nprint(train_df_original.shape)\ntrain_df_original.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.348616Z","iopub.execute_input":"2022-07-22T09:16:44.349106Z","iopub.status.idle":"2022-07-22T09:16:44.727249Z","shell.execute_reply.started":"2022-07-22T09:16:44.349076Z","shell.execute_reply":"2022-07-22T09:16:44.726054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(28, 12))\ntrain_df_original['class'].value_counts(normalize=True).plot.pie()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.729050Z","iopub.execute_input":"2022-07-22T09:16:44.729792Z","iopub.status.idle":"2022-07-22T09:16:44.902390Z","shell.execute_reply.started":"2022-07-22T09:16:44.729735Z","shell.execute_reply":"2022-07-22T09:16:44.901179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(TRAIN_ROOT_DIR + 'sample_submission.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.905268Z","iopub.execute_input":"2022-07-22T09:16:44.906029Z","iopub.status.idle":"2022-07-22T09:16:44.928296Z","shell.execute_reply.started":"2022-07-22T09:16:44.905981Z","shell.execute_reply":"2022-07-22T09:16:44.926984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(test_df) == 0:\n    DEBUG=True\n    # test_df=train_df_original.iloc[:300, :]\n    test_df=pd.read_csv(TRAIN_ROOT_DIR + 'train.csv').iloc[:10*16*3, :]\n    test_df['segmentation'] = ''\n    test_df = test_df.rename(columns={'segmentation' : 'prediction'})\nelse:\n    DEBUG=False\n    \nsubmission = test_df.copy()\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:44.930102Z","iopub.execute_input":"2022-07-22T09:16:44.930961Z","iopub.status.idle":"2022-07-22T09:16:45.283906Z","shell.execute_reply.started":"2022-07-22T09:16:44.930899Z","shell.execute_reply":"2022-07-22T09:16:45.282727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_original.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:45.286316Z","iopub.execute_input":"2022-07-22T09:16:45.286701Z","iopub.status.idle":"2022-07-22T09:16:45.298086Z","shell.execute_reply.started":"2022-07-22T09:16:45.286666Z","shell.execute_reply":"2022-07-22T09:16:45.297008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_original.head()\ntrain_df_original.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:45.299723Z","iopub.execute_input":"2022-07-22T09:16:45.300084Z","iopub.status.idle":"2022-07-22T09:16:45.312070Z","shell.execute_reply.started":"2022-07-22T09:16:45.300047Z","shell.execute_reply":"2022-07-22T09:16:45.310830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def df_preparation(df, subset=\"train\"):\n    #--------------------------------------------------------------------------\n    df[\"case\"] = df[\"id\"].apply(lambda x: int(x.split(\"_\")[0].replace(\"case\", \"\")))\n    df[\"day\"] = df[\"id\"].apply(lambda x: int(x.split(\"_\")[1].replace(\"day\", \"\")))\n    df[\"slice\"] = df[\"id\"].apply(lambda x: x.split(\"_\")[3])\n    #--------------------------------------------------------------------------\n    if (subset==\"train\") or (DEBUG):\n        DIR=\"../input/uw-madison-gi-tract-image-segmentation/train\"\n    else:\n        DIR=\"../input/uw-madison-gi-tract-image-segmentation/test\"\n    \n    all_images = glob(os.path.join(DIR, \"**\", \"*.png\"), recursive=True)\n    x = all_images[0].rsplit(\"/\", 4)[0] ## ../input/uw-madison-gi-tract-image-segmentation/train\n\n    path_partial_list = []\n    for i in range(0, df.shape[0]):\n        path_partial_list.append(os.path.join(x,\n                              \"case\"+str(df[\"case\"].values[i]),\n                              \"case\"+str(df[\"case\"].values[i])+\"_\"+ \"day\"+str(df[\"day\"].values[i]),\n                              \"scans\",\n                              \"slice_\"+str(df[\"slice\"].values[i])))\n    df[\"path_partial\"] = path_partial_list\n    #--------------------------------------------------------------------------\n    path_partial_list = []\n    for i in range(0, len(all_images)):\n        path_partial_list.append(str(all_images[i].rsplit(\"_\",4)[0]))\n\n    tmp_df = pd.DataFrame()\n    tmp_df['path_partial'] = path_partial_list\n    tmp_df['path'] = all_images\n\n    #--------------------------------------------------------------------------\n    df = df.merge(tmp_df, on=\"path_partial\").drop(columns=[\"path_partial\"])\n    #--------------------------------------------------------------------------\n    df[\"width\"] = df[\"path\"].apply(lambda x: int(x[:-4].rsplit(\"_\",4)[1]))\n    df[\"height\"] = df[\"path\"].apply(lambda x: int(x[:-4].rsplit(\"_\",4)[2]))\n    #--------------------------------------------------------------------------\n    del x, path_partial_list, tmp_df\n    #--------------------------------------------------------------------------\n    \n    df['segmentation_empty'] = df.segmentation.fillna('')\n    df['rle_len'] = df.segmentation_empty.map(len)\n    df['empty'] = (df.rle_len == 0)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:45.551188Z","iopub.execute_input":"2022-07-22T09:16:45.552430Z","iopub.status.idle":"2022-07-22T09:16:45.569970Z","shell.execute_reply.started":"2022-07-22T09:16:45.552389Z","shell.execute_reply":"2022-07-22T09:16:45.568930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df_preparation(train_df_original, subset='train')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:45.779103Z","iopub.execute_input":"2022-07-22T09:16:45.779607Z","iopub.status.idle":"2022-07-22T09:16:51.286137Z","shell.execute_reply.started":"2022-07-22T09:16:45.779543Z","shell.execute_reply":"2022-07-22T09:16:51.284809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.288531Z","iopub.execute_input":"2022-07-22T09:16:51.288928Z","iopub.status.idle":"2022-07-22T09:16:51.306696Z","shell.execute_reply.started":"2022-07-22T09:16:51.288894Z","shell.execute_reply":"2022-07-22T09:16:51.305346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.308741Z","iopub.execute_input":"2022-07-22T09:16:51.309319Z","iopub.status.idle":"2022-07-22T09:16:51.324836Z","shell.execute_reply.started":"2022-07-22T09:16:51.309269Z","shell.execute_reply":"2022-07-22T09:16:51.323842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\n\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = np.asarray(mask_rle.split(), dtype=int)\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n\ndef id2mask(id_, df=train_df):\n    idf = df[df['id']==id_]\n    wh = idf[['height','width']].iloc[0]\n    shape = (wh.height, wh.width, 3)\n    mask = np.zeros(shape, dtype=np.uint8)\n    for i, class_ in enumerate(['large_bowel', 'small_bowel', 'stomach']):\n        cdf = idf[idf['class']==class_]\n        rle = cdf.segmentation_empty.squeeze()\n        if len(cdf) and not pd.isna(rle):\n            mask[..., i] = rle_decode(rle, shape[:2])\n    return mask\n\n\ndef load_img(path) :\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    img = img.astype('float32')\n    img = (img - img.min()) / (img.max() - img.min()) * 255.0\n    img = img.astype('uint8')\n    return img\n\ndef show_img(img, mask=None):\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8,8))\n    img = clahe.apply(img)\n#     plt.figure(figsize=(10,10))\n    plt.imshow(img, cmap='bone')\n    \n    if mask is not None:\n        # plt.imshow(np.ma.masked_where(mask!=1, mask), alpha=0.5, cmap='autumn')\n        plt.imshow(mask, alpha=0.5)\n        handles = [Rectangle((0,0),1,1, color=_c) for _c in [(0.667,0.0,0.0), (0.0,0.667,0.0), (0.0,0.0,0.667)]]\n        labels = [ \"Large Bowel\", \"Small Bowel\", \"Stomach\"]\n        plt.legend(handles,labels)\n    plt.axis('off')\n    \ndef rgb2gray(mask):\n    pad_mask = np.pad(mask, pad_width=[(0,0),(0,0),(1,0)])\n    gray_mask = pad_mask.argmax(-1)\n    return gray_mask","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.327500Z","iopub.execute_input":"2022-07-22T09:16:51.328136Z","iopub.status.idle":"2022-07-22T09:16:51.350232Z","shell.execute_reply.started":"2022-07-22T09:16:51.328100Z","shell.execute_reply":"2022-07-22T09:16:51.349002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.351631Z","iopub.execute_input":"2022-07-22T09:16:51.352960Z","iopub.status.idle":"2022-07-22T09:16:51.378675Z","shell.execute_reply.started":"2022-07-22T09:16:51.352922Z","shell.execute_reply":"2022-07-22T09:16:51.377465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df['path'][0])\nprint(train_df['path'][1])\nprint(train_df['path'][2])\nprint(train_df['path'][3])","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.380266Z","iopub.execute_input":"2022-07-22T09:16:51.381445Z","iopub.status.idle":"2022-07-22T09:16:51.392825Z","shell.execute_reply.started":"2022-07-22T09:16:51.381396Z","shell.execute_reply":"2022-07-22T09:16:51.391697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check a Mask Image","metadata":{}},{"cell_type":"code","source":"row=1; col=4\nplt.figure(figsize=(5*col, 5*row))\nfor i, id_ in enumerate(train_df[~train_df.segmentation.isna()].sample(frac=1.0)['id'].unique()[:row*col]) :\n    img = load_img(train_df[train_df['id']==id_].path.iloc[0])\n    mask = id2mask(id_, df=train_df)*256\n    plt.subplot(row, col, i+1)\n    i += 1\n    show_img(img, mask=mask)\n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:51.394675Z","iopub.execute_input":"2022-07-22T09:16:51.395410Z","iopub.status.idle":"2022-07-22T09:16:52.435753Z","shell.execute_reply.started":"2022-07-22T09:16:51.395362Z","shell.execute_reply":"2022-07-22T09:16:52.434616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## WanDB Config\n\n![](https://camo.githubusercontent.com/dd842f7b0be57140e68b2ab9cb007992acd131c48284eaf6b1aca758bfea358b/68747470733a2f2f692e696d6775722e636f6d2f52557469567a482e706e67)\n\nWeights & Biases (W&B) is MLOps platform for tracking our experiemnts. \n\n1. Login to wandb.ai.\n2. Go to your Settings Page.\n3. Scroll down and select NEW_KEY in API keys tab.","metadata":{}},{"cell_type":"code","source":"import wandb\n\ntry:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    api_key = user_secrets.get_secret(\"WANDB\")\n    wandb.login(key=api_key)\nexcept:\n    wandb.login(anonymous='must',relogin=True)\n    print('To use your W&B account,\\nGo to Add-ons -> Secrets and provide your W&B access token. Use the Label name as WANDB. \\nGet your W&B access token from here: https://wandb.ai/authorize')\n    \n    id_ = wandb.util.generate_id() # generate random id","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Init wanDB","metadata":{}},{"cell_type":"code","source":"# Initialize WANDB\nrun = wandb.init(project='UW-Madison-GI-Tract-Dataset', \n                 config={},\n#                  anonymous=anonymous,\n                 name=f\"mask-data\",\n                )\n# Columns \ncolumns=[\"id\", \"case\", \"day\", \"slice\", \"empty\", \"rle_len\", \"image\"]\n# Initialize table\ntable = wandb.Table(columns=columns)\n# Labels for mask\nclass_labels = {\n    1:\"Large Bowel\",\n    2:\"Small Bowel\",\n    3:\"Stomach\",\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:52.437030Z","iopub.execute_input":"2022-07-22T09:16:52.437343Z","iopub.status.idle":"2022-07-22T09:16:55.614055Z","shell.execute_reply.started":"2022-07-22T09:16:52.437314Z","shell.execute_reply":"2022-07-22T09:16:55.612344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save wanDB Log","metadata":{}},{"cell_type":"code","source":"def save_wandb(id_,df=None, count=0):\n    idf = df[df['id']==id_]\n    mask = id2mask(id_, df=df) # mask from [0, 1] to [0, 255]\n    image_path = idf.path.iloc[0]\n    img = load_img(image_path) # load image\n    if count<=NUM_LOG:\n            img = cv2.resize(img, dsize=(160, 192), interpolation=cv2.INTER_AREA)\n            mask = cv2.resize(mask, dsize=(160, 192), interpolation=cv2.INTER_NEAREST)\n            table.add_data(id_, \n                           idf.case.iloc[0], \n                           idf.day.iloc[0],\n                           idf.slice.iloc[0],\n                           int(mask.sum()==0),\n                           idf.rle_len.iloc[0],\n                           wandb.Image(img, masks={\n                    \"ground_truth\" : {\n                    \"mask_data\" : rgb2gray(mask), # (height, width, 3) => (height, width) => may lose overlap data\n                    \"class_labels\" : class_labels\n                }}))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:16:55.615905Z","iopub.execute_input":"2022-07-22T09:16:55.616533Z","iopub.status.idle":"2022-07-22T09:16:55.629004Z","shell.execute_reply.started":"2022-07-22T09:16:55.616496Z","shell.execute_reply":"2022-07-22T09:16:55.627187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = train_df.copy()\nif NO_EMPTY:\n    tmp_df = tmp_df[~train_df.segmentation.isna()]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:17:00.177818Z","iopub.execute_input":"2022-07-22T09:17:00.178747Z","iopub.status.idle":"2022-07-22T09:17:00.231494Z","shell.execute_reply.started":"2022-07-22T09:17:00.178679Z","shell.execute_reply":"2022-07-22T09:17:00.230260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = tmp_df['id'].unique()    \n_ = Parallel(n_jobs=-1, backend='threading')(delayed(save_wandb)(id_, df=tmp_df, count=i)\\\n                                             for i, id_ in enumerate(tqdm(ids, total=len(ids))))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:17:00.233926Z","iopub.execute_input":"2022-07-22T09:17:00.234417Z","iopub.status.idle":"2022-07-22T09:22:00.743412Z","shell.execute_reply.started":"2022-07-22T09:17:00.234371Z","shell.execute_reply":"2022-07-22T09:22:00.742264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display wanDB Table","metadata":{}},{"cell_type":"code","source":"wandb.log({\"Table\":table})\nwandb.finish()\ndisplay(ipd.IFrame(run.url, width=1000, height=720))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:22:00.744912Z","iopub.execute_input":"2022-07-22T09:22:00.745226Z","iopub.status.idle":"2022-07-22T09:22:50.978948Z","shell.execute_reply.started":"2022-07-22T09:22:00.745197Z","shell.execute_reply":"2022-07-22T09:22:50.977810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with wandb.init(project='UW-Madison-GI-Tract-Dataset') as run2:\n    table2= run2.use_artifact(f\"run-{run.id}-Table:v0\").get(\"Table\")\n    \nfor idx, row in table2.iterrows():\n    break\nplt.figure(figsize=(5,5))\nplt.imshow(np.array(row[-1].image))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:22:50.980945Z","iopub.execute_input":"2022-07-22T09:22:50.981387Z","iopub.status.idle":"2022-07-22T09:23:00.172341Z","shell.execute_reply.started":"2022-07-22T09:22:50.981355Z","shell.execute_reply":"2022-07-22T09:23:00.171061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data for Segmentation and 2.5D","metadata":{}},{"cell_type":"code","source":"train_df['segmentation_empty'] = train_df.segmentation.fillna('')\ndf2 = train_df.groupby(['id'])['segmentation_empty'].agg(list).to_frame().reset_index() \ndf2 = df2.merge(train_df.groupby(['id'])['rle_len'].agg(sum).to_frame().reset_index())\ntrain_df = train_df.drop(columns=['segmentation_empty','class', 'rle_len'])\ntrain_df = train_df.groupby(['id']).head(1).reset_index(drop=True)\ntrain_df = train_df.merge(df2, on=['id'])\nfault1 = 'case7_day0'\nfault2 = 'case81_day30'\ntrain_df = train_df[~train_df['id'].str.contains(fault1) & ~train_df['id'].str.contains(fault2)].reset_index(drop=True)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:00:32.853447Z","iopub.execute_input":"2022-07-22T05:00:32.853903Z","iopub.status.idle":"2022-07-22T05:00:33.602062Z","shell.execute_reply.started":"2022-07-22T05:00:32.853869Z","shell.execute_reply":"2022-07-22T05:00:33.601069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.5D Images","metadata":{}},{"cell_type":"code","source":"for i in range(CHANNELS):\n    train_df[f'path_{i:02}'] = train_df.groupby(['case','day'])['path'].shift(-i*STRIDE).fillna(method=\"ffill\")\ntrain_df['paths'] = train_df[[f'path_{i:02d}' for i in range(CHANNELS)]].values.tolist()\ntrain_df.paths[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:00:33.603454Z","iopub.execute_input":"2022-07-22T05:00:33.603774Z","iopub.status.idle":"2022-07-22T05:00:33.980725Z","shell.execute_reply.started":"2022-07-22T05:00:33.603746Z","shell.execute_reply":"2022-07-22T05:00:33.979581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save a train data","metadata":{}},{"cell_type":"code","source":"train_df.to_csv('train_df.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:00:33.982301Z","iopub.execute_input":"2022-07-22T05:00:33.982647Z","iopub.status.idle":"2022-07-22T05:00:36.842703Z","shell.execute_reply.started":"2022-07-22T05:00:33.982617Z","shell.execute_reply":"2022-07-22T05:00:36.841736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility Function","metadata":{}},{"cell_type":"code","source":"def rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = np.asarray(mask_rle.split(), dtype=int)\n    starts = s[0::2] - 1\n    lengths = s[1::2]\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n\n\n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef id2mask(id_):\n    idf = train_df[train_df['id']==id_]\n    shape = (idf.height.item(), idf.width.item(), 3)\n    mask = np.zeros(shape, dtype=np.uint8)\n    rles = idf.segmentation_empty.squeeze()\n    for i, rle in enumerate(rles):\n        if not pd.isna(rle):\n            mask[..., i] = rle_decode(rle, shape[:2])\n    return mask\n\n\ndef rgb2gray(mask):\n    pad_mask = np.pad(mask, pad_width=[(0,0),(0,0),(1,0)])\n    gray_mask = pad_mask.argmax(-1)\n    return gray_mask\n\ndef gray2rgb(mask):\n    rgb_mask = tf.keras.utils.to_categorical(mask, num_classes=4)\n    return rgb_mask[..., 1:].astype(mask.dtype)\n\ndef load_img(path, size=IMAGE_SIZE):\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    if size is not None:\n        img = cv2.resize(img, dsize=IMAGE_SIZE, interpolation=cv2.INTER_NEAREST)\n#     img = img.astype('float32') # original is uint16\n#     img = (img - img.min())/(img.max() - img.min())*255.0 # scale image to [0, 255]\n#     img = img.astype('uint8')\n    return img\n\ndef load_imgs(paths):\n    imgs = [None]*3\n    for i, img_path in enumerate(paths):\n        img = load_img(img_path)\n        imgs[i] = img\n    return np.stack(imgs,axis=-1)\n\ndef show_img(img, mask=None):\n#     plt.figure(figsize=(10,10))\n    plt.imshow(img, cmap='bone')\n    \n    if mask is not None:\n        # plt.imshow(np.ma.masked_where(mask!=1, mask), alpha=0.5, cmap='autumn')\n        plt.imshow(mask, alpha=0.5)\n        handles = [Rectangle((0,0),1,1, color=_c) for _c in [(0.667,0.0,0.0), (0.0,0.667,0.0), (0.0,0.0,0.667)]]\n        labels = [ \"Large Bowel\", \"Small Bowel\", \"Stomach\"]\n        plt.legend(handles,labels)\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:28:07.355399Z","iopub.execute_input":"2022-07-22T05:28:07.355816Z","iopub.status.idle":"2022-07-22T05:28:07.379137Z","shell.execute_reply.started":"2022-07-22T05:28:07.355773Z","shell.execute_reply":"2022-07-22T05:28:07.377970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check a 2.5D Image","metadata":{}},{"cell_type":"code","source":"row=1; col=4\nplt.figure(figsize=(5*col, 5*row))\nfor i, id_ in enumerate(train_df[~train_df.segmentation_empty.isna()].sample(frac=1.0)['id'].unique()[:row*col]) :\n    img = load_imgs(train_df[train_df['id']==id_].squeeze().paths).astype('float32')\n    img/=img.max(axis=(0,1))\n    mask = id2mask(id_)*255\n    plt.subplot(row, col, i+1)\n    i += 1\n    show_img(img, mask=mask)\n    plt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:28:07.537499Z","iopub.execute_input":"2022-07-22T05:28:07.539823Z","iopub.status.idle":"2022-07-22T05:28:08.670054Z","shell.execute_reply.started":"2022-07-22T05:28:07.539780Z","shell.execute_reply":"2022-07-22T05:28:08.668777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split a Data using StratifiedGroupKFold","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\nskf = StratifiedGroupKFold(n_splits=FOLDS, shuffle=True, random_state=SEED)\nfor fold, (train_idx, val_idx) in enumerate(skf.split(train_df, train_df['empty'], groups = train_df[\"case\"])):\n    train_df.loc[val_idx, 'fold'] = fold\ndisplay(train_df.groupby(['fold','empty'])['id'].count())","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:08:07.316943Z","iopub.execute_input":"2022-07-22T05:08:07.317263Z","iopub.status.idle":"2022-07-22T05:08:07.502992Z","shell.execute_reply.started":"2022-07-22T05:08:07.317234Z","shell.execute_reply":"2022-07-22T05:08:07.501784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:12:45.607512Z","iopub.execute_input":"2022-07-21T16:12:45.607872Z","iopub.status.idle":"2022-07-21T16:12:45.615187Z","shell.execute_reply.started":"2022-07-21T16:12:45.607843Z","shell.execute_reply":"2022-07-21T16:12:45.614031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make a TFRecord","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ndef _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:12:51.702087Z","iopub.execute_input":"2022-07-21T16:12:51.702463Z","iopub.status.idle":"2022-07-21T16:12:51.712613Z","shell.execute_reply.started":"2022-07-21T16:12:51.702433Z","shell.execute_reply":"2022-07-21T16:12:51.711543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_serialize_example(feature0, feature1, feature2, feature3, feature4,\n                           feature5, feature6, feature7, feature8):\n    feature = {\n      'image':_bytes_feature(feature0),\n      'id':_bytes_feature(feature1),\n      'case':_int64_feature(feature2),\n      'day':_int64_feature(feature3), \n      'slice':_bytes_feature(feature4),\n      'height':_int64_feature(feature5),\n      'width':_int64_feature(feature6),\n      'empty':_int64_feature(feature7),\n      'mask':_bytes_feature(feature8),\n  }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:12:52.105736Z","iopub.execute_input":"2022-07-21T16:12:52.106135Z","iopub.status.idle":"2022-07-21T16:12:52.114057Z","shell.execute_reply.started":"2022-07-21T16:12:52.106102Z","shell.execute_reply":"2022-07-21T16:12:52.112585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Write a TFRecord","metadata":{}},{"cell_type":"code","source":"show=True\nos.makedirs('/tmp/uwmgi', exist_ok=True)\nfolds = train_df.fold.unique().tolist()\nfor fold in tqdm(folds): # create tfrecord for each fold\n    fold_df = train_df.query(\"fold==@fold\")\n    if show:\n        print(); print('Writing TFRecord of fold %i :'%(fold))  \n    with tf.io.TFRecordWriter('/tmp/uwmgi/train%.2i-%i.tfrec'%(fold,fold_df.shape[0])) as writer:\n        samples = fold_df.shape[0] # samples = 200\n        it = tqdm(range(samples)) if show else range(samples)\n        for k in it: # images in fold\n            row = fold_df.iloc[k,:]\n            image = load_imgs(row['paths'])\n            image_id = row['id']\n            case = row['case']\n            day = row['day']\n            slice_ = row['slice']\n            height = row['height']\n            width = row['width']\n            empty = row['empty']\n            mask = id2mask(image_id)*255 # [0, 1] => [0, 255]\n            example  = train_serialize_example(\n                image.tobytes(),\n                str.encode(image_id),\n                case,\n                day,\n                str.encode(slice_),\n                height,\n                width,\n                empty,\n                mask.tobytes(),\n                )\n            writer.write(example)\n        if show:\n            filepath = '/tmp/uwmgi/train%.2i-%i.tfrec'%(fold,fold_df.shape[0])\n            filename = filepath.split('/')[-1]\n            filesize = os.path.getsize(filepath)/10**6\n            print(filename,':',np.around(filesize, 2),'MB')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:12:55.022668Z","iopub.execute_input":"2022-07-21T16:12:55.023619Z","iopub.status.idle":"2022-07-21T16:28:03.970918Z","shell.execute_reply.started":"2022-07-21T16:12:55.023576Z","shell.execute_reply":"2022-07-21T16:28:03.968756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read a TFRecord Data","metadata":{}},{"cell_type":"code","source":"import re, math\ndef decode_image(data, height, width, target_size=(224, 224)):\n    image = tf.io.decode_raw(data, out_type=tf.uint16)\n    image = tf.reshape(image, [height, width, 3]) # explicit size needed for TPU\n    image = tf.image.resize(image, target_size, method='nearest')\n    image = tf.cast(image,tf.float32) \n    image = image / tf.reduce_max(image)\n    return image\n\ndef decode_mask(data, height, width, target_size=(224, 224)):    \n    mask = tf.io.decode_raw(data, out_type=tf.uint8)\n    mask = tf.reshape(mask, [height, width, 3]) # explicit size needed for TPU  \n    mask = tf.image.resize(mask, target_size, method='nearest')\n    return mask\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\" : tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"height\" : tf.io.FixedLenFeature([], tf.int64),\n        \"width\" : tf.io.FixedLenFeature([], tf.int64),\n        \"mask\" : tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    height = example['height']\n    width = example['width']\n    image = decode_image(example['image'], height, width)\n    mask = decode_mask(example['mask'], height, width)\n    return image, mask # returns a dataset of (image, label) pairs\n\ndef load_dataset(fileids, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(fileids, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(20, seed=SEED)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset\n\ndef count_data_items(fileids):\n    # the number of data items is written in the id of the .tfrec files, i.e. flowers00-230.tfrec = 230 data items\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(fileid).group(1)) for fileid in fileids]\n    return np.sum(n)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:03.974153Z","iopub.execute_input":"2022-07-21T16:28:03.974603Z","iopub.status.idle":"2022-07-21T16:28:03.992864Z","shell.execute_reply.started":"2022-07-21T16:28:03.974569Z","shell.execute_reply":"2022-07-21T16:28:03.991652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize a TFRecord Data","metadata":{}},{"cell_type":"code","source":"def display_batch(batch, size=5):\n    imgs, tars = batch\n    plt.figure(figsize=(size*5, 5))\n    for img_idx in range(size):\n        plt.subplot(1, size, img_idx+1)\n        plt.imshow(imgs[img_idx,], cmap='bone')\n        plt.imshow(tars[img_idx,], alpha=0.5)\n        plt.xticks([])\n        plt.yticks([])\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:03.994193Z","iopub.execute_input":"2022-07-21T16:28:03.994569Z","iopub.status.idle":"2022-07-21T16:28:04.010387Z","shell.execute_reply.started":"2022-07-21T16:28:03.994536Z","shell.execute_reply":"2022-07-21T16:28:04.009117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAINING_FILENAMES = tf.io.gfile.glob('/tmp/uwmgi/train*.tfrec')\nTEST_FILENAMES     = tf.io.gfile.glob('/tmp/uwmgi/test*.tfrec')\nprint('There are %i train & %i test images'%(count_data_items(TRAINING_FILENAMES), count_data_items(TEST_FILENAMES)))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:04.012841Z","iopub.execute_input":"2022-07-21T16:28:04.013309Z","iopub.status.idle":"2022-07-21T16:28:04.032538Z","shell.execute_reply.started":"2022-07-21T16:28:04.013269Z","shell.execute_reply":"2022-07-21T16:28:04.030980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize TFRecord Data","metadata":{}},{"cell_type":"code","source":"training_dataset = get_training_dataset().take(2916700)\ntraining_dataset = training_dataset.unbatch().batch(20)\ntrain_batch = next(iter(training_dataset))\ndisplay_batch(train_batch, 5);","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:04.033961Z","iopub.execute_input":"2022-07-21T16:28:04.034330Z","iopub.status.idle":"2022-07-21T16:28:06.194467Z","shell.execute_reply.started":"2022-07-21T16:28:04.034299Z","shell.execute_reply":"2022-07-21T16:28:06.193097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, label = train_batch\nnp.unique(label.numpy(), return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:06.196119Z","iopub.execute_input":"2022-07-21T16:28:06.196629Z","iopub.status.idle":"2022-07-21T16:28:06.258471Z","shell.execute_reply.started":"2022-07-21T16:28:06.196591Z","shell.execute_reply":"2022-07-21T16:28:06.257345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compress a TFRecord Data","metadata":{}},{"cell_type":"code","source":"shutil.make_archive('/kaggle/working/tfrecord',\n                    'zip',\n                    '/tmp',\n                    'uwmgi')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T16:28:06.260116Z","iopub.execute_input":"2022-07-21T16:28:06.260454Z","iopub.status.idle":"2022-07-21T16:58:31.463939Z","shell.execute_reply.started":"2022-07-21T16:28:06.260425Z","shell.execute_reply":"2022-07-21T16:58:31.462595Z"},"trusted":true},"execution_count":null,"outputs":[]}]}