{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Credits:\nhttps://www.kaggle.com/iafoss/hubmap-1024x1024\nhttps://www.kaggle.com/code/wrrosa/hubmap-tf-with-tpu-efficientunet-512x512-tfrecs/notebook\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tifffile as tiff\nimport cv2\nimport os\nfrom tqdm.notebook import tqdm\nimport zipfile\nos.makedirs('train')","metadata":{"execution":{"iopub.status.busy":"2022-07-01T09:11:47.961115Z","iopub.execute_input":"2022-07-01T09:11:47.962326Z","iopub.status.idle":"2022-07-01T09:11:48.474374Z","shell.execute_reply.started":"2022-07-01T09:11:47.962194Z","shell.execute_reply":"2022-07-01T09:11:48.473203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sz = 512#128#256   #the size of tiles\nreduce = 2\nMASKS = '../input/hubmap-organ-segmentation/train.csv'\nDATA = '../input/hubmap-organ-segmentation/train_images'","metadata":{"execution":{"iopub.status.busy":"2022-07-01T09:11:48.476457Z","iopub.execute_input":"2022-07-01T09:11:48.476875Z","iopub.status.idle":"2022-07-01T09:11:48.481769Z","shell.execute_reply.started":"2022-07-01T09:11:48.476841Z","shell.execute_reply":"2022-07-01T09:11:48.480930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#functions to convert encoding to mask and mask to encoding\ndef enc2mask1(encs, shape):\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for m,enc in enumerate(encs):\n        if isinstance(enc,np.float) and np.isnan(enc): continue\n        s = enc.split()        \n        for i in range(len(s)//2):\n            start = int(s[2*i]) - 1\n            length = int(s[2*i+1])\n            img[start:start+length] = 1 + m            \n    return img.reshape(shape).T\n#https://www.kaggle.com/code/friedchips/fast-fully-correct-hubmap-v2-rle-encoding\ndef enc2mask(rle, mask_shape):\n    ''' takes a space-delimited RLE string in column-first order\n    and turns it into a 2d boolean numpy array of shape mask_shape '''\n    \n    mask = np.zeros(np.prod(mask_shape), dtype=np.uint8) # 1d mask array\n    rle = np.array(rle.split()).astype(int) # rle values to ints\n    starts = rle[::2]\n    lengths = rle[1::2]\n    for s, l in zip(starts, lengths):\n        mask[s:s+l] = 1\n    return mask.reshape(np.flip(mask_shape)).T # flip because of column-first order\n\ndef mask2enc(mask, n=1):\n    pixels = mask.T.flatten()\n    encs = []\n    for i in range(1,n+1):\n        p = (pixels == i).astype(np.int8)\n        if p.sum() == 0: encs.append(np.nan)\n        else:\n            p = np.concatenate([[0], p, [0]])\n            runs = np.where(p[1:] != p[:-1])[0] + 1\n            runs[1::2] -= runs[::2]\n            encs.append(' '.join(str(x) for x in runs))\n    return encs\n\ndf_masks = pd.read_csv(MASKS).set_index('id')\ndf_masks.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:23:01.813795Z","iopub.execute_input":"2022-07-01T10:23:01.814913Z","iopub.status.idle":"2022-07-01T10:23:01.981056Z","shell.execute_reply.started":"2022-07-01T10:23:01.814872Z","shell.execute_reply":"2022-07-01T10:23:01.979813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-07-01T09:11:48.836050Z","iopub.execute_input":"2022-07-01T09:11:48.837024Z","iopub.status.idle":"2022-07-01T09:11:58.780036Z","shell.execute_reply.started":"2022-07-01T09:11:48.836963Z","shell.execute_reply":"2022-07-01T09:11:58.778384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The following functions can be used to convert a value to a type compatible\n# with tf.train.Example.\n\ndef _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n  \"\"\"Returns a float_list from a float / double.\"\"\"\n  return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-07-01T09:11:58.781738Z","iopub.execute_input":"2022-07-01T09:11:58.783012Z","iopub.status.idle":"2022-07-01T09:11:58.790446Z","shell.execute_reply.started":"2022-07-01T09:11:58.782950Z","shell.execute_reply":"2022-07-01T09:11:58.789099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(image, mask):\n  \"\"\"\n  Creates a tf.train.Example message ready to be written to a file.\n  \"\"\"\n  # Create a dictionary mapping the feature name to the tf.train.Example-compatible\n  # data type.\n  feature = {\n      'image': _bytes_feature(image),\n      'mask': _bytes_feature(mask),\n  }\n\n  # Create a Features message using tf.train.Example.\n\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T09:11:58.792111Z","iopub.execute_input":"2022-07-01T09:11:58.792546Z","iopub.status.idle":"2022-07-01T09:11:58.830752Z","shell.execute_reply.started":"2022-07-01T09:11:58.792501Z","shell.execute_reply":"2022-07-01T09:11:58.829537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_th = 40  #saturation blancking threshold\np_th = 200*sz//256 #threshold for the minimum number of pixels\n\nx_tot,x2_tot = [],[]\nfor index, encs in tqdm(df_masks.iterrows()):    \n    #read image and generate the mask    \n    img = tiff.imread(os.path.join(DATA,str(index)+'.tiff'))\n    if len(img.shape) == 5 or img.shape[0] == 3: \n        img = np.transpose(img.squeeze(), (1,2,0))\n    mask = enc2mask(encs['rle'],(img.shape[1],img.shape[0]))        \n    #add padding to make the image dividable into tiles\n    shape = img.shape\n    pad0 = (reduce*sz - shape[0]%(reduce*sz))%(reduce*sz)\n    pad1 = (reduce*sz - shape[1]%(reduce*sz))%(reduce*sz)\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=0)\n    mask = np.pad(mask,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2]],\n                constant_values=0)\n\n    #split image and mask into tiles using the reshape+transpose trick\n    img = cv2.resize(img,(img.shape[1]//reduce,img.shape[0]//reduce),\n                     interpolation = cv2.INTER_AREA)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    \n    mask = cv2.resize(mask,(mask.shape[1]//reduce,mask.shape[0]//reduce),\n                      interpolation = cv2.INTER_NEAREST)\n    \n    mask = mask.reshape(mask.shape[0]//sz,sz,mask.shape[1]//sz,sz)\n    mask = mask.transpose(0,2,1,3).reshape(-1,sz,sz)\n\n    #write data\n    filename = 'train/' + str(index) + '-'+str(img.shape[0]) +'.tfrec'\n    count = 0\n    with tf.io.TFRecordWriter(filename) as writer:\n\n        for i,(im,m) in enumerate(zip(img,mask)):\n            #remove black or gray images based on saturation check\n            hsv = cv2.cvtColor(im, cv2.COLOR_BGR2HSV)\n            h, s, v = cv2.split(hsv)\n            if (s>s_th).sum() <= p_th or im.sum() <= p_th: continue\n\n            x_tot.append((im/255.0).reshape(-1,3).mean(0))\n            x2_tot.append(((im/255.0)**2).reshape(-1,3).mean(0))\n\n            im = cv2.cvtColor(im, cv2.COLOR_RGB2BGR)\n            #im =cv2.cvtColor(im, cv2.COLOR_BGR2GRAY)\n            #im = cv2.equalizeHist(im)\n            #im = cv2.cvtColor(im, cv2.COLOR_GRAY2BGR)\n            #im = cv2.imencode('.png',cv2.cvtColor(im, cv2.COLOR_RGB2BGR))[1]\n            #img_out.writestr(f'{index}_{i}.png', im)\n            #m = cv2.imencode('.png',m)[1]\n            #mask_out.writestr(f'{index}_{i}.png', m)\n            example = serialize_example(im.tobytes(),m.tobytes())\n            writer.write(example)\n            count+=1\n    filename2 = 'train/' + str(index) + '-'+str(count) +'.tfrec'\n    os.rename(filename, filename2)\n    print(filename2 ,count)    \n\n#image stats\nimg_avr =  np.array(x_tot).mean(0)\nimg_std =  np.sqrt(np.array(x2_tot).mean(0) - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:23:17.283719Z","iopub.execute_input":"2022-07-01T10:23:17.284623Z","iopub.status.idle":"2022-07-01T10:26:12.630756Z","shell.execute_reply.started":"2022-07-01T10:23:17.284581Z","shell.execute_reply":"2022-07-01T10:26:12.627829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check","metadata":{}},{"cell_type":"code","source":"import re\nimport glob\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\ntrain_images = glob.glob('train/*.tfrec')\nctraini = count_data_items(train_images)\nprint(f'Num train images: {ctraini}')","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:30:42.915566Z","iopub.execute_input":"2022-07-01T10:30:42.916158Z","iopub.status.idle":"2022-07-01T10:30:42.926239Z","shell.execute_reply.started":"2022-07-01T10:30:42.916106Z","shell.execute_reply":"2022-07-01T10:30:42.925047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIM = sz\nmini_size = 256\ndef _parse_image_function(example_proto):\n    image_feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'mask': tf.io.FixedLenFeature([], tf.string)\n    }\n    single_example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = tf.reshape( tf.io.decode_raw(single_example['image'],out_type=np.dtype('uint8')), (DIM,DIM, 3))\n    mask =  tf.reshape(tf.io.decode_raw(single_example['mask'],out_type='bool'),(DIM,DIM,1))\n    \n    image = tf.image.resize(image,(mini_size,mini_size))/255.0\n    mask = tf.image.resize(tf.cast(mask,'uint8'),(mini_size,mini_size))\n    return image, mask\n\n\ndef load_dataset(filenames):\n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.map(lambda ex: _parse_image_function(ex))\n    return dataset\n\nN = 3\ndef get_dataset(FILENAME):\n    dataset = load_dataset(FILENAME)\n    dataset = dataset.batch(N*N)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:30:44.973608Z","iopub.execute_input":"2022-07-01T10:30:44.974276Z","iopub.status.idle":"2022-07-01T10:30:44.984897Z","shell.execute_reply.started":"2022-07-01T10:30:44.974224Z","shell.execute_reply":"2022-07-01T10:30:44.983744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nfrom skimage.segmentation import mark_boundaries\n\nfor imgs, masks in get_dataset(train_images[0]).take(1):\n    pass\n\nplt.figure(figsize=[20, 15])\ngs1 = gridspec.GridSpec(N,N)\n\nfor i in range(N*N):\n   # i = i + 1 # grid spec indexes from 0    \n    ax1 = plt.subplot(gs1[i])\n    plt.axis('on')\n    ax1.set_xticklabels([])\n    ax1.set_yticklabels([])\n    ax1.set_aspect('equal')\n    ax1.imshow(mark_boundaries(imgs[i], masks[i].numpy().squeeze().astype('bool')), aspect=1)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:35:34.956291Z","iopub.execute_input":"2022-07-01T10:35:34.957388Z","iopub.status.idle":"2022-07-01T10:35:36.619455Z","shell.execute_reply.started":"2022-07-01T10:35:34.957343Z","shell.execute_reply":"2022-07-01T10:35:36.618087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(masks[1].numpy().squeeze())","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:35:48.464389Z","iopub.execute_input":"2022-07-01T10:35:48.464782Z","iopub.status.idle":"2022-07-01T10:35:48.645863Z","shell.execute_reply.started":"2022-07-01T10:35:48.464748Z","shell.execute_reply":"2022-07-01T10:35:48.645062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(imgs[1].numpy().squeeze())","metadata":{"execution":{"iopub.status.busy":"2022-07-01T10:35:51.063739Z","iopub.execute_input":"2022-07-01T10:35:51.064803Z","iopub.status.idle":"2022-07-01T10:35:51.277028Z","shell.execute_reply.started":"2022-07-01T10:35:51.064764Z","shell.execute_reply":"2022-07-01T10:35:51.275856Z"},"trusted":true},"execution_count":null,"outputs":[]}]}