{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport json\nfrom shapely.geometry import Polygon\nimport glob\nimport pathlib\nfrom matplotlib import pyplot as plt\nfrom functools import partial\nfrom albumentations import (\n    Compose, RandomBrightness, HueSaturationValue, RandomContrast, HorizontalFlip,\n    Rotate\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-05T12:13:29.707988Z","iopub.execute_input":"2022-08-05T12:13:29.708519Z","iopub.status.idle":"2022-08-05T12:13:32.368921Z","shell.execute_reply.started":"2022-08-05T12:13:29.708405Z","shell.execute_reply":"2022-08-05T12:13:32.367926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This notebook prepares the data in tfrecrod format.\n## It assumes that you want to slice the images since the images are big (3000 x 3000) you can set the parameters for size of the cropped images and if you want to resize them","metadata":{}},{"cell_type":"code","source":"# put the path for data you want to \ntrain_csv_path='/kaggle/input/hubmap-organ-segmentation/train.csv'\ntrain_images_folder='/kaggle/input/hubmap-organ-segmentation/train_images/'\nannotation_train_path='/kaggle/input/hubmap-organ-segmentation/train_annotations/'\n\n\ninput_image_size=[3000,3000]\nsize_each_slice=[1000,1000]\nslice_stride=[1000,1000]\noutput_size_each_slice=[256,256] # use cv2.resize() after slicing\n\n\n\n\ntrain_images_paths=list(glob.glob(train_images_folder+'*'))\n\n\ntrain_annotation_paths=list(glob.glob(annotation_train_path+'*'))\ntrain_df=pd.read_csv(train_csv_path)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T12:13:32.374029Z","iopub.execute_input":"2022-08-05T12:13:32.376292Z","iopub.status.idle":"2022-08-05T12:13:32.791933Z","shell.execute_reply.started":"2022-08-05T12:13:32.376254Z","shell.execute_reply":"2022-08-05T12:13:32.789625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_rle_list(rle,img_shape): # to read the annotations from json files\n    mask=np.zeros([img_shape[0],img_shape[1]])\n    for i,obj in enumerate(rle):\n        tmp_mask=np.zeros([img_shape[0],img_shape[1]])\n        tmp_mask=cv2.fillPoly(tmp_mask,[np.array(obj)],color=[1])\n        mask+=tmp_mask\n           \n    return mask\n\ndef grid_slice(img,output_size=[256,256],stride=[256,256],resize_shape=None): # slice an image or mask\n    imgs=[]\n    img_shape=img.shape\n    y_step=(img_shape[0]-output_size[0])//stride[0]+1\n    x_step=(img_shape[1]-output_size[1])//stride[1]+1\n    for i in range(y_step):\n        for j in range(x_step):\n            img_slice=img[i*stride[0]:i*stride[0]+output_size[0],j*stride[1]:j*stride[1]+output_size[1]]\n            if resize_shape is not None:\n                img_slice=cv2.resize(img_slice,resize_shape)\n            imgs.append(img_slice)\n    return imgs\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T12:13:32.794837Z","iopub.execute_input":"2022-08-05T12:13:32.795954Z","iopub.status.idle":"2022-08-05T12:13:32.805294Z","shell.execute_reply.started":"2022-08-05T12:13:32.795913Z","shell.execute_reply":"2022-08-05T12:13:32.804287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\n\n# some functions to write tfrecord examples\ndef _bytes_feature(value):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\ndef _float_feature(value):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n    return tf.train.Feature(float_list=tf.train.FloatList(value=[value]))\n\ndef _int64_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\ndef _int64_feature_list(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=value))\ndef _uint8_feature(value):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n    return tf.train.Feature(uint=tf.train.Int64List(value=[value]))\ndef serialize_example(image, mask,im_id,):\n    \"\"\"\n    Creates a tf.train.Example message ready to be written to a file.\n    \"\"\"\n    # Create a dictionary mapping the feature name to the tf.train.Example-compatible\n    # data type.\n    feature = {\n        'image': _bytes_feature(image.tobytes()),\n        'mask': _bytes_feature(mask.tobytes()),\n        'id': _int64_feature(im_id),\n        'shape':  _int64_feature_list(image.shape),\n         'mask_shape':  _int64_feature_list(mask.shape)\n    }\n\n    # Create a Features message using tf.train.Example.\n\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T12:13:32.809531Z","iopub.execute_input":"2022-08-05T12:13:32.810023Z","iopub.status.idle":"2022-08-05T12:13:37.166898Z","shell.execute_reply.started":"2022-08-05T12:13:32.809996Z","shell.execute_reply":"2022-08-05T12:13:37.165934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.auto import tqdm\norgan_dict={'prostate':0, 'spleen':1, 'lung':2, 'kidney':3, 'largeintestine':4}\ntfrecord_filename = 'train_hubmap.tfrecords'\nwriter = tf.io.TFRecordWriter(tfrecord_filename)\nfor i,row in tqdm(train_df.iterrows()):\n    im_id=row['id']\n    organ=row['organ']\n    h=row['img_height']\n    w=row['img_width']\n    im_path=train_images_folder+str(im_id)+'.tiff'\n    mask_path=annotation_train_path+str(im_id)+'.json'\n    with open(mask_path) as file:\n        mask_raw=json.load(file)\n    \n    img=cv2.imread(im_path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    mask=read_rle_list(mask_raw,[h,w])\n    img=cv2.resize(img,input_image_size)\n    mask=cv2.resize(mask,input_image_size)\n    final_mask=np.zeros([input_image_size[0],input_image_size[1],5],np.uint8)\n    final_mask[...,organ_dict[organ]]=mask\n    final_mask=final_mask.astype(np.uint8)\n    imgs=grid_slice(img,size_each_slice,slice_stride,output_size_each_slice)\n    f_masks=grid_slice(final_mask,size_each_slice,slice_stride,output_size_each_slice)\n    \n    for w_img,w_mask in zip(imgs,f_masks):\n        example=serialize_example(w_img, w_mask,im_id)\n    \n        writer.write(example)\nwriter.close()\n\nprint('finish')\n\n\ndel(writer)\ndel(train_df)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T12:13:37.169242Z","iopub.execute_input":"2022-08-05T12:13:37.170300Z","iopub.status.idle":"2022-08-05T12:19:50.771592Z","shell.execute_reply.started":"2022-08-05T12:13:37.170249Z","shell.execute_reply":"2022-08-05T12:19:50.770580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"the tfrecrod is now ready you can save it and use it the following  code can be used as an example of loading it in dataset object ","metadata":{}},{"cell_type":"markdown","source":"creating a model","metadata":{}}]}