{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Credits: https://www.kaggle.com/code/aleksandrkruchinin/hubmap-hpa-images-tfrecords, https://www.kaggle.com/code/marcosnovaes/hubmap-read-data-and-build-tfrecords","metadata":{}},{"cell_type":"markdown","source":"### Related notebooks:\n* Main notebook: [HuBMAP+HPA U-Net models [Results]](https://www.kaggle.com/code/olegbaryshnikov/hubmap-hpa-u-net-models-results)\n* Training and experiment notebook: [[TF-K:TPU] HuBMAP U-Net models [Training]](https://www.kaggle.com/code/olegbaryshnikov/tf-k-tpu-hubmap-u-net-models-training)\n* Inference notebook: [[TF-K] HuBMAP U-Net models [Inference]](https://www.kaggle.com/code/olegbaryshnikov/tf-k-hubmap-u-net-models-inference)\n\n* [Utility] TF Checkpoints converter notebook: [Weights chp to model chp](https://www.kaggle.com/code/olegbaryshnikov/weights-chp-to-model-chp)\n* [Utility] Notebook to resize train images, encode masks and convert them to TFRecords (This): [HuBMAP+HPA Resized TFRecords 512x512](https://www.kaggle.com/code/olegbaryshnikov/hubmap-hpa-resized-tfrecords-512x512)","metadata":{}},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport pandas as pd\nfrom IPython.display import display\n\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport cv2\n\nfrom tqdm.notebook import tqdm\nimport gc\n\nimport tifffile as tiff\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-10T21:12:18.030735Z","iopub.execute_input":"2022-09-10T21:12:18.031154Z","iopub.status.idle":"2022-09-10T21:12:18.039280Z","shell.execute_reply.started":"2022-09-10T21:12:18.031122Z","shell.execute_reply":"2022-09-10T21:12:18.037845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#size of images\nDIM = 512\n\ntrain_csv_path = '../input/hubmap-organ-segmentation/train.csv'\ntrain_images_folder = '../input/hubmap-organ-segmentation/train_images'","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.356623Z","iopub.execute_input":"2022-09-10T20:54:42.357225Z","iopub.status.idle":"2022-09-10T20:54:42.365528Z","shell.execute_reply.started":"2022-09-10T20:54:42.357191Z","shell.execute_reply":"2022-09-10T20:54:42.364400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/code/friedchips/fast-fully-correct-hubmap-v2-rle-encoding\ndef enc2mask(rle, mask_shape):\n    ''' takes a space-delimited RLE string in column-first order\n    and turns it into a 2d boolean numpy array of shape mask_shape '''\n    \n    mask = np.zeros(np.prod(mask_shape), dtype=np.uint8) # 1d mask array\n    rle = np.array(rle.split()).astype(int) # rle values to ints\n    starts = rle[::2]\n    lengths = rle[1::2]\n    for s, l in zip(starts, lengths):\n        mask[s:s+l] = 1\n    return mask.reshape(np.flip(mask_shape)).T # flip because of column-first order","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.367918Z","iopub.execute_input":"2022-09-10T20:54:42.369165Z","iopub.status.idle":"2022-09-10T20:54:42.378574Z","shell.execute_reply.started":"2022-09-10T20:54:42.369122Z","shell.execute_reply":"2022-09-10T20:54:42.377439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv=pd.read_csv(train_csv_path)\ntrain_csv","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.380844Z","iopub.execute_input":"2022-09-10T20:54:42.381940Z","iopub.status.idle":"2022-09-10T20:54:42.615524Z","shell.execute_reply.started":"2022-09-10T20:54:42.381905Z","shell.execute_reply":"2022-09-10T20:54:42.614390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.617167Z","iopub.execute_input":"2022-09-10T20:54:42.617670Z","iopub.status.idle":"2022-09-10T20:54:42.624582Z","shell.execute_reply.started":"2022-09-10T20:54:42.617625Z","shell.execute_reply":"2022-09-10T20:54:42.623357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def serialize_example(image, mask):\n  \"\"\"\n  Creates a tf.train.Example message ready to be written to a file.\n  \"\"\"\n  # Create a dictionary mapping the feature name to the tf.train.Example-compatible\n  # data type.\n  feature = {\n      'image': _bytes_feature(image),\n      'mask': _bytes_feature(mask),\n  }\n\n  # Create a Features message using tf.train.Example.\n\n  example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n  return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.627530Z","iopub.execute_input":"2022-09-10T20:54:42.628007Z","iopub.status.idle":"2022-09-10T20:54:42.636559Z","shell.execute_reply.started":"2022-09-10T20:54:42.627957Z","shell.execute_reply":"2022-09-10T20:54:42.635607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(img,mask=None,title=None,save=False):\n    fig=plt.figure(figsize=(20, 60))\n    \n    plt.imshow(img)\n    if(mask is not None):\n        plt.imshow(mask, cmap='coolwarm', alpha=0.5)\n    \n    if(title is not None):\n        plt.title(title)\n        if(save):\n            plt.savefig(f'{title}.png')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:54:42.637844Z","iopub.execute_input":"2022-09-10T20:54:42.638156Z","iopub.status.idle":"2022-09-10T20:54:42.652561Z","shell.execute_reply.started":"2022-09-10T20:54:42.638128Z","shell.execute_reply":"2022-09-10T20:54:42.651569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids=train_csv.id\nrles=train_csv.rle\n\nos.makedirs('train')\n\nfor file_id,rle in tqdm(zip(ids,rles)):\n    img_path=os.path.join(train_images_folder,f'{file_id}.tiff')\n    \n    img = tiff.imread(img_path)\n    \n    mask = enc2mask(rle,(img.shape[0],img.shape[1]))  \n    \n    #show_image(img,mask)\n    #resizing\n    img = cv2.resize(img, (DIM, DIM),interpolation = cv2.INTER_AREA)\n    mask = cv2.resize(mask, (DIM, DIM),interpolation = cv2.INTER_NEAREST)\n    \n    #show_image(img,mask)\n    \n    #convert to tfrec\n    filename = f'train/{file_id}-{DIM}.tfrec'\n    with tf.io.TFRecordWriter(filename) as writer:\n        example = serialize_example(img.tobytes(),mask.tobytes())\n        writer.write(example)\n    writer.close()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-10T20:55:44.958182Z","iopub.execute_input":"2022-09-10T20:55:44.958916Z","iopub.status.idle":"2022-09-10T20:58:55.049680Z","shell.execute_reply.started":"2022-09-10T20:55:44.958874Z","shell.execute_reply":"2022-09-10T20:58:55.048367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!printf \"Number of files: $(ls /kaggle/working/train | wc -l)\\n\"\n\n!printf \"\\nList of files:\\n\"\n!ls /kaggle/working/train","metadata":{"execution":{"iopub.status.busy":"2022-09-10T21:04:49.271775Z","iopub.execute_input":"2022-09-10T21:04:49.272224Z","iopub.status.idle":"2022-09-10T21:04:52.672060Z","shell.execute_reply.started":"2022-09-10T21:04:49.272186Z","shell.execute_reply":"2022-09-10T21:04:52.670598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _parse_image_function(example_proto):\n    image_feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'mask': tf.io.FixedLenFeature([], tf.string)\n    }\n    single_example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = tf.reshape( tf.io.decode_raw(single_example['image'],out_type=np.dtype('uint8')), (DIM,DIM, 3))\n    mask =  tf.reshape(tf.io.decode_raw(single_example['mask'],out_type='bool'),(DIM,DIM,1))\n    return image, mask\n\n\ndef load_dataset(filenames):\n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.map(lambda ex: _parse_image_function(ex))\n    return dataset\n\nBATCH_NUM=8\ndef get_dataset(FILENAME):\n    dataset = load_dataset(FILENAME)\n    dataset = dataset.batch(BATCH_NUM)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-09-10T21:12:00.395294Z","iopub.execute_input":"2022-09-10T21:12:00.395928Z","iopub.status.idle":"2022-09-10T21:12:00.405954Z","shell.execute_reply.started":"2022-09-10T21:12:00.395893Z","shell.execute_reply":"2022-09-10T21:12:00.404585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob.glob('train/*.tfrec')\nimgs,masks=next(iter(get_dataset(train_images)))\n\nprint(\"image shape:\",imgs.shape)\nprint(\"mask shape:\",masks.shape)\n\nplt.figure(figsize=(40, 60))\nfor i in range(0,BATCH_NUM):\n    plt.subplot(1,BATCH_NUM,i+1)\n    plt.imshow(imgs[i])\n    plt.imshow(masks[i], cmap='coolwarm', alpha=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-09-10T21:13:38.990348Z","iopub.execute_input":"2022-09-10T21:13:38.990833Z","iopub.status.idle":"2022-09-10T21:13:41.135060Z","shell.execute_reply.started":"2022-09-10T21:13:38.990793Z","shell.execute_reply":"2022-09-10T21:13:41.133658Z"},"trusted":true},"execution_count":null,"outputs":[]}]}