{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf \nimport cv2\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport time\nimport random\nimport tensorflow_addons as tfa\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm import tqdm\n\n\n\nAUTO = tf.data.AUTOTUNE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-30T03:01:56.971338Z","iopub.execute_input":"2021-12-30T03:01:56.972949Z","iopub.status.idle":"2021-12-30T03:01:56.980598Z","shell.execute_reply.started":"2021-12-30T03:01:56.972878Z","shell.execute_reply":"2021-12-30T03:01:56.979956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:56.982384Z","iopub.execute_input":"2021-12-30T03:01:56.983133Z","iopub.status.idle":"2021-12-30T03:01:56.994403Z","shell.execute_reply.started":"2021-12-30T03:01:56.983095Z","shell.execute_reply":"2021-12-30T03:01:56.99358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _bytes_feature(value):         # S/O la doc tensorflow et le livre pour convertyre en byte pour le TFRECORD\n  \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n  if isinstance(value, type(tf.constant(0))):\n    value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n  return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))\n\n\ndef _int64_feature(value):\n  \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n  return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:56.997002Z","iopub.execute_input":"2021-12-30T03:01:56.997732Z","iopub.status.idle":"2021-12-30T03:01:57.007444Z","shell.execute_reply.started":"2021-12-30T03:01:56.997688Z","shell.execute_reply":"2021-12-30T03:01:57.006644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_tfrecord_2():\n    start = time.time() \n    \n    df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')    #datafrme\n    shape = (520, 704)     # all image in test data have the same shape\n    UI = df[\"id\"].unique() # Nombre de De photo / id differente        #606 \n    \n    df1 = df.groupby('id',as_index=False,sort=False).last()\n    lbl = LabelEncoder()\n    df1[\"cell_type\"] = lbl.fit_transform(df1[\"cell_type\"])\n    option = tf.io.TFRecordOptions(compression_level=2, compression_type=\"ZLIB\")\n    \n    for n in range(3):\n        with tf.io.TFRecordWriter(f\"Train_type_{n}.tfrec\", options=option) as writer:\n            for i in tqdm(range(len(UI))):   \n                if df1[\"cell_type\"].iloc[i]==n:\n                    img_masks = df.loc[df['id']==UI[i], 'annotation'].to_list() # all mask in 1 list \n                    img = cv2.imread(f\"../input/sartorius-cell-instance-segmentation/train/{UI[i]}.png\")      # Image \n                    all_masks = np.zeros(shape, dtype=np.float32)\n                    img = np.true_divide(img, 255, dtype=np.float32)\n            \n                    for mask in img_masks:\n                        all_masks += rle_decode(mask, shape) # mask\n            \n                    all_masks[all_masks > 1] = 1   #pour avoir que des 1 ou 0 pour le mask \n                    #print(img.dtype , all_masks.dtype , \"ULTRA IMPORTANT\")                             # use to check dtype\n                    #print(all_masks.dtype)\n            \n                    data = {'image': _bytes_feature(img.tobytes()),\n                        'mask': _bytes_feature(all_masks.tobytes()),\n                        'label': _int64_feature(df1[\"cell_type\"].iloc[i]),\n                        }\n        \n                    Data = tf.train.Example(features=tf.train.Features(feature=data))\n                    Data = Data.SerializeToString()\n\n                    writer.write(Data)\n                \n    \n    elapsed = time.time()\n    elapsed = elapsed - start\n    print(\"Time spent: \", elapsed)","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:57.009544Z","iopub.execute_input":"2021-12-30T03:01:57.009812Z","iopub.status.idle":"2021-12-30T03:01:57.025368Z","shell.execute_reply.started":"2021-12-30T03:01:57.009781Z","shell.execute_reply":"2021-12-30T03:01:57.024712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_mask(image_data):\n    image = tf.io.decode_raw(image_data['image'], tf.float32)\n    image = tf.reshape(image, [520,704,3])\n    mask = tf.io.decode_raw(image_data['mask'], tf.float32)\n    mask = tf.reshape(mask, [520,704,1])\n    return (image, mask)\n\n\ndef read_mask(exemple):\n    image_feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'label': tf.io.FixedLenFeature([], tf.int64),\n        'mask': tf.io.FixedLenFeature([], tf.string)\n    }\n    Exemple = tf.io.parse_single_example(exemple , image_feature_description)\n    return decode_mask(Exemple)","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:57.027149Z","iopub.execute_input":"2021-12-30T03:01:57.027786Z","iopub.status.idle":"2021-12-30T03:01:57.04228Z","shell.execute_reply.started":"2021-12-30T03:01:57.027737Z","shell.execute_reply":"2021-12-30T03:01:57.041672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dataset(path , Augment = False , Big = False , label=True):\n    dataset = tf.data.TFRecordDataset(path,compression_type=\"ZLIB\", num_parallel_reads=AUTO)\n    if label:\n        dataset = dataset.map(read_label, num_parallel_calls= AUTO)\n    else:\n        dataset = dataset.map(read_mask, num_parallel_calls= AUTO)\n    if Big:\n        dataset = dataset.repeat(6)\n    if Augment :\n        if label:\n            dataset = dataset.map(augment3L,num_parallel_calls = AUTO) \n        else:\n            dataset = dataset.map(augment3,num_parallel_calls = AUTO)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:57.043548Z","iopub.execute_input":"2021-12-30T03:01:57.044166Z","iopub.status.idle":"2021-12-30T03:01:57.059001Z","shell.execute_reply.started":"2021-12-30T03:01:57.044117Z","shell.execute_reply":"2021-12-30T03:01:57.057954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def victor(path, Augment = False ,big = False, label=True ):\n    data = load_dataset(path, Augment = Augment  , Big = big, label=label)  # class 'tensorflow.python.data.ops.dataset_ops.ParallelMapDataset'   \n    N = 5\n    i = 0\n    '''\n    #pour connaitre le nombre d'element dans un dataset\n    for _ in data:\n        i+=1\n    \n    print(i)'''\n    \n\n    ds = data.take(N)\n    if label:\n        fig, axarr = plt.subplots(N,1, figsize=(15, 40))\n        for image ,label  in ds:\n            print(image.shape , type(image))\n            print(label.shape)\n        \n            #image \n            axarr[i].imshow(image)\n            axarr[i].axis('off')\n            axarr[i].set_title(f'Masks {label}')\n            i+=1\n    else:\n        fig, axarr = plt.subplots(N,2, figsize=(15, 40))\n        for image ,mask  in ds:\n            print(image.shape , type(image))\n            print(mask.shape)\n        \n            reshape = tf.reshape(mask ,[520*704])\n            Unique = tf.unique(reshape)\n            print(mask.shape)\n            print(Unique)\n        \n            #mask\n            axarr[i, 1].imshow(mask)\n            axarr[i, 1].axis('off')\n            axarr[i, 1].set_title(f'Masks {i}')\n    \n            #image \n            axarr[i, 0].imshow(image)\n            axarr[i, 0].axis('off')\n            axarr[i, 0].set_title(f'Masks {i}')\n            i+=1\n    \n    plt.tight_layout(h_pad=0.1, w_pad=0.1)\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:57.060488Z","iopub.execute_input":"2021-12-30T03:01:57.060758Z","iopub.status.idle":"2021-12-30T03:01:57.073541Z","shell.execute_reply.started":"2021-12-30T03:01:57.060726Z","shell.execute_reply":"2021-12-30T03:01:57.072833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    path0 = \"./Train_type_0.tfrec\"\n    path1 = \"./Train_type_1.tfrec\"\n    path2 = \"./Train_type_2.tfrec\"\n    \n    build_tfrecord_2()\n    \n    '''victor(path0,Augment = False, big = True ,label=False)\n    victor(path1,Augment = False, big = True ,label=False)\n    victor(path2,Augment = False, big = True ,label=False)\n    '''","metadata":{"execution":{"iopub.status.busy":"2021-12-30T03:01:57.099295Z","iopub.execute_input":"2021-12-30T03:01:57.1001Z","iopub.status.idle":"2021-12-30T03:02:04.016819Z","shell.execute_reply.started":"2021-12-30T03:01:57.100056Z","shell.execute_reply":"2021-12-30T03:02:04.015942Z"},"trusted":true},"execution_count":null,"outputs":[]}]}