{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/34547/logos/header.png?t=2022-02-15-22-37-27\" width=1500 class=\"center\">\n<h1 >Create a 256x256 COCO Dataset for this Comp !</h1>\n<h1  style=\"background-color:#D4FFC8\">PLEASE UPVOTE THIS and ORIGINAL IF THIS HELPS! </h1>\n\n<a align=\"center\" href = \"https://www.kaggle.com/code/alejopaullier/how-to-create-a-coco-dataset\">ORIGINAL NOTEBOOK(I just modified from this)</a>\n\n<h2>This Note Book will Create val&train coco data</h2>","metadata":{}},{"cell_type":"markdown","source":"### Import libraries 📚","metadata":{}},{"cell_type":"code","source":"import cv2\nimport itertools\nimport json\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nimport random\n\n\nfrom itertools import groupby\nfrom skimage import io\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nimport gc\nimport rasterio\nfrom PIL import Image\nimport tifffile as tiff\nfrom tqdm.notebook import tqdm\nfrom rasterio.windows import Window\nfrom torch.utils.data import Dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:18.892678Z","iopub.execute_input":"2022-08-05T16:40:18.893226Z","iopub.status.idle":"2022-08-05T16:40:18.902567Z","shell.execute_reply.started":"2022-08-05T16:40:18.893182Z","shell.execute_reply":"2022-08-05T16:40:18.900836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Install pycocotools 🥥🛠","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools\nimport pycocotools.mask as mask_util","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:18.937726Z","iopub.execute_input":"2022-08-05T16:40:18.938412Z","iopub.status.idle":"2022-08-05T16:40:31.822923Z","shell.execute_reply.started":"2022-08-05T16:40:18.938358Z","shell.execute_reply":"2022-08-05T16:40:31.821386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Functions 📜","metadata":{}},{"cell_type":"code","source":"# functions to convert encoding to mask and mask to encoding\nsz = 256   # the size of tiles\nreduce = 4 # reduce the original images by 4 times \nMASKS = '../input/hubmap-organ-segmentation/train.csv'\nDATA = '../input/hubmap-organ-segmentation/train_images'\n\ndef enc2mask(encs, shape):\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for m,enc in enumerate(encs):\n        if isinstance(enc,np.float) and np.isnan(enc): continue\n        s = enc.split()\n        for i in range(len(s)//2):\n            start = int(s[2*i]) - 1\n            length = int(s[2*i+1])\n            img[start:start+length] = 1 + m\n    img = img.reshape(shape).T\n    return img\n\ndef mask2enc(mask, n=1):\n    pixels = mask.T.flatten()\n    encs = []\n    for i in range(1,n+1):\n        p = (pixels == i).astype(np.int8)\n        if p.sum() == 0: encs.append(np.nan)\n        else:\n            p = np.concatenate([[0], p, [0]])\n            runs = np.where(p[1:] != p[:-1])[0] + 1\n            runs[1::2] -= runs[::2]\n            encs.append(' '.join(str(x) for x in runs))\n    return encs\n\ndf_masks = pd.read_csv(MASKS)[['id', 'rle','organ']].set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:31.825628Z","iopub.execute_input":"2022-08-05T16:40:31.826023Z","iopub.status.idle":"2022-08-05T16:40:32.019780Z","shell.execute_reply.started":"2022-08-05T16:40:31.825984Z","shell.execute_reply":"2022-08-05T16:40:32.018495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_th = 40  # saturation blancking threshold\np_th = 1000*(sz // 256) ** 2 # threshold for the minimum number of pixels\nclass HuBMAPDataset(Dataset):\n    def __init__(self, idx, sz=sz, reduce=reduce, encs=None):\n        self.data = rasterio.open(os.path.join(DATA,str(idx)+'.tiff'),num_threads='all_cpus')\n        # some images have issues with their format \n        # and must be saved correctly before reading with rasterio\n        if self.data.count != 3:\n            subdatasets = self.data.subdatasets\n            self.layers = []\n            if len(subdatasets) > 0:\n                for i, subdataset in enumerate(subdatasets, 0):\n                    self.layers.append(rasterio.open(subdataset))\n        self.idx = idx\n        self.shape = self.data.shape\n        self.reduce = reduce\n        self.sz = reduce*sz\n        self.pad0 = (self.sz - self.shape[0]%self.sz)%self.sz\n        self.pad1 = (self.sz - self.shape[1]%self.sz)%self.sz\n        self.n0max = (self.shape[0] + self.pad0)//self.sz\n        self.n1max = (self.shape[1] + self.pad1)//self.sz\n        self.mask = enc2mask(encs,(self.shape[1],self.shape[0])) if encs is not None else None\n        \n    def __len__(self):\n        return self.n0max*self.n1max\n    \n    def __getitem__(self, idx):\n        # the code below may be a little bit difficult to understand,\n        # but the thing it does is mapping the original image to\n        # tiles created with adding padding (like in the previous version of the kernel)\n        # then the tiles are loaded with rasterio\n        # n0,n1 - are the x and y index of the tile (idx = n0*self.n1max + n1)\n        n0,n1 = idx//self.n1max, idx%self.n1max\n        # x0,y0 - are the coordinates of the lower left corner of the tile in the image\n        # negative numbers correspond to padding (which must not be loaded)\n        x0,y0 = -self.pad0//2 + n0*self.sz, -self.pad1//2 + n1*self.sz\n\n        # make sure that the region to read is within the image\n        p00,p01 = max(0,x0), min(x0+self.sz,self.shape[0])\n        p10,p11 = max(0,y0), min(y0+self.sz,self.shape[1])\n        img = np.zeros((self.sz,self.sz,3),np.uint8)\n        mask = np.zeros((self.sz,self.sz),np.uint8)\n        # mapping the loade region to the tile\n        if self.data.count == 3:\n            img[(p00-x0):(p01-x0),(p10-y0):(p11-y0)] = np.moveaxis(self.data.read([1,2,3],\n                window=Window.from_slices((p00,p01),(p10,p11))), 0, -1)\n        else:\n            for i,layer in enumerate(self.layers):\n                img[(p00-x0):(p01-x0),(p10-y0):(p11-y0),i] =\\\n                  layer.read(1,window=Window.from_slices((p00,p01),(p10,p11)))\n        if self.mask is not None: mask[(p00-x0):(p01-x0),(p10-y0):(p11-y0)] = self.mask[p00:p01,p10:p11]\n        if self.reduce != 1:\n            img = cv2.resize(img,(self.sz//reduce,self.sz//reduce),\n                             interpolation = cv2.INTER_AREA)\n            mask = cv2.resize(mask,(self.sz//reduce,self.sz//reduce),\n                             interpolation = cv2.INTER_NEAREST)\n        hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n        h,s,v = cv2.split(hsv)\n        #return -1 for empty images\n        return img, mask, (-1 if (s>s_th).sum() <= p_th or img.sum() <= p_th else idx)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:32.021967Z","iopub.execute_input":"2022-08-05T16:40:32.022319Z","iopub.status.idle":"2022-08-05T16:40:32.047336Z","shell.execute_reply.started":"2022-08-05T16:40:32.022286Z","shell.execute_reply":"2022-08-05T16:40:32.046022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%mkdir img","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:32.060512Z","iopub.execute_input":"2022-08-05T16:40:32.061184Z","iopub.status.idle":"2022-08-05T16:40:33.219990Z","shell.execute_reply.started":"2022-08-05T16:40:32.061146Z","shell.execute_reply":"2022-08-05T16:40:33.218526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle, shape):\n    \"\"\"\n    Decodes run-length encoded segmentation mask string into 2d array\n\n    Parameters\n    ----------\n    :param rle_mask (str): Run-length encoded segmentation mask string.\n    :param shape (tuple): (height, width) of array to return\n    :return mask [numpy.ndarray of shape (height, width)]: Decoded 2d segmentation mask\n    \"\"\"\n    # Splits the RLE string into a list of string by whitespaces.\n    s = mask_rle.split()\n    \n    # This creates two numpy arrays, one with the RLE starts and one with their respective lengths\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    \n    # To obtain the end point we need to substract 1 to the length or start because the initial point counts.\n    starts -= 1\n    ends = starts + lengths\n    \n    # Create a 1D array of size H*W of zeros\n    mask = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    \n    # Fill this array with ones in the positions where there is a mask using the RLE information\n    for start, end in zip(starts, ends):\n        mask[start:end] = 1\n    \n    # Reshape the 1D array into a 2D array so we can finally get the binary 2D mask.\n    mask = mask.reshape(shape)\n    return mask.T\n\n\ndef binary_mask_to_rle(binary_mask):\n    \"\"\"\n    Checkout: https://cocodataset.org/#format-results\n    :param mask [numpy.ndarray of shape (height, width)]: Decoded 2d segmentation mask\n    \n    This function returns the following dictionary:\n    {\n        \"counts\": encoded mask suggested by the official COCO dataset webpage.\n        \"size\": the size of the input mask/image\n    }\n    \"\"\"\n    # Create dictionary for the segmentation key in the COCO dataset\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    # We need to convert it to a Fortran array\n    binary_mask_fortran = np.asfortranarray(binary_mask)\n    # Encode the mask as specified by the official COCO format\n    encoded_mask = mask_util.encode(binary_mask_fortran)\n    # We must decode the byte encoded string or otherwise we cannot save it as a JSON file\n    rle[\"counts\"] = encoded_mask[\"counts\"].decode()\n    return rle\n\n\ndef create_coco_format_json(data_frame, classes, filepaths):\n    \"\"\"\n    This function creates a COCO dataset.\n    :param data_frame: pandas dataframe with an \"id\" column.\n    :param classes: list of strings where each string is a class.\n    :param filepaths: a list of strings containing all images paths\n    :return dataset_coco_format: COCO dataset (JSON).\n    \"\"\"\n    images = []\n    annotations = []\n    categories = []\n    count = 0\n    \n    # Creates a categories list, i.e: [{'id': 0, 'name': 'a'}, {'id': 1, 'name': 'b'}, {'id': 2, 'name': 'c'}] \n    for idx, class_ in enumerate(classes):\n        categories.append(\n            { \n                \"id\": idx,\n                \"name\": class_\n            }\n        )\n    \n    # Iterate over image filepaths\n    for index, encs in tqdm(data_frame.iterrows(),total=len(data_frame)):\n        # Get the image id, e.g: \"10044\"\n#         file_id = filepath.split(\"/\")[-1][:-5]\n        ds = HuBMAPDataset(index,encs=encs)#get helper dataset\n        # Get the image height, e.g: 360 (px)\n        height = 256\n        # Get the image width, e.g: 310 (px)\n        width = 256\n        # One image has many annotations associated to it (1 for each class), get a list with the indices.\n#         ids = data_frame.index[data_frame['id'] == int(file_id)].tolist()\n        # Get filename\n#         file_name = filepath.split(\"/\")[-1]\n        for i in range(len(ds)):\n            im,mk,idx = ds[i]\n            if idx < 0: continue\n            tile_id = f'{index}{idx:04d}'\n            tile_name = f'{index}_{idx:04d}.png'\n            cv2.imwrite('/kaggle/working/img/'+f'{index}_{idx:04d}.png', im)\n#             print(mk)\n            if(mk.sum()>0):\n                images.append(\n                    {\n                        \"id\": tile_id,\n                        \"width\": width,\n                        \"height\": height,\n                        \"file_name\": tile_name\n                    }\n                )\n                ys, xs = np.where(mk)\n                x1, x2 = min(xs), max(xs)\n                y1, y2 = min(ys), max(ys)\n                contours, hierarchy = cv2.findContours(mk,cv2.RETR_CCOMP,cv2.CHAIN_APPROX_NONE) \n                for id_, contour in enumerate(contours):\n                    # Image with 3 channels where H and W remain the same.\n                    mask_image = np.zeros((mk.shape[0], mk.shape[1], 3),  np.uint8)\n                    # This function takes the image and fills the contour inside it.\n                    cv2.drawContours(mask_image, [contour], -1, (255,255,255), thickness=cv2.FILLED)\n                    mask_image = cv2.cvtColor(mask_image, cv2.COLOR_BGR2GRAY)\n                    mask_image_bool = np.array(mask_image, dtype=bool).astype(np.uint8)\n                    ys, xs = np.where(mask_image_bool)\n                    x1, x2 = min(xs), max(xs)\n                    y1, y2 = min(ys), max(ys)\n                    enc = binary_mask_to_rle(mask_image_bool)\n                    data_frame_dict = data_frame.to_dict()\n#                     print(data_frame_dict[\"organ\"][index])\n                    seg = {\n                        'segmentation': enc, \n                        'bbox': [int(x1), int(y1), int(x2-x1+1), int(y2-y1+1)],\n                        'area': int(np.sum(mask_image_bool)),\n                        'image_id':tile_id, \n                        'category_id':classes.index(data_frame_dict[\"organ\"][index]), \n                        'iscrowd':0, \n                        'id': count\n                        }\n                    annotations.append(seg)\n                    count +=1\n    dataset_coco_format = {\n        \"categories\": categories,\n        \"images\": images,\n        \"annotations\": annotations,\n    }\n    \n    return dataset_coco_format\n\n\ndef sep():\n    print(\"-\"*100)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:33.222281Z","iopub.execute_input":"2022-08-05T16:40:33.222668Z","iopub.status.idle":"2022-08-05T16:40:33.250980Z","shell.execute_reply.started":"2022-08-05T16:40:33.222630Z","shell.execute_reply":"2022-08-05T16:40:33.249521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data 💾","metadata":{}},{"cell_type":"code","source":"# Setting the paths.\nDATASET_PATH = \"../input/hubmap-organ-segmentation/\"\nIMAGE_DIR = DATASET_PATH + \"train_images/\"\n\n\nsep()\ntrain_df, val_df = train_test_split(df_masks, train_size=0.85, shuffle=True, random_state=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:33.253190Z","iopub.execute_input":"2022-08-05T16:40:33.254275Z","iopub.status.idle":"2022-08-05T16:40:33.273996Z","shell.execute_reply.started":"2022-08-05T16:40:33.254226Z","shell.execute_reply":"2022-08-05T16:40:33.272687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of Train Images:{}\".format(len(train_df)))\ndisplay(train_df.head())\nsep()\nprint(\"Number of Test Images:{}\".format(len(val_df)))\ndisplay(val_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:33.275616Z","iopub.execute_input":"2022-08-05T16:40:33.276028Z","iopub.status.idle":"2022-08-05T16:40:33.303071Z","shell.execute_reply.started":"2022-08-05T16:40:33.275984Z","shell.execute_reply":"2022-08-05T16:40:33.301820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create COCO dataset","metadata":{}},{"cell_type":"code","source":"# Set classes\nclasses = train_df[\"organ\"].unique().tolist()\n\n# Get list with all image file paths\nfilepaths_train = list()\nfilepaths_val = list()\n\nfor f in train_df.itertuples():\n    filepaths_train.append(f\"{f[1]}\"+\".tiff\")\nfor f in val_df.itertuples():\n    filepaths_val.append(f\"{f[1]}\"+\".tiff\")\n\n# Create COCO Datasets\ntrain_json = create_coco_format_json(train_df, classes, filepaths_train)\nval_json = create_coco_format_json(val_df, classes, filepaths_val)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:40:33.304981Z","iopub.execute_input":"2022-08-05T16:40:33.305325Z","iopub.status.idle":"2022-08-05T16:43:51.912138Z","shell.execute_reply.started":"2022-08-05T16:40:33.305294Z","shell.execute_reply":"2022-08-05T16:43:51.909524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save results 💻","metadata":{}},{"cell_type":"code","source":"with open('train_json.json', 'w', encoding='utf-8') as f:\n    json.dump(train_json, f, indent=4)\n\nwith open('val_json.json', 'w', encoding='utf-8') as f:\n    json.dump(val_json, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T16:44:18.093402Z","iopub.execute_input":"2022-08-05T16:44:18.093917Z","iopub.status.idle":"2022-08-05T16:44:18.456562Z","shell.execute_reply.started":"2022-08-05T16:44:18.093881Z","shell.execute_reply":"2022-08-05T16:44:18.455287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}