{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":34547,"databundleVersionId":3897958,"sourceType":"competition"}],"dockerImageVersionId":30207,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/34547/logos/header.png?t=2022-02-15-22-37-27\" width=1500 class=\"center\">\n<h1 align=\"center\">How To Create a COCO Dataset 🥥</h1>\n\nCheckout the [dataset here](https://www.kaggle.com/datasets/alejopaullier/hubmap-hpa-coco-dataset)\n\nThe objective of this notebook is to create a [COCO format](https://cocodataset.org/#format-data) dataset given a `.csv` file which contains image metadata such as image IDs, segmentations, paths to images, etc, and a folder containing the images.\n\n**With some little changes this notebook can be adapted to a different problem.**\n\nIn this notebook we will:\n- Define functions to decode and encode RLE strings.\n- Define a function to create a COCO dataset.\n- Transform our data and save it.\n- Visualize some examples to check quality.\n\nThis notebook was inspired by [Cocoformat dataset creation Instance Segmentation](https://www.kaggle.com/code/mohanrobotics/cocoformat-dataset-creation-instance-segmentation/notebook).\n\nHope you like it! ⚡️\n\n<p style=\"background-color:#FFCBCB; color:Red\">Note: I don't know why but RLE masks need to be transposed. If you know the reason please leave it in the comments.</p>","metadata":{}},{"cell_type":"markdown","source":"### Import libraries 📚","metadata":{}},{"cell_type":"code","source":"import cv2\nimport itertools\nimport json\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nimport random\n\n\nfrom itertools import groupby\nfrom skimage import io\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:18.236827Z","iopub.execute_input":"2022-07-15T23:53:18.237660Z","iopub.status.idle":"2022-07-15T23:53:20.002742Z","shell.execute_reply.started":"2022-07-15T23:53:18.237618Z","shell.execute_reply":"2022-07-15T23:53:20.001604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Install pycocotools 🥥🛠","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools\nimport pycocotools.mask as mask_util","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:20.005040Z","iopub.execute_input":"2022-07-15T23:53:20.005470Z","iopub.status.idle":"2022-07-15T23:53:54.758526Z","shell.execute_reply.started":"2022-07-15T23:53:20.005429Z","shell.execute_reply":"2022-07-15T23:53:54.757412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Functions 📜","metadata":{}},{"cell_type":"code","source":"def rle_decode(mask_rle, shape):\n    \"\"\"\n    Decodes run-length encoded segmentation mask string into 2d array\n\n    Parameters\n    ----------\n    :param rle_mask (str): Run-length encoded segmentation mask string.\n    :param shape (tuple): (height, width) of array to return\n    :return mask [numpy.ndarray of shape (height, width)]: Decoded 2d segmentation mask\n    \"\"\"\n    # Splits the RLE string into a list of string by whitespaces.\n    s = mask_rle.split()\n    \n    # This creates two numpy arrays, one with the RLE starts and one with their respective lengths\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    \n    # To obtain the end point we need to substract 1 to the length or start because the initial point counts.\n    starts -= 1\n    ends = starts + lengths\n    \n    # Create a 1D array of size H*W of zeros\n    mask = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    \n    # Fill this array with ones in the positions where there is a mask using the RLE information\n    for start, end in zip(starts, ends):\n        mask[start:end] = 1\n    \n    # Reshape the 1D array into a 2D array so we can finally get the binary 2D mask.\n    mask = mask.reshape(shape)\n    return mask.T\n\n\ndef binary_mask_to_rle(binary_mask):\n    \"\"\"\n    Checkout: https://cocodataset.org/#format-results\n    :param mask [numpy.ndarray of shape (height, width)]: Decoded 2d segmentation mask\n    \n    This function returns the following dictionary:\n    {\n        \"counts\": encoded mask suggested by the official COCO dataset webpage.\n        \"size\": the size of the input mask/image\n    }\n    \"\"\"\n    # Create dictionary for the segmentation key in the COCO dataset\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    # We need to convert it to a Fortran array\n    binary_mask_fortran = np.asfortranarray(binary_mask)\n    # Encode the mask as specified by the official COCO format\n    encoded_mask = mask_util.encode(binary_mask_fortran)\n    # We must decode the byte encoded string or otherwise we cannot save it as a JSON file\n    rle[\"counts\"] = encoded_mask[\"counts\"].decode()\n    return rle\n\n\ndef create_coco_format_json(data_frame, classes, filepaths):\n    \"\"\"\n    This function creates a COCO dataset.\n    :param data_frame: pandas dataframe with an \"id\" column.\n    :param classes: list of strings where each string is a class.\n    :param filepaths: a list of strings containing all images paths\n    :return dataset_coco_format: COCO dataset (JSON).\n    \"\"\"\n    images = []\n    annotations = []\n    categories = []\n    count = 0\n    \n    # Creates a categories list, i.e: [{'id': 0, 'name': 'a'}, {'id': 1, 'name': 'b'}, {'id': 2, 'name': 'c'}] \n    for idx, class_ in enumerate(classes):\n        categories.append(\n            { \n                \"id\": idx,\n                \"name\": class_\n            }\n        )\n    \n    # Iterate over image filepaths\n    for filepath in tqdm(filepaths):\n        # Get the image id, e.g: \"10044\"\n        file_id = filepath.split(\"/\")[-1][:-5]\n        # Get the image height, e.g: 360 (px)\n        height = int(data_frame[data_frame[\"id\"]==int(file_id)][\"img_height\"].values[0])\n        # Get the image width, e.g: 310 (px)\n        width = int(data_frame[data_frame[\"id\"]==int(file_id)][\"img_width\"].values[0])\n        # One image has many annotations associated to it (1 for each class), get a list with the indices.\n        ids = data_frame.index[data_frame['id'] == int(file_id)].tolist()\n        # Get filename\n        file_name = filepath.split(\"/\")[-1]\n        \n        \n        if (len(ids) > 0):\n            # Adding images which has annotations\n            images.append(\n                {\n                    \"id\": file_id,\n                    \"width\": width,\n                    \"height\": height,\n                    \"file_name\": file_name\n                }\n            )\n            for idx in ids:\n                # Convert the RLE string into a numpy array binary mask\n                mk = rle_decode(data_frame.iloc[idx]['rle'], (height, width))\n                ys, xs = np.where(mk)\n                x1, x2 = min(xs), max(xs)\n                y1, y2 = min(ys), max(ys)\n                \"\"\"\n                Contours can be explained simply as a curve joining all the continuous points (along the boundary),\n                having same color or intensity. The function retrieves contours from the binary image using the\n                algorithm specified in the function. One RLE segmentation for a single class may have disconnected\n                shapes, like \"spots\". We will iterate over these \"spots\" thus creating a new image for each spot.\n                This image will be temporary, it will help us create annotations for each of these \"spots\".\n                \"\"\"\n                contours, hierarchy = cv2.findContours(mk,cv2.RETR_CCOMP,cv2.CHAIN_APPROX_NONE)\n                \n                for id_, contour in enumerate(contours):\n                    # Image with 3 channels where H and W remain the same.\n                    mask_image = np.zeros((mk.shape[0], mk.shape[1], 3),  np.uint8)\n                    # This function takes the image and fills the contour inside it.\n                    cv2.drawContours(mask_image, [contour], -1, (255,255,255), thickness=cv2.FILLED)\n                    mask_image = cv2.cvtColor(mask_image, cv2.COLOR_BGR2GRAY)\n                    mask_image_bool = np.array(mask_image, dtype=bool).astype(np.uint8)\n                    ys, xs = np.where(mask_image_bool)\n                    x1, x2 = min(xs), max(xs)\n                    y1, y2 = min(ys), max(ys)\n                    enc = binary_mask_to_rle(mask_image_bool)\n                    seg = {\n                        'segmentation': enc, \n                        'bbox': [int(x1), int(y1), int(x2-x1+1), int(y2-y1+1)],\n                        'area': int(np.sum(mask_image_bool)),\n                        'image_id':file_id, \n                        'category_id':classes.index(data_frame.iloc[idx]['organ']), \n                        'iscrowd':0, \n                        'id': count\n                    }\n                    annotations.append(seg)\n                    count +=1\n            \n    # Create the dataset\n    dataset_coco_format = {\n        \"categories\": categories,\n        \"images\": images,\n        \"annotations\": annotations,\n    }\n    \n    return dataset_coco_format\n\n\ndef sep():\n    print(\"-\"*100)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:54.760270Z","iopub.execute_input":"2022-07-15T23:53:54.760602Z","iopub.status.idle":"2022-07-15T23:53:54.783331Z","shell.execute_reply.started":"2022-07-15T23:53:54.760568Z","shell.execute_reply":"2022-07-15T23:53:54.781995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data 💾","metadata":{}},{"cell_type":"code","source":"# Setting the paths.\nDATASET_PATH = \"../input/hubmap-organ-segmentation/\"\nIMAGE_DIR = DATASET_PATH + \"train_images/\"\nTRAIN_CSV_PATH = DATASET_PATH + \"train.csv\"\nTEST_CSV_PATH = DATASET_PATH + \"test.csv\"\n\n# Creating a dataframe.\ntrain_df = pd.read_csv(TRAIN_CSV_PATH, sep=',')\ntest_df = pd.read_csv(TEST_CSV_PATH, sep=',')\ndisplay(train_df.head())\nsep()\ndisplay(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:54.786595Z","iopub.execute_input":"2022-07-15T23:53:54.787007Z","iopub.status.idle":"2022-07-15T23:53:55.154634Z","shell.execute_reply.started":"2022-07-15T23:53:54.786972Z","shell.execute_reply":"2022-07-15T23:53:55.153486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of Train Images:{}\".format(len(train_df)))\ndisplay(train_df.head())\nsep()\nprint(\"Number of Test Images:{}\".format(len(test_df)))\ndisplay(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:55.155861Z","iopub.execute_input":"2022-07-15T23:53:55.156192Z","iopub.status.idle":"2022-07-15T23:53:55.185895Z","shell.execute_reply.started":"2022-07-15T23:53:55.156164Z","shell.execute_reply":"2022-07-15T23:53:55.184896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create COCO dataset","metadata":{}},{"cell_type":"code","source":"# Set classes\nclasses = train_df[\"organ\"].unique().tolist()\n\n# Get list with all image file paths\nfilepaths = list()\nfor (dirpath, dirnames, filenames) in os.walk(DATASET_PATH):\n    filepaths += [os.path.join(dirpath, file) for file in filenames if file.endswith(\".tiff\")]\n    \nfilepaths_train = filepaths[:-1]\nfilepaths_test = [filepaths[-1]]\n# Create COCO Datasets\ntrain_json = create_coco_format_json(train_df, classes, filepaths_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T23:53:55.187284Z","iopub.execute_input":"2022-07-15T23:53:55.187612Z","iopub.status.idle":"2022-07-16T00:08:40.933927Z","shell.execute_reply.started":"2022-07-15T23:53:55.187581Z","shell.execute_reply":"2022-07-16T00:08:40.932647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save results 💻","metadata":{}},{"cell_type":"code","source":"with open('train_json.json', 'w', encoding='utf-8') as f:\n    json.dump(train_json, f, indent=4)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:10:45.744624Z","iopub.execute_input":"2022-07-16T00:10:45.745036Z","iopub.status.idle":"2022-07-16T00:10:45.988043Z","shell.execute_reply.started":"2022-07-16T00:10:45.745004Z","shell.execute_reply":"2022-07-16T00:10:45.986640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read COCO Dataset","metadata":{}},{"cell_type":"code","source":"from pycocotools.coco import COCO\n\ngt = COCO(\"./train_json.json\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:10:46.918934Z","iopub.execute_input":"2022-07-16T00:10:46.920023Z","iopub.status.idle":"2022-07-16T00:10:46.984954Z","shell.execute_reply.started":"2022-07-16T00:10:46.919979Z","shell.execute_reply":"2022-07-16T00:10:46.983739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's check everything is allright! ✅","metadata":{}},{"cell_type":"markdown","source":"### Plot mask on top of image function\n\nInspired by [AW-Madison: EDA & In Depth Mask Exploration](https://www.kaggle.com/code/andradaolteanu/aw-madison-eda-in-depth-mask-exploration)","metadata":{}},{"cell_type":"code","source":"from matplotlib.colors import LinearSegmentedColormap\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\n\n\n# Custom color map in matplotlib\ndef CustomCmap(rgb_color):\n\n    r1,g1,b1 = rgb_color\n\n    cdict = {'red': ((0, r1, r1),\n                   (1, r1, r1)),\n           'green': ((0, g1, g1),\n                    (1, g1, g1)),\n           'blue': ((0, b1, b1),\n                   (1, b1, b1))}\n\n    cmap = LinearSegmentedColormap('custom_cmap', cdict)\n    return cmap\n\n\ndef plot_original_mask(img, mask, alpha=1):\n\n    # Change pixels - when 1 make True, when 0 make NA\n    mask = np.ma.masked_where(mask == 0, mask)\n\n    # Plot the 2 images (Original and with Mask)\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(24, 10))\n\n    # Original\n    ax1.set_title(\"Original Image\")\n    ax1.imshow(img)\n    ax1.axis(\"off\")\n\n    # With Mask\n    ax2.set_title(\"Image with Mask\")\n    ax2.imshow(img)\n    ax2.imshow(mask, interpolation='none', cmap=CMAP2, alpha=alpha)\n#     ax2.legend(legend_colors, labels)\n    ax2.axis(\"off\")\n    \n    plt.show()\n    \n\ndef rgb(r,g,b):\n    return (r/255, g/255, b/255)\n\n# --- Custom Color Maps ---\n# Yellow Purple Red\nmask_colors = [rgb(232, 0, 255), rgb(0, 4, 255), rgb(0, 251, 255), rgb(0, 255, 15), rgb(247, 255, 0)]\nlegend_colors = [Rectangle((0,0),1,1, color=color) for color in mask_colors]\nlabels = [x.upper() for x in train_df[\"organ\"].unique().tolist()]\n\nCMAP1 = CustomCmap(mask_colors[0])\nCMAP2 = CustomCmap(mask_colors[1])\nCMAP3 = CustomCmap(mask_colors[2])\nCMAP4 = CustomCmap(mask_colors[3])\nCMAP5 = CustomCmap(mask_colors[4])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:10:50.353009Z","iopub.execute_input":"2022-07-16T00:10:50.353366Z","iopub.status.idle":"2022-07-16T00:10:50.367575Z","shell.execute_reply.started":"2022-07-16T00:10:50.353338Z","shell.execute_reply":"2022-07-16T00:10:50.366206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get random image and its mask 🩻","metadata":{}},{"cell_type":"code","source":"def read_tiff(image_path):\n    image = io.imread(image_path)\n    return image\n\n# Random idx\nrandom_image = 4 #np.random.randint(0,len(train_df))\n# Get image ID\nimg_id = list(gt.imgs)[random_image]\n# Read image\nnp_image = read_tiff(os.path.join(IMAGE_DIR, gt.imgs[img_id][\"file_name\"]))\ncv2.imwrite('img.jpg', np_image)\nimg = cv2.imread(\"img.jpg\")\n\nfor ann_id in gt.getAnnIds(imgIds = [img_id]):\n    ann_item = gt.anns[ann_id]\n    mask = mask_util.decode(ann_item[\"segmentation\"])\n    \nplot_original_mask(img, mask, alpha=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:11:30.470238Z","iopub.execute_input":"2022-07-16T00:11:30.471273Z","iopub.status.idle":"2022-07-16T00:11:34.437520Z","shell.execute_reply.started":"2022-07-16T00:11:30.471223Z","shell.execute_reply":"2022-07-16T00:11:34.436385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}