{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Efficient COCO Dataset Generator","metadata":{}},{"cell_type":"markdown","source":"This notebook was Inspired by [this](https://www.kaggle.com/slawekbiel/positive-score-with-detectron-1-3-input-data?scriptVersionId=77658860) great notebook. I made a few improvements in the `rle2mask` code to make it more efficient and used the functions provided in `pycocotools` to generate the json file. This results in massive reduction of compute time and dataset size.\n\nWhat seemed at first to be a trivial task was a bit difficult as the RLE encoding used by COCO is very different from the encoding used in this comp.  \n\nThe comp encoding is rowise and every `odd` index represent the absolute begining of the mask. In the other hand, coco format expects it to be encoded by columns and the `odd` indexes are relative to the last end of the mask.\n\nI couldn't find a trivial way to convert from those two formats without decoding the rle to mask, so the workflow is as folows:\n\n1. Decode rle (competition) to binary mask\n1. Encode the binary mask to rle (coco) using `pycocotools`\n1. Save to `.json`","metadata":{}},{"cell_type":"code","source":"%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:20.567184Z","iopub.execute_input":"2021-11-08T16:42:20.567619Z","iopub.status.idle":"2021-11-08T16:42:20.585117Z","shell.execute_reply.started":"2021-11-08T16:42:20.567566Z","shell.execute_reply":"2021-11-08T16:42:20.583979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -Uqqq pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:20.587756Z","iopub.execute_input":"2021-11-08T16:42:20.588151Z","iopub.status.idle":"2021-11-08T16:42:29.919516Z","shell.execute_reply.started":"2021-11-08T16:42:20.588068Z","shell.execute_reply":"2021-11-08T16:42:29.918549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-08T16:42:29.921336Z","iopub.execute_input":"2021-11-08T16:42:29.921656Z","iopub.status.idle":"2021-11-08T16:42:29.926548Z","shell.execute_reply.started":"2021-11-08T16:42:29.921621Z","shell.execute_reply":"2021-11-08T16:42:29.925556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the train dataframe","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:29.928091Z","iopub.execute_input":"2021-11-08T16:42:29.928422Z","iopub.status.idle":"2021-11-08T16:42:30.334807Z","shell.execute_reply.started":"2021-11-08T16:42:29.928380Z","shell.execute_reply":"2021-11-08T16:42:30.333933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function that decodes rle (for this comp) to a binary mask","metadata":{}},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    \n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype = np.uint)\n    array = array.reshape((-1,2)).T\n    array[0] = array[0] - 1\n    \n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype = np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype = np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img) ## This is important so pycocotools can handle this object\n    \n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:30.336957Z","iopub.execute_input":"2021-11-08T16:42:30.337286Z","iopub.status.idle":"2021-11-08T16:42:30.346107Z","shell.execute_reply.started":"2021-11-08T16:42:30.337255Z","shell.execute_reply":"2021-11-08T16:42:30.345016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Minor Sanity Check","metadata":{}},{"cell_type":"code","source":"rle = df.loc[0, 'annotation']\nprint(rle)\nplt.imshow(rle2mask(rle, 704, 520));","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:30.347267Z","iopub.execute_input":"2021-11-08T16:42:30.347517Z","iopub.status.idle":"2021-11-08T16:42:30.654301Z","shell.execute_reply.started":"2021-11-08T16:42:30.347454Z","shell.execute_reply":"2021-11-08T16:42:30.653390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function that builds the .json file","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\n\ndef annotate(idx, row, cat_ids):\n        mask = rle2mask(row['annotation'], row['width'], row['height']) # Binary mask\n        c_rle = maskUtils.encode(mask) # Encoding it back to rle (coco format)\n        c_rle['counts'] = c_rle['counts'].decode('utf-8') # converting from binary to utf-8\n        area = maskUtils.area(c_rle).item() # calculating the area\n        bbox = maskUtils.toBbox(c_rle).astype(int).tolist() # calculating the bboxes\n        annotation = {\n            'segmentation': c_rle,\n            'bbox': bbox,\n            'area': area,\n            'image_id':row['id'], \n            'category_id':cat_ids[row['cell_type']], \n            'iscrowd':0, \n            'id':idx\n        }\n        return annotation\n    \ndef coco_structure(df, workers = 4):\n    \n    ## Building the header\n    cat_ids = {name:id+1 for id, name in enumerate(df.cell_type.unique())}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'train/{id}.png'} for id,row in df.groupby('id').agg('first').iterrows()]\n    \n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total = len(df)))\n        \n    return {'categories':cats, 'images':images, 'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:30.655570Z","iopub.execute_input":"2021-11-08T16:42:30.655792Z","iopub.status.idle":"2021-11-08T16:42:30.669315Z","shell.execute_reply.started":"2021-11-08T16:42:30.655766Z","shell.execute_reply":"2021-11-08T16:42:30.668153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Running for the whole DF and saving it as a .json file","metadata":{}},{"cell_type":"code","source":"import json,itertools\nroot = coco_structure(df)","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:42:30.671097Z","iopub.execute_input":"2021-11-08T16:42:30.671778Z","iopub.status.idle":"2021-11-08T16:43:43.227738Z","shell.execute_reply.started":"2021-11-08T16:42:30.671729Z","shell.execute_reply":"2021-11-08T16:43:43.227007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root['annotations'][0]","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:43:43.228792Z","iopub.execute_input":"2021-11-08T16:43:43.229020Z","iopub.status.idle":"2021-11-08T16:43:43.237387Z","shell.execute_reply.started":"2021-11-08T16:43:43.228981Z","shell.execute_reply":"2021-11-08T16:43:43.236451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('annotations_train.json', 'w', encoding='utf-8') as f:\n    json.dump(root, f, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:43:43.239858Z","iopub.execute_input":"2021-11-08T16:43:43.240612Z","iopub.status.idle":"2021-11-08T16:43:46.348214Z","shell.execute_reply.started":"2021-11-08T16:43:43.240538Z","shell.execute_reply":"2021-11-08T16:43:46.347542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sanity check","metadata":{}},{"cell_type":"code","source":"from pycocotools.coco import COCO\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:43:46.349563Z","iopub.execute_input":"2021-11-08T16:43:46.350016Z","iopub.status.idle":"2021-11-08T16:43:46.355742Z","shell.execute_reply.started":"2021-11-08T16:43:46.349982Z","shell.execute_reply":"2021-11-08T16:43:46.355123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataDir=Path('../input/sartorius-cell-instance-segmentation')\nannFile = Path('./annotations_train.json')\ncoco = COCO(annFile)\nimgIds = coco.getImgIds()","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:43:46.357217Z","iopub.execute_input":"2021-11-08T16:43:46.357521Z","iopub.status.idle":"2021-11-08T16:43:47.619741Z","shell.execute_reply.started":"2021-11-08T16:43:46.357463Z","shell.execute_reply":"2021-11-08T16:43:47.618717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = coco.loadImgs(imgIds[-3:])\n_,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\nfor img, ax in zip(imgs, axs):\n    I = Image.open(dataDir/img['file_name'])\n    annIds = coco.getAnnIds(imgIds=[img['id']])\n    anns = coco.loadAnns(annIds)\n    ax[0].imshow(I)\n    ax[1].imshow(I)\n    plt.sca(ax[1])\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-08T16:43:47.621009Z","iopub.execute_input":"2021-11-08T16:43:47.621263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}