{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# COCO Dataset Generator\n> forked from [here](https://www.kaggle.com/coldfir3/efficient-coco-dataset-generator)","metadata":{}},{"cell_type":"code","source":"!pip install -Uqqq pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:27:14.826611Z","iopub.execute_input":"2021-12-30T19:27:14.827178Z","iopub.status.idle":"2021-12-30T19:27:31.884495Z","shell.execute_reply.started":"2021-12-30T19:27:14.82714Z","shell.execute_reply":"2021-12-30T19:27:31.883703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nfrom pathlib import Path\nfrom pycocotools.coco import COCO\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-30T19:39:25.57679Z","iopub.execute_input":"2021-12-30T19:39:25.577269Z","iopub.status.idle":"2021-12-30T19:39:25.582681Z","shell.execute_reply.started":"2021-12-30T19:39:25.577222Z","shell.execute_reply":"2021-12-30T19:39:25.581997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Params","metadata":{}},{"cell_type":"code","source":"FOLD = 0\ndataDir=Path('../input/sartorius-cell-instance-segmentation/train')","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:38:25.085393Z","iopub.execute_input":"2021-12-30T19:38:25.085848Z","iopub.status.idle":"2021-12-30T19:38:25.09045Z","shell.execute_reply.started":"2021-12-30T19:38:25.085802Z","shell.execute_reply":"2021-12-30T19:38:25.089895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:27:32.768154Z","iopub.execute_input":"2021-12-30T19:27:32.768504Z","iopub.status.idle":"2021-12-30T19:27:33.439369Z","shell.execute_reply.started":"2021-12-30T19:27:32.768464Z","shell.execute_reply":"2021-12-30T19:27:33.438511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Split","metadata":{}},{"cell_type":"code","source":"df = df.reset_index(drop=True)\ndf['fold'] = -1\nskf = GroupKFold(n_splits=5)\nfor fold, (_, val_idx) in enumerate(skf.split(X=df, y=df['cell_type'], groups=df['id'])):\n    df.loc[val_idx, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:35:50.689634Z","iopub.execute_input":"2021-12-30T19:35:50.690077Z","iopub.status.idle":"2021-12-30T19:35:50.780845Z","shell.execute_reply.started":"2021-12-30T19:35:50.690038Z","shell.execute_reply":"2021-12-30T19:35:50.78023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper","metadata":{}},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    \n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype = np.uint)\n    array = array.reshape((-1,2)).T\n    array[0] = array[0] - 1\n    \n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype = np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype = np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img) ## This is important so pycocotools can handle this object\n    \n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:35:52.610217Z","iopub.execute_input":"2021-12-30T19:35:52.610635Z","iopub.status.idle":"2021-12-30T19:35:52.618297Z","shell.execute_reply.started":"2021-12-30T19:35:52.610601Z","shell.execute_reply":"2021-12-30T19:35:52.617418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Minor Sanity Check","metadata":{}},{"cell_type":"code","source":"rle = df.loc[0, 'annotation']\nprint(rle)\nplt.imshow(rle2mask(rle, 704, 520));","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:35:53.887655Z","iopub.execute_input":"2021-12-30T19:35:53.888072Z","iopub.status.idle":"2021-12-30T19:35:54.152712Z","shell.execute_reply.started":"2021-12-30T19:35:53.888038Z","shell.execute_reply":"2021-12-30T19:35:54.151927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# COCO Annotations","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\nimport json,itertools\n\ndef annotate(idx, row, cat_ids):\n        mask = rle2mask(row['annotation'], row['width'], row['height']) # Binary mask\n        c_rle = maskUtils.encode(mask) # Encoding it back to rle (coco format)\n        c_rle['counts'] = c_rle['counts'].decode('utf-8') # converting from binary to utf-8\n        area = maskUtils.area(c_rle).item() # calculating the area\n        bbox = maskUtils.toBbox(c_rle).astype(int).tolist() # calculating the bboxes\n        annotation = {\n            'segmentation': c_rle,\n            'bbox': bbox,\n            'area': area,\n            'image_id':row['id'], \n            'category_id':1, # cat_ids[row['cell_type']], \n            'iscrowd':0, \n            'id':idx\n        }\n        return annotation\n    \ndef coco_structure(df, workers = 4):\n    \n    ## Building the header\n    cat_ids = {\"cell\":1}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'{id}.png'}\\\n              for id,row in df.groupby('id').agg('first').iterrows()]\n    \n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total = len(df)))\n        \n    return {'categories':cats, 'images':images, 'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:41:15.807358Z","iopub.execute_input":"2021-12-30T19:41:15.807685Z","iopub.status.idle":"2021-12-30T19:41:15.819997Z","shell.execute_reply.started":"2021-12-30T19:41:15.80764Z","shell.execute_reply":"2021-12-30T19:41:15.819397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df.query(\"fold!=@FOLD\")\nvalid_df = df.query(\"fold==@FOLD\")","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:41:21.2353Z","iopub.execute_input":"2021-12-30T19:41:21.235946Z","iopub.status.idle":"2021-12-30T19:41:21.262511Z","shell.execute_reply.started":"2021-12-30T19:41:21.235912Z","shell.execute_reply":"2021-12-30T19:41:21.261508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_json = coco_structure(train_df)\nvalid_json = coco_structure(valid_df)","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:41:21.546696Z","iopub.execute_input":"2021-12-30T19:41:21.546976Z","iopub.status.idle":"2021-12-30T19:42:28.828611Z","shell.execute_reply.started":"2021-12-30T19:41:21.546944Z","shell.execute_reply":"2021-12-30T19:42:28.827351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_json['annotations'][0])\ndisplay(valid_json['annotations'][0])","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:37:05.965259Z","iopub.execute_input":"2021-12-30T19:37:05.965515Z","iopub.status.idle":"2021-12-30T19:37:05.974016Z","shell.execute_reply.started":"2021-12-30T19:37:05.965484Z","shell.execute_reply":"2021-12-30T19:37:05.973173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('annotations_train.json', 'w', encoding='utf-8') as f:\n    json.dump(train_json, f, ensure_ascii=True, indent=4)\nwith open('annotations_valid.json', 'w', encoding='utf-8') as f:\n    json.dump(valid_json, f, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:43:53.633919Z","iopub.execute_input":"2021-12-30T19:43:53.634229Z","iopub.status.idle":"2021-12-30T19:43:56.596778Z","shell.execute_reply.started":"2021-12-30T19:43:53.634194Z","shell.execute_reply":"2021-12-30T19:43:56.595777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# COCO Image","metadata":{}},{"cell_type":"code","source":"!mkdir -p train2017\n!mkdir -p valid2017","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:43:56.598478Z","iopub.execute_input":"2021-12-30T19:43:56.598699Z","iopub.status.idle":"2021-12-30T19:43:58.101162Z","shell.execute_reply.started":"2021-12-30T19:43:56.598672Z","shell.execute_reply":"2021-12-30T19:43:58.099889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\ndef run_copy(row):\n    img_path = dataDir/f'{row.id}.png'\n    if row.fold!=FOLD:\n        shutil.copy(img_path, './train2017/')\n    else:\n        shutil.copy(img_path, './valid2017/')","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:43:58.103137Z","iopub.execute_input":"2021-12-30T19:43:58.103573Z","iopub.status.idle":"2021-12-30T19:43:58.108617Z","shell.execute_reply.started":"2021-12-30T19:43:58.103526Z","shell.execute_reply":"2021-12-30T19:43:58.107612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = df.groupby('id').agg('first').reset_index()\n_ = Parallel(n_jobs=-1,\n         backend='threading')(delayed(run_copy)(row) for _, row in tqdm(tmp_df.iterrows(),\n                                                                        total=len(tmp_df)))","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:43:58.110278Z","iopub.execute_input":"2021-12-30T19:43:58.110505Z","iopub.status.idle":"2021-12-30T19:43:59.227412Z","shell.execute_reply.started":"2021-12-30T19:43:58.110476Z","shell.execute_reply":"2021-12-30T19:43:59.226712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sanity check","metadata":{}},{"cell_type":"code","source":"# annFile = Path('./annotations_train.json')\n# coco = COCO(annFile)\n# imgIds = coco.getImgIds()\n\n# imgs = coco.loadImgs(imgIds[-3:])\n# _,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\n# for img, ax in zip(imgs, axs):\n#     I = Image.open(dataDir/img['file_name'])\n#     annIds = coco.getAnnIds(imgIds=[img['id']])\n#     anns = coco.loadAnns(annIds)\n#     ax[0].imshow(I)\n#     ax[1].imshow(I)\n#     plt.sca(ax[1])\n#     coco.showAnns(anns, draw_bbox=True)\n# plt.tight_layout()\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-30T19:44:01.862738Z","iopub.execute_input":"2021-12-30T19:44:01.863034Z","iopub.status.idle":"2021-12-30T19:47:33.208989Z","shell.execute_reply.started":"2021-12-30T19:44:01.863004Z","shell.execute_reply":"2021-12-30T19:47:33.20795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annFile = Path('./annotations_valid.json')\ncoco = COCO(annFile)\nimgIds = coco.getImgIds()\n\nimgs = coco.loadImgs(imgIds[-3:])\n_,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\nfor img, ax in zip(imgs, axs):\n    I = Image.open(dataDir/img['file_name'])\n    annIds = coco.getAnnIds(imgIds=[img['id']])\n    anns = coco.loadAnns(annIds)\n    ax[0].imshow(I)\n    ax[1].imshow(I)\n    plt.sca(ax[1])\n    coco.showAnns(anns, draw_bbox=True)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-30T17:59:53.467776Z","iopub.status.idle":"2021-12-30T17:59:53.468085Z","shell.execute_reply.started":"2021-12-30T17:59:53.467929Z","shell.execute_reply":"2021-12-30T17:59:53.467946Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compress files","metadata":{}},{"cell_type":"code","source":"shutil.make_archive('valid2017', 'zip', 'valid2017')\nshutil.make_archive('train2017', 'zip', 'train2017')","metadata":{"execution":{"iopub.status.busy":"2021-12-30T18:08:12.028221Z","iopub.execute_input":"2021-12-30T18:08:12.028505Z","iopub.status.idle":"2021-12-30T18:08:17.788762Z","shell.execute_reply.started":"2021-12-30T18:08:12.028474Z","shell.execute_reply":"2021-12-30T18:08:17.787286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r train2017\n!rm -r valid2017","metadata":{"execution":{"iopub.status.busy":"2021-12-30T18:08:38.160514Z","iopub.execute_input":"2021-12-30T18:08:38.160796Z","iopub.status.idle":"2021-12-30T18:08:39.839205Z","shell.execute_reply.started":"2021-12-30T18:08:38.160766Z","shell.execute_reply":"2021-12-30T18:08:39.838107Z"},"trusted":true},"execution_count":null,"outputs":[]}]}