{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2021-12-26T22:28:39.520453Z","iopub.execute_input":"2021-12-26T22:28:39.521212Z","iopub.status.idle":"2021-12-26T22:28:39.554997Z","shell.execute_reply.started":"2021-12-26T22:28:39.521115Z","shell.execute_reply":"2021-12-26T22:28:39.554382Z"}}},{"cell_type":"code","source":"!pip install -Uqqq pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:08.580760Z","iopub.execute_input":"2021-12-27T18:32:08.581102Z","iopub.status.idle":"2021-12-27T18:32:28.242532Z","shell.execute_reply.started":"2021-12-27T18:32:08.581015Z","shell.execute_reply":"2021-12-27T18:32:28.241689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:28.244369Z","iopub.execute_input":"2021-12-27T18:32:28.245096Z","iopub.status.idle":"2021-12-27T18:32:28.249575Z","shell.execute_reply.started":"2021-12-27T18:32:28.245058Z","shell.execute_reply":"2021-12-27T18:32:28.248783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/k/mhmdsyed/annotation-correction-v2/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:28.250684Z","iopub.execute_input":"2021-12-27T18:32:28.250970Z","iopub.status.idle":"2021-12-27T18:32:29.059216Z","shell.execute_reply.started":"2021-12-27T18:32:28.250929Z","shell.execute_reply":"2021-12-27T18:32:29.058336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    \n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype = np.uint)\n    array = array.reshape((-1,2)).T\n    array[0] = array[0] - 1\n    \n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype = np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype = np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img) ## This is important so pycocotools can handle this object\n    \n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:29.061427Z","iopub.execute_input":"2021-12-27T18:32:29.061763Z","iopub.status.idle":"2021-12-27T18:32:29.070121Z","shell.execute_reply.started":"2021-12-27T18:32:29.061721Z","shell.execute_reply":"2021-12-27T18:32:29.069519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\n\ndef annotate(idx, row, cat_ids):\n        mask = rle2mask(row['annotation'], row['width'], row['height']) # Binary mask\n        c_rle = maskUtils.encode(mask) # Encoding it back to rle (coco format)\n        c_rle['counts'] = c_rle['counts'].decode('utf-8') # converting from binary to utf-8\n        area = maskUtils.area(c_rle).item() # calculating the area\n        bbox = maskUtils.toBbox(c_rle).astype(int).tolist() # calculating the bboxes\n        annotation = {\n            'segmentation': c_rle,\n            'bbox': bbox,\n            'area': area,\n            'image_id':row['id'], \n            'category_id':cat_ids[row['cell_type']], \n            'iscrowd':0, \n            'id':idx\n        }\n        return annotation\n    \ndef coco_structure(df, workers = 4):\n    \n    ## Building the header\n    cat_ids = {name:id+1 for id, name in enumerate(df.cell_type.unique())}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'../input/sartorius-cell-instance-segmentation/train/{id}.png'} for id,row in df.groupby('id').agg('first').iterrows()]\n    \n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total = len(df)))\n        \n    return {'categories':cats, 'images':images, 'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:49.551787Z","iopub.execute_input":"2021-12-27T18:32:49.552450Z","iopub.status.idle":"2021-12-27T18:32:49.703461Z","shell.execute_reply.started":"2021-12-27T18:32:49.552409Z","shell.execute_reply":"2021-12-27T18:32:49.702592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df.id.unique()","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:51.396828Z","iopub.execute_input":"2021-12-27T18:32:51.397279Z","iopub.status.idle":"2021-12-27T18:32:51.400461Z","shell.execute_reply.started":"2021-12-27T18:32:51.397248Z","shell.execute_reply":"2021-12-27T18:32:51.399884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train,df_val = train_test_split(df.id.unique(),test_size=.03,random_state=42,shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:51.666621Z","iopub.execute_input":"2021-12-27T18:32:51.667163Z","iopub.status.idle":"2021-12-27T18:32:52.650747Z","shell.execute_reply.started":"2021-12-27T18:32:51.667127Z","shell.execute_reply":"2021-12-27T18:32:52.649646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df[df.id.isin(df_train)]\ndf_val = df[df.id.isin(df_val)]","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:52.652398Z","iopub.execute_input":"2021-12-27T18:32:52.652673Z","iopub.status.idle":"2021-12-27T18:32:52.680854Z","shell.execute_reply.started":"2021-12-27T18:32:52.652639Z","shell.execute_reply":"2021-12-27T18:32:52.680244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json,itertools\nroot_train = coco_structure(df_train)\nroot_val = coco_structure(df_val)","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:32:52.743696Z","iopub.execute_input":"2021-12-27T18:32:52.743973Z","iopub.status.idle":"2021-12-27T18:33:55.367266Z","shell.execute_reply.started":"2021-12-27T18:32:52.743944Z","shell.execute_reply":"2021-12-27T18:33:55.366582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nwith open('annotations_train.json', 'w', encoding='utf-8') as f:\n    json.dump(root_train, f, ensure_ascii=True, indent=4)\n\n\n\nwith open('annotations_val.json', 'w', encoding='utf-8') as f:\n    json.dump(root_val, f, ensure_ascii=True, indent=4)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-27T18:33:55.368429Z","iopub.execute_input":"2021-12-27T18:33:55.368658Z","iopub.status.idle":"2021-12-27T18:33:58.402869Z","shell.execute_reply.started":"2021-12-27T18:33:55.368622Z","shell.execute_reply":"2021-12-27T18:33:58.402096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# from pycocotools.coco import COCO\n# import matplotlib.pyplot as plt\n# from pathlib import Path\n# from PIL import Image\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-26T02:31:32.463921Z","iopub.execute_input":"2021-12-26T02:31:32.464173Z","iopub.status.idle":"2021-12-26T02:31:32.470625Z","shell.execute_reply.started":"2021-12-26T02:31:32.464143Z","shell.execute_reply":"2021-12-26T02:31:32.469714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataDir=Path('../input/image-filters/data')\n# annFile = Path('./annotations_train.json')\n# coco = COCO(annFile)\n# imgIds = coco.getImgIds()","metadata":{"execution":{"iopub.status.busy":"2021-12-26T02:31:32.820484Z","iopub.execute_input":"2021-12-26T02:31:32.820773Z","iopub.status.idle":"2021-12-26T02:31:35.743563Z","shell.execute_reply.started":"2021-12-26T02:31:32.820742Z","shell.execute_reply":"2021-12-26T02:31:35.742494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-12-26T02:35:02.018115Z","iopub.execute_input":"2021-12-26T02:35:02.018463Z","iopub.status.idle":"2021-12-26T02:35:02.025758Z","shell.execute_reply.started":"2021-12-26T02:35:02.018423Z","shell.execute_reply":"2021-12-26T02:35:02.024873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}