{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pycocotools","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-27T11:19:44.230289Z","iopub.execute_input":"2022-12-27T11:19:44.232127Z","iopub.status.idle":"2022-12-27T11:19:55.353944Z","shell.execute_reply.started":"2022-12-27T11:19:44.232053Z","shell.execute_reply":"2022-12-27T11:19:55.352805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Annotation for image was taken from this noteook https://www.kaggle.com/slawekbiel/positive-score-with-detectron-1-3-input-data/notebook ","metadata":{}},{"cell_type":"code","source":"# from pycocotools.coco import COCO\nimport skimage.io as io\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm\nimport json,itertools\nfrom sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.357606Z","iopub.execute_input":"2022-12-27T11:19:55.358018Z","iopub.status.idle":"2022-12-27T11:19:55.365437Z","shell.execute_reply.started":"2022-12-27T11:19:55.357979Z","shell.execute_reply":"2022-12-27T11:19:55.364135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# config\nclass CFG:\n    data_path = '../input/sartorius-cell-instance-segmentation/'\n    nfolds = 5","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.366719Z","iopub.execute_input":"2022-12-27T11:19:55.367193Z","iopub.status.idle":"2022-12-27T11:19:55.3794Z","shell.execute_reply.started":"2022-12-27T11:19:55.367145Z","shell.execute_reply":"2022-12-27T11:19:55.378301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction\n\n# From https://newbedev.com/encode-numpy-array-using-uncompressed-rle-for-coco-dataset\ndef binary_mask_to_rle(binary_mask):\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(itertools.groupby(binary_mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.38114Z","iopub.execute_input":"2022-12-27T11:19:55.381663Z","iopub.status.idle":"2022-12-27T11:19:55.3952Z","shell.execute_reply.started":"2022-12-27T11:19:55.381617Z","shell.execute_reply":"2022-12-27T11:19:55.393901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coco_structure(train_df):\n    cat_ids = {name:id+1 for id, name in enumerate(train_df.cell_type.unique())}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'train/{id}.png'} for id,row in train_df.groupby('id').agg('first').iterrows()]\n    annotations=[]\n    for idx, row in tqdm(train_df.iterrows()):\n        mk = rle_decode(row.annotation, (row.height, row.width))\n        ys, xs = np.where(mk)\n        x1, x2 = min(xs), max(xs)\n        y1, y2 = min(ys), max(ys)\n        enc =binary_mask_to_rle(mk)\n        seg = {\n            'segmentation':enc, \n            'bbox': [int(x1), int(y1), int(x2-x1+1), int(y2-y1+1)],\n            'area': int(np.sum(mk)),\n            'image_id':row.id, \n            'category_id':cat_ids[row.cell_type], \n            'iscrowd':0, \n            'id':idx\n        }\n        annotations.append(seg)\n    return {'categories':cats, 'images':images,'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.3981Z","iopub.execute_input":"2022-12-27T11:19:55.398811Z","iopub.status.idle":"2022-12-27T11:19:55.410476Z","shell.execute_reply.started":"2022-12-27T11:19:55.398773Z","shell.execute_reply":"2022-12-27T11:19:55.409026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data**\n\nSplit into folds, create annotations","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(CFG.data_path + 'train.csv')\n\ngkf = GroupKFold(n_splits = CFG.nfolds)\n\ntrain_df[\"fold\"] = -1\ny = train_df.width.values\n\nfor f, (t_, v_) in enumerate(gkf.split(X=train_df, y=y, groups=train_df.id.values)):\n    train_df.loc[v_, \"fold\"] = f\n    \nfold_id = train_df.fold.copy()\n# train_df.drop('fold', axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.411582Z","iopub.execute_input":"2022-12-27T11:19:55.411976Z","iopub.status.idle":"2022-12-27T11:19:55.845896Z","shell.execute_reply.started":"2022-12-27T11:19:55.411934Z","shell.execute_reply":"2022-12-27T11:19:55.844796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['id', 'fold', 'annotation']].to_csv('gt_fold.csv', index = False)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:55.847621Z","iopub.execute_input":"2022-12-27T11:19:55.848009Z","iopub.status.idle":"2022-12-27T11:19:56.315907Z","shell.execute_reply.started":"2022-12-27T11:19:55.847979Z","shell.execute_reply":"2022-12-27T11:19:56.314696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_ids = train_df.id.unique()\n# for fold in range(CFG.nfolds):\nfor fold in range(4,5):    \n    train_sample = train_df.loc[fold_id != fold]\n    root = coco_structure(train_sample)\n\n    with open('annotations_train_f'+str(fold)+'.json', 'w', encoding='utf-8') as f:\n        json.dump(root, f, ensure_ascii=True, indent=4)\n        \n    valid_sample = train_df.loc[fold_id == fold]\n\n    print('fold ' + str(fold) + ': produced')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T11:19:56.317675Z","iopub.execute_input":"2022-12-27T11:19:56.318157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in range(4,5):   \n    train_sample = train_df.loc[fold_id == fold]\n    root = coco_structure(train_sample)\n\n    with open('annotations_valid_f'+str(fold)+'.json', 'w', encoding='utf-8') as f:\n        json.dump(root, f, ensure_ascii=True, indent=4)\n        \n    valid_sample = train_df.loc[fold_id == fold]\n\n    print('fold ' + str(fold) + ': produced')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfold 4: produced\nroot = coco_structure(train_df)\n\nwith open('annotations_train.json', 'w', encoding='utf-8') as f:\n    json.dump(root, f, ensure_ascii=True, indent=4)\n        ","metadata":{},"execution_count":null,"outputs":[]}]}