{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pycocotools\n!pip install --upgrade scikit-learn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-16T06:11:20.641198Z","iopub.execute_input":"2021-11-16T06:11:20.641495Z","iopub.status.idle":"2021-11-16T06:11:53.2039Z","shell.execute_reply.started":"2021-11-16T06:11:20.641415Z","shell.execute_reply":"2021-11-16T06:11:53.203016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport json,itertools\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold,GroupKFold,StratifiedGroupKFold","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:11:53.206197Z","iopub.execute_input":"2021-11-16T06:11:53.206545Z","iopub.status.idle":"2021-11-16T06:11:54.096043Z","shell.execute_reply.started":"2021-11-16T06:11:53.206501Z","shell.execute_reply":"2021-11-16T06:11:54.095273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf  = StratifiedGroupKFold(n_splits = 5 )\ndf = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/train.csv')\ndf_shsy5y = df[df['cell_type'] == 'shsy5y'].copy().reset_index(drop=True)\ndf_astro = df[df['cell_type'] == 'astro'].copy().reset_index(drop=True)\ndf_cort = df[df['cell_type'] == 'cort'].copy().reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:11:54.097497Z","iopub.execute_input":"2021-11-16T06:11:54.097907Z","iopub.status.idle":"2021-11-16T06:11:54.718809Z","shell.execute_reply.started":"2021-11-16T06:11:54.097855Z","shell.execute_reply":"2021-11-16T06:11:54.718022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype=np.uint)\n    array = array.reshape((-1, 2)).T\n    array[0] = array[0] - 1\n\n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype=np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype=np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img)  ## This is important so pycocotools can handle this object\n\n    return msk_img\n\ndef annotate(idx, row, cat_ids):\n    mask = rle2mask(row['annotation'], row['width'], row['height'])  # Binary mask\n    c_rle = maskUtils.encode(mask)  # Encoding it back to rle (coco format)\n    c_rle['counts'] = c_rle['counts'].decode('utf-8')  # converting from binary to utf-8\n    area = maskUtils.area(c_rle).item()  # calculating the area\n    bbox = maskUtils.toBbox(c_rle).astype(int).tolist()  # calculating the bboxes\n    annotation = {\n        'segmentation': c_rle,\n        'bbox': bbox,\n        'area': area,\n        'image_id': row['id'],\n        'category_id': cat_ids[row['cell_type']],\n        'iscrowd': 0,\n        'id': idx\n    }\n    return annotation\n\ndef coco_structure(df, workers=4):\n    ## Building the header\n    cat_ids = {name: id + 1 for id, name in enumerate(df.cell_type.unique())}\n    cats = [{'name': name, 'id': id} for name, id in cat_ids.items()]\n    \n    images = [{'id': id, 'width': row.width, 'height': row.height, 'file_name': f'train/{id}.png'} for id, row in\n              df.groupby('id').agg('first').iterrows()]\n\n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(\n        delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total=len(df)))\n\n    return {'categories': cats, 'images': images, 'annotations': annotations}","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:12:58.524433Z","iopub.execute_input":"2021-11-16T06:12:58.524709Z","iopub.status.idle":"2021-11-16T06:12:58.537295Z","shell.execute_reply.started":"2021-11-16T06:12:58.524681Z","shell.execute_reply":"2021-11-16T06:12:58.536303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''all 3 classes'''\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df,y=np.array(df['cell_type'].to_list()),groups =  np.array(df['id'].to_list()))):\n    train_df = df.loc[train_idx].reset_index(drop=True)\n    val_df = df.loc[val_idx].reset_index(drop=True)\n    train_root = coco_structure(train_df)\n    val_root = coco_structure(val_df)\n    with open(f'annotations_train_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(train_root, f, ensure_ascii=True, indent=4)\n    with open(f'annotations_val_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(val_root, f, ensure_ascii=True, indent=4)\n    # break","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:13:01.547243Z","iopub.execute_input":"2021-11-16T06:13:01.548051Z","iopub.status.idle":"2021-11-16T06:18:36.151811Z","shell.execute_reply.started":"2021-11-16T06:13:01.548011Z","shell.execute_reply":"2021-11-16T06:18:36.150893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''only shsy5y'''\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df_shsy5y,y=np.array(df_shsy5y['cell_type'].to_list()),groups =  np.array(df_shsy5y['id'].to_list()))):\n    train_df = df_shsy5y.loc[train_idx].reset_index(drop=True)\n    val_df = df_shsy5y.loc[val_idx].reset_index(drop=True)\n\n    train_root = coco_structure(train_df)\n    val_root = coco_structure(val_df)\n    with open(f'annotations_shsy5y_train_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(train_root, f, ensure_ascii=True, indent=4)\n    with open(f'annotations_shsy5y_val_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(val_root, f, ensure_ascii=True, indent=4)\n#     break","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:18:36.153724Z","iopub.execute_input":"2021-11-16T06:18:36.154046Z","iopub.status.idle":"2021-11-16T06:22:27.742189Z","shell.execute_reply.started":"2021-11-16T06:18:36.154005Z","shell.execute_reply":"2021-11-16T06:22:27.741127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''only astro'''\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df_astro,y=np.array(df_astro['cell_type'].to_list()),groups =  np.array(df_astro['id'].to_list()))):\n    train_df = df_astro.loc[train_idx].reset_index(drop=True)\n    val_df = df_astro.loc[val_idx].reset_index(drop=True)\n    train_root = coco_structure(train_df)\n    val_root = coco_structure(val_df)\n    with open(f'annotations_astro_train_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(train_root, f, ensure_ascii=True, indent=4)\n    with open(f'annotations_astro_val_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(val_root, f, ensure_ascii=True, indent=4)\n    # break","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:22:27.745534Z","iopub.execute_input":"2021-11-16T06:22:27.745834Z","iopub.status.idle":"2021-11-16T06:23:25.318724Z","shell.execute_reply.started":"2021-11-16T06:22:27.745797Z","shell.execute_reply":"2021-11-16T06:23:25.317823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''only cort'''\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df_cort,y=np.array(df_cort['cell_type'].to_list()),groups =  np.array(df_cort['id'].to_list()))):\n    train_df = df_cort.loc[train_idx].reset_index(drop=True)\n    val_df = df_cort.loc[val_idx].reset_index(drop=True)\n    train_root = coco_structure(train_df)\n    val_root = coco_structure(val_df)\n    with open(f'annotations_cort_train_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(train_root, f, ensure_ascii=True, indent=4)\n    with open(f'annotations_cort_val_fold{fold}.json', 'w', encoding='utf-8') as f:\n        json.dump(val_root, f, ensure_ascii=True, indent=4)\n    # break","metadata":{"execution":{"iopub.status.busy":"2021-11-16T06:23:25.320629Z","iopub.execute_input":"2021-11-16T06:23:25.320893Z","iopub.status.idle":"2021-11-16T06:24:15.031402Z","shell.execute_reply.started":"2021-11-16T06:23:25.320866Z","shell.execute_reply":"2021-11-16T06:24:15.030596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}