{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -Uqqq pycocotools","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-29T12:21:17.058215Z","iopub.execute_input":"2021-11-29T12:21:17.058620Z","iopub.status.idle":"2021-11-29T12:21:35.567028Z","shell.execute_reply.started":"2021-11-29T12:21:17.058514Z","shell.execute_reply":"2021-11-29T12:21:35.566077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image, ImageEnhance\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import StratifiedKFold\n\nimport glob\nimport sys\nimport cv2\nimport imageio\nimport joblib\nimport math\nimport warnings\nimport os\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:21:35.570057Z","iopub.execute_input":"2021-11-29T12:21:35.570434Z","iopub.status.idle":"2021-11-29T12:21:36.947409Z","shell.execute_reply.started":"2021-11-29T12:21:35.570387Z","shell.execute_reply":"2021-11-29T12:21:36.946420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HEIGHT = 520\nWIDTH = 704\n\ntrain = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:21:57.213125Z","iopub.execute_input":"2021-11-29T12:21:57.213445Z","iopub.status.idle":"2021-11-29T12:21:57.576895Z","shell.execute_reply.started":"2021-11-29T12:21:57.213412Z","shell.execute_reply":"2021-11-29T12:21:57.575964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add image file path\ndef get_file_path(image_id):\n    return f'/kaggle/input/sartorius-cell-instance-segmentation/train/{image_id}.png'\n\ntrain['file_path'] = train['id'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:21:58.920115Z","iopub.execute_input":"2021-11-29T12:21:58.921189Z","iopub.status.idle":"2021-11-29T12:21:58.965525Z","shell.execute_reply.started":"2021-11-29T12:21:58.921151Z","shell.execute_reply":"2021-11-29T12:21:58.964804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['shape'] = train[['height', 'width']].apply(tuple, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:03.588677Z","iopub.execute_input":"2021-11-29T12:22:03.589222Z","iopub.status.idle":"2021-11-29T12:22:04.675170Z","shell.execute_reply.started":"2021-11-29T12:22:03.589187Z","shell.execute_reply":"2021-11-29T12:22:04.673804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head())","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:05.806828Z","iopub.execute_input":"2021-11-29T12:22:05.807178Z","iopub.status.idle":"2021-11-29T12:22:05.831943Z","shell.execute_reply.started":"2021-11-29T12:22:05.807140Z","shell.execute_reply":"2021-11-29T12:22:05.831274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\ntrain['cell_type'].value_counts().plot(kind='pie', autopct='%1.1f%%', title='Cell Type Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:14.297232Z","iopub.execute_input":"2021-11-29T12:22:14.297800Z","iopub.status.idle":"2021-11-29T12:22:14.484828Z","shell.execute_reply.started":"2021-11-29T12:22:14.297764Z","shell.execute_reply":"2021-11-29T12:22:14.483974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kfold\n\nBelow codes are copied from [this discussion](https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/285546) by [Gunes Evitan\n](https://www.kaggle.com/gunesevitan)","metadata":{}},{"cell_type":"code","source":"df_images = train.groupby('id').first().reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:18.788039Z","iopub.execute_input":"2021-11-29T12:22:18.789012Z","iopub.status.idle":"2021-11-29T12:22:18.911495Z","shell.execute_reply.started":"2021-11-29T12:22:18.788959Z","shell.execute_reply":"2021-11-29T12:22:18.910322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_images","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:32.479441Z","iopub.execute_input":"2021-11-29T12:22:32.480382Z","iopub.status.idle":"2021-11-29T12:22:32.506854Z","shell.execute_reply.started":"2021-11-29T12:22:32.480334Z","shell.execute_reply":"2021-11-29T12:22:32.505711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nfor fold, (_, val_idx) in enumerate(skf.split(X=df_images, y=df_images['cell_type']), 1):\n    df_images.loc[val_idx, 'fold'] = fold\ndf_images['fold'] = df_images['fold'].astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:22:54.993109Z","iopub.execute_input":"2021-11-29T12:22:54.994231Z","iopub.status.idle":"2021-11-29T12:22:55.005897Z","shell.execute_reply.started":"2021-11-29T12:22:54.994171Z","shell.execute_reply":"2021-11-29T12:22:55.005005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = './fold'\n!mkdir $DATA_PATH","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:23:43.976603Z","iopub.execute_input":"2021-11-29T12:23:43.976973Z","iopub.status.idle":"2021-11-29T12:23:44.765793Z","shell.execute_reply.started":"2021-11-29T12:23:43.976937Z","shell.execute_reply":"2021-11-29T12:23:44.764469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_images[['id', 'fold']].to_csv(f'{DATA_PATH}/train_folds.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:27:27.014330Z","iopub.execute_input":"2021-11-29T12:27:27.014654Z","iopub.status.idle":"2021-11-29T12:27:27.024907Z","shell.execute_reply.started":"2021-11-29T12:27:27.014620Z","shell.execute_reply":"2021-11-29T12:27:27.023853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf_train_folds = pd.read_csv(f'{DATA_PATH}/train_folds.csv')\ndf_train = df_train.merge(df_train_folds, how='left', on='id')","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:28:24.438205Z","iopub.execute_input":"2021-11-29T12:28:24.438770Z","iopub.status.idle":"2021-11-29T12:28:24.836572Z","shell.execute_reply.started":"2021-11-29T12:28:24.438722Z","shell.execute_reply":"2021-11-29T12:28:24.835835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0, 5):\n    print(len(df_images[df_images.fold == i+1]))","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:29:20.389072Z","iopub.execute_input":"2021-11-29T12:29:20.389876Z","iopub.status.idle":"2021-11-29T12:29:20.400433Z","shell.execute_reply.started":"2021-11-29T12:29:20.389839Z","shell.execute_reply":"2021-11-29T12:29:20.399759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folds = []\nval_folds = []\nfor i in range(1, 6):\n    train_folds.append(df_train[df_train['fold'] != i])\n    val_folds.append(df_train[df_train['fold'] == i])","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:25:47.573445Z","iopub.execute_input":"2021-11-29T12:25:47.573853Z","iopub.status.idle":"2021-11-29T12:25:47.642168Z","shell.execute_reply.started":"2021-11-29T12:25:47.573816Z","shell.execute_reply":"2021-11-29T12:25:47.641066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_folds)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:33:09.366645Z","iopub.execute_input":"2021-11-29T12:33:09.367032Z","iopub.status.idle":"2021-11-29T12:33:09.374131Z","shell.execute_reply.started":"2021-11-29T12:33:09.366996Z","shell.execute_reply":"2021-11-29T12:33:09.373167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Coco Generator\n\nBelow code are copied from https://www.kaggle.com/coldfir3/efficient-coco-dataset-generator by [Adriano Passos](https://www.kaggle.com/coldfir3)","metadata":{}},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    \n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype = np.uint)\n    array = array.reshape((-1,2)).T\n    array[0] = array[0] - 1\n    \n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype = np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype = np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img) ## This is important so pycocotools can handle this object\n    \n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:29:46.989619Z","iopub.execute_input":"2021-11-29T12:29:46.990641Z","iopub.status.idle":"2021-11-29T12:29:46.999190Z","shell.execute_reply.started":"2021-11-29T12:29:46.990592Z","shell.execute_reply":"2021-11-29T12:29:46.998182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\n\ndef annotate(idx, row, cat_ids):\n        mask = rle2mask(row['annotation'], row['width'], row['height']) # Binary mask\n        c_rle = maskUtils.encode(mask) # Encoding it back to rle (coco format)\n        c_rle['counts'] = c_rle['counts'].decode('utf-8') # converting from binary to utf-8\n        area = maskUtils.area(c_rle).item() # calculating the area\n        bbox = maskUtils.toBbox(c_rle).astype(int).tolist() # calculating the bboxes\n        annotation = {\n            'segmentation': c_rle,\n            'bbox': bbox,\n            'area': area,\n            'image_id':row['id'], \n            'category_id':cat_ids[row['cell_type']], \n            'iscrowd':0, \n            'id':idx\n        }\n        return annotation\n    \ndef coco_structure(df, workers = 4):\n    \n    ## Building the header\n    cat_ids = {name:id+1 for id, name in enumerate(df.cell_type.unique())}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'train/{id}.png'} for id,row in df.groupby('id').agg('first').iterrows()]\n    \n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total = len(df)))\n        \n    return {'categories':cats, 'images':images, 'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:29:49.129838Z","iopub.execute_input":"2021-11-29T12:29:49.130161Z","iopub.status.idle":"2021-11-29T12:29:49.148446Z","shell.execute_reply.started":"2021-11-29T12:29:49.130130Z","shell.execute_reply":"2021-11-29T12:29:49.147724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## To COCO json","metadata":{}},{"cell_type":"code","source":"import json,itertools\n\ntrain_fold_json = [coco_structure(fold) for fold in train_folds]\nval_fold_json = [coco_structure(fold) for fold in val_folds]","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:33:19.869476Z","iopub.execute_input":"2021-11-29T12:33:19.870354Z","iopub.status.idle":"2021-11-29T12:39:38.197116Z","shell.execute_reply.started":"2021-11-29T12:33:19.870306Z","shell.execute_reply":"2021-11-29T12:39:38.195985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco_json = './fold_json'\n!mkdir $coco_json","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:47:09.712664Z","iopub.execute_input":"2021-11-29T12:47:09.713633Z","iopub.status.idle":"2021-11-29T12:47:10.490458Z","shell.execute_reply.started":"2021-11-29T12:47:09.713575Z","shell.execute_reply":"2021-11-29T12:47:10.489559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, (train_fold, val_fold) in enumerate(zip(train_fold_json, val_fold_json)):\n    with open(f'{coco_json}/fold_{idx+1}_train.json', 'w+', encoding='utf-8') as f:\n        json.dump(train_fold, f, ensure_ascii=True, indent=4)\n    with open(f'{coco_json}/fold_{idx+1}_val.json', 'w+', encoding='utf-8') as f:\n        json.dump(val_fold, f, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:47:11.425770Z","iopub.execute_input":"2021-11-29T12:47:11.426826Z","iopub.status.idle":"2021-11-29T12:47:35.246115Z","shell.execute_reply.started":"2021-11-29T12:47:11.426762Z","shell.execute_reply":"2021-11-29T12:47:35.245127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools.coco import COCO\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:47:35.256371Z","iopub.execute_input":"2021-11-29T12:47:35.256689Z","iopub.status.idle":"2021-11-29T12:47:35.270231Z","shell.execute_reply.started":"2021-11-29T12:47:35.256651Z","shell.execute_reply":"2021-11-29T12:47:35.269198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataDir=Path('../input/sartorius-cell-instance-segmentation')\nannFile = Path(f'{coco_json}/fold_1_train.json')\ncoco = COCO(annFile)\nimgIds = coco.getImgIds()","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:47:35.731455Z","iopub.execute_input":"2021-11-29T12:47:35.732353Z","iopub.status.idle":"2021-11-29T12:47:37.823821Z","shell.execute_reply.started":"2021-11-29T12:47:35.732316Z","shell.execute_reply":"2021-11-29T12:47:37.822821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ./fold_json","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:50:26.634795Z","iopub.execute_input":"2021-11-29T12:50:26.635769Z","iopub.status.idle":"2021-11-29T12:50:27.410563Z","shell.execute_reply.started":"2021-11-29T12:50:26.635715Z","shell.execute_reply":"2021-11-29T12:50:27.409503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = coco.loadImgs(imgIds[-3:])\n_,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\nfor img, ax in zip(imgs, axs):\n    I = Image.open(dataDir/img['file_name'])\n    annIds = coco.getAnnIds(imgIds=[img['id']])\n    anns = coco.loadAnns(annIds)\n    ax[0].imshow(I)\n    ax[1].imshow(I)\n    plt.sca(ax[1])\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-29T12:50:36.528171Z","iopub.execute_input":"2021-11-29T12:50:36.528467Z","iopub.status.idle":"2021-11-29T12:53:04.048420Z","shell.execute_reply.started":"2021-11-29T12:50:36.528436Z","shell.execute_reply":"2021-11-29T12:53:04.046693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}