{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -Uqqq pycocotools","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-28T08:13:20.876838Z","iopub.execute_input":"2021-11-28T08:13:20.877161Z","iopub.status.idle":"2021-11-28T08:13:29.957902Z","shell.execute_reply.started":"2021-11-28T08:13:20.877067Z","shell.execute_reply":"2021-11-28T08:13:29.956753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image, ImageEnhance\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import StratifiedKFold\n\nimport glob\nimport sys\nimport cv2\nimport imageio\nimport joblib\nimport math\nimport warnings\nimport os\n\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:29.959828Z","iopub.execute_input":"2021-11-28T08:13:29.960108Z","iopub.status.idle":"2021-11-28T08:13:30.384096Z","shell.execute_reply.started":"2021-11-28T08:13:29.960076Z","shell.execute_reply":"2021-11-28T08:13:30.383194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HEIGHT = 520\nWIDTH = 704\n\ntrain = pd.read_csv('/kaggle/input/sartorius-cell-instance-segmentation/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:30.385564Z","iopub.execute_input":"2021-11-28T08:13:30.385892Z","iopub.status.idle":"2021-11-28T08:13:30.736802Z","shell.execute_reply.started":"2021-11-28T08:13:30.385850Z","shell.execute_reply":"2021-11-28T08:13:30.736115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add image file path\ndef get_file_path(image_id):\n    return f'/kaggle/input/sartorius-cell-instance-segmentation/train/{image_id}.png'\n\ntrain['file_path'] = train['id'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:30.739488Z","iopub.execute_input":"2021-11-28T08:13:30.740030Z","iopub.status.idle":"2021-11-28T08:13:30.778423Z","shell.execute_reply.started":"2021-11-28T08:13:30.739982Z","shell.execute_reply":"2021-11-28T08:13:30.777334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['shape'] = train[['height', 'width']].apply(tuple, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:30.779975Z","iopub.execute_input":"2021-11-28T08:13:30.780531Z","iopub.status.idle":"2021-11-28T08:13:31.884807Z","shell.execute_reply.started":"2021-11-28T08:13:30.780478Z","shell.execute_reply":"2021-11-28T08:13:31.883819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head())","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:31.886380Z","iopub.execute_input":"2021-11-28T08:13:31.886641Z","iopub.status.idle":"2021-11-28T08:13:31.908641Z","shell.execute_reply.started":"2021-11-28T08:13:31.886608Z","shell.execute_reply":"2021-11-28T08:13:31.907258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\ntrain['cell_type'].value_counts().plot(kind='pie', autopct='%1.1f%%', title='Cell Type Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:31.910835Z","iopub.execute_input":"2021-11-28T08:13:31.911701Z","iopub.status.idle":"2021-11-28T08:13:32.055138Z","shell.execute_reply.started":"2021-11-28T08:13:31.911649Z","shell.execute_reply":"2021-11-28T08:13:32.054568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kfold\n\nBelow codes are copied from [this discussion](https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/285546) by [Gunes Evitan\n](https://www.kaggle.com/gunesevitan)","metadata":{}},{"cell_type":"code","source":"df_images = train.groupby('id').first().reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:39.488286Z","iopub.execute_input":"2021-11-28T08:13:39.489393Z","iopub.status.idle":"2021-11-28T08:13:39.578243Z","shell.execute_reply.started":"2021-11-28T08:13:39.489349Z","shell.execute_reply":"2021-11-28T08:13:39.577413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_images.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:39.581105Z","iopub.execute_input":"2021-11-28T08:13:39.581365Z","iopub.status.idle":"2021-11-28T08:13:39.600067Z","shell.execute_reply.started":"2021-11-28T08:13:39.581336Z","shell.execute_reply":"2021-11-28T08:13:39.599133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nfor fold, (_, val_idx) in enumerate(skf.split(X=df_images, y=df_images['cell_type']), 1):\n    df_images.loc[val_idx, 'fold'] = fold\ndf_images['fold'] = df_images['fold'].astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:39.601403Z","iopub.execute_input":"2021-11-28T08:13:39.602021Z","iopub.status.idle":"2021-11-28T08:13:39.619689Z","shell.execute_reply.started":"2021-11-28T08:13:39.601987Z","shell.execute_reply":"2021-11-28T08:13:39.618679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = '/kaggle/fold'\n!mkdir $DATA_PATH","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:39.620842Z","iopub.execute_input":"2021-11-28T08:13:39.621193Z","iopub.status.idle":"2021-11-28T08:13:40.429787Z","shell.execute_reply.started":"2021-11-28T08:13:39.621163Z","shell.execute_reply":"2021-11-28T08:13:40.428844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_images[['id', 'fold']].to_csv(f'{DATA_PATH}/train_folds.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.431113Z","iopub.execute_input":"2021-11-28T08:13:40.431387Z","iopub.status.idle":"2021-11-28T08:13:40.450381Z","shell.execute_reply.started":"2021-11-28T08:13:40.431355Z","shell.execute_reply":"2021-11-28T08:13:40.449550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf_train_folds = pd.read_csv(f'{DATA_PATH}/train_folds.csv')\ndf_train = df_train.merge(df_train_folds, how='left', on='id')","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.453523Z","iopub.execute_input":"2021-11-28T08:13:40.454936Z","iopub.status.idle":"2021-11-28T08:13:40.859684Z","shell.execute_reply.started":"2021-11-28T08:13:40.454882Z","shell.execute_reply":"2021-11-28T08:13:40.858494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_folds = []\nval_folds = []\nfor i in range(1, 6):\n    train_folds.append(df_train[df_train['fold'] != i])\n    val_folds.append(df_train[df_train['fold'] == i])","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.864007Z","iopub.execute_input":"2021-11-28T08:13:40.864285Z","iopub.status.idle":"2021-11-28T08:13:40.917948Z","shell.execute_reply.started":"2021-11-28T08:13:40.864253Z","shell.execute_reply":"2021-11-28T08:13:40.917025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Coco Generator\n\nBelow code are copied from https://www.kaggle.com/coldfir3/efficient-coco-dataset-generator by [Adriano Passos](https://www.kaggle.com/coldfir3)","metadata":{}},{"cell_type":"code","source":"## Based on: https://www.kaggle.com/eigrad/convert-rle-to-bounding-box-x0-y0-x1-y1\ndef rle2mask(rle, img_w, img_h):\n    \n    ## transforming the string into an array of shape (2, N)\n    array = np.fromiter(rle.split(), dtype = np.uint)\n    array = array.reshape((-1,2)).T\n    array[0] = array[0] - 1\n    \n    ## decompressing the rle encoding (ie, turning [3, 1, 10, 2] into [3, 4, 10, 11, 12])\n    # for faster mask construction\n    starts, lenghts = array\n    mask_decompressed = np.concatenate([np.arange(s, s + l, dtype = np.uint) for s, l in zip(starts, lenghts)])\n\n    ## Building the binary mask\n    msk_img = np.zeros(img_w * img_h, dtype = np.uint8)\n    msk_img[mask_decompressed] = 1\n    msk_img = msk_img.reshape((img_h, img_w))\n    msk_img = np.asfortranarray(msk_img) ## This is important so pycocotools can handle this object\n    \n    return msk_img","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.919493Z","iopub.execute_input":"2021-11-28T08:13:40.919818Z","iopub.status.idle":"2021-11-28T08:13:40.928767Z","shell.execute_reply.started":"2021-11-28T08:13:40.919776Z","shell.execute_reply":"2021-11-28T08:13:40.927898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nfrom pycocotools import mask as maskUtils\nfrom joblib import Parallel, delayed\n\ndef annotate(idx, row, cat_ids):\n        mask = rle2mask(row['annotation'], row['width'], row['height']) # Binary mask\n        c_rle = maskUtils.encode(mask) # Encoding it back to rle (coco format)\n        c_rle['counts'] = c_rle['counts'].decode('utf-8') # converting from binary to utf-8\n        area = maskUtils.area(c_rle).item() # calculating the area\n        bbox = maskUtils.toBbox(c_rle).astype(int).tolist() # calculating the bboxes\n        annotation = {\n            'segmentation': c_rle,\n            'bbox': bbox,\n            'area': area,\n            'image_id':row['id'], \n            'category_id':cat_ids[row['cell_type']], \n            'iscrowd':0, \n            'id':idx\n        }\n        return annotation\n    \ndef coco_structure(df, workers = 4):\n    \n    ## Building the header\n    cat_ids = {name:id+1 for id, name in enumerate(df.cell_type.unique())}    \n    cats =[{'name':name, 'id':id} for name,id in cat_ids.items()]\n    images = [{'id':id, 'width':row.width, 'height':row.height, 'file_name':f'train/{id}.png'} for id,row in df.groupby('id').agg('first').iterrows()]\n    \n    ## Building the annotations\n    annotations = Parallel(n_jobs=workers)(delayed(annotate)(idx, row, cat_ids) for idx, row in tqdm(df.iterrows(), total = len(df)))\n        \n    return {'categories':cats, 'images':images, 'annotations':annotations}","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.930341Z","iopub.execute_input":"2021-11-28T08:13:40.930854Z","iopub.status.idle":"2021-11-28T08:13:40.947034Z","shell.execute_reply.started":"2021-11-28T08:13:40.930809Z","shell.execute_reply":"2021-11-28T08:13:40.946327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## To COCO json","metadata":{}},{"cell_type":"code","source":"import json,itertools\n\ntrain_fold_json = [coco_structure(fold) for fold in train_folds]\nval_fold_json = [coco_structure(fold) for fold in val_folds]","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:13:40.948503Z","iopub.execute_input":"2021-11-28T08:13:40.949004Z","iopub.status.idle":"2021-11-28T08:19:40.084588Z","shell.execute_reply.started":"2021-11-28T08:13:40.948959Z","shell.execute_reply":"2021-11-28T08:19:40.083811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coco_json = '/kaggle/working/fold_json'\n!mkdir $coco_json","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:19:40.085572Z","iopub.execute_input":"2021-11-28T08:19:40.085808Z","iopub.status.idle":"2021-11-28T08:19:40.855405Z","shell.execute_reply.started":"2021-11-28T08:19:40.085777Z","shell.execute_reply":"2021-11-28T08:19:40.854271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, (train_fold, val_fold) in enumerate(zip(train_fold_json, val_fold_json)):\n    with open(f'{coco_json}/fold_{idx+1}_train.json', 'w+', encoding='utf-8') as f:\n        json.dump(train_fold, f, ensure_ascii=True, indent=4)\n    with open(f'{coco_json}/fold_{idx+1}_val.json', 'w+', encoding='utf-8') as f:\n        json.dump(val_fold, f, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:19:40.857486Z","iopub.execute_input":"2021-11-28T08:19:40.857979Z","iopub.status.idle":"2021-11-28T08:20:05.107666Z","shell.execute_reply.started":"2021-11-28T08:19:40.857828Z","shell.execute_reply":"2021-11-28T08:20:05.106699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pycocotools.coco import COCO\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:21:04.938843Z","iopub.execute_input":"2021-11-28T08:21:04.939224Z","iopub.status.idle":"2021-11-28T08:21:04.946279Z","shell.execute_reply.started":"2021-11-28T08:21:04.939185Z","shell.execute_reply":"2021-11-28T08:21:04.945547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataDir=Path('../input/sartorius-cell-instance-segmentation')\nannFile = Path(f'{coco_json}/fold_1_train.json')\ncoco = COCO(annFile)\nimgIds = coco.getImgIds()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:21:12.651258Z","iopub.execute_input":"2021-11-28T08:21:12.651621Z","iopub.status.idle":"2021-11-28T08:21:14.063376Z","shell.execute_reply.started":"2021-11-28T08:21:12.651582Z","shell.execute_reply":"2021-11-28T08:21:14.062510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = coco.loadImgs(imgIds[-3:])\n_,axs = plt.subplots(len(imgs),2,figsize=(40,15 * len(imgs)))\nfor img, ax in zip(imgs, axs):\n    I = Image.open(dataDir/img['file_name'])\n    annIds = coco.getAnnIds(imgIds=[img['id']])\n    anns = coco.loadAnns(annIds)\n    ax[0].imshow(I)\n    ax[1].imshow(I)\n    plt.sca(ax[1])\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T08:24:42.215929Z","iopub.execute_input":"2021-11-28T08:24:42.216261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}