{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **[HuBMAP 2023] K-fold CV COCO Dataset Generator**","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools -Uqq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-19T02:55:17.157563Z","iopub.execute_input":"2023-07-19T02:55:17.158014Z","iopub.status.idle":"2023-07-19T02:56:04.091772Z","shell.execute_reply.started":"2023-07-19T02:55:17.157979Z","shell.execute_reply":"2023-07-19T02:56:04.090016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### importing libraries","metadata":{}},{"cell_type":"code","source":"import json, cv2, numpy as np, itertools, random, pandas as pd\nfrom pycocotools.coco import COCO\nfrom pycocotools import mask as maskUtils\nfrom skimage import io\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn import model_selection\n\nimport matplotlib.pyplot as plt\nfrom skimage import io\nfrom pycocotools.coco import COCO\nimport matplotlib.patches as mpatches\n","metadata":{"execution":{"iopub.status.busy":"2023-07-19T02:56:04.094592Z","iopub.execute_input":"2023-07-19T02:56:04.095011Z","iopub.status.idle":"2023-07-19T02:56:05.747934Z","shell.execute_reply.started":"2023-07-19T02:56:04.094972Z","shell.execute_reply":"2023-07-19T02:56:05.746560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Functions","metadata":{}},{"cell_type":"code","source":"def coordinates_to_masks(coordinates, shape):\n    masks = []\n    for coord in coordinates:\n        mask = np.zeros(shape, dtype=np.uint8)\n        cv2.fillPoly(mask, [np.array(coord)], 1)\n        masks.append(mask)\n    return masks\n\ndef binary_mask_to_rle(binary_mask):\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(itertools.groupby(binary_mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle\n\ndef rle_to_binary_mask(mask_rle, shape=(512, 512)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) \n                       for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction","metadata":{"execution":{"iopub.status.busy":"2023-07-19T02:56:05.750220Z","iopub.execute_input":"2023-07-19T02:56:05.751311Z","iopub.status.idle":"2023-07-19T02:56:05.766010Z","shell.execute_reply.started":"2023-07-19T02:56:05.751246Z","shell.execute_reply":"2023-07-19T02:56:05.764458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading Dataset","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv')\ndf = df.query('dataset != 3')\n#df=df.head(500)\ndf.reset_index(inplace=True,drop=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T02:56:05.769611Z","iopub.execute_input":"2023-07-19T02:56:05.770035Z","iopub.status.idle":"2023-07-19T02:56:05.860283Z","shell.execute_reply.started":"2023-07-19T02:56:05.769990Z","shell.execute_reply":"2023-07-19T02:56:05.859050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spliting training & Valid","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\nn_splits=5\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\nfor fold, (_, val_idx) in enumerate(skf.split(X=df, y=df['source_wsi']), 1):\n    df.loc[val_idx, 'fold'] = fold\n    \ndf['fold'] = df['fold'].astype(np.uint8)\ndf.groupby('fold').size()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T03:10:35.460607Z","iopub.execute_input":"2023-07-19T03:10:35.461174Z","iopub.status.idle":"2023-07-19T03:10:35.490301Z","shell.execute_reply.started":"2023-07-19T03:10:35.461133Z","shell.execute_reply":"2023-07-19T03:10:35.489072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coco_structure(images_ids):\n    idx=1\n    annotations=[]\n    images=[]\n    for item in tqdm(data,total=int(len(images_ids))):\n        image_id=item[\"id\"]\n        if image_id in images_ids:\n            image = {\"id\": image_id, \"file_name\": image_id + \".tif\", \"height\": 512, \"width\": 512}\n            images.append(image)\n        else:continue\n        #-----------------------------\n        anns=item[\"annotations\"]\n        for an in anns:\n            category_type=an[\"type\"]\n            category_id=categories_ids[category_type]\n            segmentation=an[\"coordinates\"]\n            mask_img = coordinates_to_masks(segmentation, (512, 512))[0]\n            ys, xs = np.where(mask_img)\n            x1, x2 = min(xs), max(xs)\n            y1, y2 = min(ys), max(ys)\n\n            rle = binary_mask_to_rle(mask_img)\n\n            seg = {\n                \"id\": idx,\n                \"image_id\": image_id,\n                \"category_id\": category_id,\n                \"segmentation\": rle,\n                \"bbox\": [int(x1), int(y1), int(x2 - x1 + 1), int(y2 - y1 + 1)],\n                \"area\": int(np.sum(mask_img)),\n                \"iscrowd\": 0,\n            }\n            if image_id in images_ids:\n                annotations.append(seg)\n                idx=idx+1\n                \n    return {\"info\": {}, \"licenses\": [], \"categories\": categories, \"images\": images, \"annotations\": annotations}","metadata":{"execution":{"iopub.status.busy":"2023-07-19T03:10:37.275634Z","iopub.execute_input":"2023-07-19T03:10:37.276131Z","iopub.status.idle":"2023-07-19T03:10:37.290528Z","shell.execute_reply.started":"2023-07-19T03:10:37.276095Z","shell.execute_reply":"2023-07-19T03:10:37.289289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold in range(n_splits):\n    selected_fold=fold + 1\n    train_ids = df.query(f'fold != {selected_fold}')['id'].values.tolist()\n    valid_ids = df.query(f'fold == {selected_fold}')['id'].values.tolist()\n    print(len(train_ids), len(valid_ids))\n    \n    jsonl_file_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\n    data = []\n    with open(jsonl_file_path, \"r\") as file:\n        for line in file:\n            data.append(json.loads(line))\n    categories_list=[]\n#------------------------------------------------------------------------------\n    for i,item in enumerate(tqdm(data)):  \n        ann=item[\"annotations\"]\n        for an in ann:\n            category_type=an[\"type\"]\n            if category_type not in categories_list:\n                categories_list.append(category_type)\n    #------------------------------------------------------------------------------\n    categories_ids = {name:id+1 for id, name in enumerate(categories_list)}  \n    ids_categories = {id+1:name for id, name in enumerate(categories_list)}  \n    categories =[{'id':id,'name':name} for name,id in categories_ids.items()]\n\n    print(categories_ids)\n    print(ids_categories)\n    print(categories)\n    train_coco_data = coco_structure(train_ids)\n    valid_coco_data = coco_structure(valid_ids)\n    output_file_path = f\"coco_annotations_train_all_fold{selected_fold}.json\"\n    with open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n        json.dump(train_coco_data, output_file, ensure_ascii=True, indent=4)\n\n    output_file_path = f\"coco_annotations_valid_all_fold{selected_fold}.json\"\n    with open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n        json.dump(valid_coco_data, output_file, ensure_ascii=True, indent=4)\n    dataDir = Path(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\")\n    annFile = Path(f\"./coco_annotations_valid_all_fold{selected_fold}.json\")\n\n    colors = ['Set1', 'Set3_r', 'Set3'] \n    legend = ids_categories #{1: 'glomerulus', 2: 'blood_vessel', 3: 'unsure'}\n\n    coco = COCO(annFile)\n    imgIds = coco.getImgIds()\n    imgs = coco.loadImgs(imgIds[0:4])\n\n    fig, axs = plt.subplots(len(imgs), 2, figsize=(10, 5*len(imgs)))\n    for img, ax_row in zip(imgs, axs):\n        ax = ax_row[0]  # Access the first axis in each row\n        I = io.imread(dataDir / img[\"file_name\"])\n        annIds = coco.getAnnIds(imgIds=[img[\"id\"]])\n        anns = coco.loadAnns(annIds)\n        ax.imshow(I)\n        ax = ax_row[1]  # Access the second axis in each row\n        ax.imshow(I)\n        plt.sca(ax)\n        for i, ann in enumerate(anns):\n            category_id = ann['category_id']\n            color = colors[category_id-1]\n            #-----------------------------------------\n            mask = coco.annToMask(ann)\n            mask = np.ma.masked_where(mask == 0, mask)\n            ax.imshow(mask, cmap=color, alpha=1)\n            #-----------------------------------------\n            handles = []\n            for category_id in legend:\n                color = colors[category_id - 1]\n                handles.append(mpatches.Patch(color=plt.colormaps.get_cmap(color)(0)))\n            ax.legend(handles, legend.values(), bbox_to_anchor=(1.05, 1), loc='upper left')\n\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-19T03:11:12.881107Z","iopub.execute_input":"2023-07-19T03:11:12.881623Z","iopub.status.idle":"2023-07-19T04:23:42.163816Z","shell.execute_reply.started":"2023-07-19T03:11:12.881588Z","shell.execute_reply":"2023-07-19T04:23:42.162378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading polygons.jsonl","metadata":{}},{"cell_type":"markdown","source":"### Categories","metadata":{}},{"cell_type":"markdown","source":"### Creating COCO","metadata":{}},{"cell_type":"markdown","source":"### Saving COCO","metadata":{}},{"cell_type":"markdown","source":"### Visualization","metadata":{}}]}