{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pycocotools -Uqq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-31T08:40:37.234434Z","iopub.execute_input":"2023-05-31T08:40:37.234789Z","iopub.status.idle":"2023-05-31T08:41:10.261463Z","shell.execute_reply.started":"2023-05-31T08:40:37.234765Z","shell.execute_reply":"2023-05-31T08:41:10.259846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json, cv2, numpy as np, itertools, random, pandas as pd\nfrom pycocotools.coco import COCO\nfrom pycocotools import mask as maskUtils\nfrom skimage import io\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn import model_selection\nfrom copy import deepcopy","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:30:26.159653Z","iopub.execute_input":"2023-05-31T10:30:26.160396Z","iopub.status.idle":"2023-05-31T10:30:26.174239Z","shell.execute_reply.started":"2023-05-31T10:30:26.160333Z","shell.execute_reply":"2023-05-31T10:30:26.173059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:28:51.437779Z","iopub.execute_input":"2023-05-31T10:28:51.438254Z","iopub.status.idle":"2023-05-31T10:28:51.464442Z","shell.execute_reply.started":"2023-05-31T10:28:51.438219Z","shell.execute_reply":"2023-05-31T10:28:51.462926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids = df.query('dataset == 2')['id'].values.tolist()\nvalid_ids = df.query('dataset == 1')['id'].values.tolist()\nprint(len(train_ids), len(valid_ids))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:30:08.217861Z","iopub.execute_input":"2023-05-31T10:30:08.218308Z","iopub.status.idle":"2023-05-31T10:30:08.234374Z","shell.execute_reply.started":"2023-05-31T10:30:08.218274Z","shell.execute_reply":"2023-05-31T10:30:08.231596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coordinates_to_masks(coordinates, shape):\n    masks = []\n    for coord in coordinates:\n        mask = np.zeros(shape, dtype=np.uint8)\n        cv2.fillPoly(mask, [np.array(coord)], 1)\n        masks.append(mask)\n    return masks\n\ndef binary_mask_to_rle(binary_mask):\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(itertools.groupby(binary_mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jsonl_file_path = \"/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl\"\ndata = []\nwith open(jsonl_file_path, \"r\") as file:\n    for line in file:\n        data.append(json.loads(line))\n\ncoco_data = {\"info\": {}, \"licenses\": [], \"categories\": [], \"images\": [], \"annotations\": []}\n\ncategories = []\nfor item in tqdm(data, dynamic_ncols=True):\n    annotations = item[\"annotations\"]\n    for annotation in annotations:\n        annotation_type = annotation[\"type\"]\n        if annotation_type not in categories:\n            categories.append(annotation_type)\n            coco_data[\"categories\"].append({\"id\": len(categories), \"name\": annotation_type})\n            \ntrain_coco_data = deepcopy(coco_data)\nvalid_coco_data = deepcopy(coco_data)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:31:00.399473Z","iopub.execute_input":"2023-05-31T10:31:00.400432Z","iopub.status.idle":"2023-05-31T10:31:10.033534Z","shell.execute_reply.started":"2023-05-31T10:31:00.400318Z","shell.execute_reply":"2023-05-31T10:31:10.031817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for item in tqdm(data, dynamic_ncols=True):\n    image_id = item[\"id\"]\n\n    if image_id in train_ids:\n        ds = train_coco_data\n    elif image_id in valid_ids:\n        ds = valid_coco_data\n    else:\n        raise NotImplementedError()\n\n    image_info = {\"id\": image_id, \"file_name\": item[\"id\"] + \".tif\", \"height\": 512, \"width\": 512}\n    ds[\"images\"].append(image_info)\n\n    for annotation in item[\"annotations\"]:\n        category_id = categories.index(annotation[\"type\"]) + 1\n\n        segmentation = annotation[\"coordinates\"]\n        mask_img = coordinates_to_masks(segmentation, (512, 512))[0]\n\n        ys, xs = np.where(mask_img)\n        x1, x2 = min(xs), max(xs)\n        y1, y2 = min(ys), max(ys)\n\n        rle = binary_mask_to_rle(mask_img)\n\n        annotation_info = {\n            \"id\": len(ds[\"annotations\"]) + 1,\n            \"image_id\": image_id,\n            \"category_id\": category_id,\n            \"segmentation\": rle,\n            \"bbox\": [int(x1), int(y1), int(x2 - x1 + 1), int(y2 - y1 + 1)],\n            \"area\": int(np.sum(mask_img)),\n            \"iscrowd\": 0,\n        }\n        ds[\"annotations\"].append(annotation_info)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:40:53.681683Z","iopub.execute_input":"2023-05-31T10:40:53.682145Z","iopub.status.idle":"2023-05-31T10:46:27.722003Z","shell.execute_reply.started":"2023-05-31T10:40:53.682113Z","shell.execute_reply":"2023-05-31T10:46:27.720744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_file_path = \"coco_annotations_train_all.json\"\nwith open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n    json.dump(train_coco_data, output_file, ensure_ascii=True, indent=4)\n    \noutput_file_path = \"coco_annotations_valid_all.json\"\nwith open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n    json.dump(valid_coco_data, output_file, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:48:26.369567Z","iopub.execute_input":"2023-05-31T10:48:26.370086Z","iopub.status.idle":"2023-05-31T10:48:30.016804Z","shell.execute_reply.started":"2023-05-31T10:48:26.370025Z","shell.execute_reply":"2023-05-31T10:48:30.014658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataDir = Path(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\")\nannFile = Path(\"./coco_annotations_train_all.json\")\ncoco = COCO(annFile)\n\nimgIds = coco.getImgIds()\nprint(len(imgIds))\nimgs = coco.loadImgs(random.sample(imgIds, 3))\n\nimgs = coco.loadImgs(imgIds[-3:])\nfig, axs = plt.subplots(len(imgs), 2, figsize=(10, 5*len(imgs)))\n\nfor img, ax_row in zip(imgs, axs):\n    ax = ax_row[0]  # Access the first axis in each row\n    I = io.imread(dataDir / img[\"file_name\"])\n    annIds = coco.getAnnIds(imgIds=[img[\"id\"]])\n    anns = coco.loadAnns(annIds)\n    ax.imshow(I)\n\n    ax = ax_row[1]  # Access the second axis in each row\n    ax.imshow(I)\n    plt.sca(ax)\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:48:30.019437Z","iopub.execute_input":"2023-05-31T10:48:30.019844Z","iopub.status.idle":"2023-05-31T10:48:37.292191Z","shell.execute_reply.started":"2023-05-31T10:48:30.019808Z","shell.execute_reply":"2023-05-31T10:48:37.291073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataDir = Path(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train\")\nannFile = Path(\"./coco_annotations_valid_all.json\")\ncoco = COCO(annFile)\n\nimgIds = coco.getImgIds()\nprint(len(imgIds))\nimgs = coco.loadImgs(random.sample(imgIds, 3))\n\nimgs = coco.loadImgs(imgIds[-3:])\nfig, axs = plt.subplots(len(imgs), 2, figsize=(10, 5*len(imgs)))\n\nfor img, ax_row in zip(imgs, axs):\n    ax = ax_row[0]  # Access the first axis in each row\n    I = io.imread(dataDir / img[\"file_name\"])\n    annIds = coco.getAnnIds(imgIds=[img[\"id\"]])\n    anns = coco.loadAnns(annIds)\n    ax.imshow(I)\n\n    ax = ax_row[1]  # Access the second axis in each row\n    ax.imshow(I)\n    plt.sca(ax)\n    coco.showAnns(anns, draw_bbox=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T10:48:42.068592Z","iopub.execute_input":"2023-05-31T10:48:42.069859Z","iopub.status.idle":"2023-05-31T10:48:45.397727Z","shell.execute_reply.started":"2023-05-31T10:48:42.069818Z","shell.execute_reply":"2023-05-31T10:48:45.396555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}