{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pycocotools -Uqq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-28T20:07:24.850541Z","iopub.execute_input":"2023-06-28T20:07:24.851107Z","iopub.status.idle":"2023-06-28T20:08:00.416797Z","shell.execute_reply.started":"2023-06-28T20:07:24.851067Z","shell.execute_reply":"2023-06-28T20:08:00.415480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json, cv2, numpy as np, itertools, random, pandas as pd\nfrom pycocotools.coco import COCO\nfrom pycocotools import mask as maskUtils\nfrom skimage import io\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn import model_selection\nfrom copy import deepcopy","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:08:00.419303Z","iopub.execute_input":"2023-06-28T20:08:00.419722Z","iopub.status.idle":"2023-06-28T20:08:01.632349Z","shell.execute_reply.started":"2023-06-28T20:08:00.419677Z","shell.execute_reply":"2023-06-28T20:08:01.631578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:08:01.633348Z","iopub.execute_input":"2023-06-28T20:08:01.633674Z","iopub.status.idle":"2023-06-28T20:08:01.686717Z","shell.execute_reply.started":"2023-06-28T20:08:01.633646Z","shell.execute_reply":"2023-06-28T20:08:01.685801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_1_ids = df.query('source_wsi == 1')['id'].values.tolist()\ntrain_2_ids = df.query('source_wsi == 2')['id'].values.tolist()\ntrain_3_ids = df.query('source_wsi == 3')['id'].values.tolist()\ntrain_4_ids = df.query('source_wsi == 4')['id'].values.tolist()\nprint(len(train_1_ids), len(train_2_ids), len(train_3_ids), len(train_4_ids))","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:08:01.688687Z","iopub.execute_input":"2023-06-28T20:08:01.689033Z","iopub.status.idle":"2023-06-28T20:08:01.712078Z","shell.execute_reply.started":"2023-06-28T20:08:01.689008Z","shell.execute_reply":"2023-06-28T20:08:01.711286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def coordinates_to_masks(coordinates, shape):\n    masks = []\n    for coord in coordinates:\n        mask = np.zeros(shape, dtype=np.uint8)\n        cv2.fillPoly(mask, [np.array(coord)], 1)\n        masks.append(mask)\n    return masks\n\ndef binary_mask_to_rle(binary_mask):\n    rle = {'counts': [], 'size': list(binary_mask.shape)}\n    counts = rle.get('counts')\n    for i, (value, elements) in enumerate(itertools.groupby(binary_mask.ravel(order='F'))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:08:01.713053Z","iopub.execute_input":"2023-06-28T20:08:01.713335Z","iopub.status.idle":"2023-06-28T20:08:01.720078Z","shell.execute_reply.started":"2023-06-28T20:08:01.713310Z","shell.execute_reply":"2023-06-28T20:08:01.719137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dataset 2 all for training, dataset 1 split kfold, \"unsure\" treat as \"blood_vessel\" in training and not use in validate\nfrom sklearn.model_selection import KFold\njsonl_file_path = \"/kaggle/input/de-duplicate-labels/cleaned_polygons.jsonl\"\ntrain_data = [[], [], [], []]\nval_data = [[], [], [], []]\nwith open(jsonl_file_path, \"r\") as file:\n    for line in file:\n        info = json.loads(line)\n        if info[\"id\"] in train_1_ids:\n            val_data[0].append(info)\n            train_data[1].append(info)\n            train_data[2].append(info)\n            train_data[3].append(info)\n        elif info[\"id\"] in train_2_ids:\n            val_data[1].append(info)\n            train_data[0].append(info)\n            train_data[2].append(info)\n            train_data[3].append(info)\n        elif info[\"id\"] in train_3_ids:\n            val_data[2].append(info)\n            train_data[0].append(info)\n            train_data[1].append(info)\n            train_data[3].append(info)\n        elif info[\"id\"] in train_4_ids:\n            val_data[3].append(info)\n            train_data[0].append(info)\n            train_data[2].append(info)\n            train_data[1].append(info)\n\ncoco_data = {\"info\": {}, \"licenses\": [], \"categories\": [], \"images\": [], \"annotations\": []}\n\ncategories = []\nmax_len = 0\nfor item in tqdm(train_data[0], dynamic_ncols=True):\n    annotations = item[\"annotations\"]\n    if max_len < len(annotations):\n        max_len = len(annotations)\n    for annotation in annotations:\n        annotation_type = annotation[\"type\"]\n        if annotation_type != \"unsure\" and annotation_type != \"glomerulus\":\n            if annotation_type not in categories:\n                categories.append(annotation_type)\n                coco_data[\"categories\"].append({\"id\": len(categories), \"name\": annotation_type})\nprint(max_len, coco_data)\nmax_len_segmentation = -1\nfor fold, (train, val) in enumerate(zip(train_data, val_data)):\n    train_coco_data = deepcopy(coco_data)\n    valid_coco_data = deepcopy(coco_data)\n    split_index = 0\n    for data, ds in zip([train, val], [train_coco_data, valid_coco_data]):\n        for item in tqdm(data, dynamic_ncols=True):\n            image_id = item[\"id\"]\n\n            image_info = {\"id\": image_id, \"file_name\": item[\"id\"] + \".tif\", \"height\": 512, \"width\": 512}\n            ds[\"images\"].append(image_info)\n            anno_index = 0\n            for annotation in item[\"annotations\"]:\n                if annotation[\"type\"] == 'glomerulus' or annotation[\"type\"] == 'unsure':\n                    continue\n                        \n                category_id = categories.index(annotation[\"type\"]) + 1\n\n                segmentation = annotation[\"coordinates\"]\n                mask_img = coordinates_to_masks(segmentation, (512, 512))[0]\n\n                ys, xs = np.where(mask_img)\n                x1, x2 = min(xs), max(xs)\n                y1, y2 = min(ys), max(ys)\n                if max_len_segmentation < len(segmentation):\n                    max_len_segmentation = len(segmentation)\n                polygon = []\n    \n                for seg in segmentation:\n                    polygon.append([])\n                    for point in seg:\n                        polygon[-1].append(point[0])\n                        polygon[-1].append(point[1])\n                        \n                if(x2 - x1 == 0):\n                    print(\"Assert: \", x2 - x1)\n                if(y2 - y1 == 0):\n                    print(\"Assert: \", y2 - y1)\n                annotation_info = {\n                    \"id\": len(ds[\"annotations\"]) + 1,\n                    \"image_id\": image_id,\n                    \"category_id\": category_id,\n                    \"segmentation\": polygon,\n                    \"bbox\": [int(x1), int(y1), int(x2 - x1), int(y2 - y1)],\n                    \"bbox_mode\": 1,\n                    \"area\": int(np.sum(mask_img)),\n                    \"iscrowd\": 0,\n                }\n                ds[\"annotations\"].append(annotation_info)\n                anno_index += 1\n        split_index += 1\n    print(max_len_segmentation)\n    output_file_path = \"hubmap_train_fold_{}.json\".format(fold)\n    with open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n        json.dump(train_coco_data, output_file, ensure_ascii=True, indent=4)\n\n    output_file_path = \"hubmap_val_fold_{}.json\".format(fold)\n    with open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n        json.dump(valid_coco_data, output_file, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:09:44.928191Z","iopub.execute_input":"2023-06-28T20:09:44.929269Z","iopub.status.idle":"2023-06-28T20:13:18.039820Z","shell.execute_reply.started":"2023-06-28T20:09:44.929234Z","shell.execute_reply":"2023-06-28T20:13:18.038867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data[0] + val_data[0]\nds = deepcopy(coco_data)\nfor item in tqdm(train_data, dynamic_ncols=True):\n    image_id = item[\"id\"]\n    image_info = {\"id\": image_id, \"file_name\": item[\"id\"] + \".tif\", \"height\": 512, \"width\": 512}\n    ds[\"images\"].append(image_info)\n    anno_index = 0\n    for annotation in item[\"annotations\"]:\n        if annotation[\"type\"] == 'glomerulus' or annotation[\"type\"] == 'unsure':\n            continue\n\n        category_id = categories.index(annotation[\"type\"]) + 1\n\n        segmentation = annotation[\"coordinates\"]\n        mask_img = coordinates_to_masks(segmentation, (512, 512))[0]\n\n        ys, xs = np.where(mask_img)\n        x1, x2 = min(xs), max(xs)\n        y1, y2 = min(ys), max(ys)\n#                 print(segmentation)\n        polygon = []\n        for seg in segmentation:\n            polygon.append([])\n            for point in seg:\n                polygon[-1].append(point[0])\n                polygon[-1].append(point[1])\n\n        annotation_info = {\n            \"id\": len(ds[\"annotations\"]) + 1,\n            \"image_id\": image_id,\n            \"category_id\": category_id,\n            \"segmentation\": polygon,\n            \"bbox\": [int(x1), int(y1), int(x2 - x1), int(y2 - y1)],\n            \"bbox_mode\": 1,\n            \"area\": int(np.sum(mask_img)),\n            \"iscrowd\": 0,\n        }\n        ds[\"annotations\"].append(annotation_info)\n        anno_index += 1\n\noutput_file_path = \"hubmap_train_all.json\"\nwith open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n    json.dump(ds, output_file, ensure_ascii=True, indent=4)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T20:08:15.444360Z","iopub.status.idle":"2023-06-28T20:08:15.444742Z","shell.execute_reply.started":"2023-06-28T20:08:15.444540Z","shell.execute_reply":"2023-06-28T20:08:15.444556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids = df.query('dataset == 2')['id'].values.tolist()\nvalid_ids = df.query('dataset == 1')['id'].values.tolist()\ndata = []\nwith open(jsonl_file_path, \"r\") as file:\n    for line in file:\n        info = json.loads(line)\n        data.append(info)\n        \ntrain_coco_data = deepcopy(coco_data)\nvalid_coco_data = deepcopy(coco_data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_train = 0\ntotal_val = 0\nfor item in tqdm(data, dynamic_ncols=True):\n    image_id = item[\"id\"]\n\n    if image_id in train_ids:\n        ds = train_coco_data\n        total_train += 1\n    elif image_id in valid_ids:\n        total_val += 1\n        ds = valid_coco_data\n    else:\n        raise NotImplementedError()\n    image_info = {\"id\": image_id, \"file_name\": item[\"id\"] + \".tif\", \"height\": 512, \"width\": 512}\n    ds[\"images\"].append(image_info)\n    anno_index = 0\n    for annotation in item[\"annotations\"]:\n        if annotation[\"type\"] == 'glomerulus' or annotation[\"type\"] == 'unsure':\n            continue\n\n        category_id = categories.index(annotation[\"type\"]) + 1\n\n        segmentation = annotation[\"coordinates\"]\n        mask_img = coordinates_to_masks(segmentation, (512, 512))[0]\n\n        ys, xs = np.where(mask_img)\n        x1, x2 = min(xs), max(xs)\n        y1, y2 = min(ys), max(ys)\n#                 print(segmentation)\n        polygon = []\n        for seg in segmentation:\n            polygon.append([])\n            for point in seg:\n                polygon[-1].append(point[0])\n                polygon[-1].append(point[1])\n\n        annotation_info = {\n            \"id\": len(ds[\"annotations\"]) + 1,\n            \"image_id\": image_id,\n            \"category_id\": category_id,\n            \"segmentation\": polygon,\n            \"bbox\": [int(x1), int(y1), int(x2 - x1), int(y2 - y1)],\n            \"bbox_mode\": 1,\n            \"area\": int(np.sum(mask_img)),\n            \"iscrowd\": 0,\n        }\n        ds[\"annotations\"].append(annotation_info)\n        anno_index += 1\n\noutput_file_path = \"hubmap_train_data2.json\"\nwith open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n    json.dump(train_coco_data, output_file, ensure_ascii=True, indent=4)\n    \noutput_file_path = \"hubmap_val_data1.json\"\nwith open(output_file_path, \"w\", encoding=\"utf-8\") as output_file:\n    json.dump(valid_coco_data, output_file, ensure_ascii=True, indent=4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(total_train, total_val)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}