{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"reference:https://www.kaggle.com/code/finlay/hubmap-eda-coco-yolov7-train-and-validate#YOLOV7-%E8%AE%AD%E7%BB%83","metadata":{}},{"cell_type":"markdown","source":"# Load libraries","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd\nfrom tqdm import tqdm, notebook\nfrom collections import Counter\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.396135Z","iopub.execute_input":"2023-08-01T21:53:58.397116Z","iopub.status.idle":"2023-08-01T21:53:58.404901Z","shell.execute_reply.started":"2023-08-01T21:53:58.397054Z","shell.execute_reply":"2023-08-01T21:53:58.403725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.EDA","metadata":{}},{"cell_type":"markdown","source":"https://www.kaggle.com/code/huangzeyuzheng/eda-for-hubmap-2023","metadata":{}},{"cell_type":"code","source":"#files path\npolygons_path='/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl'\nsample_submission_path='/kaggle/input/hubmap-hacking-the-human-vasculature/sample_submission.csv'\ntile_meta_path='/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv'\nwsi_meta_path='/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv'\n\n#folders path\ntrain_path='/kaggle/input/hubmap-hacking-the-human-vasculature/train/'\ntest_path='/kaggle/input/hubmap-hacking-the-human-vasculature/test/'","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.411820Z","iopub.execute_input":"2023-08-01T21:53:58.412135Z","iopub.status.idle":"2023-08-01T21:53:58.449295Z","shell.execute_reply.started":"2023-08-01T21:53:58.412108Z","shell.execute_reply":"2023-08-01T21:53:58.448265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission=pd.read_csv(sample_submission_path)\ntile_meta=pd.read_csv(tile_meta_path)\nwsi_meta=pd.read_csv(wsi_meta_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.451278Z","iopub.execute_input":"2023-08-01T21:53:58.451726Z","iopub.status.idle":"2023-08-01T21:53:58.500406Z","shell.execute_reply.started":"2023-08-01T21:53:58.451694Z","shell.execute_reply":"2023-08-01T21:53:58.499402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Number of tiles in each known WSI\ntile_count=tile_meta['source_wsi'].value_counts()\ntile_count","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.501755Z","iopub.execute_input":"2023-08-01T21:53:58.503476Z","iopub.status.idle":"2023-08-01T21:53:58.519220Z","shell.execute_reply.started":"2023-08-01T21:53:58.503441Z","shell.execute_reply":"2023-08-01T21:53:58.518227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get number of tiles belonging to each wsi,every tiles has their own annotations(experted or non-experted)\nwsi4=tile_count.values[-1]\nwsi3=tile_count.values[-2]\nwsi2=tile_count.values[-3]\nwsi1=tile_count.values[-4]\nprint(wsi1,wsi2,wsi3,wsi4)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.521994Z","iopub.execute_input":"2023-08-01T21:53:58.522374Z","iopub.status.idle":"2023-08-01T21:53:58.531257Z","shell.execute_reply.started":"2023-08-01T21:53:58.522343Z","shell.execute_reply":"2023-08-01T21:53:58.530290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The competition data includes small pieces called \"tiles\" that come from five big pictures called \"Whole Slide Images\" (WSI). These WSIs are split into two groups called datasets. In **Dataset 1**, the **tiles have been looked at by experts who reviewed and marked them**. In **Dataset 2**, the tiles are from the same big pictures but **they don't have as many marks, and the marks they do have haven't been reviewed by experts**.","metadata":{}},{"cell_type":"markdown","source":"### WSI source Distribution in dataset1","metadata":{}},{"cell_type":"code","source":"#WSI1\ntile_meta.loc[(tile_meta['source_wsi']==1)& (tile_meta['dataset']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.532877Z","iopub.execute_input":"2023-08-01T21:53:58.533340Z","iopub.status.idle":"2023-08-01T21:53:58.561308Z","shell.execute_reply.started":"2023-08-01T21:53:58.533307Z","shell.execute_reply":"2023-08-01T21:53:58.560377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of tiles of wsi1 in the dataset1\nnum_w1d1=len(tile_meta.loc[(tile_meta['source_wsi']==1)& (tile_meta['dataset']==1)])\n#Store all ids of the WSI1 tiles in dataset1 respectively\nwsi1_inds1=tile_meta.loc[(tile_meta['source_wsi']==1)& (tile_meta['dataset']==1)]['id'].tolist()\n#number of tiles of wsi1 in the dataset2\nnum_w1d2=wsi1-num_w1d1","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.562789Z","iopub.execute_input":"2023-08-01T21:53:58.563140Z","iopub.status.idle":"2023-08-01T21:53:58.571225Z","shell.execute_reply.started":"2023-08-01T21:53:58.563112Z","shell.execute_reply":"2023-08-01T21:53:58.569906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#WSI2\ntile_meta.loc[(tile_meta['source_wsi']==2)& (tile_meta['dataset']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.572974Z","iopub.execute_input":"2023-08-01T21:53:58.573631Z","iopub.status.idle":"2023-08-01T21:53:58.594162Z","shell.execute_reply.started":"2023-08-01T21:53:58.573597Z","shell.execute_reply":"2023-08-01T21:53:58.593156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of tiles of WSI2 in the dataset1\nnum_w2d1=len(tile_meta.loc[(tile_meta['source_wsi']==2)& (tile_meta['dataset']==1)])\n#Store all ids of the WSI1 tiles in dataset1 respectively\nwsi2_inds1=tile_meta.loc[(tile_meta['source_wsi']==2)& (tile_meta['dataset']==1)]['id'].tolist()\n#number of tiles of wsi2 in the dataset2\nnum_w2d2=wsi2-num_w2d1\n","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.597487Z","iopub.execute_input":"2023-08-01T21:53:58.598389Z","iopub.status.idle":"2023-08-01T21:53:58.607744Z","shell.execute_reply.started":"2023-08-01T21:53:58.598354Z","shell.execute_reply":"2023-08-01T21:53:58.606643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#WSI3\ntile_meta.loc[(tile_meta['source_wsi']==3)& (tile_meta['dataset']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.609119Z","iopub.execute_input":"2023-08-01T21:53:58.609665Z","iopub.status.idle":"2023-08-01T21:53:58.625760Z","shell.execute_reply.started":"2023-08-01T21:53:58.609630Z","shell.execute_reply":"2023-08-01T21:53:58.624805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of tiles of WSI3 in the dataset1\nnum_w3d1=len(tile_meta.loc[(tile_meta['source_wsi']==3)& (tile_meta['dataset']==1)])\n#Store all ids of the WSI1 tiles in dataset1 respectively\nwsi3_inds1=tile_meta.loc[(tile_meta['source_wsi']==3)& (tile_meta['dataset']==1)]['id'].tolist()\n#number of tiles of wsi3 in the dataset2\nnum_w3d2=wsi3-num_w3d1\n","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.627521Z","iopub.execute_input":"2023-08-01T21:53:58.628321Z","iopub.status.idle":"2023-08-01T21:53:58.637801Z","shell.execute_reply.started":"2023-08-01T21:53:58.628287Z","shell.execute_reply":"2023-08-01T21:53:58.637150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#WSI4\ntile_meta.loc[(tile_meta['source_wsi']==4)& (tile_meta['dataset']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.639988Z","iopub.execute_input":"2023-08-01T21:53:58.641587Z","iopub.status.idle":"2023-08-01T21:53:58.654416Z","shell.execute_reply.started":"2023-08-01T21:53:58.641561Z","shell.execute_reply":"2023-08-01T21:53:58.653473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of tiles of WSI4 in the dataset1\nnum_w4d1=len(tile_meta.loc[(tile_meta['source_wsi']==3)& (tile_meta['dataset']==1)])\n#Store all ids of the WSI1 tiles in dataset1 respectively\nwsi4_inds1=tile_meta.loc[(tile_meta['source_wsi']==3)& (tile_meta['dataset']==1)]['id'].tolist()\n#number of tiles of wsi4 in the dataset2\nnum_w4d2=wsi4-num_w4d1","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.656360Z","iopub.execute_input":"2023-08-01T21:53:58.657117Z","iopub.status.idle":"2023-08-01T21:53:58.667340Z","shell.execute_reply.started":"2023-08-01T21:53:58.657082Z","shell.execute_reply":"2023-08-01T21:53:58.666393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds1_samples=wsi1_inds1+wsi2_inds1\nlen(ds1_samples)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:53:58.671093Z","iopub.execute_input":"2023-08-01T21:53:58.671422Z","iopub.status.idle":"2023-08-01T21:53:58.681301Z","shell.execute_reply.started":"2023-08-01T21:53:58.671396Z","shell.execute_reply":"2023-08-01T21:53:58.680167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Load Data","metadata":{}},{"cell_type":"markdown","source":"## Get Testing set\nAll the Test samples are randomly sampled from Dataset1, which is annotated by experts","metadata":{}},{"cell_type":"code","source":"import random\nrandom.seed(42)\n\ntest_set=random.sample(ds1_samples,163)\nlen(test_set)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:02.121646Z","iopub.execute_input":"2023-08-01T21:54:02.122011Z","iopub.status.idle":"2023-08-01T21:54:02.129852Z","shell.execute_reply.started":"2023-08-01T21:54:02.121978Z","shell.execute_reply":"2023-08-01T21:54:02.128710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set[:3]","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:03.218232Z","iopub.execute_input":"2023-08-01T21:54:03.218594Z","iopub.status.idle":"2023-08-01T21:54:03.225095Z","shell.execute_reply.started":"2023-08-01T21:54:03.218560Z","shell.execute_reply":"2023-08-01T21:54:03.223982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(data=test_set)\ndf.to_csv('/kaggle/working/testset_trainv11.csv',encoding='utf-8',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:09.240314Z","iopub.execute_input":"2023-08-01T21:54:09.240921Z","iopub.status.idle":"2023-08-01T21:54:09.253937Z","shell.execute_reply.started":"2023-08-01T21:54:09.240887Z","shell.execute_reply":"2023-08-01T21:54:09.253086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Polygons","metadata":{}},{"cell_type":"code","source":"# !pip install pycocotools\n!cp -r /kaggle/input/pycocotools/ /kaggle/working/pycocotools\n!pip install /kaggle/working/pycocotools/pycocotools-2.0.6  --no-index --find-links=/kaggle/working/pycocotools/ ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:12.876343Z","iopub.execute_input":"2023-08-01T21:54:12.877380Z","iopub.status.idle":"2023-08-01T21:54:48.021742Z","shell.execute_reply.started":"2023-08-01T21:54:12.877335Z","shell.execute_reply":"2023-08-01T21:54:48.020579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dataset","metadata":{}},{"cell_type":"code","source":"#Libraries\nfrom torch.utils.data import Dataset\nimport cv2 #openCV \nimport yaml #for configure file\nimport json\nfrom PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:53.660664Z","iopub.execute_input":"2023-08-01T21:54:53.661141Z","iopub.status.idle":"2023-08-01T21:54:57.107382Z","shell.execute_reply.started":"2023-08-01T21:54:53.661099Z","shell.execute_reply":"2023-08-01T21:54:57.106173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Annotation Config for present different types of tissues\ndataset_config = {\n    \"background\": {\n        \"apply_mask\": None,\n        \"label\": 0,\n        \"rgb\": (0, 0, 0),\n        \"loss_weight\": None\n    },\n    \"blood_vessel\": {\n        \"apply_mask\": True,#decide whether load this type's mask\n        \"label\": 1,\n        \"rgb\": (25, 8, 8),\n        \"loss_weight\": None\n    },\n    \"glomerulus\": {\n        \"apply_mask\":True,\n        \"label\": 2,\n        \"rgb\": (8, 12, 255),\n        \"loss_weight\": None\n    },\n    \"unsure\": {\n        \"apply_mask\": True,\n        \"label\": 3,\n        \"rgb\": (8, 255, 20),\n        \"loss_weight\": None\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:54:59.849998Z","iopub.execute_input":"2023-08-01T21:54:59.850624Z","iopub.status.idle":"2023-08-01T21:54:59.858402Z","shell.execute_reply.started":"2023-08-01T21:54:59.850591Z","shell.execute_reply":"2023-08-01T21:54:59.857390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HuBMAPDataset(Dataset):#build dataset\n    def __init__(self,\n                annotation_path:str,#the path to the polygon annotation json file\n                image_path:str):#path to the folder /train, which contains the images\n        self.__image_path=image_path\n        self.__samples=self.parse_jsonl(annotation_path)#the whole polygon annotation data stored in samples\n    \n    def __len__(self) -> int:\n        return len(self.__samples)#each image has its related annotation list among the whole list,num_sample=num_images_with_annotaiton\n    \n    def __getitem__(self, idx:int) -> tuple[np.ndarray, np.ndarray]:\n        image=self.__get_image(idx)#get array of an image \n        mask=self.__get_mask(idx)#get image-size mask with multiple instances\n        return image,mask\n    \n    @staticmethod\n    def parse_jsonl(path: str) -> list[dict, ...]:#read json file\n        with open(path, 'r') as json_file:\n            jsonl_labels = [\n                json.loads(line)\n                for line in notebook.tqdm(\n                    json_file, desc=\"Processing polygons\", total=1633\n                )\n            ]\n        return jsonl_labels\n    \n    @staticmethod\n    def load_config(path: str) -> dict: \n        with open(path,'r') as f:\n            data=yaml.load(stream=f, Loader=yaml.SafeLoader)#transfer a YAML document into a Python object，stream:file stream，Loader:load method，返回值:return a Python dict obejct\n        return data\n    \n    def __get_image_path(self, id:str) ->str: #get path to each specific image\n        path=os.path.join(\n            self.__image_path, f\"{id}.tif\"\n        )\n        return path\n    \n    def __get_image(self, idx:int) -> np.ndarray:#get single image\n        id=self.__samples[idx][\"id\"]#id of per image\n        image_path=self.__get_image_path(id)#get the complete path of a image file\n        image=Image.open(image_path)#load a image\n        image=np.asarray(image)#change it into a array\n        return image\n    \n    def __get_mask(self, idx:int) ->np.ndarray: #\n        mask=np.zeros((512,512),dtype=np.uint8)#image size is 512x512, thus the size of image-mask is 512x512\n        annotations = self.__samples[idx][\"annotations\"]#get the mask list of an image\n        \n        for vessel in annotations:#every instance has its dict including its type and list corrdinates of annotation\n            vessel_type=vessel[\"type\"]  #get instance type\n            config=dataset_config[vessel_type]#load this type relative config for futher presentation\n            \n            if config[\"apply_mask\"]:#only the 1 valid tissues:blood_vessel\n                coordinates=np.array(vessel[\"coordinates\"])#convert list of coordinate into array\n                mask=cv2.fillPoly(mask, pts=coordinates, color=config[\"rgb\"])#crop the all the polygons and fill the area with their relative color config\n        return mask\n        ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:55:03.844449Z","iopub.execute_input":"2023-08-01T21:55:03.844852Z","iopub.status.idle":"2023-08-01T21:55:03.860935Z","shell.execute_reply.started":"2023-08-01T21:55:03.844820Z","shell.execute_reply":"2023-08-01T21:55:03.859838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = HuBMAPDataset(polygons_path, train_path)#initiate the dataset","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:55:04.567682Z","iopub.execute_input":"2023-08-01T21:55:04.568034Z","iopub.status.idle":"2023-08-01T21:55:09.210217Z","shell.execute_reply.started":"2023-08-01T21:55:04.568004Z","shell.execute_reply":"2023-08-01T21:55:09.209253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image, mask = dataset[1000]#_get__items function \nfig, (ax1, ax2) = plt.subplots(1, 2)\n\nax1.imshow(image)#show the image\nax2.imshow(mask)#show the mask\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:55:10.325391Z","iopub.execute_input":"2023-08-01T21:55:10.325758Z","iopub.status.idle":"2023-08-01T21:55:10.846912Z","shell.execute_reply.started":"2023-08-01T21:55:10.325729Z","shell.execute_reply":"2023-08-01T21:55:10.845987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert original tiles into COCO format\nreference: https://www.kaggle.com/code/fnands/convert-training-data-to-coco-format","metadata":{}},{"cell_type":"code","source":"#import json\nfrom pathlib import Path\nimport shutil\nimport itertools\nfrom typing import List, Dict\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-08-01T21:55:16.008138Z","iopub.execute_input":"2023-08-01T21:55:16.009110Z","iopub.status.idle":"2023-08-01T21:55:16.553478Z","shell.execute_reply.started":"2023-08-01T21:55:16.009076Z","shell.execute_reply":"2023-08-01T21:55:16.552287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baseline","metadata":{}},{"cell_type":"code","source":"# Baseline\n# with open(polygons_path, \"r\") as json_file:#load polygons annotation file\n#     json_list=list(json_file)\n    \n# tiles_dicts=[]\n# test=[]\n\n# for json_str in json_list:\n#     item=json.loads(json_str)\n#     if item['id'] in test_set:#split test set and train&val set\n#         test.append(json.loads(json_str))\n#     else:\n#         tiles_dicts.append(json.loads(json_str))\n    \n# # Conversion between class name and index\n# id_dict={\"blood_vessel\":0, \"glomerulus\":1, \"unsure\":2}#in COCO format, the class is reperesented by id(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-31T14:23:59.565099Z","iopub.execute_input":"2023-07-31T14:23:59.565478Z","iopub.status.idle":"2023-07-31T14:24:05.368747Z","shell.execute_reply.started":"2023-07-31T14:23:59.565449Z","shell.execute_reply":"2023-07-31T14:24:05.367742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-processing1: using different number of types as training data","metadata":{}},{"cell_type":"code","source":"# #only blood_vessel annotation\n# with open(polygons_path, \"r\") as json_file:#load polygons annotation file\n#     json_list=list(json_file)\n\n# blood_vessel_dicts=[]\n# for json_str in json_list:\n#     annotation=[]\n#     item=json.loads(json_str)\n#     for ann in item['annotations']:\n#         if ann['type']=='blood_vessel':\n#             annotation.append({'type':ann['type'],'coordinates':ann['coordinates']})\n#     blood_vessel_dicts.append({'id':item['id'],'annotations':annotation})\n\n            \n    \n# tiles_dicts=[]\n# test=[]\n\n# for bv in blood_vessel_dicts:\n#     if bv['id'] in test_set:#split test set and train&val set\n#         test.append(bv)\n#     else:\n#         tiles_dicts.append(bv)\n    \n# # Conversion between class name and index\n# id_dict={\"blood_vessel\":0, \"glomerulus\":1, \"unsure\":2}#in COCO format, the class is reperesented by id(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-31T19:37:35.973277Z","iopub.execute_input":"2023-07-31T19:37:35.973631Z","iopub.status.idle":"2023-07-31T19:37:38.852478Z","shell.execute_reply.started":"2023-07-31T19:37:35.973604Z","shell.execute_reply":"2023-07-31T19:37:38.851483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #only blood_vessel&glomerulus annotation\n# with open(polygons_path, \"r\") as json_file:#load polygons annotation file\n#     json_list=list(json_file)\n\n# annotation_dicts=[]\n# for json_str in json_list:\n#     annotation=[]\n#     item=json.loads(json_str)\n#     for ann in item[\"annotations\"]:\n#         if ann[\"type\"]=='blood_vessel' or ann['type']=='glomerulus':\n#             annotation.append({'type':ann[\"type\"],'coordinates':ann[\"coordinates\"]})\n#     annotation_dicts.append({'id':item[\"id\"],'annotations':annotation})\n\n            \n# # for dic in annotation_dicts:\n# #     for anno in dic[\"annotations\"]:\n# #         if anno[\"type\"]==\"unsure\":\n# #             print(\"bugs here!\")\n\n# tiles_dicts=[]\n# test=[]\n\n# for bv in annotation_dicts:\n#     if bv['id'] in test_set:#split test set and train&val set\n#         test.append(bv)\n#     else:\n#         tiles_dicts.append(bv)\n    \n# # Conversion between class name and index\n# id_dict={\"blood_vessel\":0, \"glomerulus\":1, \"unsure\":2}#in COCO format, the class is reperesented by id(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T15:44:15.127555Z","iopub.execute_input":"2023-08-01T15:44:15.128252Z","iopub.status.idle":"2023-08-01T15:44:19.174902Z","shell.execute_reply.started":"2023-08-01T15:44:15.128219Z","shell.execute_reply":"2023-08-01T15:44:19.173916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#consider \"unsure\" as blood_vessel,only blood_vessel&glomerulus annotation\nwith open(polygons_path, \"r\") as json_file:#load polygons annotation file\n    json_list=list(json_file)\n\nannotation_dicts=[]\n#i=0\nfor json_str in json_list:\n    annotation=[]\n    item=json.loads(json_str)\n    for ann in item[\"annotations\"]:\n        if ann[\"type\"]=='blood_vessel' or ann['type']=='glomerulus':\n            annotation.append({'type':ann[\"type\"],'coordinates':ann[\"coordinates\"]})\n        elif ann['type']=='unsure':\n            ann['type']='blood_vessel'\n            annotation.append({'type':ann[\"type\"],'coordinates':ann[\"coordinates\"]})\n    annotation_dicts.append({'id':item[\"id\"],'annotations':annotation})\n\n#print(i):897            \n# for dic in annotation_dicts:\n#     for anno in dic[\"annotations\"]:\n#         if anno[\"type\"]==\"unsure\":\n#             print(\"bugs here!\")\n\ntiles_dicts=[]\ntest=[]\n\nfor bv in annotation_dicts:\n    if bv['id'] in test_set:#split test set and train&val set\n        test.append(bv)\n    else:\n        tiles_dicts.append(bv)\n    \n# Conversion between class name and index\nid_dict={\"blood_vessel\":0, \"glomerulus\":1, \"unsure\":2}#in COCO format, the class is reperesented by id(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:04:22.056564Z","iopub.execute_input":"2023-08-01T22:04:22.056961Z","iopub.status.idle":"2023-08-01T22:04:25.440637Z","shell.execute_reply.started":"2023-08-01T22:04:22.056931Z","shell.execute_reply":"2023-08-01T22:04:25.439623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# blood_vessel_dicts=[]\n# for json_str in json_list[:1]:\n#     annotation=[]\n#     item=json.loads(json_str)\n#     for ann in item['annotations']:\n#         if ann['type']=='blood_vessel':\n#             annotation.append({'type':ann['type'],'coordinates':ann['coordinates']})\n#     blood_vessel_dicts.append({'id':item['id'],'annotations':annotation})\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T14:24:10.873752Z","iopub.execute_input":"2023-07-31T14:24:10.874439Z","iopub.status.idle":"2023-07-31T14:24:10.878810Z","shell.execute_reply.started":"2023-07-31T14:24:10.874404Z","shell.execute_reply":"2023-07-31T14:24:10.877886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for ann in blood_vessel_dicts:\n#     print(ann['id'])","metadata":{"execution":{"iopub.status.busy":"2023-07-26T11:34:38.506354Z","iopub.execute_input":"2023-07-26T11:34:38.506922Z","iopub.status.idle":"2023-07-26T11:34:38.514772Z","shell.execute_reply.started":"2023-07-26T11:34:38.506887Z","shell.execute_reply":"2023-07-26T11:34:38.513314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  [x[\"coordinates\"][0][0] for x in tiles_dicts[1][\"annotations\"]]#has 2 instances with different types,return the first point of each tissue\n# #[x[\"type\"] for x in tiles_dicts[1][\"annotations\"]]#has 2 instance","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-25T11:54:48.298610Z","iopub.execute_input":"2023-07-25T11:54:48.299124Z","iopub.status.idle":"2023-07-25T11:54:48.307445Z","shell.execute_reply.started":"2023-07-25T11:54:48.299087Z","shell.execute_reply":"2023-07-25T11:54:48.306351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split into train and valalidation\ntrain,val=train_test_split(tiles_dicts,test_size=0.11,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:04:30.983284Z","iopub.execute_input":"2023-08-01T22:04:30.984459Z","iopub.status.idle":"2023-08-01T22:04:30.991290Z","shell.execute_reply.started":"2023-08-01T22:04:30.984416Z","shell.execute_reply":"2023-08-01T22:04:30.990384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train),len(val),len(test))","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:04:37.153326Z","iopub.execute_input":"2023-08-01T22:04:37.153709Z","iopub.status.idle":"2023-08-01T22:04:37.160784Z","shell.execute_reply.started":"2023-08-01T22:04:37.153681Z","shell.execute_reply":"2023-08-01T22:04:37.159583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set=[]\nfor t in train:\n    train_set.append(t['id'])\n    #val_set.append(v['id'])\n    \ndf_t=pd.DataFrame(data=train_set)\ndf_t.to_csv('/kaggle/working/trainset_trainv11.csv',encoding='utf-8',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:04:55.764424Z","iopub.execute_input":"2023-08-01T22:04:55.764832Z","iopub.status.idle":"2023-08-01T22:04:55.777837Z","shell.execute_reply.started":"2023-08-01T22:04:55.764799Z","shell.execute_reply":"2023-08-01T22:04:55.776882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_set=[]\nfor v in val:\n    val_set.append(v['id'])\n    \n    \ndf_v=pd.DataFrame(data=val_set)\ndf_v.to_csv('/kaggle/working/valset_trainv11.csv',encoding='utf-8',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:04:59.013327Z","iopub.execute_input":"2023-08-01T22:04:59.013708Z","iopub.status.idle":"2023-08-01T22:04:59.022274Z","shell.execute_reply.started":"2023-08-01T22:04:59.013678Z","shell.execute_reply":"2023-08-01T22:04:59.021250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile_to_coco(tile: List[Dict], output_path: Path):#convert tile into COCO format,tile is list of dict which contains id,list of instance's type&coordinates\n    tile_id=tile[\"id\"]#get the id of tile(image)\n    \n    shutil.copyfile(train_path+f\"{tile_id}.tif\", output_path+f\"{tile_id}.tif\")#create output image file\n    \n    with open(output_path+f\"{tile_id}.txt\",\"w\") as text_file:#COCO mask is extend with .txt,this file will be used to store the annotations in the COCO format.\n        for annotation in tile[\"annotations\"]:#each annotation in a tile\n            class_id=id_dict[annotation['type']]#convert class name(str) into id(int)\n            flat_mask_polygon=list(itertools.chain(*annotation['coordinates'][0]))#used to flatten a list of lists or a sequence of sequences,'flat_mask_polygon' is a list of number\n            array=np.array(flat_mask_polygon)/512. #coco labels is relative posistion between 0 and 1 instead of pixel indices\n            text_file.write(f'{class_id} {\" \".join(map(str, array))}\\n')#write into the file\n        ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:01.673841Z","iopub.execute_input":"2023-08-01T22:05:01.674253Z","iopub.status.idle":"2023-08-01T22:05:01.682446Z","shell.execute_reply.started":"2023-08-01T22:05:01.674194Z","shell.execute_reply":"2023-08-01T22:05:01.681257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs('./COCO/', exist_ok=True)\nos.makedirs('./COCO/train/', exist_ok=True)\nos.makedirs('./COCO/val/', exist_ok=True)\nos.makedirs('./COCO/test/', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:03.748685Z","iopub.execute_input":"2023-08-01T22:05:03.749105Z","iopub.status.idle":"2023-08-01T22:05:03.756730Z","shell.execute_reply.started":"2023-08-01T22:05:03.749066Z","shell.execute_reply":"2023-08-01T22:05:03.755287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for train_dict in train:#each tile is transferred into tile_to_coco function\n    tile_to_coco(train_dict, './COCO/train/')\n    \nfor val_dict in val:\n    tile_to_coco(val_dict,'./COCO/val/')\n    \nfor test_dict in test:\n    tile_to_coco(test_dict,'./COCO/test/')","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:05.251707Z","iopub.execute_input":"2023-08-01T22:05:05.252891Z","iopub.status.idle":"2023-08-01T22:05:35.102503Z","shell.execute_reply.started":"2023-08-01T22:05:05.252849Z","shell.execute_reply":"2023-08-01T22:05:35.101401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a yaml file as expected by YOLOv7 ---data yaml file\nyaml_text = \"\"\"\n# HuBMAP - Hacking the Human Vasculature dataset \n# https://www.kaggle.com/competitions/hubmap-hacking-the-human-vasculature\n\n\n# train and val data as 1) directory: path/images/, 2) file: path/images.txt, or 3) list: [path1/images/, path2/images/]\n#path of training set,including image file and its relative coco annotation .txt file, total=1308\ntrain: /kaggle/working/COCO/train \n#path of validation set, total=162\nval: /kaggle/working/COCO/val \n\n# class names\nnames: \n  0: blood_vessel\n  1: glomerulus\n  2: unsure\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:50.540024Z","iopub.execute_input":"2023-08-01T22:05:50.540434Z","iopub.status.idle":"2023-08-01T22:05:50.545939Z","shell.execute_reply.started":"2023-08-01T22:05:50.540403Z","shell.execute_reply":"2023-08-01T22:05:50.544725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/working/COCO/hubmap-coco.yaml', 'w') as text_file:\n    text_file.write(yaml_text)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:52.137964Z","iopub.execute_input":"2023-08-01T22:05:52.139066Z","iopub.status.idle":"2023-08-01T22:05:52.144959Z","shell.execute_reply.started":"2023-08-01T22:05:52.139023Z","shell.execute_reply":"2023-08-01T22:05:52.143883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.Training\nreference:https://www.kaggle.com/code/fnands/a-quick-yolov7-baseline","metadata":{}},{"cell_type":"markdown","source":"### 1.Clone the repository from git hub","metadata":{}},{"cell_type":"code","source":"!git clone -b u7 --single-branch https://github.com/WongKinYiu/yolov7.git ","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:05:55.349270Z","iopub.execute_input":"2023-08-01T22:05:55.349643Z","iopub.status.idle":"2023-08-01T22:05:57.677499Z","shell.execute_reply.started":"2023-08-01T22:05:55.349613Z","shell.execute_reply":"2023-08-01T22:05:57.676146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.set up the environment and import the package","metadata":{}},{"cell_type":"code","source":"!pip install -r yolov7/seg/requirements.txt","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-01T22:06:00.657125Z","iopub.execute_input":"2023-08-01T22:06:00.658022Z","iopub.status.idle":"2023-08-01T22:06:15.533476Z","shell.execute_reply.started":"2023-08-01T22:06:00.657984Z","shell.execute_reply":"2023-08-01T22:06:15.532181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.download the pre-trained weight","metadata":{}},{"cell_type":"code","source":"!wget https://github.com/WongKinYiu/yolov7/releases/download/v0.1/yolov7-seg.pt .\n\n#!\\rm -rf /kaggle/working/wandb /kaggle/working/runs","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:06:20.169381Z","iopub.execute_input":"2023-08-01T22:06:20.169787Z","iopub.status.idle":"2023-08-01T22:06:21.944021Z","shell.execute_reply.started":"2023-08-01T22:06:20.169754Z","shell.execute_reply":"2023-08-01T22:06:21.942738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.delete the folder","metadata":{}},{"cell_type":"code","source":"!\\rm -rf /kaggle/working/wandb /kaggle/working/runs","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:06:25.634010Z","iopub.execute_input":"2023-08-01T22:06:25.634467Z","iopub.status.idle":"2023-08-01T22:06:26.660192Z","shell.execute_reply.started":"2023-08-01T22:06:25.634431Z","shell.execute_reply":"2023-08-01T22:06:26.658747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5. creating the yaml file for hyperparameters","metadata":{}},{"cell_type":"code","source":"# Create a yaml file as expected by YOLOv7 ---hyperparameter yaml file\nyaml_text = \"\"\"\n# YOLOv5 🚀 by Ultralytics, GPL-3.0 license\n# Hyperparameters for high-augmentation COCO training from scratch\n# python train.py --batch 32 --cfg yolov5m6.yaml --weights '' --data coco.yaml --img 1280 --epochs 300\n# See tutorials for hyperparameter evolution https://github.com/ultralytics/yolov5#tutorials\n\nlr0: 0.0005  # initial learning rate (SGD=1E-2, Adam=1E-3)\nlrf: 0.1  # final OneCycleLR learning rate (lr0 * lrf)\nmomentum: 0.937  # SGD momentum/Adam beta1\nweight_decay: 0.0005  # optimizer weight decay 5e-4\nwarmup_epochs: 3.0  # warmup epochs (fractions ok)\nwarmup_momentum: 0.8  # warmup initial momentum\nwarmup_bias_lr: 0.1  # warmup initial bias lr\nbox: 0.05  # box loss gain\ncls: 0.3  # cls loss gain\ncls_pw: 1.0  # cls BCELoss positive_weight\nobj: 0.7  # obj loss gain (scale with pixels)\nobj_pw: 1.0  # obj BCELoss positive_weight\niou_t: 0.20  # IoU training threshold\nanchor_t: 4.0  # anchor-multiple threshold\n# anchors: 3  # anchors per output layer (0 to ignore)\nfl_gamma: 0.0  # focal loss gamma (efficientDet default gamma=1.5)\nhsv_h: 0.015  # image HSV-Hue augmentation (fraction)\nhsv_s: 0.7  # image HSV-Saturation augmentation (fraction)\nhsv_v: 0.4  # image HSV-Value augmentation (fraction)\ndegrees: 0.0  # image rotation (+/- deg)\ntranslate: 0.1  # image translation (+/- fraction)\nscale: 0.9  # image scale (+/- gain)\nshear: 0.0  # image shear (+/- deg)\nperspective: 0.0  # image perspective (+/- fraction), range 0-0.001\nflipud: 0.0  # image flip up-down (probability)\nfliplr: 0.5  # image flip left-right (probability)\nmosaic: 1.0  # image mosaic (probability)\nmixup: 0.1  # image mixup (probability)\ncopy_paste: 0.1  # segment copy-paste (probability)\n\"\"\"\nwith open('/kaggle/working/hyp.yaml', 'w') as text_file:\n    text_file.write(yaml_text)","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:06:28.816290Z","iopub.execute_input":"2023-08-01T22:06:28.816737Z","iopub.status.idle":"2023-08-01T22:06:28.824818Z","shell.execute_reply.started":"2023-08-01T22:06:28.816701Z","shell.execute_reply":"2023-08-01T22:06:28.823852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 6. Training","metadata":{}},{"cell_type":"code","source":"#load train.py\nfrom yolov7.seg.segment import train\n\ntrain.run(data=\"/kaggle/working/COCO/hubmap-coco.yaml\",#dataset's yaml\n         imgsz=512,#image size\n         batch=16, #batch size for all gpu\n         weights='yolov7-seg.pt',#pre-trained weight\n         cfg='/kaggle/working/yolov7/seg/models/segment/yolov7-seg.yaml',#IS's parameters and backbone\n         epochs=35,#total training epoch number\n         name='trainv11_2class_with_unsure',\n         project='yolov7-fine-tune_EX3_trainv9',\n         hyp='/kaggle/working/hyp.yaml',#hyperparameter's path\n         optimizer='Adam'\n         )","metadata":{"execution":{"iopub.status.busy":"2023-08-01T22:06:58.591824Z","iopub.execute_input":"2023-08-01T22:06:58.592279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport datetime\n\ndef file2zip(packagePath, zipPath):\n    '''\n  :param packagePath: 文件夹路径\n  :param zipPath: 压缩包路径\n  :return:\n  '''\n    zip = zipfile.ZipFile(zipPath, 'w', zipfile.ZIP_DEFLATED)\n    for path, dirNames, fileNames in os.walk(packagePath):\n        fpath = path.replace(packagePath, '')\n        for name in fileNames:\n            fullName = os.path.join(path, name)\n            name = fpath + '\\\\' + name\n            zip.write(fullName, name)\n    zip.close()\n\n\nif __name__ == \"__main__\":\n    # 文件夹路径\n    packagePath = '/kaggle/working/COCO/test/'\n    zipPath = '/kaggle/working/test.zip'\n    if os.path.exists(zipPath):\n        os.remove(zipPath)\n    file2zip(packagePath, zipPath)\n    print(\"打包完成\")\n    print(datetime.datetime.utcnow())\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T15:56:17.240432Z","iopub.execute_input":"2023-07-25T15:56:17.240912Z","iopub.status.idle":"2023-07-25T15:56:29.685850Z","shell.execute_reply.started":"2023-07-25T15:56:17.240877Z","shell.execute_reply":"2023-07-25T15:56:29.684760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}