{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# A Quick YOLOv7 Baseline [Inference Edition]\n\nThis is the inference notebook for this YOLOv7 [training notebook](https://www.kaggle.com/code/fnands/a-quick-yolov7-baseline).  \n\nThe inference code in the main repo [doesn't seem to actually export segmentation masks](https://github.com/WongKinYiu/yolov7/issues/1483), so I modified the inference loop.   \n","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/pycocotools/ /kaggle/working/pycocotools\n!pip install /kaggle/working/pycocotools/pycocotools-2.0.6  --no-index --find-links=/kaggle/working/pycocotools/ ","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:42:40.180335Z","iopub.execute_input":"2023-07-06T07:42:40.180627Z","iopub.status.idle":"2023-07-06T07:43:18.238813Z","shell.execute_reply.started":"2023-07-06T07:42:40.180601Z","shell.execute_reply":"2023-07-06T07:43:18.237261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import base64\nimport numpy as np\nfrom pycocotools import _mask as coco_mask\nfrom typing import Text, Dict, Tuple\nimport zlib","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:43:18.242299Z","iopub.execute_input":"2023-07-06T07:43:18.242731Z","iopub.status.idle":"2023-07-06T07:43:18.257162Z","shell.execute_reply.started":"2023-07-06T07:43:18.242689Z","shell.execute_reply":"2023-07-06T07:43:18.256002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/yolov7-weights-and-wheels/yolo_wheel/yolov7-0.0.1-py37.py38.py39-none-any.whl --no-index --find-links=/kaggle/input/yolov7-weights-and-wheels/yolo_wheel","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:43:18.259153Z","iopub.execute_input":"2023-07-06T07:43:18.259854Z","iopub.status.idle":"2023-07-06T07:43:54.138174Z","shell.execute_reply.started":"2023-07-06T07:43:18.259817Z","shell.execute_reply":"2023-07-06T07:43:54.136975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r /kaggle/input/yolov7-weights-and-wheels/yolov7 yolo","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:43:54.142052Z","iopub.execute_input":"2023-07-06T07:43:54.142483Z","iopub.status.idle":"2023-07-06T07:43:56.468043Z","shell.execute_reply.started":"2023-07-06T07:43:54.142437Z","shell.execute_reply":"2023-07-06T07:43:56.466717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\ndata_path = '/kaggle/input/hubmap-hacking-the-human-vasculature'\ntile_meta_path = os.path.join(data_path, 'tile_meta.csv')\ntile_meta = pd.read_csv(tile_meta_path)\ntile_meta_dataset_2 = tile_meta[tile_meta['dataset'] != 1]\ntile_meta_dataset_2","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:43:56.469747Z","iopub.execute_input":"2023-07-06T07:43:56.470679Z","iopub.status.idle":"2023-07-06T07:43:56.528370Z","shell.execute_reply.started":"2023-07-06T07:43:56.470616Z","shell.execute_reply":"2023-07-06T07:43:56.527016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\n# Set your directories\noriginal_folder = '/kaggle/input/hubmap-hacking-the-human-vasculature/train'\nnew_folder = '/kaggle/working/coco_datasets_2'\n\n# Ensure the new directory exists\nos.makedirs(new_folder, exist_ok=True)\n\n# Loop over the IDs in dataset 1 and copy the corresponding files\nfor image_id in tile_meta_dataset_2['id']:\n    # You may need to adjust the file extension if your images are not .jpg\n    original_path = os.path.join(original_folder, image_id + '.tif')\n    new_path = os.path.join(new_folder, image_id + '.tif')\n    \n    # Check if the file exists and then copy\n    if os.path.exists(original_path):\n        shutil.copyfile(original_path, new_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:43:56.530559Z","iopub.execute_input":"2023-07-06T07:43:56.531410Z","iopub.status.idle":"2023-07-06T07:45:35.973896Z","shell.execute_reply.started":"2023-07-06T07:43:56.531371Z","shell.execute_reply":"2023-07-06T07:45:35.972101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from yolo.seg.segment import predict","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:35.977393Z","iopub.execute_input":"2023-07-06T07:45:35.977819Z","iopub.status.idle":"2023-07-06T07:45:41.191236Z","shell.execute_reply.started":"2023-07-06T07:45:35.977780Z","shell.execute_reply":"2023-07-06T07:45:41.190194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a yaml file as expected by YOLOv7 (and others)\nyaml_text = \"\"\"\n# class names\nnames: \n  0: blood_vessel\n  1: glomerulus\n  2: unsure\n\"\"\"\nwith open('/kaggle/working/hubmap-coco.yaml', 'w') as text_file:\n    text_file.write(yaml_text)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:41.192899Z","iopub.execute_input":"2023-07-06T07:45:41.193780Z","iopub.status.idle":"2023-07-06T07:45:41.199389Z","shell.execute_reply.started":"2023-07-06T07:45:41.193745Z","shell.execute_reply":"2023-07-06T07:45:41.198410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import argparse\nimport os\nimport platform\nimport sys\nfrom pathlib import Path\nimport json\n\nimport torch\nimport torch.backends.cudnn as cudnn\n\n\n\nfrom models.common import DetectMultiBackend\nfrom utils.dataloaders import IMG_FORMATS, VID_FORMATS, LoadImages, LoadStreams\nfrom utils.general import (LOGGER, Profile, check_file, check_img_size, check_imshow, check_requirements, colorstr, cv2,\n                           increment_path, non_max_suppression, print_args, scale_coords, strip_optimizer, xyxy2xywh)\nfrom utils.plots import Annotator, colors, save_one_box\nfrom utils.segment.general import process_mask, scale_masks\nfrom utils.segment.plots import plot_masks\nfrom utils.torch_utils import select_device, smart_inference_mode","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:41.201064Z","iopub.execute_input":"2023-07-06T07:45:41.201675Z","iopub.status.idle":"2023-07-06T07:45:41.211749Z","shell.execute_reply.started":"2023-07-06T07:45:41.201641Z","shell.execute_reply":"2023-07-06T07:45:41.210733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_binary_mask(mask: np.ndarray) -> Text:\n    \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n    # check input mask --\n    if mask.dtype != np.bool:\n        raise ValueError(\n            \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n            mask.dtype)\n\n    mask = np.squeeze(mask)\n    if len(mask.shape) != 2:\n        raise ValueError(\n            \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n            mask.shape)\n\n    # convert input mask to expected COCO API input --\n    mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n    mask_to_encode = mask_to_encode.astype(np.uint8)\n    mask_to_encode = np.asfortranarray(mask_to_encode)\n\n    # RLE encode mask --\n    encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n    # compress and base64 encoding --\n    binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n    base64_str = base64.b64encode(binary_str)\n    return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:41.215608Z","iopub.execute_input":"2023-07-06T07:45:41.215916Z","iopub.status.idle":"2023-07-06T07:45:41.228802Z","shell.execute_reply.started":"2023-07-06T07:45:41.215893Z","shell.execute_reply":"2023-07-06T07:45:41.227932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read .jsonl file and convert it to a list of dicts\n# The dicts contain IDs, class names and segmentation masks\n# from https://www.kaggle.com/code/leonidkulyk/eda-hubmap-hhv-interactive-annotations\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r') as json_file:\n    json_list = list(json_file)\n    \ntiles_dicts = []\nfor json_str in json_list:\n    tiles_dicts.append(json.loads(json_str))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:41.230098Z","iopub.execute_input":"2023-07-06T07:45:41.230703Z","iopub.status.idle":"2023-07-06T07:45:46.125724Z","shell.execute_reply.started":"2023-07-06T07:45:41.230671Z","shell.execute_reply":"2023-07-06T07:45:46.124577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_of_tiles = {}\nfor tile in tiles_dicts:\n    dict_of_tiles[tile['id']] = tile['annotations']","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:46.127437Z","iopub.execute_input":"2023-07-06T07:45:46.127842Z","iopub.status.idle":"2023-07-06T07:45:46.135083Z","shell.execute_reply.started":"2023-07-06T07:45:46.127805Z","shell.execute_reply":"2023-07-06T07:45:46.133855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_glomerulus_mask(annotations: Dict, mask_shape: Tuple = (512, 512)) -> np.ndarray:\n    \"\"\" Converts glomerulus labels into boolean mask \"\"\"\n    mask = np.ones(shape=mask_shape, dtype=np.uint8)\n    \n    for annotation in annotations: \n        if annotation['type'] == 'glomerulus':            \n            coords = np.array(annotation['coordinates'])\n            cv2.fillPoly(mask, pts=coords, color=0)\n        \n\n    return mask.astype(bool)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:46.136755Z","iopub.execute_input":"2023-07-06T07:45:46.137204Z","iopub.status.idle":"2023-07-06T07:45:46.147489Z","shell.execute_reply.started":"2023-07-06T07:45:46.137166Z","shell.execute_reply":"2023-07-06T07:45:46.146499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_json(binary_mask, image_id, classification):\n    # 创建坐标列表\n    coordinates = np.where(binary_mask)\n    coordinates = [(int(x), int(y)) for x, y in zip(coordinates[0], coordinates[1])]\n    coordinates = [list(item) for item in coordinates]  # 将每个坐标转换为列表\n\n    # 创建分类字典\n    class_dict = {\n        0: \"blood_vessel\",\n        1: \"glomerulus\",\n        2: \"unsure\",\n    }\n\n    # 创建标注\n    annotation = {\n        \"type\": class_dict[int(classification)],\n        \"coordinates\": [coordinates]  # 坐标列表直接赋值\n    }\n\n    return annotation  # 返回单个annotation\n","metadata":{"execution":{"iopub.status.busy":"2023-07-06T08:53:38.315230Z","iopub.execute_input":"2023-07-06T08:53:38.316242Z","iopub.status.idle":"2023-07-06T08:53:38.324048Z","shell.execute_reply.started":"2023-07-06T08:53:38.316202Z","shell.execute_reply":"2023-07-06T08:53:38.322633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install jsonlines","metadata":{"execution":{"iopub.status.busy":"2023-07-06T07:45:46.160770Z","iopub.execute_input":"2023-07-06T07:45:46.161690Z","iopub.status.idle":"2023-07-06T07:45:58.737908Z","shell.execute_reply.started":"2023-07-06T07:45:46.161654Z","shell.execute_reply":"2023-07-06T07:45:58.736752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jsonlines\ndef segment(\n        weights,\n        source,\n        data,\n        project,\n        imgsz=(640, 640),  # inference size (height, width)\n        conf_thres=0.25,  # confidence threshold\n        iou_thres=0.45,  # NMS IOU threshold\n        max_det=1000,  # maximum detections per image\n        device='',  # cuda device, i.e. 0 or 0,1,2,3 or cpu\n        view_img=False,  # show results\n        save_txt=False,  # save results to *.txt\n        save_conf=False,  # save confidences in --save-txt labels\n        save_crop=False,  # save cropped prediction boxes\n        nosave=False,  # do not save images/videos\n        classes=None,  # filter by class: --class 0, or --class 0 2 3\n        agnostic_nms=False,  # class-agnostic NMS\n        augment=False,  # augmented inference\n        visualize=False,  # visualize features\n        update=False,  # update all models\n        name='exp',  # save results to project/name\n        exist_ok=False,  # existing project/name ok, do not increment\n        line_thickness=3,  # bounding box thickness (pixels)\n        hide_labels=False,  # hide labels\n        hide_conf=False,  # hide confidences\n        half=False,  # use FP16 half-precision inference\n        dnn=False,  # use OpenCV DNN for ONNX inference\n    ):\n    json_data_list = []\n    with jsonlines.open('/kaggle/working/polygons_2.jsonl', mode='w') as writer:\n        device = select_device(device)\n        model = DetectMultiBackend(weights, device=device, dnn=dnn, data=data, fp16=half)\n        stride, names, pt = model.stride, model.names, model.pt\n        imgsz = check_img_size(imgsz, s=stride)  # check image size\n\n        dataset = LoadImages(source, img_size=imgsz, stride=stride, auto=pt)\n        bs = 1  # batch_size\n\n        model.warmup(imgsz=(1 if pt else bs, 3, *imgsz))  # warmup\n        seen, windows, dt = 0, [], (Profile(), Profile(), Profile())\n        for path, im, im0s, vid_cap, s in dataset: \n            # Write id and size\n            image_id = Path(path).stem\n            \n            if image_id in dict_of_tiles:\n                annotations = dict_of_tiles[image_id]\n                glomerulus_mask = get_glomerulus_mask(annotations)\n            else:\n                annotations = []\n                glomerulus_mask = np.ones(shape=imgsz).astype(bool)\n                \n                \n            with dt[0]:\n                im = torch.from_numpy(im).to(device)\n                im = im.half() if model.fp16 else im.float()  # uint8 to fp16/32\n                im /= 255  # 0 - 255 to 0.0 - 1.0\n                if len(im.shape) == 3:\n                    im = im[None]  # expand for batch dim\n\n            # Inference\n            with dt[1]:\n                visualize = increment_path(save_dir / Path(path).stem, mkdir=True) if visualize else False\n                pred, out = model(im, augment=augment, visualize=visualize)\n                proto = out[1]\n\n            # NMS\n            with dt[2]:\n                pred = non_max_suppression(pred, conf_thres, iou_thres, classes, agnostic_nms, max_det=max_det, nm=32)\n\n            for i, det in enumerate(pred):  # per image\n                seen += 1\n\n                if len(det):\n                    masks = process_mask(proto[i], det[:, 6:], det[:, :4], im.shape[2:], upsample=True)  # HWC\n                    confs = det[:, 4]\n                    clasf = det[:, 5]\n                    \n                    annotation_list = []\n                    threshold=0.6\n                    for mask, confidence, classification in zip(masks, confs, clasf):\n                        if confidence < threshold:\n                            continue\n                        binary_mask = mask.cpu().numpy()\n                        binary_mask = cv2.resize(binary_mask, (512, 512))\n                        kernel = np.ones(shape=(3, 3), dtype=np.uint8)\n                        binary_mask = cv2.dilate(binary_mask, kernel, 3)\n                        glomerulus_mask = glomerulus_mask.astype(np.uint8)\n                        glomerulus_mask=cv2.resize(glomerulus_mask,(512, 512))\n                        glomerulus_mask = glomerulus_mask.astype(bool)\n                        binary_mask = binary_mask.astype(bool)\n                        binary_mask = binary_mask & glomerulus_mask\n                        annotation = create_json(binary_mask, image_id, classification.item())  # 创建annotation\n                        annotation_list.append(annotation)\n                        \n                    json_data = {\n                        \"id\": image_id,\n                        \"annotations\": annotation_list  # 如果有更多的标注，它们可以作为额外的元素添加到这个列表中\n                    }\n                    writer.write(json_data)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T08:55:02.284663Z","iopub.execute_input":"2023-07-06T08:55:02.285624Z","iopub.status.idle":"2023-07-06T08:55:02.322126Z","shell.execute_reply.started":"2023-07-06T08:55:02.285575Z","shell.execute_reply":"2023-07-06T08:55:02.321104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segment(source='/kaggle/working/coco_datasets_2',\n            data='/kaggle/working/hubmap-coco.yaml',\n            imgsz=(736, 736), \n            classes=0,\n            weights='/kaggle/input/dataset1/best_dataset1.pt',\n            name='yolov7-predict',\n            project='yolov7-predict',\n            exist_ok=True,\n            nosave=True,\n            save_txt=True,\n            view_img=True,\n            )","metadata":{"execution":{"iopub.status.busy":"2023-07-06T08:14:08.703258Z","iopub.execute_input":"2023-07-06T08:14:08.703708Z","iopub.status.idle":"2023-07-06T08:23:47.264709Z","shell.execute_reply.started":"2023-07-06T08:14:08.703673Z","shell.execute_reply":"2023-07-06T08:23:47.263394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-06T07:24:04.838939Z","iopub.execute_input":"2023-07-06T07:24:04.839832Z","iopub.status.idle":"2023-07-06T07:24:05.848234Z","shell.execute_reply.started":"2023-07-06T07:24:04.839797Z","shell.execute_reply":"2023-07-06T07:24:05.846953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}