{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nsys.path.append(\"/kaggle/input/yolov8/ultralytics-main\")","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:07:29.865444Z","iopub.execute_input":"2023-07-31T12:07:29.866215Z","iopub.status.idle":"2023-07-31T12:07:29.878111Z","shell.execute_reply.started":"2023-07-31T12:07:29.866181Z","shell.execute_reply":"2023-07-31T12:07:29.876897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from dataclasses import dataclass\n\nimport albumentations as A\n\nimport warnings\n\nimport shutil\nimport os\nimport torch\nimport pandas as pd\nimport numpy as np\nimport tifffile as tiff\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\nfrom pathlib import Path\nfrom glob import glob\nimport gc\nfrom collections import defaultdict\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nfrom IPython.display import Image as show_image\n\nimport ultralytics\nfrom ultralytics import YOLO\nimport ultralytics\nimport torch\n\nultralytics.checks()\n\nfrom skimage.morphology import binary_dilation","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:07:29.880337Z","iopub.execute_input":"2023-07-31T12:07:29.880814Z","iopub.status.idle":"2023-07-31T12:07:47.589302Z","shell.execute_reply.started":"2023-07-31T12:07:29.880778Z","shell.execute_reply":"2023-07-31T12:07:47.588419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# utils","metadata":{}},{"cell_type":"code","source":"def set_seed(seed=42):\n    import random\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\nset_seed()","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:07:47.590985Z","iopub.execute_input":"2023-07-31T12:07:47.591692Z","iopub.status.idle":"2023-07-31T12:07:47.602252Z","shell.execute_reply.started":"2023-07-31T12:07:47.591656Z","shell.execute_reply":"2023-07-31T12:07:47.601411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nHOME = os.getcwd()\n!mkdir /kaggle/working/packages\n!cp -r /kaggle/input/pycocotools/* /kaggle/working/packages\nos.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n!python setup.py install\n!pip install . --no-index --find-links /kaggle/working/packages/\nos.chdir(\"/kaggle/working\")\n\nimport base64\nimport numpy as np\nimport pycocotools\nimport pycocotools.mask\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\nfrom PIL import Image\nimport cv2\nimport pandas as pd\nimport os\nfrom itertools import groupby\nfrom skimage.measure import label, regionprops\n\ndef encode_binary_mask(mask: np.ndarray) -> t.Text:\n    \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n    # check input mask --\n    if mask.dtype != bool:\n        raise ValueError(\n            \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n            mask.dtype)\n\n    mask = np.squeeze(mask)\n    if len(mask.shape) != 2:\n        raise ValueError(\n            \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n            mask.shape)\n\n    # convert input mask to expected COCO API input --\n    mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n    mask_to_encode = mask_to_encode.astype(np.uint8)\n    mask_to_encode = np.asfortranarray(mask_to_encode)\n\n    # RLE encode mask --\n    encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n    # compress and base64 encoding --\n    binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n    base64_str = base64.b64encode(binary_str)\n    return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:07:47.606365Z","iopub.execute_input":"2023-07-31T12:07:47.606619Z","iopub.status.idle":"2023-07-31T12:08:42.876294Z","shell.execute_reply.started":"2023-07-31T12:07:47.606596Z","shell.execute_reply":"2023-07-31T12:08:42.875159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble","metadata":{}},{"cell_type":"code","source":"%%writefile wbf_tracking.py\n# coding: utf-8\n\n# __author__ = 'ZFTurbo: https://kaggle.com/zfturbo'\n# Modified by sir-timio : https://www.kaggle.com/sirtimio\n\nimport sys\nimport warnings\nimport numpy as np\nfrom numba import jit\n\n@jit(nopython=True)\ndef bb_intersection_over_union(A, B) -> float:\n    xA = max(A[0], B[0])\n    yA = max(A[1], B[1])\n    xB = min(A[2], B[2])\n    yB = min(A[3], B[3])\n\n    # compute the area of intersection rectangle\n    interArea = max(0, xB - xA) * max(0, yB - yA)\n\n    if interArea == 0:\n        return 0.0\n\n    # compute the area of both the prediction and ground-truth rectangles\n    boxAArea = (A[2] - A[0]) * (A[3] - A[1])\n    boxBArea = (B[2] - B[0]) * (B[3] - B[1])\n\n    iou = interArea / float(boxAArea + boxBArea - interArea)\n    return iou\n\n\ndef prefilter_boxes(boxes, scores, labels, weights, thr):\n    # Create dict with boxes stored by its label\n    new_boxes = dict()\n\n    for t in range(len(boxes)):\n\n        if len(boxes[t]) != len(scores[t]):\n            print('Error. Length of boxes arrays not equal to length of scores array: {} != {}'.format(len(boxes[t]), len(scores[t])))\n            sys.exit()\n\n        if len(boxes[t]) != len(labels[t]):\n            print('Error. Length of boxes arrays not equal to length of labels array: {} != {}'.format(len(boxes[t]), len(labels[t])))\n            sys.exit()\n\n        for j in range(len(boxes[t])):\n            score = scores[t][j]\n            if score < thr:\n                continue\n            label = int(labels[t][j])\n            box_part = boxes[t][j]\n            x1 = max(float(box_part[0]), 0.)\n            y1 = max(float(box_part[1]), 0.)\n            x2 = max(float(box_part[2]), 0.)\n            y2 = max(float(box_part[3]), 0.)\n\n            # Box data checks\n            if x2 < x1:\n                warnings.warn('X2 < X1 value in box. Swap them.')\n                x1, x2 = x2, x1\n            if y2 < y1:\n                warnings.warn('Y2 < Y1 value in box. Swap them.')\n                y1, y2 = y2, y1\n            if x1 > 1:\n                warnings.warn('X1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x1 = 1\n            if x2 > 1:\n                warnings.warn('X2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x2 = 1\n            if y1 > 1:\n                warnings.warn('Y1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y1 = 1\n            if y2 > 1:\n                warnings.warn('Y2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y2 = 1\n            if (x2 - x1) * (y2 - y1) == 0.0:\n                warnings.warn(\"Zero area box skipped: {}.\".format(box_part))\n                continue\n\n            # [label, score, weight, model index, x1, y1, x2, y2]\n            b = [int(label), float(score) * weights[t], weights[t], t, x1, y1, x2, y2]\n            if label not in new_boxes:\n                new_boxes[label] = []\n            new_boxes[label].append(b)\n\n    # Sort each list in dict by score and transform it to numpy array\n    for k in new_boxes:\n        current_boxes = np.array(new_boxes[k])\n        new_boxes[k] = current_boxes[current_boxes[:, 1].argsort()[::-1]]\n\n    return new_boxes\n\n\ndef get_weighted_box(boxes, conf_type='avg'):\n    \"\"\"\n    Create weighted box for set of boxes\n    :param boxes: set of boxes to fuse\n    :param conf_type: type of confidence one of 'avg' or 'max'\n    :return: weighted box (label, score, weight, x1, y1, x2, y2)\n    \"\"\"\n\n    box = np.zeros(8, dtype=np.float32)\n    conf = 0\n    conf_list = []\n    w = 0\n    for b in boxes:\n        box[4:] += (b[1] * b[4:])\n        conf += b[1]\n        conf_list.append(b[1])\n        w += b[2]\n    box[0] = boxes[0][0]\n    if conf_type == 'avg':\n        box[1] = conf / len(boxes)\n    elif conf_type == 'max':\n        box[1] = np.array(conf_list).max()\n    elif conf_type in ['box_and_model_avg', 'absent_model_aware_avg']:\n        box[1] = conf / len(boxes)\n    box[2] = w\n    box[3] = -1 # model index field is retained for consistensy but is not used.\n    box[4:] /= conf\n    return box\n\n\ndef find_matching_box(boxes_list, new_box, match_iou):\n    best_iou = match_iou\n    best_index = -1\n    for i in range(len(boxes_list)):\n        box = boxes_list[i]\n        if box[0] != new_box[0]:\n            continue\n        iou = bb_intersection_over_union(box[4:], new_box[4:])\n        if iou > best_iou:\n            best_index = i\n            best_iou = iou\n\n    return best_index, best_iou\n\n\n\ndef weighted_boxes_fusion_tracking(boxes_list, scores_list, labels_list, weights=None, iou_thr=0.55, skip_box_thr=0.0, conf_type='avg', allows_overflow=False):\n    '''\n    :param boxes_list: list of boxes predictions from each model, each box is 4 numbers.\n    It has 3 dimensions (models_number, model_preds, 4)\n    Order of boxes: x1, y1, x2, y2. We expect float normalized coordinates [0; 1]\n    :param scores_list: list of scores for each model\n    :param labels_list: list of labels for each model\n    :param weights: list of weights for each model. Default: None, which means weight == 1 for each model\n    :param iou_thr: IoU value for boxes to be a match\n    :param skip_box_thr: exclude boxes with score lower than this variable\n    :param conf_type: how to calculate confidence in weighted boxes. 'avg': average value, 'max': maximum value, 'box_and_model_avg': box and model wise hybrid weighted average, 'absent_model_aware_avg': weighted average that takes into account the absent model.\n    :param allows_overflow: false if we want confidence score not exceed 1.0\n\n    :return: boxes: boxes coordinates (Order of boxes: x1, y1, x2, y2).\n    :return: scores: confidence scores\n    :return: labels: boxes labels\n    :return: wbfo: original boxes coordinates for each fused box\n    '''\n\n    if weights is None:\n        weights = np.ones(len(boxes_list))\n    if len(weights) != len(boxes_list):\n        print('Warning: incorrect number of weights {}. Must be: {}. Set weights equal to 1.'.format(len(weights), len(boxes_list)))\n        weights = np.ones(len(boxes_list))\n    weights = np.array(weights)\n\n    if conf_type not in ['avg', 'max', 'box_and_model_avg', 'absent_model_aware_avg']:\n        print('Unknown conf_type: {}. Must be \"avg\", \"max\" or \"box_and_model_avg\", or \"absent_model_aware_avg\"'.format(conf_type))\n        sys.exit()\n\n    filtered_boxes = prefilter_boxes(boxes_list, scores_list, labels_list, weights, skip_box_thr)\n    if len(filtered_boxes) == 0:\n        return np.zeros((0, 4)), np.zeros((0,)), np.zeros((0,)), np.zeros((0, 4))\n    \n    overall_boxes = []\n    original_boxes = []\n    for label in filtered_boxes:\n        boxes = filtered_boxes[label]\n        new_boxes = []\n        weighted_boxes = []\n        # Clusterize boxes\n        for j in range(0, len(boxes)):\n            index, best_iou = find_matching_box(weighted_boxes, boxes[j], iou_thr)\n            if index != -1:\n                new_boxes[index].append(boxes[j])\n                weighted_boxes[index] = get_weighted_box(new_boxes[index], conf_type)\n            else:\n                new_boxes.append([boxes[j].copy()])\n                weighted_boxes.append(boxes[j].copy())\n        # Rescale confidence based on number of models and boxes\n        original_boxes.append(new_boxes)\n        for i in range(len(new_boxes)):\n            clustered_boxes = np.array(new_boxes[i])\n            if conf_type == 'box_and_model_avg':\n                # weighted average for boxes\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weighted_boxes[i][2]\n                # identify unique model index by model index column\n                _, idx = np.unique(clustered_boxes[:, 3], return_index=True)\n                # rescale by unique model weights\n                weighted_boxes[i][1] = weighted_boxes[i][1] *  clustered_boxes[idx, 2].sum() / weights.sum()\n            elif conf_type == 'absent_model_aware_avg':\n                # get unique model index in the cluster\n                models = np.unique(clustered_boxes[:, 3]).astype(int)\n                # create a mask to get unused model weights\n                mask = np.ones(len(weights), dtype=bool)\n                mask[models] = False\n                # absent model aware weighted average\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / (weighted_boxes[i][2] + weights[mask].sum())\n            elif conf_type == 'max':\n                weighted_boxes[i][1] = weighted_boxes[i][1] / weights.max()\n            elif not allows_overflow:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * min(len(weights), len(clustered_boxes)) / weights.sum()\n            else:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weights.sum()\n        overall_boxes.append(np.array(weighted_boxes))\n    overall_boxes = np.concatenate(overall_boxes, axis=0)\n    sidx = overall_boxes[:, 1].argsort()\n    overall_boxes = overall_boxes[sidx[::-1]]\n    boxes = overall_boxes[:, 4:]\n    scores = overall_boxes[:, 1]\n    labels = overall_boxes[:, 0]\n    # sort originals accoring to wbf\n    original_boxes = original_boxes[0]\n    wbfo = [original_boxes[i] for i in sidx[::-1]]\n    return boxes, scores, labels, wbfo\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-31T12:08:42.879618Z","iopub.execute_input":"2023-07-31T12:08:42.880273Z","iopub.status.idle":"2023-07-31T12:08:42.897105Z","shell.execute_reply.started":"2023-07-31T12:08:42.880236Z","shell.execute_reply":"2023-07-31T12:08:42.895354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def combine_results_nmms(masks, boxes, scores, iou_nms=0.5):\n    # assert list(np.argsort(boxes[:, 4])[::-1]) == list(range(len(boxes)))\n    order = np.argsort(scores)[::-1]\n    masks = masks[order]\n    boxes = boxes[order]\n\n    rle_pred = [pycocotools.mask.encode(np.asarray(m, order='F')) for m in masks]\n    ious = pycocotools.mask.iou(rle_pred, rle_pred, [0] * len(rle_pred))\n\n    picks = []\n    idxs = list(range(len(ious)))\n    # removed = []\n\n    while len(idxs) > 0:\n        idx = idxs[0]\n        overlapping = np.where(ious[idx] > iou_nms)[0]\n\n        # removed += [v for v in overlapping if v > idx]\n\n        if len(overlapping):\n            picks.append(idx)\n            idxs = [i for i in idxs if i not in overlapping]\n        else:\n            idxs = idxs[1:]\n\n    masks = masks[picks]\n    boxes = boxes[picks]\n    return masks, boxes, scores[picks]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-31T12:08:42.898566Z","iopub.execute_input":"2023-07-31T12:08:42.899521Z","iopub.status.idle":"2023-07-31T12:08:42.919570Z","shell.execute_reply.started":"2023-07-31T12:08:42.899489Z","shell.execute_reply":"2023-07-31T12:08:42.918437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile wbf_tracking.py\n\n# coding: utf-8\n__author__ = 'ZFTurbo: https://kaggle.com/zfturbo'\n# Modified by Mista G: https://www.kaggle.com/mistag\n\nimport sys\nimport warnings\nimport numpy as np\nfrom numba import jit\n\n@jit(nopython=True)\ndef bb_intersection_over_union(A, B) -> float:\n    xA = max(A[0], B[0])\n    yA = max(A[1], B[1])\n    xB = min(A[2], B[2])\n    yB = min(A[3], B[3])\n\n    # compute the area of intersection rectangle\n    interArea = max(0, xB - xA) * max(0, yB - yA)\n\n    if interArea == 0:\n        return 0.0\n\n    # compute the area of both the prediction and ground-truth rectangles\n    boxAArea = (A[2] - A[0]) * (A[3] - A[1])\n    boxBArea = (B[2] - B[0]) * (B[3] - B[1])\n\n    iou = interArea / float(boxAArea + boxBArea - interArea)\n    return iou\n\n\ndef prefilter_boxes(boxes, scores, labels, weights, thr):\n    # Create dict with boxes stored by its label\n    new_boxes = dict()\n\n    for t in range(len(boxes)):\n\n        if len(boxes[t]) != len(scores[t]):\n            print('Error. Length of boxes arrays not equal to length of scores array: {} != {}'.format(len(boxes[t]), len(scores[t])))\n            sys.exit()\n\n        if len(boxes[t]) != len(labels[t]):\n            print('Error. Length of boxes arrays not equal to length of labels array: {} != {}'.format(len(boxes[t]), len(labels[t])))\n            sys.exit()\n\n        for j in range(len(boxes[t])):\n            score = scores[t][j]\n            if score < thr:\n                continue\n            label = int(labels[t][j])\n            box_part = boxes[t][j]\n            x1 = max(float(box_part[0]), 0.)\n            y1 = max(float(box_part[1]), 0.)\n            x2 = max(float(box_part[2]), 0.)\n            y2 = max(float(box_part[3]), 0.)\n\n            # Box data checks\n            if x2 < x1:\n                warnings.warn('X2 < X1 value in box. Swap them.')\n                x1, x2 = x2, x1\n            if y2 < y1:\n                warnings.warn('Y2 < Y1 value in box. Swap them.')\n                y1, y2 = y2, y1\n            if x1 > 1:\n                warnings.warn('X1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x1 = 1\n            if x2 > 1:\n                warnings.warn('X2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x2 = 1\n            if y1 > 1:\n                warnings.warn('Y1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y1 = 1\n            if y2 > 1:\n                warnings.warn('Y2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y2 = 1\n            if (x2 - x1) * (y2 - y1) == 0.0:\n                warnings.warn(\"Zero area box skipped: {}.\".format(box_part))\n                continue\n\n            # [label, score, weight, model index, x1, y1, x2, y2]\n            b = [int(label), float(score) * weights[t], weights[t], t, x1, y1, x2, y2]\n            if label not in new_boxes:\n                new_boxes[label] = []\n            new_boxes[label].append(b)\n\n    # Sort each list in dict by score and transform it to numpy array\n    for k in new_boxes:\n        current_boxes = np.array(new_boxes[k])\n        new_boxes[k] = current_boxes[current_boxes[:, 1].argsort()[::-1]]\n\n    return new_boxes\n\n\ndef get_weighted_box(boxes, conf_type='avg'):\n    \"\"\"\n    Create weighted box for set of boxes\n    :param boxes: set of boxes to fuse\n    :param conf_type: type of confidence one of 'avg' or 'max'\n    :return: weighted box (label, score, weight, x1, y1, x2, y2)\n    \"\"\"\n\n    box = np.zeros(8, dtype=np.float32)\n    conf = 0\n    conf_list = []\n    w = 0\n    for b in boxes:\n        box[4:] += (b[1] * b[4:])\n        conf += b[1]\n        conf_list.append(b[1])\n        w += b[2]\n    box[0] = boxes[0][0]\n    if conf_type == 'avg':\n        box[1] = conf / len(boxes)\n    elif conf_type == 'max':\n        box[1] = np.array(conf_list).max()\n    elif conf_type in ['box_and_model_avg', 'absent_model_aware_avg']:\n        box[1] = conf / len(boxes)\n    box[2] = w\n    box[3] = -1 # model index field is retained for consistensy but is not used.\n    box[4:] /= conf\n    return box\n\n\ndef find_matching_box(boxes_list, new_box, match_iou):\n    best_iou = match_iou\n    best_index = -1\n    for i in range(len(boxes_list)):\n        box = boxes_list[i]\n        if box[0] != new_box[0]:\n            continue\n        iou = bb_intersection_over_union(box[4:], new_box[4:])\n        if iou > best_iou:\n            best_index = i\n            best_iou = iou\n\n    return best_index, best_iou\n\n\n\ndef weighted_boxes_fusion_tracking(boxes_list, scores_list, labels_list, weights=None, iou_thr=0.55, skip_box_thr=0.0, conf_type='avg', allows_overflow=False):\n    '''\n    :param boxes_list: list of boxes predictions from each model, each box is 4 numbers.\n    It has 3 dimensions (models_number, model_preds, 4)\n    Order of boxes: x1, y1, x2, y2. We expect float normalized coordinates [0; 1]\n    :param scores_list: list of scores for each model\n    :param labels_list: list of labels for each model\n    :param weights: list of weights for each model. Default: None, which means weight == 1 for each model\n    :param iou_thr: IoU value for boxes to be a match\n    :param skip_box_thr: exclude boxes with score lower than this variable\n    :param conf_type: how to calculate confidence in weighted boxes. 'avg': average value, 'max': maximum value, 'box_and_model_avg': box and model wise hybrid weighted average, 'absent_model_aware_avg': weighted average that takes into account the absent model.\n    :param allows_overflow: false if we want confidence score not exceed 1.0\n\n    :return: boxes: boxes coordinates (Order of boxes: x1, y1, x2, y2).\n    :return: scores: confidence scores\n    :return: labels: boxes labels\n    :return: wbfo: original boxes coordinates for each fused box\n    '''\n\n    if weights is None:\n        weights = np.ones(len(boxes_list))\n    if len(weights) != len(boxes_list):\n        print('Warning: incorrect number of weights {}. Must be: {}. Set weights equal to 1.'.format(len(weights), len(boxes_list)))\n        weights = np.ones(len(boxes_list))\n    weights = np.array(weights)\n\n    if conf_type not in ['avg', 'max', 'box_and_model_avg', 'absent_model_aware_avg']:\n        print('Unknown conf_type: {}. Must be \"avg\", \"max\" or \"box_and_model_avg\", or \"absent_model_aware_avg\"'.format(conf_type))\n        sys.exit()\n\n    filtered_boxes = prefilter_boxes(boxes_list, scores_list, labels_list, weights, skip_box_thr)\n    if len(filtered_boxes) == 0:\n        return np.zeros((0, 4)), np.zeros((0,)), np.zeros((0,)), np.zeros((0, 4))\n    \n    overall_boxes = []\n    original_boxes = []\n    for label in filtered_boxes:\n        boxes = filtered_boxes[label]\n        new_boxes = []\n        weighted_boxes = []\n        # Clusterize boxes\n        for j in range(0, len(boxes)):\n            index, best_iou = find_matching_box(weighted_boxes, boxes[j], iou_thr)\n            if index != -1:\n                new_boxes[index].append(boxes[j])\n                weighted_boxes[index] = get_weighted_box(new_boxes[index], conf_type)\n            else:\n                new_boxes.append([boxes[j].copy()])\n                weighted_boxes.append(boxes[j].copy())\n        # Rescale confidence based on number of models and boxes\n        original_boxes.append(new_boxes)\n        for i in range(len(new_boxes)):\n            clustered_boxes = np.array(new_boxes[i])\n            if conf_type == 'box_and_model_avg':\n                # weighted average for boxes\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weighted_boxes[i][2]\n                # identify unique model index by model index column\n                _, idx = np.unique(clustered_boxes[:, 3], return_index=True)\n                # rescale by unique model weights\n                weighted_boxes[i][1] = weighted_boxes[i][1] *  clustered_boxes[idx, 2].sum() / weights.sum()\n            elif conf_type == 'absent_model_aware_avg':\n                # get unique model index in the cluster\n                models = np.unique(clustered_boxes[:, 3]).astype(int)\n                # create a mask to get unused model weights\n                mask = np.ones(len(weights), dtype=bool)\n                mask[models] = False\n                # absent model aware weighted average\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / (weighted_boxes[i][2] + weights[mask].sum())\n            elif conf_type == 'max':\n                weighted_boxes[i][1] = weighted_boxes[i][1] / weights.max()\n            elif not allows_overflow:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * min(len(weights), len(clustered_boxes)) / weights.sum()\n            else:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weights.sum()\n        overall_boxes.append(np.array(weighted_boxes))\n    overall_boxes = np.concatenate(overall_boxes, axis=0)\n    sidx = overall_boxes[:, 1].argsort()\n    overall_boxes = overall_boxes[sidx[::-1]]\n    boxes = overall_boxes[:, 4:]\n    scores = overall_boxes[:, 1]\n    labels = overall_boxes[:, 0]\n    # sort originals accoring to wbf\n    original_boxes = original_boxes[0]\n    wbfo = [original_boxes[i] for i in sidx[::-1]]\n    return boxes, scores, labels, wbfo","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-31T12:08:42.921098Z","iopub.execute_input":"2023-07-31T12:08:42.921630Z","iopub.status.idle":"2023-07-31T12:08:42.941089Z","shell.execute_reply.started":"2023-07-31T12:08:42.921598Z","shell.execute_reply":"2023-07-31T12:08:42.940074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wbf_tracking import weighted_boxes_fusion_tracking as wbf_tracking\n\ndef bbox_to_key(bbox):\n    return ','.join(map(str, np.round(bbox, 4)))\n    \n\ndef combine_results_wbf(masks, boxes, scores, iou_thr=0.55, skip_box_thr=0.0, label_thr=0, min_votes=2, cut_by_box=False, weights=None):\n    bbox_to_idx = {}\n    h, w = masks.shape[-2:]\n    \n    for i, (mask, box) in enumerate(zip(masks, boxes)):\n        bbox_to_idx[bbox_to_key(box)] = i\n    \n    labels = np.ones(len(boxes))  \n    wbf_boxes, _, _, wbf_ogs = wbf_tracking(\n        [boxes],\n        [scores],\n        labels_list=[labels],\n        iou_thr=iou_thr,\n        skip_box_thr=skip_box_thr,\n        weights=weights,\n    )\n\n    wbf_masks, wbf_scores = [], []\n    for i, wbf_box in enumerate(wbf_boxes):\n        \n        if cut_by_box:\n            filter_box = np.zeros((h, w), dtype=np.uint8)\n\n            x1 = max(0, int(h * wbf_box[0]))\n            y1 = max(0, int(w * wbf_box[1]))\n            x2 = min(h, int(h * wbf_box[2]))\n            y2 = min(w, int(w * wbf_box[3]))\n            filter_box[y1:y2, x1:x2] = 1\n        \n        keep = []\n        skiped = 0\n        for og_box in wbf_ogs[i]:\n            key = bbox_to_key(og_box[4:])\n            if key in bbox_to_idx:\n                keep.append(bbox_to_idx[key])\n            else:\n                skiped += 1\n                continue\n        if len(keep) < min_votes-skiped:\n            wbf_mask = np.zeros((h, w), dtype=np.uint8)\n            wbf_score = 0\n        else:\n            wbf_mask = (np.mean(masks[keep], axis=0) > label_thr).astype(bool)\n            if cut_by_box:\n                wbf_mask = wbf_mask & filter_box # remove pixels outside wbf\n            wbf_score = np.mean(scores[keep])\n\n        wbf_masks.append(wbf_mask)\n        wbf_scores.append(wbf_score)\n    \n    wbf_masks = np.stack(wbf_masks)\n    wbf_scores = np.array(wbf_scores) \n    return wbf_masks, wbf_boxes, wbf_scores","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-07-31T12:08:42.942737Z","iopub.execute_input":"2023-07-31T12:08:42.943572Z","iopub.status.idle":"2023-07-31T12:08:43.603693Z","shell.execute_reply.started":"2023-07-31T12:08:42.943540Z","shell.execute_reply":"2023-07-31T12:08:43.602773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# postprocessing","metadata":{}},{"cell_type":"code","source":"def remove_overlap_naive(masks, ious=None):\n    if ious is None:\n        rles = [pycocotools.mask.encode(np.asarray(m, order='F')) for m in masks]\n        ious = pycocotools.mask.iou(rles, rles, [0] * len(rles))\n\n    if not len(ious):\n        return masks\n    \n    for i in range(len(ious)):\n        ious[i, i] = 0\n    to_process = np.where(ious.sum(0) > 0)[0]\n\n    if not len(to_process):\n        return masks\n\n    masks = torch.from_numpy(masks).cuda()\n    overlapping_masks = masks[to_process]\n\n    for idx, i in enumerate(to_process):\n        if idx == 0:\n            continue\n        others = overlapping_masks[:idx].max(0)[0]\n        masks[i] *= ~others\n\n    return masks.cpu().numpy()\n\ndef remove_small_masks(masks, boxes, min_size=0):\n    if min_size == 0 or len(masks) == 0:\n        return masks, boxes\n\n    sizes = masks.sum(-1).sum(-1) / 512**2\n    to_keep = sizes > min_size\n\n    if to_keep.min() == 1:\n        return masks, boxes\n\n    smallest = sizes.min()\n    to_keep = sizes > smallest\n\n    return masks[to_keep], boxes[to_keep]\n\ndef degrade_mask(mask):\n    cont, hier = cv2.findContours(mask.astype(np.uint8), cv2.RETR_TREE, cv2.CHAIN_APPROX_SIMPLE)\n\n    img_cont = np.zeros((mask.shape[0], mask.shape[1], 3), dtype=np.uint8)\n    img_cont = cv2.drawContours(img_cont, cont, -1, (255, 255, 255), 1)\n    img_cont = img_cont[:, :, 0]\n\n    conv_mask = np.zeros((mask.shape[0], mask.shape[1], 3), dtype=np.uint8)\n\n    for c in cont:\n        conv_mask = cv2.fillConvexPoly(conv_mask, points=c, color=(1, 1, 1))\n    conv_mask = conv_mask[:, :, 0].astype(mask.dtype)\n\n    return conv_mask, img_cont\n\ndef postprocess_masks(masks, boxes,\n                      iou_nms=0.6,\n                      conf_thresh=0,\n                      min_size=0,\n                      dilation_n_iter=1,\n                      remove_overlap=True,\n                      corrupt=False,\n                     ):\n\n    # Sort by decreasing conf\n    order = np.argsort(boxes[:, 4])[::-1]\n    masks = masks[order]\n    boxes = boxes[order]\n\n    # Remove low confidence\n    last = (\n        np.argmax(boxes[:, 4] < conf_thresh) if np.min(boxes[:, 4]) < conf_thresh\n        else len(boxes)\n    )\n    masks = masks[:last]\n    boxes = boxes[:last]\n\n    # Remove small masks\n    if min_size:\n        masks, boxes = remove_small_masks(masks, boxes, min_size=min_size)\n       \n    # Corrupt\n    if corrupt:\n        masks = np.array([degrade_mask(mask)[0] for mask in masks])\n\n    # Remove overlap\n    if remove_overlap:\n        masks = remove_overlap_naive(masks)\n    \n    # Dilate\n    if dilation_n_iter != 0:\n        for i in range(len(masks)):\n            for _ in range(dilation_n_iter):\n                masks[i] = binary_dilation(masks[i])\n\n    return masks, boxes","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:08:43.604905Z","iopub.execute_input":"2023-07-31T12:08:43.605269Z","iopub.status.idle":"2023-07-31T12:08:43.626477Z","shell.execute_reply.started":"2023-07-31T12:08:43.605234Z","shell.execute_reply":"2023-07-31T12:08:43.624093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:08:43.630198Z","iopub.execute_input":"2023-07-31T12:08:43.630604Z","iopub.status.idle":"2023-07-31T12:08:43.643040Z","shell.execute_reply.started":"2023-07-31T12:08:43.630578Z","shell.execute_reply":"2023-07-31T12:08:43.642128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@dataclass\nclass CFG:\n    # inference\n    conf: float = 0.01\n    imgsz: int = (512, 512)\n    retina_masks: bool = True\n    iou_nms: float = 0.5\n#     transforms: tuple = (None,)\n    transforms: tuple = (None, A.HorizontalFlip(p=1.0), A.VerticalFlip(p=1.0), A.Rotate(limit=(180,180), p=1.0))\n        \n    \n    # ensemble\n    method: str = 'wbf' # 'nms' or 'wbf'\n    iou_ensemble: float = 0.5\n\n    wbf_min_votes: int = 2\n    wbf_label_thr: float = 0.6\n    wbf_cut_by_box: bool = 1\n    # postproccess\n    conf_thresh: float = 0.0\n    min_size: float = 0.001\n    dilation_n_iter: int = 1\n    remove_overlap: bool = False\n    corrupt: bool = False\n        \n    # models\n    models = [\n        '/kaggle/input/models-hubmap-vasculative/mskf_yolov8x-seg-fold0.pt',\n        '/kaggle/input/models-hubmap-vasculative/mskf_yolov8x-seg-fold1.pt',\n        '/kaggle/input/models-hubmap-vasculative/mskf_yolov8x-seg-fold2.pt',\n        '/kaggle/input/models-hubmap-vasculative/mskf_yolov8x-seg-fold3.pt',\n#         '/kaggle/input/models-hubmap-vasculative/mskf_yolov8x-seg-fold4.pt',\n        \n    ]\n\n    def __repr__(self):\n        params = '\\n'.join(f'{k}={v}' for k, v in self.__dict__.items())\n        return params","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:40:13.567582Z","iopub.execute_input":"2023-07-31T12:40:13.567997Z","iopub.status.idle":"2023-07-31T12:40:13.578454Z","shell.execute_reply.started":"2023-07-31T12:40:13.567959Z","shell.execute_reply":"2023-07-31T12:40:13.577519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(\n    image, # image in bgr\n    models,\n    transforms=[None],\n    imgsz=512,\n    conf=0.01,\n    iou_nms=0.5,\n    retina_masks=True,\n):\n    boxes, masks, scores = [], [], []\n\n    og_img = np.array(image.copy()).astype(np.uint8)\n    h, w = og_img.shape[:2]\n\n    for model in models:\n        for transform in transforms:\n            if transform is not None:\n                img = transform(image=og_img)[\"image\"]\n            else:\n                img = og_img.copy()\n\n            pred = model.predict(\n                img, imgsz=imgsz, conf=conf, iou=iou_nms, retina_masks=retina_masks, verbose=False, \n            )[0]\n\n            if pred.masks is None:\n                continue\n\n            pred_masks = pred.masks.data[pred.boxes.cls == 0].detach().cpu().numpy()\n\n            if len(pred_masks) == 0:\n                continue\n            pred_confs = pred.boxes.conf[pred.boxes.cls == 0].detach().cpu().numpy()\n            pred_boxes = pred.boxes.xyxyn[pred.boxes.cls == 0].detach().cpu().numpy()\n\n            if transform is not None:\n                out = transform(image=img, bboxes=pred_boxes, masks=pred_masks)\n                pred_boxes = out[\"bboxes\"]\n                pred_masks  = out[\"masks\"]\n\n            boxes.append(np.array(pred_boxes))\n            masks.append(pred_masks)\n            scores.append(np.array(pred_confs))\n            del pred, pred_boxes, pred_masks, img\n    del og_img\n    if len(boxes) == 0:\n        return np.array([]), np.array([]), np.array([])\n    boxes, masks, scores = np.concatenate(boxes), np.concatenate(masks).astype(np.uint8), np.concatenate(scores)\n#     boxes = np.concatenate((boxes, scores[:, None]), axis=1) # add scores to boxes\n    \n    return masks, boxes, scores","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:39:29.130708Z","iopub.execute_input":"2023-07-31T12:39:29.131084Z","iopub.status.idle":"2023-07-31T12:39:29.145492Z","shell.execute_reply.started":"2023-07-31T12:39:29.131050Z","shell.execute_reply":"2023-07-31T12:39:29.144568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = [\n    YOLO(p) for p in CFG.models\n]","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:39:29.433700Z","iopub.execute_input":"2023-07-31T12:39:29.434343Z","iopub.status.idle":"2023-07-31T12:39:30.115652Z","shell.execute_reply.started":"2023-07-31T12:39:29.434307Z","shell.execute_reply":"2023-07-31T12:39:30.114637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_rows = []\nresults_csv = pd.DataFrame([], columns=[\"id\",\"height\",\"width\",\"prediction_string\"])\nresults_csv.set_index(\"id\")\n\nfor dirname, _, filenames in os.walk('/kaggle/input/hubmap-hacking-the-human-vasculature/test'):\n    for filename in filenames:\n        \n        image = cv2.imread(os.path.join(dirname, filename))\n        width, height = image.shape[:2]\n        \n        row = dict()\n        row[\"id\"] = filename[:-4]\n        row[\"height\"] = height\n        row[\"width\"] = width\n        row[\"prediction_string\"] = \"\"\n        \n\n\n        raw_masks, raw_boxes, raw_scores = predict(\n            image=image,\n            models=models,\n            transforms=CFG.transforms,\n            imgsz=CFG.imgsz,\n            conf=CFG.conf,\n            iou_nms=CFG.iou_nms,\n            retina_masks=CFG.retina_masks\n        )\n\n        if len(raw_boxes):\n            if CFG.method == 'wbf':\n                masks, boxes, scores = combine_results_wbf(\n                    raw_masks, raw_boxes, raw_scores,\n                    iou_thr=CFG.iou_ensemble,\n                    min_votes=CFG.wbf_min_votes, \n                    label_thr=CFG.wbf_label_thr,\n                    cut_by_box=CFG.wbf_cut_by_box,                    \n                )\n            elif CFG.method == 'nms':\n                masks, boxes, scores = combine_results_nmms(\n                    raw_masks, raw_boxes,\n                    raw_scores, iou_nms=CFG.iou_ensemble\n                )\n            else:\n                masks, boxes, scores = raw_masks, raw_boxes, raw_scores\n        else:\n            masks, boxes, scores = raw_masks, raw_boxes, raw_scores\n\n        if len(boxes):\n            boxes = np.concatenate((boxes, scores[:, None]), axis=1)  # add conf to boxes\n            masks, boxes = postprocess_masks(\n                masks,\n                boxes,\n                conf_thresh=CFG.conf_thresh,\n                min_size=CFG.min_size,\n                dilation_n_iter=CFG.dilation_n_iter,\n                remove_overlap=CFG.remove_overlap,\n                corrupt=CFG.corrupt\n            )\n            scores = boxes[:, 4]\n        else:\n            masks = np.zeros((1, 512, 512), dtype=np.uint8)\n            scores = np.array([0])\n\n        non_zero_masks = []\n        for i, (mask, score) in enumerate(sorted(zip(masks, scores), key=lambda x: x[1], reverse=True)):\n            if score == 0:\n                continue\n            non_zero_masks.append(mask)\n            coded_len = encode_binary_mask(mask.astype(bool)).decode('utf-8')\n            row[\"prediction_string\"] += '0 ' + str(score)+' '+ coded_len+' '\n        \n        new_row = pd.DataFrame(row, index=[0])\n        results_csv = pd.concat([new_row, results_csv.loc[:]]).reset_index(drop=True)\n\nresults_csv.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:39:30.117903Z","iopub.execute_input":"2023-07-31T12:39:30.118361Z","iopub.status.idle":"2023-07-31T12:39:34.600721Z","shell.execute_reply.started":"2023-07-31T12:39:30.118326Z","shell.execute_reply":"2023-07-31T12:39:34.599756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(filenames) == 1:\n    masks = np.stack(non_zero_masks)\n    print(masks.shape)\n    plt.imshow(np.any(masks, 0))","metadata":{"execution":{"iopub.status.busy":"2023-07-31T12:39:34.602733Z","iopub.execute_input":"2023-07-31T12:39:34.603201Z","iopub.status.idle":"2023-07-31T12:39:34.908807Z","shell.execute_reply.started":"2023-07-31T12:39:34.603165Z","shell.execute_reply":"2023-07-31T12:39:34.907835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}