{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## HuBMAP - Hacking the Human Vasculature - Inference","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/addict-2.4.0-py3-none-any.whl\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/mmengine-0.7.4-py3-none-any.whl\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/mmcv-2.0.0-cp310-cp310-linux_x86_64.whl\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/terminaltables-3.1.10-py2.py3-none-any.whl\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/pycocotools-2.0.6-cp310-cp310-linux_x86_64.whl --no-index --find-links /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/ensemble_boxes-1.0.9-py3-none-any.whl\n!pip install /kaggle/input/hubmap-hacking-the-human-vasculature-dataset/packages/mmdet-3.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:07:38.044824Z","iopub.execute_input":"2023-07-31T13:07:38.045537Z","iopub.status.idle":"2023-07-31T13:11:05.064147Z","shell.execute_reply.started":"2023-07-31T13:07:38.045498Z","shell.execute_reply":"2023-07-31T13:11:05.063007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/hubmap-hacking-the-human-vasculature-dataset/modules')","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:05.068177Z","iopub.execute_input":"2023-07-31T13:11:05.068540Z","iopub.status.idle":"2023-07-31T13:11:05.074826Z","shell.execute_reply.started":"2023-07-31T13:11:05.068507Z","shell.execute_reply":"2023-07-31T13:11:05.073974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nfrom glob import glob\nimport yaml\nimport base64\nimport zlib\nimport warnings\n\nimport numpy as np\nimport pandas as pd\nimport networkx as nx\nimport cv2\nimport tifffile\n\nfrom numba import jit\nimport torch\nimport torch.nn as nn\nimport mmcv\nfrom mmengine.config import Config\nfrom mmdet.apis import init_detector, inference_detector\nfrom pycocotools import _mask as mask_util\nimport albumentations as A\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:05.076164Z","iopub.execute_input":"2023-07-31T13:11:05.076774Z","iopub.status.idle":"2023-07-31T13:11:10.604962Z","shell.execute_reply.started":"2023-07-31T13:11:05.076739Z","shell.execute_reply":"2023-07-31T13:11:10.603993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_dataset = Path('/kaggle/input/hubmap-hacking-the-human-vasculature')\nexternal_dataset = Path('/kaggle/input/hubmap-hacking-the-human-vasculature-dataset')\n\nnp.set_printoptions(suppress=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:10.607629Z","iopub.execute_input":"2023-07-31T13:11:10.608031Z","iopub.status.idle":"2023-07-31T13:11:10.613342Z","shell.execute_reply.started":"2023-07-31T13:11:10.607987Z","shell.execute_reply":"2023-07-31T13:11:10.612267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_directory = competition_dataset / 'test'\ndf = pd.read_csv(competition_dataset / 'sample_submission.csv')\npolygons = pd.read_json(competition_dataset / 'polygons.jsonl', lines=True)\n\nverbose = df.shape[0] == 1\nvisualize = df.shape[0] == 1\n\nprint(f'Dataset Shape: {df.shape} - Image Count: {len(os.listdir(image_directory))}')","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:10.614905Z","iopub.execute_input":"2023-07-31T13:11:10.615567Z","iopub.status.idle":"2023-07-31T13:11:15.070951Z","shell.execute_reply.started":"2023-07-31T13:11:10.615534Z","shell.execute_reply":"2023-07-31T13:11:15.069955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Annotation Utilities","metadata":{}},{"cell_type":"code","source":"def decode_rle_mask(rle_mask, shape):\n\n    \"\"\"\n    Decode run-length encoded segmentation mask string into 2d array\n\n    Parameters\n    ----------\n    rle_mask: str\n        Run-length encoded segmentation mask string\n\n    shape: tuple of shape (2)\n        Height and width of the mask\n\n    Returns\n    -------\n    mask: numpy.ndarray of shape (height, width)\n        Decoded 2d segmentation mask\n    \"\"\"\n\n    rle_mask = rle_mask.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (rle_mask[0:][::2], rle_mask[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n\n    mask = np.zeros((shape[0] * shape[1]), dtype=np.uint8)\n    for start, end in zip(starts, ends):\n        mask[start:end] = 1\n\n    mask = mask.reshape(shape[0], shape[1])\n    return mask\n\n\ndef encode_binary_mask(mask):\n    \n    \"\"\"\n    Encode 2d array into run-length encoded segmentation mask string\n\n    Parameters\n    ----------\n    mask: numpy.ndarray of shape (height, width)\n        2d segmentation mask\n\n    Returns\n    -------\n    rle_mask: str\n        Run-length encoded segmentation mask string\n    \"\"\"\n\n    mask = mask.reshape(mask.shape[0], mask.shape[1], 1).astype(np.uint8)\n    mask = np.asfortranarray(mask)\n    \n    rle_mask = mask_util.encode(mask)[0]['counts']\n    rle_mask = zlib.compress(rle_mask, zlib.Z_BEST_COMPRESSION)\n    rle_mask = base64.b64encode(rle_mask)\n    \n    return rle_mask\n\n\ndef binary_to_multi_object_mask(binary_masks):\n\n    \"\"\"\n    Encode multiple 2d binary masks into a single 2d multi-object segmentation mask\n\n    Parameters\n    ----------\n    binary_masks: numpy.ndarray of shape (n_objects, height, width)\n        2d binary masks\n\n    Returns\n    -------\n    multi_object_mask: numpy.ndarray of shape (height, width)\n        2d multi-object mask\n    \"\"\"\n\n    multi_object_mask = np.zeros((binary_masks.shape[1], binary_masks.shape[2]))\n    for i, binary_mask in enumerate(binary_masks):\n        non_zero_idx = binary_mask == 1\n        multi_object_mask[non_zero_idx] = i + 1\n\n    return multi_object_mask\n\n\ndef polygon_to_mask(polygon, shape):\n\n    \"\"\"\n    Create binary segmentation mask from polygon\n\n    Parameters\n    ----------\n    polygon: list of shape (n_polygons, n_points, 2)\n        List of polygons\n\n    shape: tuple of shape (2)\n        Height and width of the mask\n\n    Returns\n    -------\n    mask: numpy.ndarray of shape (height, width)\n        2d segmentation mask\n    \"\"\"\n\n    mask = np.zeros(shape)\n    # Convert list of points to tuple pairs of X and Y coordinates\n    points = np.array(polygon).reshape(-1, 2)\n    # Draw mask from the polygon\n    cv2.fillPoly(mask, [points], 1, lineType=cv2.LINE_8, shift=0)\n    mask = np.array(mask).astype(np.uint8)\n\n    return mask\n\n\ndef mask_to_bounding_box(mask):\n\n    \"\"\"\n    Get bounding box from a binary segmentation mask\n\n    Parameters\n    ----------\n    mask: numpy.ndarray of shape (height, width)\n        2d binary mask\n\n    Returns\n    -------\n    bounding_box: list of shape (4)\n        Bounding box\n    \"\"\"\n\n    non_zero_idx = np.where(mask == 1)\n    bounding_box = [\n        int(np.min(non_zero_idx[1])),\n        int(np.min(non_zero_idx[0])),\n        int(np.max(non_zero_idx[1])),\n        int(np.max(non_zero_idx[0]))\n    ]\n\n    return bounding_box\n\n\ndef decode_hhthv_annotations(annotations, shape):\n\n    \"\"\"\n    Create arrays of mask binary segmentation masks and labels\n\n    Parameters\n    ----------\n    annotations: list of dictionaries (n_annotations)\n        List of dictionaries of polygons and labels\n\n    shape: tuple of shape (2)\n        Height and width of the masks\n\n    Returns\n    -------\n    masks: numpy.ndarray of shape (n_annotations, height, width)\n        Array of 2d segmentation masks\n\n    bounding_boxes: numpy.ndarray of shape (n_annotations, 4)\n        Array of bounding boxes\n\n    labels: numpy.ndarray of shape (n_annotations)\n        Array of labels\n    \"\"\"\n\n    masks = []\n    bounding_boxes = []\n    labels = []\n\n    for annotation in annotations:\n\n        mask = polygon_to_mask(polygon=annotation['coordinates'], shape=shape)\n        bounding_box = mask_to_bounding_box(mask=mask)\n\n        masks.append(mask)\n        bounding_boxes.append(bounding_box)\n        labels.append(annotation['type'])\n\n    masks = np.array(masks)\n    bounding_boxes = np.array(bounding_boxes)\n    labels = np.array(labels)\n\n    return masks, bounding_boxes, labels\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.072431Z","iopub.execute_input":"2023-07-31T13:11:15.073019Z","iopub.status.idle":"2023-07-31T13:11:15.094801Z","shell.execute_reply.started":"2023-07-31T13:11:15.072984Z","shell.execute_reply":"2023-07-31T13:11:15.091827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Visualization","metadata":{}},{"cell_type":"code","source":"def visualize_predictions(image, ground_truth, predictions, metadata, path=None):\n\n    \"\"\"\n    Visualize image along with its annotations\n\n    Parameters\n    ----------\n    image: path-like str or numpy.ndarray of shape (height, width, 3)\n        Image path or image array\n        \n    ground_truth: dict\n        Dictionary of ground truth labels\n        \n    predictions: dict\n        Dictionary of model predictions\n        \n    metadata: dict\n        Dictionary of metadata\n\n    path: (path-like str or None)\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n\n    labels = {\n        0: 'blood_vessel',\n        1: 'glomerulus',\n        2: 'unsure'\n    }\n\n    if isinstance(image, Path) or isinstance(image, str):\n        # Read image from the given path\n        image_path = image\n        if image_path.endswith('.tif') or image_path.endswith('.tiff'):\n            image = tifffile.imread(image_path)\n        else:\n            image = cv2.imread(image_path, -1)\n\n    elif isinstance(image, np.ndarray):\n        title = ''\n\n    else:\n        # Raise TypeError if image argument is not an array-like object or a path-like string\n        raise TypeError('Image is not an array or path')\n\n    fig, axes = plt.subplots(figsize=(48, 20), ncols=2)\n\n    image_ground_truth = np.copy(image)\n    \n    if ground_truth is not None:\n        for bounding_box, label in zip(ground_truth['boxes'], ground_truth['labels']):\n            # Draw bounding box and its label to the image\n            image_ground_truth = cv2.rectangle(image_ground_truth, (int(bounding_box[0]), int(bounding_box[1])), (int(bounding_box[2]), int(bounding_box[3])), (36, 255, 12), 2)\n            (w, h), _ = cv2.getTextSize(labels[label], cv2.FONT_HERSHEY_SIMPLEX, 0.6, 1)\n            image_ground_truth = cv2.rectangle(image_ground_truth, (int(bounding_box[0]), int(bounding_box[1]) - 20), (int(bounding_box[0]) + w, int(bounding_box[1])), (36, 255, 12), -1)\n            image_ground_truth = cv2.putText(image_ground_truth, labels[label], (int(bounding_box[0]), int(bounding_box[1]) - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255, 255, 255), 1)\n\n    axes[0].imshow(image_ground_truth)\n    \n    if ground_truth is not None:\n        mask = binary_to_multi_object_mask(binary_masks=ground_truth['masks'])\n        axes[0].imshow(mask, alpha=0.5)\n\n    image_predictions = np.copy(image)\n    for bounding_box, label, score in zip(predictions['boxes'], predictions['labels'], predictions['scores']):\n        # Draw bounding box and its label to the image\n        image_predictions = cv2.rectangle(image_predictions, (int(bounding_box[0]), int(bounding_box[1])), (int(bounding_box[2]), int(bounding_box[3])), (36, 255, 12), 2)\n        text = f'{score:.2f}'\n        (w, h), _ = cv2.getTextSize(text, cv2.FONT_HERSHEY_SIMPLEX, 0.6, 1)\n        image_predictions = cv2.rectangle(image_predictions, (int(bounding_box[0]), int(bounding_box[1]) - 20), (int(bounding_box[0]) + w, int(bounding_box[1])), (36, 255, 12), -1)\n        image_predictions = cv2.putText(image_predictions, text, (int(bounding_box[2]), int(bounding_box[3]) - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255, 255, 255), 1)\n\n    axes[1].imshow(image_predictions)\n\n    mask = binary_to_multi_object_mask(binary_masks=predictions['masks'])\n    axes[1].imshow(mask, alpha=0.5)\n\n    for i in range(2):\n        axes[i].set_xlabel('')\n        axes[i].set_ylabel('')\n        axes[i].tick_params(axis='x', labelsize=15, pad=10)\n        axes[i].tick_params(axis='y', labelsize=15, pad=10)\n    \n    if ground_truth is not None:\n        axes[0].set_title(f'Image + Ground truth ({ground_truth[\"labels\"].shape[0]} objects)', size=20, pad=15)\n    else:\n        axes[0].set_title(f'Image', size=25, pad=15)\n        \n    axes[1].set_title(\n        f'''\n        Image + Predictions ({predictions['labels'].shape[0]} objects)\n        Segmentation AP: {metadata['score']:.4f}\n        NMS IoU Threshold: {metadata['nms_iou_threshold']:.2f} - Score Threshold: {metadata['score_threshold']:.2f}\n        ''',\n        size=25, pad=15\n    )\n\n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.099627Z","iopub.execute_input":"2023-07-31T13:11:15.099911Z","iopub.status.idle":"2023-07-31T13:11:15.140260Z","shell.execute_reply.started":"2023-07-31T13:11:15.099886Z","shell.execute_reply":"2023-07-31T13:11:15.139034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Semantic Segmentation","metadata":{}},{"cell_type":"code","source":"class CoaTDAFormer(nn.Module):\n\n    def __init__(self, encoder_name, decoder_args, encoder_weights=None):\n\n        super(CoaTDAFormer, self).__init__()\n\n        self.encoder = getattr(coat, encoder_name)()\n        if encoder_weights is not None:\n            if encoder_name == 'coat_lite_medium':\n                state_dict = torch.load(encoder_weights)['model']\n            else:\n                state_dict = torch.load(encoder_weights)\n            self.encoder.load_state_dict(\n                state_dict=state_dict,\n                strict=False\n            )\n        self.decoder = daformer.DAFormerDecoder(**decoder_args)\n        self.conv_head = nn.Sequential(\n            nn.Conv2d(self.decoder.decoder_dim, 1, kernel_size=1),\n            nn.Upsample(scale_factor=4, mode='bilinear', align_corners=False),\n        )\n\n    def forward(self, x):\n\n        x = self.encoder(x)\n        last, decoder = self.decoder(x)\n        out = self.conv_head(last)\n\n        return out\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.141981Z","iopub.execute_input":"2023-07-31T13:11:15.142354Z","iopub.status.idle":"2023-07-31T13:11:15.155511Z","shell.execute_reply.started":"2023-07-31T13:11:15.142324Z","shell.execute_reply":"2023-07-31T13:11:15.154399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_semantic_segmentation_model(model_directory, model_filenames, device):\n\n    \"\"\"\n    Load semantic segmentation models from given paths\n\n    Parameters\n    ----------\n    model_directory: pathlib.Path or str\n        Path of the model directory\n\n    model_filenames: list of shape (n_models)\n        List of model filenames\n\n    device: torch.device\n        Location of the model and inputs\n\n    Returns\n    -------\n    models: dict\n        Dictionary of pretrained models\n\n    config: dict\n        Dictionary of model configurations\n    \"\"\"\n\n    config = yaml.load(open(model_directory / 'config.yaml', 'r'), Loader=yaml.FullLoader)\n\n    if (config['model']['model_module'] == 'smp') or (config['model']['model_module'] == 'coat_daformer'):\n        # Set encoder_weights to None if model module is smp (segmentation_models_pytorch)\n        config['model']['model_args']['encoder_weights'] = None\n    else:\n        raise ValueError('Invalid Model Module')\n\n    models = {}\n\n    for model_filename in model_filenames:\n        if config['model']['model_module'] == 'smp':\n            model = getattr(smp, config['model']['model_class'])(**config['model']['model_args'])\n        elif config['model']['model_module'] == 'coat_daformer':\n            model = CoaTDAFormer(**config['model']['model_args'])\n        else:\n            model = None\n\n        model.load_state_dict(torch.load(model_directory / model_filename))\n        model.to(device)\n        model.eval()\n        models[model_filename] = model\n\n    return models, config\n\n\ndef predict_semantic_segmentation_model(inputs, model, transforms, device, amp=False, tta=False):\n\n    \"\"\"\n    Predict given inputs with given model\n\n    Parameters\n    ----------\n    inputs: numpy.ndarray of shape (channel, height, width)\n        Path of the model directory\n\n    model: torch.nn.Module\n        Model\n\n    transforms: albumentations.Compose\n        Transforms pipeline\n\n    device: torch.device\n        Location of the model and inputs\n\n    amp: bool\n        Whether to use auto mixed precision or not\n\n    tta: bool\n        Whether to use test time augmentations or not\n\n    Returns\n    -------\n    outputs: numpy.ndarray of shape (height, width)\n        2D array of soft semantic segmentation predictions\n    \"\"\"\n\n    inputs = transforms(image=inputs)['image']\n    inputs = torch.unsqueeze(inputs, dim=0)\n    inputs = inputs.to(device)\n\n    with torch.no_grad():\n        if amp:\n            with torch.autocast(device_type=device.type, dtype=torch.float16):\n                outputs = model(inputs.half()).float().cpu()\n        else:\n            outputs = model(inputs).cpu()\n\n    if tta:\n        with torch.no_grad():\n            vertical_flipped_inputs = inputs.flip(dims=(2,))\n            if amp:\n                with torch.autocast(device_type=device.type, dtype=torch.float16):\n                    vertical_flipped_outputs = model(vertical_flipped_inputs.half()).float().cpu()\n            else:\n                vertical_flipped_outputs = model(vertical_flipped_inputs).cpu()\n\n            vertical_flipped_outputs = vertical_flipped_outputs.flip(dims=(2,))\n\n            horizontal_flipped_inputs = inputs.flip(dims=(3,))\n            if amp:\n                with torch.autocast(device_type=device.type, dtype=torch.float16):\n                    horizontal_flipped_outputs = model(horizontal_flipped_inputs.half()).float().cpu()\n            else:\n                horizontal_flipped_outputs = model(horizontal_flipped_inputs).cpu()\n\n            horizontal_flipped_outputs = horizontal_flipped_outputs.flip(dims=(3,))\n\n            vertical_horizontal_flipped_inputs = inputs.flip(dims=(2, 3))\n            if amp:\n                with torch.autocast(device_type=device.type, dtype=torch.float16):\n                    vertical_horizontal_flipped_outputs = model(vertical_horizontal_flipped_inputs.half()).float().cpu()\n            else:\n                vertical_horizontal_flipped_outputs = model(vertical_horizontal_flipped_inputs).cpu()\n\n            vertical_horizontal_flipped_outputs = vertical_horizontal_flipped_outputs.flip(dims=(2, 3))\n\n        # Concatenate and merge test time augmentations\n        outputs = torch.cat([\n            outputs.unsqueeze(dim=0),\n            vertical_flipped_outputs.unsqueeze(dim=0),\n            horizontal_flipped_outputs.unsqueeze(dim=0),\n            vertical_horizontal_flipped_outputs.unsqueeze(dim=0)\n        ], dim=0).mean(dim=0)\n\n    outputs = torch.sigmoid(outputs)\n    outputs = torch.squeeze(torch.squeeze(outputs, dim=0), dim=0).numpy()\n\n    return outputs\n\n\ndef get_semantic_segmentation_transforms(**transform_parameters):\n\n    \"\"\"\n    Get transforms for semantic segmentation dataset\n\n    Parameters\n    ----------\n    transform_parameters: dict\n        Dictionary of transform parameters\n\n    Returns\n    -------\n    transforms: dict\n        Transforms of training and test sets\n    \"\"\"\n\n    test_transforms = A.Compose([\n        A.Resize(\n            height=transform_parameters['resize_height'],\n            width=transform_parameters['resize_width'],\n            interpolation=cv2.INTER_NEAREST,\n            always_apply=True\n        ),\n        A.Normalize(\n            mean=transform_parameters['normalize_mean'],\n            std=transform_parameters['normalize_std'],\n            max_pixel_value=255,\n            always_apply=True\n        ),\n        ToTensorV2(always_apply=True)\n    ])\n\n    semantic_segmentation_transforms = {'test': test_transforms}\n    return semantic_segmentation_transforms\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.156978Z","iopub.execute_input":"2023-07-31T13:11:15.157566Z","iopub.status.idle":"2023-07-31T13:11:15.183233Z","shell.execute_reply.started":"2023-07-31T13:11:15.157531Z","shell.execute_reply":"2023-07-31T13:11:15.181900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"semantic_segmentation_inference = False\n\nif semantic_segmentation_inference:\n\n    coat_daformer_models, coat_daformer_config = load_semantic_segmentation_model(\n        model_directory=external_dataset / 'coat_daformer',\n        model_filenames=[\n            'model_fold1_best.pt',\n            'model_fold2_best.pt',\n            'model_fold3_best.pt',\n            'model_fold4_best.pt',\n        ],\n        device=torch.device('cuda')\n    )\n    coat_daformer_transforms = get_semantic_segmentation_transforms(**coat_daformer_config['transforms'])['test']\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.187766Z","iopub.execute_input":"2023-07-31T13:11:15.188217Z","iopub.status.idle":"2023-07-31T13:11:15.199197Z","shell.execute_reply.started":"2023-07-31T13:11:15.188189Z","shell.execute_reply":"2023-07-31T13:11:15.197971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Instance Segmentation","metadata":{}},{"cell_type":"code","source":"def predict_mmdetection(image, model, tta=False):\n\n    \"\"\"\n    Predict given image with given model\n\n    Parameters\n    ----------\n    image: numpy.ndarray of shape (height, width, channel)\n        Image array\n\n    model: torch.nn.Module\n        Model\n\n    tta: bool\n        Whether to use test time augmentations or not\n\n    Returns\n    -------\n    predictions: Dict of dicts\n        Dictionaries of boxes, masks, labels and scores from single or multiple test time augmentations\n    \"\"\"\n\n    predictions = {}\n\n    outputs = inference_detector(model, imgs=image)\n    detections = {\n        'boxes': outputs.pred_instances.bboxes.cpu().numpy(),\n        'masks': outputs.pred_instances.masks.cpu().numpy(),\n        'labels': outputs.pred_instances.labels.cpu().numpy(),\n        'scores': outputs.pred_instances.scores.cpu().numpy()\n    }\n    del outputs\n    blood_vessel_mask = detections['labels'] == 0\n    for k in detections.keys():\n        detections[k] = detections[k][blood_vessel_mask]\n    predictions['raw'] = detections\n\n    if tta:\n        vertical_flipped_image = np.flip(image, axis=0)\n        vertical_flipped_outputs = inference_detector(model, imgs=vertical_flipped_image)\n        vertical_flipped_detections = {\n            'boxes': vertical_flipped_outputs.pred_instances.bboxes.cpu().numpy(),\n            'masks': vertical_flipped_outputs.pred_instances.masks.cpu().numpy(),\n            'labels': vertical_flipped_outputs.pred_instances.labels.cpu().numpy(),\n            'scores': vertical_flipped_outputs.pred_instances.scores.cpu().numpy()\n        }\n        del vertical_flipped_outputs\n        vertical_flipped_detections['boxes'][:, 1] = image.shape[0] - vertical_flipped_detections['boxes'][:, 1]\n        vertical_flipped_detections['boxes'][:, 3] = image.shape[0] - vertical_flipped_detections['boxes'][:, 3]\n        vertical_flipped_detections['boxes'][:, 1], vertical_flipped_detections['boxes'][:, 3] = vertical_flipped_detections['boxes'][:, 3], vertical_flipped_detections['boxes'][:, 1].copy()\n        vertical_flipped_detections['masks'] = np.flip(vertical_flipped_detections['masks'], axis=1)\n        blood_vessel_mask = vertical_flipped_detections['labels'] == 0\n        for k in vertical_flipped_detections.keys():\n            vertical_flipped_detections[k] = vertical_flipped_detections[k][blood_vessel_mask]\n        predictions['vertical_flip'] = vertical_flipped_detections\n\n        horizontal_flipped_image = np.flip(image, axis=1)\n        horizontal_flipped_outputs = inference_detector(model, imgs=horizontal_flipped_image)\n        horizontal_flipped_detections = {\n            'boxes': horizontal_flipped_outputs.pred_instances.bboxes.cpu().numpy(),\n            'masks': horizontal_flipped_outputs.pred_instances.masks.cpu().numpy(),\n            'labels': horizontal_flipped_outputs.pred_instances.labels.cpu().numpy(),\n            'scores': horizontal_flipped_outputs.pred_instances.scores.cpu().numpy()\n        }\n        del horizontal_flipped_outputs\n        horizontal_flipped_detections['boxes'][:, 0] = image.shape[1] - horizontal_flipped_detections['boxes'][:, 0]\n        horizontal_flipped_detections['boxes'][:, 2] = image.shape[1] - horizontal_flipped_detections['boxes'][:, 2]\n        horizontal_flipped_detections['boxes'][:, 0], horizontal_flipped_detections['boxes'][:, 2] = horizontal_flipped_detections['boxes'][:, 2], horizontal_flipped_detections['boxes'][:, 0].copy()\n        horizontal_flipped_detections['masks'] = np.flip(horizontal_flipped_detections['masks'], axis=2)\n        blood_vessel_mask = horizontal_flipped_detections['labels'] == 0\n        for k in horizontal_flipped_detections.keys():\n            horizontal_flipped_detections[k] = horizontal_flipped_detections[k][blood_vessel_mask]\n        predictions['horizontal_flip'] = horizontal_flipped_detections\n\n        diagonal_flipped_image = np.flip(image, axis=(0, 1))\n        diagonal_flipped_outputs = inference_detector(model, imgs=diagonal_flipped_image)\n        diagonal_flipped_detections = {\n            'boxes': diagonal_flipped_outputs.pred_instances.bboxes.cpu().numpy(),\n            'masks': diagonal_flipped_outputs.pred_instances.masks.cpu().numpy(),\n            'labels': diagonal_flipped_outputs.pred_instances.labels.cpu().numpy(),\n            'scores': diagonal_flipped_outputs.pred_instances.scores.cpu().numpy()\n        }\n        del diagonal_flipped_outputs\n        diagonal_flipped_detections['boxes'][:, 0] = image.shape[1] - diagonal_flipped_detections['boxes'][:, 0]\n        diagonal_flipped_detections['boxes'][:, 1] = image.shape[0] - diagonal_flipped_detections['boxes'][:, 1]\n        diagonal_flipped_detections['boxes'][:, 2] = image.shape[1] - diagonal_flipped_detections['boxes'][:, 2]\n        diagonal_flipped_detections['boxes'][:, 3] = image.shape[0] - diagonal_flipped_detections['boxes'][:, 3]\n        diagonal_flipped_detections['boxes'][:, 0], diagonal_flipped_detections['boxes'][:, 2] = diagonal_flipped_detections['boxes'][:, 2], diagonal_flipped_detections['boxes'][:, 0].copy()\n        diagonal_flipped_detections['boxes'][:, 1], diagonal_flipped_detections['boxes'][:, 3] = diagonal_flipped_detections['boxes'][:, 3], diagonal_flipped_detections['boxes'][:, 1].copy()\n        diagonal_flipped_detections['masks'] = np.flip(diagonal_flipped_detections['masks'], axis=(1, 2))\n        blood_vessel_mask = diagonal_flipped_detections['labels'] == 0\n        for k in diagonal_flipped_detections.keys():\n            diagonal_flipped_detections[k] = diagonal_flipped_detections[k][blood_vessel_mask]\n        predictions['diagonal_flip'] = diagonal_flipped_detections\n        \n    for k in predictions.keys():\n        predictions[k]['mask_lookup'] = {str(np.round(box, 2)): idx for idx, box in enumerate(predictions[k]['boxes'])}\n\n    return predictions\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.201020Z","iopub.execute_input":"2023-07-31T13:11:15.201381Z","iopub.status.idle":"2023-07-31T13:11:15.229517Z","shell.execute_reply.started":"2023-07-31T13:11:15.201349Z","shell.execute_reply":"2023-07-31T13:11:15.228244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda:0'\n\nmmdetection_mask_rcnn_directory = external_dataset / 'mmdetection_mask_rcnn'\nmmdetection_mask_rcnn_config = Config.fromfile(str(mmdetection_mask_rcnn_directory / 'config.py'))\nmmdetection_mask_rcnn_models = {\n    1: init_detector(mmdetection_mask_rcnn_config, checkpoint=str(mmdetection_mask_rcnn_directory / 'model_fold1_best.pth'), device=device),\n    2: init_detector(mmdetection_mask_rcnn_config, checkpoint=str(mmdetection_mask_rcnn_directory / 'model_fold2_best.pth'), device=device),\n    3: init_detector(mmdetection_mask_rcnn_config, checkpoint=str(mmdetection_mask_rcnn_directory / 'model_fold3_best.pth'), device=device),\n    4: init_detector(mmdetection_mask_rcnn_config, checkpoint=str(mmdetection_mask_rcnn_directory / 'model_fold4_best.pth'), device=device)\n}\n\nmmdetection_cascade_mask_rcnn_directory = external_dataset / 'mmdetection_cascade_mask_rcnn'\nmmdetection_cascade_mask_rcnn_config = Config.fromfile(str(mmdetection_cascade_mask_rcnn_directory / 'config.py'))\nmmdetection_cascade_mask_rcnn_models = {\n    1: init_detector(mmdetection_cascade_mask_rcnn_config, checkpoint=str(mmdetection_cascade_mask_rcnn_directory / 'model_fold1_best.pth'), device=device),\n    2: init_detector(mmdetection_cascade_mask_rcnn_config, checkpoint=str(mmdetection_cascade_mask_rcnn_directory / 'model_fold2_best.pth'), device=device),\n    3: init_detector(mmdetection_cascade_mask_rcnn_config, checkpoint=str(mmdetection_cascade_mask_rcnn_directory / 'model_fold3_best.pth'), device=device),\n    4: init_detector(mmdetection_cascade_mask_rcnn_config, checkpoint=str(mmdetection_cascade_mask_rcnn_directory / 'model_fold4_best.pth'), device=device)\n}","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:11:15.231459Z","iopub.execute_input":"2023-07-31T13:11:15.231838Z","iopub.status.idle":"2023-07-31T13:12:20.499814Z","shell.execute_reply.started":"2023-07-31T13:11:15.231806Z","shell.execute_reply":"2023-07-31T13:12:20.498819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append('/kaggle/input/yolov8')\nfrom ultralytics import YOLO\n\nyolo_model = YOLO('/kaggle/input/model-weights/yolov8m-0.395_fulldata.pt')","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:20.501603Z","iopub.execute_input":"2023-07-31T13:12:20.501968Z","iopub.status.idle":"2023-07-31T13:12:31.143542Z","shell.execute_reply.started":"2023-07-31T13:12:20.501918Z","shell.execute_reply":"2023-07-31T13:12:31.142493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TTA_new(im, model, tta_dict, predictions):\n    for aug_name, aug in tta_dict.items():\n        transform = aug\n        ima = transform(image=im)['image']\n        pred = model.predict(\n            ima,\n            imgsz=1024,\n            conf = 0.001,\n            iou = 0.6\n        )[0]\n        conf_scores = pred.boxes.conf.detach().cpu().numpy()\n        classes = pred.boxes.cls.detach().cpu().numpy()\n        pred_masks = pred.masks.data.detach().cpu().numpy()\n        pred_boxes = pred.boxes.xyxyn.detach().cpu().numpy()\n        del pred\n        pred_boxes = transform(image=ima, bboxes=pred_boxes)['bboxes']\n        pred_masks = transform(image=ima, masks=pred_masks)['masks']\n            \n        preds = {\n            \"boxes\" : np.array(pred_boxes),\n            \"scores\" : np.array(conf_scores),\n            \"masks\" : np.array(pred_masks, dtype=np.uint8),\n            \"labels\" : np.array(classes)\n        }\n        predictions[aug_name] = preds\n\n        del pred_boxes, pred_masks, ima\n    \n    return predictions\n\ndef scaling_boxes(box, scale):\n    #simetrik\n    box[0] = box[0]*scale\n    box[1] = box[1]*scale\n    box[2] = box[2]*scale\n    box[3] = box[3]*scale\n    return box\n\ndef post_process_yolo(predictions, img0_shape=(1024,1024), img1_shape=(512,512)):\n    scale= 512\n    for k in predictions.keys():\n        predictions[k][\"boxes\"] = np.array([scaling_boxes(x, scale) for x in predictions[k][\"boxes\"]])\n        predictions[k][\"masks\"] = np.array([(cv2.resize(x, dsize=img1_shape,interpolation=cv2.INTER_AREA) > 0).astype(np.uint8) for x in predictions[k][\"masks\"]])\n        \n    return predictions\n\ndef predict_yolo(image, model, tta=False):\n    \n    predictions = {}\n    \n    pred = model.predict(\n            image,\n            imgsz = 1024,\n            conf = 0.001,\n            iou = 0.6\n    )[0]\n    preds = {\n        \"scores\" : pred.boxes.conf.detach().cpu().numpy(),\n        \"labels\" : pred.boxes.cls.detach().cpu().numpy(),\n        \"masks\" : pred.masks.data.detach().cpu().numpy(),\n        \"boxes\" :  pred.boxes.xyxyn.detach().cpu().numpy()\n    }\n    predictions[\"raw\"] = preds\n    \n    if tta:\n        tta_augs = {\n            \"horizontal_flip\": A.HorizontalFlip(p=1.0),\n            \"vertical_flip\" : A.VerticalFlip(p=1.0),\n            \"diagonal_flip\" : A.Rotate(limit=(180,180), p=1.0)\n        }\n        \n        predictions = TTA_new(image, model, tta_augs, predictions)\n        \n    \n    predictions = post_process_yolo(predictions)\n    \n    for k in predictions.keys():\n        predictions[k]['mask_lookup'] = {str(np.round(box, 2)): idx for idx, box in enumerate(predictions[k]['boxes'])}\n        \n        \n    return predictions\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:31.145025Z","iopub.execute_input":"2023-07-31T13:12:31.145409Z","iopub.status.idle":"2023-07-31T13:12:31.165919Z","shell.execute_reply.started":"2023-07-31T13:12:31.145370Z","shell.execute_reply":"2023-07-31T13:12:31.164839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Post-processing","metadata":{}},{"cell_type":"code","source":"# coding: utf-8\n__author__ = 'ZFTurbo: https://kaggle.com/zfturbo'\n\n\ndef prepare_boxes(boxes, scores, labels):\n    result_boxes = boxes.copy()\n\n    cond = (result_boxes < 0)\n    cond_sum = cond.astype(np.int32).sum()\n    if cond_sum > 0:\n        print('Warning. Fixed {} boxes coordinates < 0'.format(cond_sum))\n        result_boxes[cond] = 0\n\n    cond = (result_boxes > 1)\n    cond_sum = cond.astype(np.int32).sum()\n    if cond_sum > 0:\n        print('Warning. Fixed {} boxes coordinates > 1. Check that your boxes was normalized at [0, 1]'.format(cond_sum))\n        result_boxes[cond] = 1\n\n    boxes1 = result_boxes.copy()\n    result_boxes[:, 0] = np.min(boxes1[:, [0, 2]], axis=1)\n    result_boxes[:, 2] = np.max(boxes1[:, [0, 2]], axis=1)\n    result_boxes[:, 1] = np.min(boxes1[:, [1, 3]], axis=1)\n    result_boxes[:, 3] = np.max(boxes1[:, [1, 3]], axis=1)\n\n    area = (result_boxes[:, 2] - result_boxes[:, 0]) * (result_boxes[:, 3] - result_boxes[:, 1])\n    cond = (area == 0)\n    cond_sum = cond.astype(np.int32).sum()\n    if cond_sum > 0:\n        print('Warning. Removed {} boxes with zero area!'.format(cond_sum))\n        result_boxes = result_boxes[area > 0]\n        scores = scores[area > 0]\n        labels = labels[area > 0]\n\n    return result_boxes, scores, labels\n\n\ndef cpu_soft_nms_float(dets, sc, Nt, sigma, thresh, method):\n    \"\"\"\n    Based on: https://github.com/DocF/Soft-NMS/blob/master/soft_nms.py\n    It's different from original soft-NMS because we have float coordinates on range [0; 1]\n\n    :param dets:   boxes format [x1, y1, x2, y2]\n    :param sc:     scores for boxes\n    :param Nt:     required iou \n    :param sigma:  \n    :param thresh: \n    :param method: 1 - linear soft-NMS, 2 - gaussian soft-NMS, 3 - standard NMS\n    :return: index of boxes to keep\n    \"\"\"\n\n    # indexes concatenate boxes with the last column\n    N = dets.shape[0]\n    indexes = np.array([np.arange(N)])\n    dets = np.concatenate((dets, indexes.T), axis=1)\n\n    # the order of boxes coordinate is [y1, x1, y2, x2]\n    y1 = dets[:, 1]\n    x1 = dets[:, 0]\n    y2 = dets[:, 3]\n    x2 = dets[:, 2]\n    scores = sc\n    areas = (x2 - x1) * (y2 - y1)\n\n    for i in range(N):\n        # intermediate parameters for later parameters exchange\n        tBD = dets[i, :].copy()\n        tscore = scores[i].copy()\n        tarea = areas[i].copy()\n        pos = i + 1\n\n        #\n        if i != N - 1:\n            maxscore = np.max(scores[pos:], axis=0)\n            maxpos = np.argmax(scores[pos:], axis=0)\n        else:\n            maxscore = scores[-1]\n            maxpos = 0\n        if tscore < maxscore:\n            dets[i, :] = dets[maxpos + i + 1, :]\n            dets[maxpos + i + 1, :] = tBD\n            tBD = dets[i, :]\n\n            scores[i] = scores[maxpos + i + 1]\n            scores[maxpos + i + 1] = tscore\n            tscore = scores[i]\n\n            areas[i] = areas[maxpos + i + 1]\n            areas[maxpos + i + 1] = tarea\n            tarea = areas[i]\n\n        # IoU calculate\n        xx1 = np.maximum(dets[i, 1], dets[pos:, 1])\n        yy1 = np.maximum(dets[i, 0], dets[pos:, 0])\n        xx2 = np.minimum(dets[i, 3], dets[pos:, 3])\n        yy2 = np.minimum(dets[i, 2], dets[pos:, 2])\n\n        w = np.maximum(0.0, xx2 - xx1)\n        h = np.maximum(0.0, yy2 - yy1)\n        inter = w * h\n        ovr = inter / (areas[i] + areas[pos:] - inter)\n\n        # Three methods: 1.linear 2.gaussian 3.original NMS\n        if method == 1:  # linear\n            weight = np.ones(ovr.shape)\n            weight[ovr > Nt] = weight[ovr > Nt] - ovr[ovr > Nt]\n        elif method == 2:  # gaussian\n            weight = np.exp(-(ovr * ovr) / sigma)\n        else:  # original NMS\n            weight = np.ones(ovr.shape)\n            weight[ovr > Nt] = 0\n\n        scores[pos:] = weight * scores[pos:]\n\n    # select the boxes and keep the corresponding indexes\n    inds = dets[:, 4][scores > thresh]\n    keep = inds.astype(int)\n    return keep\n\n\n@jit(nopython=True)\ndef nms_float_fast(dets, scores, thresh):\n    \"\"\"\n    # It's different from original nms because we have float coordinates on range [0; 1]\n    :param dets: numpy array of boxes with shape: (N, 5). Order: x1, y1, x2, y2, score. All variables in range [0; 1]\n    :param thresh: IoU value for boxes\n    :return: index of boxes to keep\n    \"\"\"\n    x1 = dets[:, 0]\n    y1 = dets[:, 1]\n    x2 = dets[:, 2]\n    y2 = dets[:, 3]\n\n    areas = (x2 - x1) * (y2 - y1)\n    order = scores.argsort()[::-1]\n\n    keep = []\n    while order.size > 0:\n        i = order[0]\n        keep.append(i)\n        xx1 = np.maximum(x1[i], x1[order[1:]])\n        yy1 = np.maximum(y1[i], y1[order[1:]])\n        xx2 = np.minimum(x2[i], x2[order[1:]])\n        yy2 = np.minimum(y2[i], y2[order[1:]])\n\n        w = np.maximum(0.0, xx2 - xx1)\n        h = np.maximum(0.0, yy2 - yy1)\n        inter = w * h\n        ovr = inter / (areas[i] + areas[order[1:]] - inter)\n        inds = np.where(ovr <= thresh)[0]\n        order = order[inds + 1]\n\n    return keep\n\n\ndef nms_method(boxes, scores, labels, method=3, iou_thr=0.5, sigma=0.5, thresh=0.001, weights=None):\n    \"\"\"\n    :param boxes: list of boxes predictions from each model, each box is 4 numbers. \n    It has 3 dimensions (models_number, model_preds, 4)\n    Order of boxes: x1, y1, x2, y2. We expect float normalized coordinates [0; 1] \n    :param scores: list of scores for each model \n    :param labels: list of labels for each model\n    :param method: 1 - linear soft-NMS, 2 - gaussian soft-NMS, 3 - standard NMS\n    :param iou_thr: IoU value for boxes to be a match \n    :param sigma: Sigma value for SoftNMS\n    :param thresh: threshold for boxes to keep (important for SoftNMS)\n    :param weights: list of weights for each model. Default: None, which means weight == 1 for each model\n\n    :return: boxes: boxes coordinates (Order of boxes: x1, y1, x2, y2). \n    :return: scores: confidence scores\n    :return: labels: boxes labels\n    \"\"\"\n\n    # If weights are specified\n    if weights is not None:\n        if len(boxes) != len(weights):\n            print('Incorrect number of weights: {}. Must be: {}. Skip it'.format(len(weights), len(boxes)))\n        else:\n            weights = np.array(weights)\n            for i in range(len(weights)):\n                scores[i] = (np.array(scores[i]) * weights[i]) / weights.sum()\n\n    # Do the checks and skip empty predictions\n    filtered_boxes = []\n    filtered_scores = []\n    filtered_labels = []\n    for i in range(len(boxes)):\n        if len(boxes[i]) != len(scores[i]) or len(boxes[i]) != len(labels[i]):\n            print('Check length of boxes and scores and labels: {} {} {} at position: {}. Boxes are skipped!'.format(len(boxes[i]), len(scores[i]), len(labels[i]), i))\n            continue\n        if len(boxes[i]) == 0:\n            # print('Empty boxes!')\n            continue\n        filtered_boxes.append(boxes[i])\n        filtered_scores.append(scores[i])\n        filtered_labels.append(labels[i])\n\n    # We concatenate everything\n    boxes = np.concatenate(filtered_boxes)\n    scores = np.concatenate(filtered_scores)\n    labels = np.concatenate(filtered_labels)\n    \n\n    # Fix coordinates and removed zero area boxes\n    boxes, scores, labels = prepare_boxes(boxes, scores, labels)\n\n    # Run NMS independently for each label\n    unique_labels = np.unique(labels)\n    final_boxes = []\n    final_scores = []\n    final_labels = []\n    for l in unique_labels:\n        condition = (labels == l)\n        boxes_by_label = boxes[condition]\n        scores_by_label = scores[condition]\n        labels_by_label = np.array([l] * len(boxes_by_label))\n\n        if method != 3:\n            keep = cpu_soft_nms_float(boxes_by_label.copy(), scores_by_label.copy(), Nt=iou_thr, sigma=sigma, thresh=thresh, method=method)\n        else:\n            # Use faster function\n            keep = nms_float_fast(boxes_by_label, scores_by_label, thresh=iou_thr)\n\n        final_boxes.append(boxes_by_label[keep])\n        final_scores.append(scores_by_label[keep])\n        final_labels.append(labels_by_label[keep])\n    final_boxes = np.concatenate(final_boxes)\n    final_scores = np.concatenate(final_scores)\n    final_labels = np.concatenate(final_labels)\n\n    return final_boxes, final_scores, final_labels, keep\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:31.167916Z","iopub.execute_input":"2023-07-31T13:12:31.168592Z","iopub.status.idle":"2023-07-31T13:12:31.302206Z","shell.execute_reply.started":"2023-07-31T13:12:31.168558Z","shell.execute_reply":"2023-07-31T13:12:31.301207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@jit(nopython=True)\ndef bb_intersection_over_union(A, B) -> float:\n    xA = max(A[0], B[0])\n    yA = max(A[1], B[1])\n    xB = min(A[2], B[2])\n    yB = min(A[3], B[3])\n\n    # compute the area of intersection rectangle\n    interArea = max(0, xB - xA) * max(0, yB - yA)\n\n    if interArea == 0:\n        return 0.0\n\n    # compute the area of both the prediction and ground-truth rectangles\n    boxAArea = (A[2] - A[0]) * (A[3] - A[1])\n    boxBArea = (B[2] - B[0]) * (B[3] - B[1])\n\n    iou = interArea / float(boxAArea + boxBArea - interArea)\n    return iou\n\n\ndef prefilter_boxes(boxes, scores, labels, weights, thr):\n    # Create dict with boxes stored by its label\n    new_boxes = dict()\n\n    for t in range(len(boxes)):\n\n        if len(boxes[t]) != len(scores[t]):\n            print('Error. Length of boxes arrays not equal to length of scores array: {} != {}'.format(len(boxes[t]), len(scores[t])))\n            sys.exit()\n\n        if len(boxes[t]) != len(labels[t]):\n            print('Error. Length of boxes arrays not equal to length of labels array: {} != {}'.format(len(boxes[t]), len(labels[t])))\n            sys.exit()\n\n        for j in range(len(boxes[t])):\n            score = scores[t][j]\n            if score < thr:\n                continue\n            label = int(labels[t][j])\n            box_part = boxes[t][j]\n            x1 = max(float(box_part[0]), 0.)\n            y1 = max(float(box_part[1]), 0.)\n            x2 = max(float(box_part[2]), 0.)\n            y2 = max(float(box_part[3]), 0.)\n\n            # Box data checks\n            if x2 < x1:\n                warnings.warn('X2 < X1 value in box. Swap them.')\n                x1, x2 = x2, x1\n            if y2 < y1:\n                warnings.warn('Y2 < Y1 value in box. Swap them.')\n                y1, y2 = y2, y1\n            if x1 > 1:\n                warnings.warn('X1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x1 = 1\n            if x2 > 1:\n                warnings.warn('X2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                x2 = 1\n            if y1 > 1:\n                warnings.warn('Y1 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y1 = 1\n            if y2 > 1:\n                warnings.warn('Y2 > 1 in box. Set it to 1. Check that you normalize boxes in [0, 1] range.')\n                y2 = 1\n            if (x2 - x1) * (y2 - y1) == 0.0:\n                warnings.warn(\"Zero area box skipped: {}.\".format(box_part))\n                continue\n\n            # [label, score, weight, model index, x1, y1, x2, y2]\n            b = [int(label), float(score) * weights[t], weights[t], t, x1, y1, x2, y2]\n            if label not in new_boxes:\n                new_boxes[label] = []\n            new_boxes[label].append(b)\n\n    # Sort each list in dict by score and transform it to numpy array\n    for k in new_boxes:\n        current_boxes = np.array(new_boxes[k])\n        new_boxes[k] = current_boxes[current_boxes[:, 1].argsort()[::-1]]\n\n    return new_boxes\n\n\ndef get_weighted_box(boxes, conf_type='avg'):\n    \"\"\"\n    Create weighted box for set of boxes\n    :param boxes: set of boxes to fuse\n    :param conf_type: type of confidence one of 'avg' or 'max'\n    :return: weighted box (label, score, weight, x1, y1, x2, y2)\n    \"\"\"\n\n    box = np.zeros(8, dtype=np.float32)\n    conf = 0\n    conf_list = []\n    w = 0\n    for b in boxes:\n        box[4:] += (b[1] * b[4:])\n        conf += b[1]\n        conf_list.append(b[1])\n        w += b[2]\n    box[0] = boxes[0][0]\n    if conf_type == 'avg':\n        box[1] = conf / len(boxes)\n    elif conf_type == 'max':\n        box[1] = np.array(conf_list).max()\n    elif conf_type in ['box_and_model_avg', 'absent_model_aware_avg']:\n        box[1] = conf / len(boxes)\n    box[2] = w\n    box[3] = -1 # model index field is retained for consistensy but is not used.\n    box[4:] /= conf\n    return box\n\n\ndef find_matching_box(boxes_list, new_box, match_iou):\n    best_iou = match_iou\n    best_index = -1\n    for i in range(len(boxes_list)):\n        box = boxes_list[i]\n        if box[0] != new_box[0]:\n            continue\n        iou = bb_intersection_over_union(box[4:], new_box[4:])\n        if iou > best_iou:\n            best_index = i\n            best_iou = iou\n\n    return best_index, best_iou\n\n\ndef weighted_boxes_fusion_tracking(boxes_list, scores_list, labels_list, weights=None, iou_thr=0.6, skip_box_thr=0.0, conf_type='avg', allows_overflow=False):\n    '''\n    :param boxes_list: list of boxes predictions from each model, each box is 4 numbers.\n    It has 3 dimensions (models_number, model_preds, 4)\n    Order of boxes: x1, y1, x2, y2. We expect float normalized coordinates [0; 1]\n    :param scores_list: list of scores for each model\n    :param labels_list: list of labels for each model\n    :param weights: list of weights for each model. Default: None, which means weight == 1 for each model\n    :param iou_thr: IoU value for boxes to be a match\n    :param skip_box_thr: exclude boxes with score lower than this variable\n    :param conf_type: how to calculate confidence in weighted boxes. 'avg': average value, 'max': maximum value, 'box_and_model_avg': box and model wise hybrid weighted average, 'absent_model_aware_avg': weighted average that takes into account the absent model.\n    :param allows_overflow: false if we want confidence score not exceed 1.0\n\n    :return: boxes: boxes coordinates (Order of boxes: x1, y1, x2, y2).\n    :return: scores: confidence scores\n    :return: labels: boxes labels\n    :return: wbfo: original boxes coordinates for each fused box\n    '''\n\n    if weights is None:\n        weights = np.ones(len(boxes_list))\n    if len(weights) != len(boxes_list):\n        print('Warning: incorrect number of weights {}. Must be: {}. Set weights equal to 1.'.format(len(weights), len(boxes_list)))\n        weights = np.ones(len(boxes_list))\n    weights = np.array(weights)\n\n    if conf_type not in ['avg', 'max', 'box_and_model_avg', 'absent_model_aware_avg']:\n        print('Unknown conf_type: {}. Must be \"avg\", \"max\" or \"box_and_model_avg\", or \"absent_model_aware_avg\"'.format(conf_type))\n        sys.exit()\n\n    filtered_boxes = prefilter_boxes(boxes_list, scores_list, labels_list, weights, skip_box_thr)\n    if len(filtered_boxes) == 0:\n        return np.zeros((0, 4)), np.zeros((0,)), np.zeros((0,)), np.zeros((0, 4))\n    \n    overall_boxes = []\n    original_boxes = []\n    for label in filtered_boxes:\n        boxes = filtered_boxes[label]\n        new_boxes = []\n        weighted_boxes = []\n        # Clusterize boxes\n        for j in range(0, len(boxes)):\n            index, best_iou = find_matching_box(weighted_boxes, boxes[j], iou_thr)\n            if index != -1:\n                new_boxes[index].append(boxes[j])\n                weighted_boxes[index] = get_weighted_box(new_boxes[index], conf_type)\n            else:\n                new_boxes.append([boxes[j].copy()])\n                weighted_boxes.append(boxes[j].copy())\n        # Rescale confidence based on number of models and boxes\n        original_boxes.append(new_boxes)\n        for i in range(len(new_boxes)):\n            clustered_boxes = np.array(new_boxes[i])\n            if conf_type == 'box_and_model_avg':\n                # weighted average for boxes\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weighted_boxes[i][2]\n                # identify unique model index by model index column\n                _, idx = np.unique(clustered_boxes[:, 3], return_index=True)\n                # rescale by unique model weights\n                weighted_boxes[i][1] = weighted_boxes[i][1] *  clustered_boxes[idx, 2].sum() / weights.sum()\n            elif conf_type == 'absent_model_aware_avg':\n                # get unique model index in the cluster\n                models = np.unique(clustered_boxes[:, 3]).astype(int)\n                # create a mask to get unused model weights\n                mask = np.ones(len(weights), dtype=bool)\n                mask[models] = False\n                # absent model aware weighted average\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / (weighted_boxes[i][2] + weights[mask].sum())\n            elif conf_type == 'max':\n                weighted_boxes[i][1] = weighted_boxes[i][1] / weights.max()\n            elif not allows_overflow:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * min(len(weights), len(clustered_boxes)) / weights.sum()\n            else:\n                weighted_boxes[i][1] = weighted_boxes[i][1] * len(clustered_boxes) / weights.sum()\n        overall_boxes.append(np.array(weighted_boxes))\n    overall_boxes = np.concatenate(overall_boxes, axis=0)\n    sidx = overall_boxes[:, 1].argsort()\n    overall_boxes = overall_boxes[sidx[::-1]]\n    boxes = overall_boxes[:, 4:]\n    scores = overall_boxes[:, 1]\n    labels = overall_boxes[:, 0]\n    # sort originals accoring to wbf\n    original_boxes = original_boxes[0]\n    wbfo = [original_boxes[i] for i in sidx[::-1]]\n    return boxes, scores, labels, wbfo\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:31.303977Z","iopub.execute_input":"2023-07-31T13:12:31.304685Z","iopub.status.idle":"2023-07-31T13:12:31.345058Z","shell.execute_reply.started":"2023-07-31T13:12:31.304653Z","shell.execute_reply":"2023-07-31T13:12:31.344083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_iou_matrix(masks1, masks2):\n\n    \"\"\"\n    Calculate IoU matrix between two sets of masks\n\n    Parameters\n    ----------\n    masks1: numpy.ndarray of shape (n_objects, height, width)\n        Binary masks 1\n\n    masks2: numpy.ndarray of shape (m_objects, height, width)\n        Binary masks 2\n\n    Returns\n    -------\n    iou_matrix: numpy.ndarray of shape (n_objects, m_objects)\n        IoU matrix between two sets of masks\n    \"\"\"\n\n    if len(list(masks1)) == 0 or len(list(masks2)) == 0:\n        return np.array([[]])\n        \n    encoded_masks1 = mask_util.encode(np.asfortranarray(np.swapaxes(masks1, 0, -1)))\n    encoded_masks2 = mask_util.encode(np.asfortranarray(np.swapaxes(masks2, 0, -1)))\n    is_crowd = [0] * len(encoded_masks2)\n    iou_matrix = mask_util.iou(encoded_masks1, encoded_masks2, is_crowd)\n\n    return iou_matrix\n\n\ndef blend_masks(masks, mask_shape, scores=None, iou_threshold=0.9, label_threshold=0.5, drop_single_components=True):\n\n    \"\"\"\n    Blend set of multiple masks based on IoU\n\n    Parameters\n    ----------\n    masks: list of numpy.ndarray of shape (n_objects, height, width)\n        List of multiple object masks\n\n    mask_shape: tuple\n        Height and width of masks\n\n    iou_threshold: float\n        IoU threshold for merging masks (0 <= iou_threshold <= 1)\n\n    label_threshold: float\n        Label threshold for converting soft predictions to labels (0 <= iou_threshold <= 1)\n\n    drop_single_components: bool\n        Whether to discard predictions without connections or not\n\n    Returns\n    -------\n    blended_masks: numpy.ndarray of shape (n_objects, height, width)\n        Blended binary masks\n\n    blended_scores: numpy.ndarray of shape (n_objects)\n        Blended mask scores\n    \"\"\"\n\n    iou_matrices = {}\n\n    # Create pairwise IoU matrices from given list of masks\n    for i in range(len(masks)):\n        for j in range(i + 1, len(masks)):\n            iou_matrix = get_iou_matrix(masks[i], masks[j])\n            iou_matrices[f'{i}_{j}'] = iou_matrix\n\n    # Create a graph to store connected masks\n    graph = nx.Graph()\n\n    # Add all masks from all lists as nodes\n    for list_idx, masks_ in enumerate(masks):\n        nodes = [f'{list_idx}_{mask_idx}' for mask_idx in np.arange(len(masks_))]\n        graph.add_nodes_from(nodes)\n\n    # Add edges between nodes with IoU >= iou_threshold\n    for pair, iou_matrix in iou_matrices.items():\n        matching_idx = np.where(iou_matrix >= iou_threshold)\n        list1_idx, list2_idx = pair.split('_')\n        edges = [(f'{list1_idx}_{mask1_idx}', f'{list2_idx}_{mask2_idx}') for mask1_idx, mask2_idx in zip(*matching_idx)]\n        graph.add_edges_from(edges)\n\n    blended_masks = []\n    if scores is not None:\n        blended_scores = []\n\n    for connections in nx.connected_components(graph):\n        if len(connections) == 1:\n            if drop_single_components:\n                # Skip mask if it isn't connected to any other mask\n                continue\n            else:\n                # Append mask directly if it isn't connected to any other mask\n                list_idx, mask_idx = list(connections)[0].split('_')\n                list_idx, mask_idx = int(list_idx), int(mask_idx)\n                blended_masks.append(masks[list_idx][mask_idx])\n                if scores is not None:\n                    blended_scores.append(scores[list_idx][mask_idx])\n        else:\n            # Blend mask with its connections and append\n            blended_mask = np.zeros(mask_shape, dtype=np.float32)\n            if scores is not None:\n                blended_score = []\n            for connection in connections:\n                list_idx, mask_idx = connection.split('_')\n                list_idx, mask_idx = int(list_idx), int(mask_idx)\n                # Divide soft predictions with number of connections and accumulate on blended_mask\n                blended_mask += (masks[list_idx][mask_idx] / len(connections))\n                if scores is not None:\n                    # Divide score with number of connections and accumulate on blended_score\n                    blended_score.append(scores[list_idx][mask_idx])\n\n            blended_masks.append(blended_mask)\n            if scores is not None:\n                blended_scores.append((blended_score))\n\n    blended_masks = np.stack(blended_masks)\n    # Convert soft predictions to binary labels\n    blended_masks = np.uint8(blended_masks >= label_threshold)\n\n    if scores is not None:\n        return blended_masks, blended_scores\n    else:\n        return blended_masks\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:31.346733Z","iopub.execute_input":"2023-07-31T13:12:31.347883Z","iopub.status.idle":"2023-07-31T13:12:31.369185Z","shell.execute_reply.started":"2023-07-31T13:12:31.347857Z","shell.execute_reply":"2023-07-31T13:12:31.368264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_wsf_mask(wbf_box, wbf_org, shape, predictions, thres=0, scale=False):\n    \n    mask = np.zeros(shape, dtype=np.uint8)\n\n    for i in range(len(wbf_org)):\n        key = str(np.round(wbf_org[i][4:] * shape[0], 2))\n        model = int(wbf_org[i][3])\n        if key in predictions[model]['mask_lookup']:\n            ind = predictions[model]['mask_lookup'][key]\n            mask = mask + predictions[model]['masks'][ind]\n\n    # convert thres to integer based on number of boxes\n    threshold = max(1, int(thres*len(wbf_org)))\n    # remove pixels outside WBF box\n    m2 = np.zeros(shape, dtype=np.uint8)\n    x1 = max(0, int(shape[0] * wbf_box[0]))\n    y1 = max(0, int(shape[1] * wbf_box[1]))\n    x2 = min(shape[0], int(shape[0] * wbf_box[2]))\n    y2 = min(shape[1], int(shape[1] * wbf_box[3]))\n\n    m2[y1:y2, x1:x2] = 1\n    mask = (mask >= threshold) * m2\n\n    return mask.astype(np.uint8)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:12:31.371563Z","iopub.execute_input":"2023-07-31T13:12:31.371972Z","iopub.status.idle":"2023-07-31T13:12:31.384212Z","shell.execute_reply.started":"2023-07-31T13:12:31.371923Z","shell.execute_reply":"2023-07-31T13:12:31.383240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Inference","metadata":{}},{"cell_type":"code","source":"blend = False\narea_threshold = 60\n\nimage_paths = glob(str(image_directory / '*'))\n\nfor image_path in image_paths:\n\n    image_id = image_path.split('/')[-1].split('.')[0]\n    image = mmcv.imread(image_path)\n    \n    predictions = []\n    \n    for models in [mmdetection_mask_rcnn_models, mmdetection_cascade_mask_rcnn_models]:\n    \n        for fold in [1, 2, 3, 4]:\n\n            fold_predictions = predict_mmdetection(image, models[fold], tta=True)\n            if verbose:\n                print(\n                    f'''\n                    Fold: {fold} Predictions: {len(fold_predictions)}\n                    Predictions Raw: {fold_predictions['raw']['scores'].shape[0]} Mean Score: {fold_predictions['raw']['scores'].mean():.4f}\n                    Predictions Horizontal Flip: {fold_predictions['horizontal_flip']['scores'].shape[0]} Mean Score: {fold_predictions['horizontal_flip']['scores'].mean():.4f}\n                    Predictions Vertical Flip: {fold_predictions['vertical_flip']['scores'].shape[0]} Mean Score: {fold_predictions['vertical_flip']['scores'].mean():.4f}\n                    Predictions Diagonal Flip: {fold_predictions['diagonal_flip']['scores'].shape[0]} Mean Score: {fold_predictions['diagonal_flip']['scores'].mean():.4f}\n                    '''\n                )\n\n            fold_wbf_boxes, fold_wbf_scores, fold_wbf_labels, fold_wbf_originals = weighted_boxes_fusion_tracking(\n                boxes_list=[\n                    fold_predictions['raw']['boxes'] / 512,\n                    fold_predictions['horizontal_flip']['boxes'] / 512,\n                    fold_predictions['vertical_flip']['boxes'] / 512,\n                    fold_predictions['diagonal_flip']['boxes'] / 512\n                ],\n                scores_list=[\n                    fold_predictions['raw']['scores'],\n                    fold_predictions['horizontal_flip']['scores'],\n                    fold_predictions['vertical_flip']['scores'],\n                    fold_predictions['diagonal_flip']['scores']\n                ],\n                labels_list=[\n                    fold_predictions['raw']['labels'],\n                    fold_predictions['horizontal_flip']['labels'],\n                    fold_predictions['vertical_flip']['labels'],\n                    fold_predictions['diagonal_flip']['labels']\n                ],\n                weights=None,\n                iou_thr=0.8,\n                skip_box_thr=0.,\n                conf_type='avg',\n                allows_overflow=False\n            )\n\n            if verbose:\n                print(f'Fold: {fold} WBF: {fold_wbf_scores.shape[0]} Mean Score: {fold_wbf_scores.mean():.4f}')\n\n            if fold_wbf_scores.shape[0] > 0:\n\n                fold_wsf_masks = []\n\n                for wbf_box, wbf_score, wbf_original in zip(fold_wbf_boxes, fold_wbf_scores, fold_wbf_originals):\n                    if len(wbf_original) > 0:\n                        mask = get_wsf_mask(\n                            wbf_box,\n                            wbf_original,\n                            shape=image.shape[:2],\n                            predictions=[\n                                fold_predictions['raw'],\n                                fold_predictions['horizontal_flip'],\n                                fold_predictions['vertical_flip'],\n                                fold_predictions['diagonal_flip']\n                            ],\n                            thres=0,\n                            scale=True\n                        )\n                        fold_wsf_masks.append(mask)\n\n                fold_wsf_masks = np.stack(fold_wsf_masks)\n            else:\n                fold_wsf_masks = np.array([])\n\n            fold_predictions = {\n                'boxes': fold_wbf_boxes,\n                'masks': fold_wsf_masks,\n                'scores': fold_wbf_scores,\n                'labels': fold_wbf_labels,\n                'mask_lookup': {str(np.round(box * 512, 2)): idx for idx, box in enumerate(fold_wbf_boxes)}\n            }\n            predictions.append(fold_predictions)\n            \n    yolo_predictions = predict_yolo(image=image, model=yolo_model, tta=True)\n    for k in yolo_predictions.keys():\n        yolo_predictions[k]['boxes'] /= 512.\n        predictions.append(yolo_predictions[k])\n    \n    if verbose:\n        print(\n            f'''\n            YOLO Predictions: {len(yolo_predictions)}\n            Predictions Raw: {yolo_predictions['raw']['scores'].shape[0]} Mean Score: {yolo_predictions['raw']['scores'].mean():.4f}\n            Predictions Horizontal Flip: {yolo_predictions['horizontal_flip']['scores'].shape[0]} Mean Score: {yolo_predictions['horizontal_flip']['scores'].mean():.4f}\n            Predictions Vertical Flip: {yolo_predictions['vertical_flip']['scores'].shape[0]} Mean Score: {yolo_predictions['vertical_flip']['scores'].mean():.4f}\n            Predictions Diagonal Flip: {yolo_predictions['diagonal_flip']['scores'].shape[0]} Mean Score: {yolo_predictions['diagonal_flip']['scores'].mean():.4f}\n            '''\n        )\n                \n    wbf_boxes, wbf_scores, wbf_labels, wbf_originals = weighted_boxes_fusion_tracking(\n        boxes_list=[prediction['boxes'] for prediction in predictions],\n        scores_list=[prediction['scores'] for prediction in predictions],\n        labels_list=[prediction['labels'] for prediction in predictions],\n        weights=None,\n        iou_thr=0.8,\n        skip_box_thr=0.,\n        conf_type='avg',\n        allows_overflow=False\n    )\n    \n    if verbose:\n        print(f'Predictions WBF: {wbf_scores.shape[0]} Mean Score: {wbf_scores.mean():.4f}')\n    \n    if wbf_scores.shape[0] > 0:\n\n        wsf_masks = []\n\n        for wbf_box, wbf_score, wbf_original in zip(wbf_boxes, wbf_scores, wbf_originals):\n            if len(wbf_original) > 0:\n                mask = get_wsf_mask(\n                    wbf_box,\n                    wbf_original,\n                    shape=image.shape[:2],\n                    predictions=predictions,\n                    thres=0\n                )\n                wsf_masks.append(mask)\n\n        wsf_masks = np.stack(wsf_masks)\n    else:\n        wsf_masks = np.array([])\n            \n    if image_id in polygons['id'].values:\n        glomerulus_masks, _, _ = decode_hhthv_annotations(polygons.loc[polygons['id'] == image_id, 'annotations'].values[0], shape=(512, 512))\n        glomerulus_masks = 1 - np.any(glomerulus_masks == 1, axis=0).astype(np.uint8)\n    else:\n        glomerulus_masks = np.ones((512, 512))\n        \n    masks = wsf_masks\n    scores = wbf_scores\n    \n    prediction_string = ''\n    \n    if len(scores) > 0:\n        \n        masks = np.stack([mask * glomerulus_masks for mask in masks])\n        non_zero_mask_idx = [area >= area_threshold for area in np.sum(masks, axis=(1, 2))]\n        masks = masks[non_zero_mask_idx]\n        scores = np.array(scores)[non_zero_mask_idx]\n        for mask, score in zip(masks, scores):\n            \n            score *= 1.05\n            \n            #kernel = np.ones(shape=(3, 3), dtype=np.uint8)\n            #mask = cv2.dilate(mask.astype(np.uint8), kernel, 1).astype(bool)\n            \n            #if score >= 0.95:\n                #score = 1.0\n            \n            encoded_mask = encode_binary_mask(mask).decode()\n            prediction_string += f'0 {score} {encoded_mask} '\n        \n        prediction_string = prediction_string.strip()\n        \n    df.loc[df['id'] == image_id, 'prediction_string'] = prediction_string\n    \n    if verbose:\n        visualization_outputs = {\n            'boxes': [mask_to_bounding_box(mask) for mask in masks],\n            'masks': masks.astype(np.uint8),\n            'scores': scores,\n            'labels': np.zeros(len(masks)),\n        }\n        visualize_predictions(\n            image=image,\n            ground_truth=None,\n            predictions=visualization_outputs,\n            metadata={\n                'nms_iou_threshold': 1,\n                'score_threshold': 0,\n                'score': np.nan\n            },\n            path=None\n        )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:15:55.591440Z","iopub.execute_input":"2023-07-31T13:15:55.591829Z","iopub.status.idle":"2023-07-31T13:16:15.872847Z","shell.execute_reply.started":"2023-07-31T13:15:55.591799Z","shell.execute_reply":"2023-07-31T13:16:15.871183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[0, 'prediction_string']","metadata":{"execution":{"iopub.status.busy":"2023-07-31T13:16:15.875123Z","iopub.execute_input":"2023-07-31T13:16:15.876343Z","iopub.status.idle":"2023-07-31T13:16:15.885556Z","shell.execute_reply.started":"2023-07-31T13:16:15.876291Z","shell.execute_reply":"2023-07-31T13:16:15.884737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T13:57:53.360661Z","iopub.execute_input":"2023-07-30T13:57:53.361712Z","iopub.status.idle":"2023-07-30T13:57:53.369788Z","shell.execute_reply.started":"2023-07-30T13:57:53.361662Z","shell.execute_reply":"2023-07-30T13:57:53.368813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}