{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## HuBMAP Inference\n\n### version1,2: legacy version\n\n### version3: [my public train code](https://www.kaggle.com/code/itsuki9180/hubmap-train)\n\n### version4: my code with a few changes. and using dilation. Please read [this discussion on dilation](https://www.kaggle.com/competitions/hubmap-hacking-the-human-vasculature/discussion/416901).","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport sys\nimport json\nfrom PIL import Image\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport tifffile as tiff\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport torch\nimport cv2\nfrom skimage.morphology import binary_dilation\n\nimport pandas as pd\n\nfrom sklearn.model_selection import KFold\n\nsys.path.append(\"/kaggle/input/detection-wheel\")","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:43:05.283183Z","iopub.execute_input":"2023-06-29T06:43:05.283806Z","iopub.status.idle":"2023-06-29T06:43:10.658866Z","shell.execute_reply.started":"2023-06-29T06:43:05.283743Z","shell.execute_reply":"2023-06-29T06:43:10.657890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Install pycocotools package\nimport os\n!mkdir /kaggle/working/packages\n!cp -r /kaggle/input/pycocotools/* /kaggle/working/packages\nos.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n!python setup.py install -q\n!pip install . --no-index --find-links /kaggle/working/packages/ -q\nos.chdir(\"/kaggle/working\")","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:43:10.660723Z","iopub.execute_input":"2023-06-29T06:43:10.661425Z","iopub.status.idle":"2023-06-29T06:44:01.414482Z","shell.execute_reply.started":"2023-06-29T06:43:10.661392Z","shell.execute_reply":"2023-06-29T06:44:01.413255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import base64\nimport numpy as np\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\n\ndef encode_binary_mask(mask: np.ndarray) -> t.Text:\n  \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n  # check input mask --\n  if mask.dtype != np.bool:\n    raise ValueError(\n        \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n        mask.dtype)\n\n  mask = np.squeeze(mask)\n  if len(mask.shape) != 2:\n    raise ValueError(\n        \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n        mask.shape)\n\n  # convert input mask to expected COCO API input --\n  mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n  mask_to_encode = mask_to_encode.astype(np.uint8)\n  mask_to_encode = np.asfortranarray(mask_to_encode)\n\n  # RLE encode mask --\n  encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n  # compress and base64 encoding --\n  binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n  base64_str = base64.b64encode(binary_str)\n  return base64_str","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.416629Z","iopub.execute_input":"2023-06-29T06:44:01.417472Z","iopub.status.idle":"2023-06-29T06:44:01.439187Z","shell.execute_reply.started":"2023-06-29T06:44:01.417403Z","shell.execute_reply":"2023-06-29T06:44:01.438156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport torch\nfrom PIL import Image\n\n\nclass PennFudanDataset(torch.utils.data.Dataset):\n    def __init__(self, imgs, transforms):\n        self.transforms = transforms\n        # load all image files, sorting them to\n        # ensure that they are aligned\n        self.imgs = imgs\n        self.name_indices = [os.path.splitext(os.path.basename(i))[0] for i in imgs]\n\n    def __getitem__(self, idx):\n        # load images and masks\n        img_path = self.imgs[idx]\n        name = self.name_indices[idx]\n        array = tiff.imread(img_path)\n        img = Image.fromarray(array)\n        \n        img, _ = self.transforms(img, img)\n\n        return img, name\n\n    def __len__(self):\n        return len(self.imgs)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.442221Z","iopub.execute_input":"2023-06-29T06:44:01.442596Z","iopub.status.idle":"2023-06-29T06:44:01.453413Z","shell.execute_reply.started":"2023-06-29T06:44:01.442563Z","shell.execute_reply":"2023-06-29T06:44:01.452492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision\nimport torchvision\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection.mask_rcnn import MaskRCNNPredictor\n\ndef get_model_instance_segmentation(num_classes):\n    # load an instance segmentation model pre-trained on COCO\n    model = torchvision.models.detection.maskrcnn_resnet50_fpn_v2(weights=None, weights_backbone=None)\n\n    # get number of input features for the classifier\n    in_features = model.roi_heads.box_predictor.cls_score.in_features\n    # replace the pre-trained head with a new one\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n\n    # now get the number of input features for the mask classifier\n    in_features_mask = model.roi_heads.mask_predictor.conv5_mask.in_channels\n    hidden_layer = 256\n    # and replace the mask predictor with a new one\n    model.roi_heads.mask_predictor = MaskRCNNPredictor(in_features_mask,\n                                                       hidden_layer,\n                                                       num_classes)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.455067Z","iopub.execute_input":"2023-06-29T06:44:01.455851Z","iopub.status.idle":"2023-06-29T06:44:01.704205Z","shell.execute_reply.started":"2023-06-29T06:44:01.455818Z","shell.execute_reply":"2023-06-29T06:44:01.703247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transforms as T\n\ndef get_transform(train):\n    transforms = []\n    transforms.append(T.PILToTensor())\n    transforms.append(T.ConvertImageDtype(torch.float))\n    return T.Compose(transforms)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.705805Z","iopub.execute_input":"2023-06-29T06:44:01.706147Z","iopub.status.idle":"2023-06-29T06:44:01.731024Z","shell.execute_reply.started":"2023-06-29T06:44:01.706113Z","shell.execute_reply":"2023-06-29T06:44:01.730175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from engine import train_one_epoch, evaluate\nimport utils","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.732318Z","iopub.execute_input":"2023-06-29T06:44:01.732673Z","iopub.status.idle":"2023-06-29T06:44:01.759972Z","shell.execute_reply.started":"2023-06-29T06:44:01.732640Z","shell.execute_reply":"2023-06-29T06:44:01.759098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda') if torch.cuda.is_available() else torch.device('cpu')","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.761138Z","iopub.execute_input":"2023-06-29T06:44:01.761477Z","iopub.status.idle":"2023-06-29T06:44:01.795823Z","shell.execute_reply.started":"2023-06-29T06:44:01.761424Z","shell.execute_reply":"2023-06-29T06:44:01.794748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model_instance_segmentation(num_classes=2)\nmodel.to(device)\nmodel.load_state_dict(torch.load('/kaggle/input/hubmap-well-tuned/well_tuned_weight.pth'))\nmodel.eval()\nprint()","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:01.797333Z","iopub.execute_input":"2023-06-29T06:44:01.797942Z","iopub.status.idle":"2023-06-29T06:44:07.875345Z","shell.execute_reply.started":"2023-06-29T06:44:01.797907Z","shell.execute_reply":"2023-06-29T06:44:07.874243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_imgs = glob.glob('/kaggle/input/hubmap-hacking-the-human-vasculature/test/*.tif')\ndataset_test = PennFudanDataset(all_imgs, get_transform(train=False))\ntest_dl = torch.utils.data.DataLoader(\n        dataset_test, batch_size=1, shuffle=False, num_workers=os.cpu_count(), pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:07.880026Z","iopub.execute_input":"2023-06-29T06:44:07.881727Z","iopub.status.idle":"2023-06-29T06:44:07.890002Z","shell.execute_reply.started":"2023-06-29T06:44:07.881687Z","shell.execute_reply":"2023-06-29T06:44:07.888955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\nheights = []\nwidths = []\nprediction_strings = []","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:07.891755Z","iopub.execute_input":"2023-06-29T06:44:07.892182Z","iopub.status.idle":"2023-06-29T06:44:07.900195Z","shell.execute_reply.started":"2023-06-29T06:44:07.892144Z","shell.execute_reply":"2023-06-29T06:44:07.898911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = None\nwith torch.no_grad():\n    for img, idx in test_dl:\n        img = img.to(device)\n        pred = model(img)\n        if sample is None: sample=pred\n        pred_string = ''\n        for m in range(len(pred[0]['masks'])):\n            mask = pred[0]['masks'][m].detach().permute(1,2,0).cpu().numpy()\n            mask = np.where(mask>0.5, 1, 0).astype(np.bool)\n            mask = binary_dilation(mask)\n            \n            score = pred[0]['scores'][m].detach().cpu().numpy()\n            encoded = encode_binary_mask(mask)\n            if m==0:\n                pred_string += f\"0 {score} {encoded.decode('utf-8')}\"\n\n            else:\n                pred_string += f\" 0 {score} {encoded.decode('utf-8')}\"\n        b, c, h, w = img.shape\n        ids.append(idx[0])\n        heights.append(h)\n        widths.append(w)\n        prediction_strings.append(pred_string)","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:07.902066Z","iopub.execute_input":"2023-06-29T06:44:07.902596Z","iopub.status.idle":"2023-06-29T06:44:13.292029Z","shell.execute_reply.started":"2023-06-29T06:44:07.902561Z","shell.execute_reply":"2023-06-29T06:44:13.290840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"array = tiff.imread('/kaggle/input/hubmap-hacking-the-human-vasculature/test/72e40acccadf.tif')\nplt.imshow(array)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:13.293805Z","iopub.execute_input":"2023-06-29T06:44:13.294195Z","iopub.status.idle":"2023-06-29T06:44:13.682109Z","shell.execute_reply.started":"2023-06-29T06:44:13.294157Z","shell.execute_reply":"2023-06-29T06:44:13.681198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(all_imgs)==1:\n    top20 = [sample[0]['masks'][i].cpu().numpy().reshape(512, 512) for i in range(min(20,len(sample[0]['masks'])))]\n    \n    pred_img = np.zeros((512,512), dtype=np.float32)\n    for i, j in enumerate(top20):\n        pred_img += j * (1 - 1/len(top20)*i)\n        pred_img = np.clip(pred_img, 0, 1)\n        print(sample[0]['scores'][i].cpu().numpy())\n        plt.imshow(j)\n        plt.show()\n        \n    plt.imshow(pred_img)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:13.714614Z","iopub.execute_input":"2023-06-29T06:44:13.714984Z","iopub.status.idle":"2023-06-29T06:44:18.939343Z","shell.execute_reply.started":"2023-06-29T06:44:13.714950Z","shell.execute_reply":"2023-06-29T06:44:18.938343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['id'] = ids\nsubmission['height'] = heights\nsubmission['width'] = widths\nsubmission['prediction_string'] = prediction_strings\nsubmission = submission.set_index('id')\nsubmission.to_csv(\"submission.csv\")\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-29T06:44:18.940976Z","iopub.execute_input":"2023-06-29T06:44:18.941625Z","iopub.status.idle":"2023-06-29T06:44:18.974050Z","shell.execute_reply.started":"2023-06-29T06:44:18.941590Z","shell.execute_reply":"2023-06-29T06:44:18.973156Z"},"trusted":true},"execution_count":null,"outputs":[]}]}