{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/efficientnet-pytorch/EfficientNet-PyTorch-master')\nsys.path.append('/kaggle/input/pretrainedmodels/pretrainedmodels-0.7.4')\nsys.path.append('/kaggle/input/pytorch-image-models/pytorch-image-models')\nsys.path.append('/kaggle/input/segmentation-models-pytorch/segmentation_models_pytorch')\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-20T12:07:57.378527Z","iopub.execute_input":"2023-07-20T12:07:57.379401Z","iopub.status.idle":"2023-07-20T12:07:57.392431Z","shell.execute_reply.started":"2023-07-20T12:07:57.379366Z","shell.execute_reply":"2023-07-20T12:07:57.391216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!mkdir /kaggle/working/packages\n!cp -r /kaggle/input/pycocotools/* /kaggle/working/packages\nos.chdir(\"/kaggle/working/packages/pycocotools-2.0.6/\")\n!python setup.py install\n!pip install . --no-index --find-links /kaggle/working/packages/\nos.chdir(\"/kaggle/working\")\n# from pycocotools import _mask as coco_mask","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:07:57.393634Z","iopub.execute_input":"2023-07-20T12:07:57.394163Z","iopub.status.idle":"2023-07-20T12:08:50.835941Z","shell.execute_reply.started":"2023-07-20T12:07:57.394132Z","shell.execute_reply":"2023-07-20T12:08:50.834684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tifffile as tiff\nfrom glob import glob\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom collections import defaultdict\nfrom collections import OrderedDict\nfrom sklearn.model_selection import train_test_split\nimport os\nfrom sklearn.model_selection import GroupKFold\nfrom torch.cuda import amp\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n# from torchinfo import summary\nimport pdb\nimport segmentation_models_pytorch as smp\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport base64\nimport numpy as np\nfrom pycocotools import _mask as coco_mask\nimport typing as t\nimport zlib\nimport cv2\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:08:50.838577Z","iopub.execute_input":"2023-07-20T12:08:50.839700Z","iopub.status.idle":"2023-07-20T12:08:58.615467Z","shell.execute_reply.started":"2023-07-20T12:08:50.839658Z","shell.execute_reply":"2023-07-20T12:08:58.614519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_transforms(CFG):\n    data_transforms = {\n        \"train\": A.Compose([\n            # A.OneOf([\n            #     A.Resize(*CFG.img_size, interpolation=cv2.INTER_NEAREST, p=1.0),\n            # ], p=1),\n            A.HorizontalFlip(p=0.5),\n            # A.VerticalFlip(p=0.5),\n            # A.ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.05, rotate_limit=10, p=0.5),\n            A.OneOf([\n                A.GridDistortion(num_steps=5, distort_limit=0.05, p=1.0),\n                # A.OpticalDistortion(distort_limit=0.05, shift_limit=0.05, p=1.0),\n                A.ElasticTransform(alpha=1, sigma=50, alpha_affine=50, p=1.0)\n            ], p=0.1),\n            A.CoarseDropout(max_holes=8, max_height=CFG.img_size[0]//20, max_width=CFG.img_size[1]//20,\n                            min_holes=5, fill_value=0, mask_fill_value=0, p=0.1),\n#             A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n            ToTensorV2(p=1),\n            ],p=1.0),\n        \n        \"valid_test\": A.Compose([\n            # A.Resize(*CFG.img_size, interpolation=cv2.INTER_NEAREST),\n#             A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n            ToTensorV2(p=1),\n            ], p=1.0)\n        }\n    return data_transforms\n\ndef load_image(path):\n    image = tiff.imread(path)\n    return image\n\ndef transform_into_mask(coordinates):\n    mask = np.zeros(shape=(512, 512))\n    \n    # transform coordinate into mask image\n    for coordinate in coordinates:\n        for axis in coordinate[0]:\n            mask[axis[1], axis[0]] = 1\n    \n    # add channel\n    mask = mask[np.newaxis,:, :]\n    \n    return mask\n\n# Define Dataset class\nclass Hubmap(Dataset):\n    def __init__(self, paths, annotations_dict=None, transforms=None):\n        self.paths = paths\n#         pdb.set_trace()\n        self.ids = [os.path.basename(i).split(\".\")[0]  for i in self.paths]\n        self.annotations_dict = annotations_dict\n        self.length = len(self.paths)\n        self.transforms = transforms\n    \n    def __getitem__(self, idx):\n        path = self.paths[idx]\n        id_ = Path(path).stem\n#         print(\"id_\",id_)\n#         pdb.set_trace()\n        # Load band images\n        image = load_image(path)\n        \n        # Normalization\n        image = image / 255\n        \n        if self.annotations_dict:\n            coordinates = self.annotations_dict[id_]\n            mask = transform_into_mask(coordinates)\n            \n            if self.transforms:\n                transformed = self.transforms(image=image, mask=mask)\n            \n            return {'image': transformed['image'], 'mask': transformed['mask']}\n        else:\n            if self.transforms:\n                transformed = self.transforms(image=image)\n#             print(\"de\",id_)\n            return (transformed['image'],id_)\n    \n    def __len__(self):\n        return self.length\n\ndef build_model(CFG, test_flag=False):\n    if test_flag:\n        pretrain_weights = None\n    else:\n        pretrain_weights = \"imagenet\"\n    model = smp.Unet(\n            encoder_name=CFG.backbone,\n            encoder_weights=pretrain_weights, \n            in_channels = 3,             \n            classes=CFG.num_classes,   \n            activation=None,\n        )\n    model.to(CFG.device)\n    return model\n\ndef encode_binary_mask(mask: np.ndarray) -> t.Text:\n      \"\"\"Converts a binary mask into OID challenge encoding ascii text.\"\"\"\n\n      # check input mask --\n      if mask.dtype != np.bool:\n        raise ValueError(\n            \"encode_binary_mask expects a binary mask, received dtype == %s\" %\n            mask.dtype)\n\n      mask = np.squeeze(mask)\n      if len(mask.shape) != 2:\n        raise ValueError(\n            \"encode_binary_mask expects a 2d mask, received shape == %s\" %\n            mask.shape)\n\n      # convert input mask to expected COCO API input --\n      mask_to_encode = mask.reshape(mask.shape[0], mask.shape[1], 1)\n      mask_to_encode = mask_to_encode.astype(np.uint8)\n      mask_to_encode = np.asfortranarray(mask_to_encode)\n\n      # RLE encode mask --\n      encoded_mask = coco_mask.encode(mask_to_encode)[0][\"counts\"]\n\n      # compress and base64 encoding --\n      binary_str = zlib.compress(encoded_mask, zlib.Z_BEST_COMPRESSION)\n      base64_str = base64.b64encode(binary_str)\n      return base64_str\n\n@torch.no_grad()\ndef test(test_loader,ckpt_paths_dict,CFG):\n    \n    ids = []\n    heights = []\n    widths = []\n    prediction_strings = []\n#     showmask = np.zeros_like(binary_mask)\n    flips = [[-1],[-2],[-2,-1]]    \n    pbar = tqdm(enumerate(test_loader), total=len(test_loader), desc='Test: ')\n    for _, data in pbar:\n        images = data[0]\n        idd = data[1][0]\n#         pdb.set_trace()\n        images  = images.to(CFG.device, dtype=torch.float) # [b, c, h, w]\n        size = images.size()\n        masks = torch.zeros((size[0], 1, size[2], size[3]), device=CFG.device, dtype=torch.float32) # [b, c, w, h]\n        \n        for backbone_name,ckpt_paths in ckpt_paths_dict.items():\n            CFG.backbone = backbone_name\n            for sub_ckpt_path in ckpt_paths:\n                model = build_model(CFG, test_flag=True)\n                model.load_state_dict(torch.load(sub_ckpt_path))\n                model.eval()\n                y_preds = model(images) # [b, c, w, h]\n                y_preds   = torch.nn.Sigmoid()(y_preds)\n                masks += y_preds\n                if CFG.tta:\n                    for f in flips:\n                        image_f = torch.flip(images,f)\n#                         pdb.set_trace()\n                        y_preds = model(image_f)\n                        y_preds = torch.flip(y_preds,f)\n                        y_preds = torch.nn.Sigmoid()(y_preds)\n                        masks  += y_preds\n                \n        if CFG.tta:\n            total_ckpt_len = len(ckpt_paths)*(len(flips)+1)\n        else:\n            total_ckpt_len = len(ckpt_paths_dict) * CFG.n_fold\n            \n        masks /= total_ckpt_len\n        masks = F.interpolate(masks, size=[CFG.org_size, CFG.org_size], mode='bilinear', align_corners=False)\n        pred_string = ''\n        pred = (masks > CFG.thr).float().cpu().numpy()\n        for m in range(len(pred)):\n            kernel = np.ones(shape=(3, 3), dtype=np.uint8)\n            binary_mask = cv2.dilate((pred[m][0] * 255), kernel, 3)\n#             showmask = np.zeros_like(binary_mask).astype(np.uint8)\n            num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(binary_mask.astype(np.uint8),connectivity=8)\n#             print(num_labels)\n#             showmask = np.zeros_like(binary_mask)\n\n\n\n#             for i in range(1,num_labels):\n#                 showmask[labels == i] = i\n#             pdb.set_trace()\n#             showmask = showmask[:,:,np.newaxis]\n#             fig, ax = plt.subplots(1, 3, figsize=(16, 16))\n#             ax0 = ax[0].imshow(images[0,:,:,:].permute(1, 2, 0).to('cpu'), cmap='viridis')\n#             ax[0].set_title(\"{}_image\".format(idd))\n#             ax1 = ax[1].imshow(pred[0,:,:,:].transpose(1, 2, 0), cmap='viridis')\n#             ax[1].set_title(\"pred\")\n# #             ax2 = ax[2].imshow(groudtrue[0,:,:,:].permute(1, 2, 0).to('cpu'), cmap='viridis')\n# #             ax[2].set_title(\"groudtrue\")\n#             ax3 = ax[2].imshow(showmask,cmap='viridis')\n#             ax[2].set_title(\"showmask\")\n            \n            for i in range(1, num_labels):\n                mask_i = np.zeros_like(binary_mask)\n#                 pdb.set_trace()\n                mask_i[labels == i] = 1\n#                 showmask[labels == i] = i\n                mask = mask_i[:, :, np.newaxis].astype(np.bool)\n                score = 1.0\n                encoded = encode_binary_mask(mask)\n                if i==1:\n                    pred_string = f\"0 {score} {encoded.decode('utf-8')}\"\n                else:\n                    pred_string += f\" 0 {score} {encoded.decode('utf-8')}\"\n        b, c, h, w = images.shape\n        ids.append(idd)\n        heights.append(h)\n        widths.append(w)\n        prediction_strings.append(pred_string)\n#         showmask = showmask[:,:,np.newaxis]\n    return ids, heights, widths, prediction_strings","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:08:58.618445Z","iopub.execute_input":"2023-07-20T12:08:58.618825Z","iopub.status.idle":"2023-07-20T12:08:58.655852Z","shell.execute_reply.started":"2023-07-20T12:08:58.618792Z","shell.execute_reply":"2023-07-20T12:08:58.654784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    class CFG:\n        # step1: hyper-parameter\n        polygons_path = '../hubmap-hacking-the-human-vasculature/polygons.jsonl'\n        seed = 42  # birthday\n        num_worker = 0 # debug => 0\n        device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        \n        \n        # step2: data\n        org_size = 512\n        n_fold = 4\n        img_size = [512, 512]\n        train_bs = 16\n        valid_bs = train_bs\n\n        # step3: model\n        backbone = 'efficientnet-b5'  #resnet50 efficientnet-b3\n        num_classes = 1\n\n        # step4: optimizer\n        epoch = 15\n        lr = 1e-3\n        wd = 1e-5\n        lr_drop = 8\n        warmup_epochs = 1\n        # step5: infer\n        thr = 0.4\n        train = False\n        test = True\n        \n        ckpt_fold = \"ckpt-hao\"\n        backbone_1 = 'efficientnet-b5'\n        ckpt_name_1 = \"thr04-efficeinetnet-b5-fold4-e25\"  # for submit. # \n        tta = False\n#         backbone_2 = \"\"\n#         ckpt_name_2 = \"\"\n        \n    # pdb.set_trace()\n    \n        BASE_DIR = Path('/kaggle/input/hubmap-hacking-the-human-vasculature')\n\n        test_paths = glob(f'{BASE_DIR}/test/*')\n    \n    test_dataset = Hubmap(CFG.test_paths, transforms=build_transforms(CFG)[\"valid_test\"])\n    test_dataloader = DataLoader(test_dataset,\n                                 batch_size = 1,\n                                 num_workers = 1,)\n    ckpt_path_1 = f\"../input/{CFG.ckpt_fold}/{CFG.ckpt_name_1}\"\n#     ckpt_path_2 = f\"../input/{CFG.ckpt_fold}/{CFG.ckpt_name_2}\"\n    ckpt_paths_1  = glob(f'{ckpt_path_1}/best*')\n#     ckpt_paths_2  = glob(f'{ckpt_path_2}/best*')\n    assert len(ckpt_path_1)!=0,\"error!\"\n#     assert len(ckpt_paths_1) == CFG.n_fold and len(ckpt_paths_2) == CFG.n_fold, \"ckpt path error!\"\n    ckpt_paths_dict = {CFG.backbone_1:ckpt_paths_1,}\n    ids, heights, widths, prediction_strings = test(test_dataloader,ckpt_paths_dict,CFG)\n#     print(showmask)\n#     plt.imshow(showmask,cmap ='viridis')\n#     plt.show()\n    submission = pd.DataFrame()\n    submission['id'] = ids\n    submission['height'] = heights\n    submission['width'] = widths\n    submission['prediction_string'] = prediction_strings\n    submission = submission.set_index('id')\n    submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:08:58.659008Z","iopub.execute_input":"2023-07-20T12:08:58.659347Z","iopub.status.idle":"2023-07-20T12:09:02.598282Z","shell.execute_reply.started":"2023-07-20T12:08:58.659321Z","shell.execute_reply":"2023-07-20T12:09:02.596454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:09:02.600345Z","iopub.status.idle":"2023-07-20T12:09:02.601190Z","shell.execute_reply.started":"2023-07-20T12:09:02.600909Z","shell.execute_reply":"2023-07-20T12:09:02.600940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if __name__ == \"__main__\":\n#     class CFG:\n#         # step1: hyper-parameter\n#         polygons_path = '../hubmap-hacking-the-human-vasculature/polygons.jsonl'\n#         seed = 42  # birthday\n#         num_worker = 16 # debug => 0\n#         device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        \n        \n#         # step2: data\n#         org_size = 512\n#         n_fold = 4\n#         img_size = [512, 512]\n#         train_bs = 16\n#         valid_bs = train_bs\n\n#         # step3: model\n#         backbone = 'efficientnet-b5'  #resnet50 efficientnet-b3\n#         num_classes = 1\n\n#         # step4: optimizer\n#         epoch = 15\n#         lr = 1e-3\n#         wd = 1e-5\n#         lr_drop = 8\n#         warmup_epochs = 1\n#         # step5: infer\n#         thr = 0.1\n#         train = False\n#         test = True\n        \n#         ckpt_fold = \"ckpt-hao\"\n#         backbone_1 = 'efficientnet-b5'\n#         ckpt_name_1 = \"eff-b5-thr01-e20-nomean\"  # for submit. # \n#         backbone_2 = \"\"\n#         ckpt_name_2 = \"\"\n        \n#     # pdb.set_trace()\n    \n#         BASE_DIR = Path('/kaggle/input/hubmap-hacking-the-human-vasculature')\n\n#         test_paths = glob(f'{BASE_DIR}/test/*')\n    \n#     test_dataset = Hubmap(CFG.test_paths, transforms=build_transforms(CFG)[\"valid_test\"])\n#     test_dataloader = DataLoader(test_dataset,\n#                                  batch_size = 1,\n#                                  num_workers = 1,)\n#     ckpt_path_1 = f\"../input/{CFG.ckpt_fold}/{CFG.ckpt_name_1}\"\n#     ckpt_path_2 = f\"../input/{CFG.ckpt_fold}/{CFG.ckpt_name_2}\"\n#     ckpt_paths_1  = glob(f'{ckpt_path_1}/best*')\n#     ckpt_paths_2  = glob(f'{ckpt_path_2}/best*')\n#     assert len(ckpt_paths_1) == CFG.n_fold and len(ckpt_paths_2) == CFG.n_fold, \"ckpt path error!\"\n#     ckpt_paths_dict = {CFG.backbone_1:ckpt_paths_1, CFG.backbone_2:ckpt_paths_2}\n#     ids, heights, widths, prediction_strings, = test(test_dataloader,ckpt_paths_dict,CFG)\n#     submission = pd.DataFrame()\n#     submission['id'] = ids\n#     submission['height'] = heights\n#     submission['width'] = widths\n#     submission['prediction_string'] = prediction_strings\n#     submission = submission.set_index('id')\n#     submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-07-20T12:09:02.602716Z","iopub.status.idle":"2023-07-20T12:09:02.603504Z","shell.execute_reply.started":"2023-07-20T12:09:02.603248Z","shell.execute_reply":"2023-07-20T12:09:02.603271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}