{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"},{"sourceId":150248402,"sourceType":"kernelVersion"},{"sourceId":150386064,"sourceType":"kernelVersion"}],"dockerImageVersionId":30579,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Abstract","metadata":{}},{"cell_type":"markdown","source":"### **Note**\nAs a result of [changes in the scoring system](https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/455220), submitting this notebook as is now will result in a Scoring Error.  \nThe root cause is unknown, but it seems to be more likely to occur when the prediction results are complex (e.g., inaccurate or predicting a lot of small noise).\n\nCurrent typical countermeasures are as follows  \n1. Remove small masks  \n2. Adjusting the probability threshold  \n3. Training a model with better accuracy  \n\nThe first has already been implemented in this notebook.  \n`pred = remove_small_objects(pred, 10)`  \nYou try to remove regions with area less than 10, but this parameter can be adjusted.  \n(Reference: https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/456033)  \n\nThe second, probability threshold, is also set to 0.5 by default in this notebook, but can be changed to suit your model.  \n`preds = (nn.Sigmoid()(preds)>0.5).double()`  \n\nFor the third, a training notebook is also [available](https://www.kaggle.com/code/kashiwaba/sennet-hoa-train-unet-simple-baseline).  \nIt is fairly basic, so you should be able to easily increase your score by incorporating your own innovations. (Perhaps just changing the value of config will produce a model with a score of 0.4 ~.)","metadata":{}},{"cell_type":"markdown","source":"**Simple baseline with Unet**  \n* Training notebook is [here](https://www.kaggle.com/code/kashiwaba/sennet-hoa-train-unet-simple-baseline)    \n* Please upvote if it helps you.","metadata":{}},{"cell_type":"markdown","source":"# Prepare","metadata":{}},{"cell_type":"code","source":"!python -m pip install --no-index --find-links=/kaggle/input/pip-download-for-segmentation-models-pytorch segmentation-models-pytorch","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-11-18T00:41:29.490126Z","iopub.execute_input":"2023-11-18T00:41:29.490896Z","iopub.status.idle":"2023-11-18T00:41:48.694898Z","shell.execute_reply.started":"2023-11-18T00:41:29.490864Z","shell.execute_reply":"2023-11-18T00:41:48.693830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom glob import glob\nfrom tqdm import tqdm\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport albumentations as A\nimport segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:48.696762Z","iopub.execute_input":"2023-11-18T00:41:48.697073Z","iopub.status.idle":"2023-11-18T00:41:56.452727Z","shell.execute_reply.started":"2023-11-18T00:41:48.697043Z","shell.execute_reply":"2023-11-18T00:41:56.451863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed          = 42\n    debug         = False # set debug=False for Full Training\n    exp_name      = 'baseline'\n    comment       = 'unet-efficientnet_b1-512x512'\n    model_name    = 'Unet'\n    backbone      = 'efficientnet-b1'\n    ckpt_path     = '/kaggle/input/sennet-hoa-train-unet-simple-baseline/best_epoch.bin'\n    valid_bs      = 32\n    img_size      = [512, 512]\n    num_classes   = 1\n    device        = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n    \n    data_transforms = {\n        \"train\": A.Compose([\n            A.Resize(*img_size, interpolation=cv2.INTER_NEAREST),\n            A.HorizontalFlip(p=0.5),\n        ], p=1.0),\n        \n        \"valid\": A.Compose([\n            A.Resize(*img_size, interpolation=cv2.INTER_NEAREST),\n        ], p=1.0)\n    }","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.453874Z","iopub.execute_input":"2023-11-18T00:41:56.454159Z","iopub.status.idle":"2023-11-18T00:41:56.485805Z","shell.execute_reply.started":"2023-11-18T00:41:56.454134Z","shell.execute_reply":"2023-11-18T00:41:56.484844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    rle = ' '.join(str(x) for x in runs)\n    if rle=='':\n        rle = '1 0'\n    return rle","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.487856Z","iopub.execute_input":"2023-11-18T00:41:56.488133Z","iopub.status.idle":"2023-11-18T00:41:56.505489Z","shell.execute_reply.started":"2023-11-18T00:41:56.488108Z","shell.execute_reply":"2023-11-18T00:41:56.504564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"def load_img(path):\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    img = np.tile(img[...,None], [1, 1, 3]) # gray to rgb\n    img = img.astype('float32') # original is uint16\n    mx = np.max(img)\n    if mx:\n        img/=mx # scale image to [0, 1]\n    return img\n\ndef load_msk(path):\n    msk = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    msk = msk.astype('float32')\n    msk/=255.0\n    return msk","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.506666Z","iopub.execute_input":"2023-11-18T00:41:56.507025Z","iopub.status.idle":"2023-11-18T00:41:56.516946Z","shell.execute_reply.started":"2023-11-18T00:41:56.506991Z","shell.execute_reply":"2023-11-18T00:41:56.516080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BuildDataset(torch.utils.data.Dataset):\n    def __init__(self, img_paths, msk_paths=[], transforms=None):\n        self.img_paths  = img_paths\n        self.msk_paths  = msk_paths\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.img_paths)\n    \n    def __getitem__(self, index):\n        img_path  = self.img_paths[index]\n        img = load_img(img_path)\n        \n        if len(self.msk_paths)>0:\n            msk_path = self.msk_paths[index]\n            msk = load_msk(msk_path)\n            if self.transforms:\n                data = self.transforms(image=img, mask=msk)\n                img  = data['image']\n                msk  = data['mask']\n            img = np.transpose(img, (2, 0, 1))\n            return torch.tensor(img), torch.tensor(msk)\n        else:\n            orig_size = img.shape\n            if self.transforms:\n                data = self.transforms(image=img)\n                img  = data['image']\n            img = np.transpose(img, (2, 0, 1))\n            return torch.tensor(img), torch.tensor(np.array([orig_size[0], orig_size[1]]))","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.518350Z","iopub.execute_input":"2023-11-18T00:41:56.518705Z","iopub.status.idle":"2023-11-18T00:41:56.529456Z","shell.execute_reply.started":"2023-11-18T00:41:56.518669Z","shell.execute_reply":"2023-11-18T00:41:56.528627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_FOLDER = \"/kaggle/input/blood-vessel-segmentation\"\nls_images = glob(os.path.join(DATASET_FOLDER, \"test\", \"*\", \"*\", \"*.tif\"))\nprint(f\"found images: {len(ls_images)}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.530633Z","iopub.execute_input":"2023-11-18T00:41:56.531295Z","iopub.status.idle":"2023-11-18T00:41:56.561382Z","shell.execute_reply.started":"2023-11-18T00:41:56.531242Z","shell.execute_reply":"2023-11-18T00:41:56.560515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BuildDataset(ls_images, [], transforms=CFG.data_transforms['valid'])\ntest_loader = DataLoader(test_dataset, batch_size=CFG.valid_bs, num_workers=0, shuffle=False, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.562591Z","iopub.execute_input":"2023-11-18T00:41:56.562909Z","iopub.status.idle":"2023-11-18T00:41:56.569132Z","shell.execute_reply.started":"2023-11-18T00:41:56.562883Z","shell.execute_reply":"2023-11-18T00:41:56.568266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def build_model(backbone, num_classes, device):\n    model = smp.Unet(\n        encoder_name=backbone,      # choose encoder, e.g. mobilenet_v2 or efficientnet-b7\n        encoder_weights=None,     # use `imagenet` pre-trained weights for encoder initialization\n        in_channels=3,                  # model input channels (1 for gray-scale images, 3 for RGB, etc.)\n        classes=num_classes,        # model output channels (number of classes in your dataset)\n        activation=None,\n    )\n    model.to(device)\n    return model\n\ndef load_model(backbone, num_classes, device, path):\n    model = build_model(backbone, num_classes, device)\n    model.load_state_dict(torch.load(path))\n    model.eval()\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.570309Z","iopub.execute_input":"2023-11-18T00:41:56.570570Z","iopub.status.idle":"2023-11-18T00:41:56.578088Z","shell.execute_reply.started":"2023-11-18T00:41:56.570546Z","shell.execute_reply":"2023-11-18T00:41:56.577125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = load_model(CFG.backbone, CFG.num_classes, CFG.device, CFG.ckpt_path)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:41:56.580739Z","iopub.execute_input":"2023-11-18T00:41:56.580997Z","iopub.status.idle":"2023-11-18T00:42:00.349233Z","shell.execute_reply.started":"2023-11-18T00:41:56.580974Z","shell.execute_reply":"2023-11-18T00:42:00.348286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"def remove_small_objects(img, min_size):\n    # Find all connected components (labels)\n    num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(img, connectivity=8)\n\n    # Create a mask where small objects are removed\n    new_img = np.zeros_like(img)\n    for label in range(1, num_labels):\n        if stats[label, cv2.CC_STAT_AREA] >= min_size:\n            new_img[labels == label] = 1\n\n    return new_img","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:42:00.350414Z","iopub.execute_input":"2023-11-18T00:42:00.350698Z","iopub.status.idle":"2023-11-18T00:42:00.356583Z","shell.execute_reply.started":"2023-11-18T00:42:00.350673Z","shell.execute_reply":"2023-11-18T00:42:00.355549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rles = []\npbar = tqdm(enumerate(test_loader), total=len(test_loader), desc='Inference ')\nfor step, (images, shapes) in pbar:\n    shapes = shapes.numpy()\n    images = images.to(CFG.device, dtype=torch.float)\n    with torch.no_grad():\n        preds = model(images)\n        preds = (nn.Sigmoid()(preds)>0.5).double()\n    preds = preds.cpu().numpy().astype(np.uint8)\n\n    for pred, shape in zip(preds, shapes):\n        pred = cv2.resize(pred[0], (shape[1], shape[0]), cv2.INTER_NEAREST)\n        pred = remove_small_objects(pred, 10)\n        rle = rle_encode(pred)\n        rles.append(rle)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:42:00.357979Z","iopub.execute_input":"2023-11-18T00:42:00.358320Z","iopub.status.idle":"2023-11-18T00:42:05.537992Z","shell.execute_reply.started":"2023-11-18T00:42:00.358288Z","shell.execute_reply":"2023-11-18T00:42:05.537043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\nfor p_img in tqdm(ls_images):\n    path_ = p_img.split(os.path.sep)\n    # parse the submission ID\n    dataset = path_[-3]\n    slice_id, _ = os.path.splitext(path_[-1])\n    ids.append(f\"{dataset}_{slice_id}\")","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:42:05.539343Z","iopub.execute_input":"2023-11-18T00:42:05.539689Z","iopub.status.idle":"2023-11-18T00:42:05.549616Z","shell.execute_reply.started":"2023-11-18T00:42:05.539655Z","shell.execute_reply":"2023-11-18T00:42:05.548586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame.from_dict({\n    \"id\": ids,\n    \"rle\": rles\n})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:42:05.550897Z","iopub.execute_input":"2023-11-18T00:42:05.551177Z","iopub.status.idle":"2023-11-18T00:42:05.572286Z","shell.execute_reply.started":"2023-11-18T00:42:05.551153Z","shell.execute_reply":"2023-11-18T00:42:05.571334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:42:05.573507Z","iopub.execute_input":"2023-11-18T00:42:05.573802Z","iopub.status.idle":"2023-11-18T00:42:05.590311Z","shell.execute_reply.started":"2023-11-18T00:42:05.573777Z","shell.execute_reply":"2023-11-18T00:42:05.589122Z"},"trusted":true},"execution_count":null,"outputs":[]}]}