{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What makes a \"baseline\"?\n\nA lot of competitors share a common mentality: \"start simple, then gain in complexity\". The best baselines are the most simple and easy to implement techniques, as they define the lower bound on engineering effort (and often performance as well). However, in this competition, most competitors (myself included) quickly started throwing ML at the problem without questioning if there are simpler techniques. After training a few models, I started to question if there is a simpler traditional technique that could be performant for this task.\n\nIn this notebook, we look at a very simple conventional image processing technique that does surprisingly well on the Leaderboard. This approach clearly has it's flaws, and I haven't even started playing around with modifications (I'm in a rush to get to a holiday party as I type this), but the implications are really cool. Could you define some heuristics to clean this naive mask up? Could you use this mask as additional information in network training/post-processing?  \n\n## TODO:\n- [ ] Investigate improving the masks\n- [ ] Apply the technique to the high-resolution partition\n- [ ] Investigate folding this in to network training (additional channel in input imagery?)\n- [ ] Investigate fusing this intelligently with network output\n\nI will update this notebook as I continue to play around.\n____","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport skimage as skimage\nimport rasterio\n\nfrom glob import glob\nfrom torch.utils.data import DataLoader\nimport scipy.spatial\nimport skimage\n\n# This is a required patch because scipy appears to be broken in the current conda env\nclass QhullError(Exception):\n    pass\n\nscipy.spatial.QhullError = QhullError\n\nfrom skimage.feature import canny","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-24T01:22:31.805549Z","iopub.execute_input":"2023-11-24T01:22:31.805918Z","iopub.status.idle":"2023-11-24T01:22:38.052040Z","shell.execute_reply.started":"2023-11-24T01:22:31.805885Z","shell.execute_reply":"2023-11-24T01:22:38.050686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle: str, img_shape: tuple = None) -> np.ndarray:\n    seq = mask_rle.split()\n    starts = np.array(list(map(int, seq[0::2])))\n    lengths = np.array(list(map(int, seq[1::2])))\n    assert len(starts) == len(lengths)\n    ends = starts + lengths\n    img = np.zeros((np.product(img_shape),), dtype=np.uint8)\n    for begin, end in zip(starts, ends):\n        img[begin:end] = 1\n    # https://stackoverflow.com/a/46574906/4521646\n    img.shape = img_shape\n    return img","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-24T01:22:38.053835Z","iopub.execute_input":"2023-11-24T01:22:38.054509Z","iopub.status.idle":"2023-11-24T01:22:38.060872Z","shell.execute_reply.started":"2023-11-24T01:22:38.054477Z","shell.execute_reply":"2023-11-24T01:22:38.059663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    rle = ' '.join(str(x) for x in runs)\n    if rle=='':\n        rle = '1 0'\n    return rle","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-11-24T01:22:38.063067Z","iopub.execute_input":"2023-11-24T01:22:38.064345Z","iopub.status.idle":"2023-11-24T01:22:38.080457Z","shell.execute_reply.started":"2023-11-24T01:22:38.064301Z","shell.execute_reply":"2023-11-24T01:22:38.078529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n\n#https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/456033\ndef remove_small_objects(mask, min_size):\n    # Find all connected components (labels)\n    num_label, label, stats, centroid = cv2.connectedComponentsWithStats(mask, connectivity=8)\n\n    # create a mask where small objects are removed\n    processed = np.zeros_like(mask)\n    for l in range(1, num_label):\n        if stats[l, cv2.CC_STAT_AREA] >= min_size:\n            processed[label == l] = 255\n\n    return processed","metadata":{"execution":{"iopub.status.busy":"2023-11-24T01:22:38.082852Z","iopub.execute_input":"2023-11-24T01:22:38.083178Z","iopub.status.idle":"2023-11-24T01:22:38.408310Z","shell.execute_reply.started":"2023-11-24T01:22:38.083149Z","shell.execute_reply":"2023-11-24T01:22:38.406325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Demonstration of Canny + Filling on Training Data","metadata":{}},{"cell_type":"code","source":"from skimage import color\nfrom PIL import Image\n\nfolder = 'train'\nDATASET_FOLDER = \"/kaggle/input/blood-vessel-segmentation\"\nIMG_IDX = 490\n\nls_images = list(sorted(glob(os.path.join(DATASET_FOLDER, folder, \"*\", \"*\", \"*.tif\"))))\n\nimg_fpath = ls_images[IMG_IDX]\ndataset_name = img_fpath.split('/')[-3]\nreader = rasterio.open(img_fpath)\nimg = reader.read()\n\nimg = ((img - img.min()) / (img.max() - img.min())).transpose(1, 2, 0)\n    \nedges = canny(img[:, :, 0], sigma=1.1)\nfill_img = scipy.ndimage.binary_fill_holes(edges)\nprediction = fill_img ^ edges\n\nprediction = remove_small_objects(prediction.astype(np.uint8) * 255, 20)\nprediction[prediction == 255] = 1\nprediction = prediction.astype(bool)\nfig, axarr = plt.subplots(ncols=5, figsize=(12, 6))\n\nrle = rle_encode(prediction)\nreconstructed_mask = rle_decode(rle, img_shape=img.shape)\n\naxarr[0].imshow(img, cmap=\"gray\")\naxarr[1].imshow(color.label2rgb(prediction, img.repeat(3, -1), bg_label=0, bg_color=(1.,1.,1.), alpha=0.25))\naxarr[2].imshow(prediction, vmin=0, interpolation='antialiased', interpolation_stage='rgba')\naxarr[3].imshow(Image.open(os.path.join(DATASET_FOLDER, folder, dataset_name, 'labels', os.path.basename(img_fpath))), vmin=0, interpolation='antialiased', interpolation_stage='rgba')\naxarr[4].imshow(reconstructed_mask, vmin=0, interpolation='antialiased', interpolation_stage='rgba')\n\nfor i in range(3):\n    axarr[i].set_axis_off()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T01:22:38.412826Z","iopub.execute_input":"2023-11-24T01:22:38.413216Z","iopub.status.idle":"2023-11-24T01:22:43.857737Z","shell.execute_reply.started":"2023-11-24T01:22:38.413181Z","shell.execute_reply":"2023-11-24T01:22:43.856899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Running Inference","metadata":{}},{"cell_type":"code","source":"from glob import glob\nfrom torch.utils.data import DataLoader\nimport scipy.spatial\nimport skimage\nimport os\n\n# This is a required patch because scipy appears to be broken in the current conda env\nclass QhullError(Exception):\n    pass\n\nscipy.spatial.QhullError = QhullError\n\nfrom skimage.feature import canny\nfolder = 'test'\nDATASET_FOLDER = \"/kaggle/input/blood-vessel-segmentation\"\n\nls_images = [(y[0], y[2].split('.')[0]) for y in [x.split('/')[-3:] for x in sorted(glob(os.path.join(DATASET_FOLDER, folder, \"*\", \"*\", \"*.tif\")))]]\npath_image_dir = os.path.join(DATASET_FOLDER, folder)\ntest_df = pd.DataFrame([['_'.join(x), '1 0', x[0], x[1]] for x in ls_images], columns=['id', 'rle', 'dataset', 'slice'])\n\n\nfor idx, row in test_df.iterrows():\n    img_fpath = os.path.join(DATASET_FOLDER, folder, row[\"dataset\"], \"images\", f'{row[\"slice\"]}.tif')\n    \n    reader = rasterio.open(img_fpath)\n    img = reader.read()\n\n    img = ((img - img.min()) / (img.max() - img.min())).transpose(1, 2, 0)\n    \n    edges = canny(img[:, :, 0],\n                              sigma=1)\n    fill_img = scipy.ndimage.binary_fill_holes(edges)\n    prediction = fill_img ^ edges\n\n    prediction = remove_small_objects(prediction.astype(np.uint8) * 255, 20)\n    prediction[prediction == 255] = 1\n    prediction = prediction.astype(bool)\n    \n    rle = rle_encode(prediction)\n    test_df.loc[idx, 'rle'] = rle","metadata":{"execution":{"iopub.status.busy":"2023-11-24T01:22:43.858715Z","iopub.execute_input":"2023-11-24T01:22:43.859072Z","iopub.status.idle":"2023-11-24T01:22:44.841982Z","shell.execute_reply.started":"2023-11-24T01:22:43.859039Z","shell.execute_reply":"2023-11-24T01:22:44.841282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = test_df.loc[:, ['id', 'rle']]\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T01:22:44.842949Z","iopub.execute_input":"2023-11-24T01:22:44.844378Z","iopub.status.idle":"2023-11-24T01:22:44.866300Z","shell.execute_reply.started":"2023-11-24T01:22:44.844305Z","shell.execute_reply":"2023-11-24T01:22:44.863963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2023-11-24T01:22:44.869083Z","iopub.execute_input":"2023-11-24T01:22:44.869638Z","iopub.status.idle":"2023-11-24T01:22:44.887490Z","shell.execute_reply.started":"2023-11-24T01:22:44.869559Z","shell.execute_reply":"2023-11-24T01:22:44.886397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}