{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(dirname)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-18T19:39:29.650629Z","iopub.execute_input":"2023-11-18T19:39:29.651094Z","iopub.status.idle":"2023-11-18T19:39:38.187465Z","shell.execute_reply.started":"2023-11-18T19:39:29.651043Z","shell.execute_reply":"2023-11-18T19:39:38.186106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nimport pathlib\nimport traceback\nimport pandas as pd\nimport numpy as np\nimport tifffile as tiff\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2023-11-18T19:39:38.191365Z","iopub.execute_input":"2023-11-18T19:39:38.192509Z","iopub.status.idle":"2023-11-18T19:39:38.464192Z","shell.execute_reply.started":"2023-11-18T19:39:38.192427Z","shell.execute_reply":"2023-11-18T19:39:38.462983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef remove_small_objects(img, min_size):\n    img = img > 0\n    img = img.astype(np.uint8)\n    # Find all connected components (labels)\n    num_labels, labels, stats, centroids = cv2.connectedComponentsWithStats(img, connectivity=8)\n\n    # Create a mask where small objects are removed\n    new_img = np.zeros_like(img)\n    for label in range(1, num_labels):\n        if stats[label, cv2.CC_STAT_AREA] >= min_size:\n            new_img[labels == label] = 1\n    new_img = new_img.astype(np.uint8)\n    return new_img\n\ndef readimage(mypath):\n    image = tiff.imread(mypath)\n    assert(len(image.shape)==2)\n    image = image.astype(np.float64)\n    return image\n\n# https://www.kaggle.com/code/paulorzp/run-length-encode-and-decode/script\n# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    mystr = ' '.join(str(x) for x in runs)\n    if mystr == \"\":\n        mystr = \"1 1\"\n    return mystr\n\ndef rle_decode(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T19:39:38.465779Z","iopub.execute_input":"2023-11-18T19:39:38.466281Z","iopub.status.idle":"2023-11-18T19:39:38.483815Z","shell.execute_reply.started":"2023-11-18T19:39:38.466235Z","shell.execute_reply":"2023-11-18T19:39:38.482418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# success.\ndef method0(tif_path):\n    return \"1 1\" \n\n# error.\ndef method1(tif_path):\n    image = readimage(tif_path)\n    output_mask = np.zeros_like(image)==0\n    output_mask = output_mask.astype(np.uint8)\n    rle_str = rle_encode(output_mask)\n    return rle_str\n\n# success - meaning no oom when rle_encode called.\ndef method2(tif_path):\n    image = readimage(tif_path)\n    output_mask = np.zeros_like(image)==0\n    output_mask = output_mask.astype(np.uint8)\n    rle_str = rle_encode(output_mask)\n    rle_str = \"1 1\"\n    return rle_str\n\n# pending submission\ndef method3(tif_path):\n    image = readimage(tif_path)\n    output_mask = np.zeros_like(image)==0\n    output_mask = output_mask.astype(np.uint8)\n    output_mask[:10,:]=0\n    output_mask[:,:10]=0\n    output_mask[-10:,:]=0\n    output_mask[:,-10:]=0\n    rle_str = rle_encode(output_mask)\n    return rle_str\n\n#\n# https://www.kaggle.com/code/hengck23/lb0-534-baseline-simple-unet-seresnext26d-32x4\n#\n#https://www.kaggle.com/competitions/blood-vessel-segmentation/discussion/456033\ndef remove_small_objects(mask, min_size):\n    # Find all connected components (labels)\n    num_label, label, stats, centroid = cv2.connectedComponentsWithStats(mask, connectivity=8)\n\n    # create a mask where small objects are removed\n    processed = np.zeros_like(mask)\n    for l in range(1, num_label):\n        if stats[l, cv2.CC_STAT_AREA] >= min_size:\n            processed[label == l] = 255\n\n    return processed\n\ndef rle_encode_alt(mask): # same as prior rle_encode except for '1 0'\n    pixel = mask.flatten()\n    pixel = np.concatenate([[0], pixel, [0]])\n    run = np.where(pixel[1:] != pixel[:-1])[0] + 1\n    run[1::2] -= run[::2]\n    rle = ' '.join(str(r) for r in run)\n    if rle == '':\n        rle = '1 0'\n    return rle\n\ndef method4(tif_path):\n    image = readimage(tif_path)\n    mask = np.zeros_like(image)==0\n    mask = mask.astype(np.uint8)\n    mask = remove_small_objects(mask, 10) # this line should be doing nothing since we have a mask with all ones.\n    rle_str = rle_encode_alt(mask) # this is same `rle_encode` except if blank mask, it'll return `1 0`\n    return rle_str\n\n\ndef submit():\n    test_folder = \"/kaggle/input/blood-vessel-segmentation/test\"\n    tif_list = sorted([str(x) for x in pathlib.Path(test_folder).rglob(\"*.tif\")])\n    mylist = []\n    for n,tif_path in enumerate(tif_list):\n        dataset_id = os.path.basename(os.path.dirname(os.path.dirname(tif_path)))\n        slice_id = os.path.basename(tif_path).replace(\".tif\",\"\")\n        case_id = f'{dataset_id}_{slice_id}'\n        print(case_id)\n        print(n,tif_path)\n        rle_str = method4(tif_path)\n        myitem={\n            \"id\":case_id,\n            \"rle\":rle_str\n        }\n        mylist.append(myitem)\n\n\n    df = pd.DataFrame(mylist)\n    df.to_csv(\"submission.csv\",index=False)\n\nif __name__ == \"__main__\":\n    submit()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T15:57:37.788045Z","iopub.execute_input":"2023-11-19T15:57:37.788923Z","iopub.status.idle":"2023-11-19T15:57:38.262527Z","shell.execute_reply.started":"2023-11-19T15:57:37.788882Z","shell.execute_reply":"2023-11-19T15:57:38.260951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}