{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\nfrom pathlib import Path\n\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-13T08:00:53.186296Z","iopub.execute_input":"2023-11-13T08:00:53.186742Z","iopub.status.idle":"2023-11-13T08:00:53.193899Z","shell.execute_reply.started":"2023-11-13T08:00:53.186708Z","shell.execute_reply":"2023-11-13T08:00:53.192588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Paths","metadata":{}},{"cell_type":"code","source":"KAGGLE_DIR = Path(\"/\") / \"kaggle\"\nINPUT_DIR = KAGGLE_DIR / \"input\"\nCOMPETITION_DATA_DIR = INPUT_DIR / \"blood-vessel-segmentation\"","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.196976Z","iopub.execute_input":"2023-11-13T08:00:53.198216Z","iopub.status.idle":"2023-11-13T08:00:53.214056Z","shell.execute_reply.started":"2023-11-13T08:00:53.198166Z","shell.execute_reply":"2023-11-13T08:00:53.212616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test 1: Sample submission --> works","metadata":{}},{"cell_type":"code","source":"sample_submission_df = pd.read_csv(COMPETITION_DATA_DIR / \"sample_submission.csv\")\nsample_submission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.216001Z","iopub.execute_input":"2023-11-13T08:00:53.216529Z","iopub.status.idle":"2023-11-13T08:00:53.241333Z","shell.execute_reply.started":"2023-11-13T08:00:53.216403Z","shell.execute_reply":"2023-11-13T08:00:53.240185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample_submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.242851Z","iopub.execute_input":"2023-11-13T08:00:53.244060Z","iopub.status.idle":"2023-11-13T08:00:53.251619Z","shell.execute_reply.started":"2023-11-13T08:00:53.244013Z","shell.execute_reply":"2023-11-13T08:00:53.250476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test 2: Create empty submission with all test images --> fails","metadata":{}},{"cell_type":"code","source":"test_image_paths = sorted((COMPETITION_DATA_DIR / \"test\").glob(\"*/images/*.tif\"))\n\nsubmission = defaultdict(list)\nfor test_image_path in test_image_paths:\n    kidney_id = test_image_path.parent.parent.name\n    slice_idx = test_image_path.stem\n    \n    submission[\"id\"].append(f\"{kidney_id}_{slice_idx}\")\n    submission[\"rle\"].append(\"1 0\")\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.255293Z","iopub.execute_input":"2023-11-13T08:00:53.256065Z","iopub.status.idle":"2023-11-13T08:00:53.277007Z","shell.execute_reply.started":"2023-11-13T08:00:53.256019Z","shell.execute_reply":"2023-11-13T08:00:53.275842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.278532Z","iopub.execute_input":"2023-11-13T08:00:53.279510Z","iopub.status.idle":"2023-11-13T08:00:53.284299Z","shell.execute_reply.started":"2023-11-13T08:00:53.279464Z","shell.execute_reply":"2023-11-13T08:00:53.283231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test 3: Create empty submission with all test images, us '.tif*' globbing --> failed","metadata":{}},{"cell_type":"code","source":"test_image_paths = sorted((COMPETITION_DATA_DIR / \"test\").glob(\"*/images/*.tif*\"))\n\nsubmission = defaultdict(list)\nfor test_image_path in test_image_paths:\n    kidney_id = test_image_path.parent.parent.name\n    slice_idx = test_image_path.stem\n    \n    submission[\"id\"].append(f\"{kidney_id}_{slice_idx}\")\n    submission[\"rle\"].append(\"1 0\")\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.286058Z","iopub.execute_input":"2023-11-13T08:00:53.286728Z","iopub.status.idle":"2023-11-13T08:00:53.307523Z","shell.execute_reply.started":"2023-11-13T08:00:53.286684Z","shell.execute_reply":"2023-11-13T08:00:53.306674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.308833Z","iopub.execute_input":"2023-11-13T08:00:53.309199Z","iopub.status.idle":"2023-11-13T08:00:53.314123Z","shell.execute_reply.started":"2023-11-13T08:00:53.309166Z","shell.execute_reply":"2023-11-13T08:00:53.313022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test 4: Jirka's demo submission (https://www.kaggle.com/code/jirkaborovec/sennet-hoa-rle-decode-encode-demo-submission) --> works","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nDATASET_FOLDER = \"/kaggle/input/blood-vessel-segmentation\"","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.315668Z","iopub.execute_input":"2023-11-13T08:00:53.316094Z","iopub.status.idle":"2023-11-13T08:00:53.325290Z","shell.execute_reply.started":"2023-11-13T08:00:53.316050Z","shell.execute_reply":"2023-11-13T08:00:53.324152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(mask, bg = 0) -> dict:\n    vec = mask.flatten()\n    nb = len(vec)\n    where = np.flatnonzero\n    starts = np.r_[0, where(~np.isclose(vec[1:], vec[:-1], equal_nan=True)) + 1]\n    lengths = np.diff(np.r_[starts, nb])\n    values = vec[starts]\n    assert len(starts) == len(lengths) == len(values)\n    rle = []\n    for start, length, val in zip(starts, lengths, values):\n        if val == bg:\n            continue\n        rle += [str(start), length]\n    # post-processing\n    return \" \".join(map(str, rle))","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.327590Z","iopub.execute_input":"2023-11-13T08:00:53.328070Z","iopub.status.idle":"2023-11-13T08:00:53.338124Z","shell.execute_reply.started":"2023-11-13T08:00:53.328000Z","shell.execute_reply":"2023-11-13T08:00:53.336849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls_images = glob.glob(os.path.join(DATASET_FOLDER, \"test\", \"*\", \"*\", \"*.tif\"))\n\nsubmission = []\nfor p_img in ls_images:\n    path_ = p_img.split(os.path.sep)\n    \n    # parse the submission ID\n    dataset = path_[-3]\n    slice_id, _ = os.path.splitext(path_[-1])\n    \n    # load image to get dimension\n    img = plt.imread(p_img)\n    \n    # sample mask with rectangle\n    mask = np.zeros(img.shape[:2])\n    i, j = int(img.shape[0] / 3), int(img.shape[1] / 4)\n    mask[i:i+50, j:j+100] = 1\n    \n    \n    # submission entry\n    submission.append({\n        \"id\": f\"{dataset}_{slice_id}\",\n        \"rle\": rle_encode(mask)\n    })\n    \ndf_sub = pd.DataFrame(submission)\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.341806Z","iopub.execute_input":"2023-11-13T08:00:53.342241Z","iopub.status.idle":"2023-11-13T08:00:53.517721Z","shell.execute_reply.started":"2023-11-13T08:00:53.342197Z","shell.execute_reply":"2023-11-13T08:00:53.516530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.519173Z","iopub.execute_input":"2023-11-13T08:00:53.519506Z","iopub.status.idle":"2023-11-13T08:00:53.524166Z","shell.execute_reply.started":"2023-11-13T08:00:53.519474Z","shell.execute_reply":"2023-11-13T08:00:53.523104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test 5: Same as Test 2 other dummy RLE","metadata":{}},{"cell_type":"code","source":"test_image_paths = sorted((COMPETITION_DATA_DIR / \"test\").glob(\"*/images/*.tif\"))\n\nsubmission = defaultdict(list)\nfor test_image_path in test_image_paths:\n    kidney_id = test_image_path.parent.parent.name\n    slice_idx = test_image_path.stem\n    \n    submission[\"id\"].append(f\"{kidney_id}_{slice_idx}\")\n    submission[\"rle\"].append(\"1 1 100 10\")\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.526078Z","iopub.execute_input":"2023-11-13T08:00:53.526504Z","iopub.status.idle":"2023-11-13T08:00:53.547184Z","shell.execute_reply.started":"2023-11-13T08:00:53.526461Z","shell.execute_reply":"2023-11-13T08:00:53.546017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T08:00:53.548502Z","iopub.execute_input":"2023-11-13T08:00:53.548842Z","iopub.status.idle":"2023-11-13T08:00:53.555187Z","shell.execute_reply.started":"2023-11-13T08:00:53.548810Z","shell.execute_reply":"2023-11-13T08:00:53.554200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}