{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"},{"sourceId":7455059,"sourceType":"datasetVersion","datasetId":4339413},{"sourceId":159975318,"sourceType":"kernelVersion"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# simple exploratory data analysis (EDA)","metadata":{}},{"cell_type":"code","source":"import os, glob\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sys\n\nsys.path.append(\"/kaggle/input/rle-run-length-encoding-py-module\")  # this will be path to this notebook\nDATASET_FOLDER = \"/kaggle/input/blood-vessel-segmentation\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-01-22T12:20:53.351687Z","iopub.execute_input":"2024-01-22T12:20:53.352444Z","iopub.status.idle":"2024-01-22T12:20:53.899566Z","shell.execute_reply.started":"2024-01-22T12:20:53.352388Z","shell.execute_reply":"2024-01-22T12:20:53.898107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(os.path.join(DATASET_FOLDER, \"train_rles.csv\"))\ndf_train[[\"dataset\", \"slice\"]] = df_train['id'].str.rsplit(pat='_', n=1, expand=True)\ndisplay(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:20:53.901553Z","iopub.execute_input":"2024-01-22T12:20:53.902077Z","iopub.status.idle":"2024-01-22T12:20:55.406948Z","shell.execute_reply.started":"2024-01-22T12:20:53.902039Z","shell.execute_reply":"2024-01-22T12:20:55.405266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uq_datasets = df_train[\"dataset\"].unique()\nprint(f\"parsed {len(uq_datasets)} unique datasets: {uq_datasets}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:20:55.408550Z","iopub.execute_input":"2024-01-22T12:20:55.409086Z","iopub.status.idle":"2024-01-22T12:20:55.424756Z","shell.execute_reply.started":"2024-01-22T12:20:55.409036Z","shell.execute_reply":"2024-01-22T12:20:55.422807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RLE decoding & encoding\n\n> In order to reduce the submission file size, teams must submit segmentation results using run-length encoding on the pixel values. That is, instead of submitting an exhaustive list of indices for your segmentation, you will submit pairs of values that contain a start position and a run length. E.g. '0 3' implies starting at pixel 0 and running a total of 3 pixels (0,1,2). The competition format requires a space delimited list of pairs. For example, '0 3 10 5' implies pixels 0,1,2, and 10,11,12,13,14 are to be included in the mask. The metric checks that the pairs are sorted, positive, and the decoded pixel values are not duplicated. The pixels are numbered from top to bottom, then left to right: 0 is pixel (0,0), 1 is pixel (1,0), and 2 is pixel (2,0) etc. [source](https://www.kaggle.com/code/leahscherschel/run-length-encoding)","metadata":{}},{"cell_type":"code","source":"from rle import rle_decode, rle_encode","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:20:55.429698Z","iopub.execute_input":"2024-01-22T12:20:55.430219Z","iopub.status.idle":"2024-01-22T12:20:55.448118Z","shell.execute_reply.started":"2024-01-22T12:20:55.430169Z","shell.execute_reply":"2024-01-22T12:20:55.446945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### show some images","metadata":{}},{"cell_type":"code","source":"from skimage import color\n\nshow_args = dict(vmin=0, interpolation='antialiased', interpolation_stage='rgba') \n\nfor _, spl in df_train.sample(5).iterrows():\n    p_img = os.path.join(DATASET_FOLDER, \"train\", spl[\"dataset\"], \"images\", f'{spl[\"slice\"]}.tif')\n    if not os.path.isfile(p_img):\n        print(f\"missing image: {p_img}\")\n        continue\n    fig, axarr = plt.subplots(ncols=3, figsize=(12, 6))\n    img = plt.imread(p_img)\n    rle_mask = rle_decode(spl[\"rle\"], img_shape=img.shape)\n    axarr[0].imshow(img, cmap=\"gray\")\n    axarr[1].imshow(color.label2rgb(rle_mask, img, bg_label=0, bg_color=(1.,1.,1.), alpha=0.25))\n    axarr[2].imshow(rle_mask, **show_args)\n\n    for i in range(3):\n        axarr[i].set_axis_off()","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:20:55.449688Z","iopub.execute_input":"2024-01-22T12:20:55.450246Z","iopub.status.idle":"2024-01-22T12:21:05.626852Z","shell.execute_reply.started":"2024-01-22T12:20:55.450179Z","shell.execute_reply":"2024-01-22T12:21:05.625726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Demo submission","metadata":{}},{"cell_type":"code","source":"!head /kaggle/input/blood-vessel-segmentation/sample_submission.csv","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-01-22T12:21:05.628286Z","iopub.execute_input":"2024-01-22T12:21:05.628691Z","iopub.status.idle":"2024-01-22T12:21:06.766059Z","shell.execute_reply.started":"2024-01-22T12:21:05.628655Z","shell.execute_reply":"2024-01-22T12:21:06.764052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls_images = glob.glob(os.path.join(DATASET_FOLDER, \"test\", \"*\", \"*\", \"*.tif\"))\nprint(f\"found images: {len(ls_images)}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:21:06.768263Z","iopub.execute_input":"2024-01-22T12:21:06.768819Z","iopub.status.idle":"2024-01-22T12:21:06.794923Z","shell.execute_reply.started":"2024-01-22T12:21:06.768761Z","shell.execute_reply":"2024-01-22T12:21:06.793791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.auto import tqdm\n\nsubmission = []\nfor p_img in tqdm(ls_images):\n    path_ = p_img.split(os.path.sep)\n    # parse the submission ID\n    dataset = path_[-3]\n    slice_id, _ = os.path.splitext(path_[-1])\n    # load image to get dimension\n    img = plt.imread(p_img)\n    # sample mask with rectangle\n    mask = np.zeros(img.shape[:2])\n    i, j = int(img.shape[0] / 3), int(img.shape[1] / 4)\n    mask[i:i+50, j:j+100] = 1\n    # submission entry\n    submission.append({\n        \"id\": f\"{dataset}_{slice_id}\",\n        \"rle\": rle_encode(mask)[1]\n    })","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:24:02.217460Z","iopub.execute_input":"2024-01-22T12:24:02.219234Z","iopub.status.idle":"2024-01-22T12:24:02.671354Z","shell.execute_reply.started":"2024-01-22T12:24:02.219141Z","shell.execute_reply":"2024-01-22T12:24:02.670094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.DataFrame(submission)\ndisplay(df_sub.head())\ndf_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T12:24:06.018477Z","iopub.execute_input":"2024-01-22T12:24:06.020164Z","iopub.status.idle":"2024-01-22T12:24:06.040100Z","shell.execute_reply.started":"2024-01-22T12:24:06.020104Z","shell.execute_reply":"2024-01-22T12:24:06.038788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-01-22T12:24:09.707681Z","iopub.execute_input":"2024-01-22T12:24:09.708147Z","iopub.status.idle":"2024-01-22T12:24:10.828730Z","shell.execute_reply.started":"2024-01-22T12:24:09.708109Z","shell.execute_reply":"2024-01-22T12:24:10.827027Z"},"trusted":true},"execution_count":null,"outputs":[]}]}