{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nfrom PIL import Image\n\nimport matplotlib.pyplot as plt \n%matplotlib inline \nplt.style.use(\"bmh\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-27T03:29:46.865390Z","iopub.execute_input":"2022-06-27T03:29:46.866192Z","iopub.status.idle":"2022-06-27T03:29:46.874867Z","shell.execute_reply.started":"2022-06-27T03:29:46.866151Z","shell.execute_reply":"2022-06-27T03:29:46.873995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imgs_root = Path(\"../input/hubmap-organ-segmentation/train_images\")\ndf = pd.read_csv(\"../input/hubmap-organ-segmentation/train.csv\")\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-27T03:29:46.949324Z","iopub.execute_input":"2022-06-27T03:29:46.949938Z","iopub.status.idle":"2022-06-27T03:29:47.132868Z","shell.execute_reply.started":"2022-06-27T03:29:46.949903Z","shell.execute_reply":"2022-06-27T03:29:47.131732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Understanding Data\n- we have 351 biopsy slides from several different organs. (kidney, prostate, large intestine, spleen, lung)\n- image are of very large size and have high resolution. ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8*2, 3.5*3), nrows=3, ncols=2)\n\norgan = df[\"organ\"].value_counts()\nax.flat[0].bar(organ.index, organ.values)\nax.flat[0].set_title(\"organ counts\")\n\nax.flat[1].scatter(df[\"img_height\"].values, df[\"img_width\"].values)\nax.flat[1].set_title(\"img height width\")\n\nax.flat[2].hist(df[\"pixel_size\"].values)\nax.flat[2].set_title(\"pixel_size\")\n\nax.flat[3].hist(df[\"age\"].values)\nax.flat[3].set_title(\"age\")\n\nax.flat[4].hist(df[\"tissue_thickness\"].values)\nax.flat[4].set_title(\"tissue_thicknees\")\n\nsex = df[\"sex\"].value_counts()\nax.flat[5].bar(sex.index, sex.values)\nax.flat[5].set_title(\"sex\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-27T03:29:47.135065Z","iopub.execute_input":"2022-06-27T03:29:47.135378Z","iopub.status.idle":"2022-06-27T03:29:48.051065Z","shell.execute_reply.started":"2022-06-27T03:29:47.135349Z","shell.execute_reply":"2022-06-27T03:29:48.049869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle2mask(rle_string, img_shape):\n    rle = [int(i) for i in rle_string.split(' ')]\n    pairs = list(zip(rle[0::2],rle[1::2]))\n\n    p_loc = []\n\n    for start, length in pairs:\n        for p_pos in range(start, start + length):\n            p_loc.append((p_pos % img_shape[1], p_pos // img_shape[0]))\n\n    canvas = np.zeros(img_shape).T\n    canvas[tuple(zip(*p_loc))] = 1.0\n\n    return canvas","metadata":{"execution":{"iopub.status.busy":"2022-06-27T03:29:48.053011Z","iopub.execute_input":"2022-06-27T03:29:48.053710Z","iopub.status.idle":"2022-06-27T03:29:48.061890Z","shell.execute_reply.started":"2022-06-27T03:29:48.053675Z","shell.execute_reply":"2022-06-27T03:29:48.060329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing images ","metadata":{}},{"cell_type":"code","source":"def get_random_img_mask(df):\n    idx = np.random.randint(len(df))\n    info = df.iloc[idx].to_dict()\n    hpa_id = info[\"id\"]\n    organ = info[\"organ\"]\n    img_height, img_width = info[\"img_height\"], info[\"img_width\"]\n    rle = info[\"rle\"]\n    mask = rle2mask(rle, (img_width, img_height))\n    img = np.asarray(Image.open(train_imgs_root/(str(hpa_id)+\".tiff\")))\n    return img, mask, info\n\ndef vis_mask_img(img, mask, info):\n    fig, ax = plt.subplots(figsize=(8*2, 3.5*3), nrows=1, ncols=3)\n\n    ax.flat[0].imshow(img)\n    ax.flat[0].set_title(info[\"organ\"])\n\n    ax.flat[1].imshow(mask)\n    ax.flat[1].set_title(\"mask\")\n    \n    #https://www.kaggle.com/code/sohaibanwaar1203/polygons-and-masks-visualisation\n    ax.flat[2].imshow(np.dstack((mask, np.zeros(mask.shape), np.argmax(img, axis=-1))))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-27T03:29:48.064810Z","iopub.execute_input":"2022-06-27T03:29:48.065571Z","iopub.status.idle":"2022-06-27T03:29:48.078234Z","shell.execute_reply.started":"2022-06-27T03:29:48.065524Z","shell.execute_reply":"2022-06-27T03:29:48.077031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organs = df[\"organ\"].unique()\nfor organ in organs:\n    dd = df[df[\"organ\"] == organ].reset_index(drop=True)\n    print(f\"visualizing: {organ}\")\n    for _ in range(2):\n        img, mask, info = get_random_img_mask(dd)\n        vis_mask_img(img, mask, info)","metadata":{"execution":{"iopub.status.busy":"2022-06-27T03:29:48.080894Z","iopub.execute_input":"2022-06-27T03:29:48.081349Z","iopub.status.idle":"2022-06-27T03:30:58.404678Z","shell.execute_reply.started":"2022-06-27T03:29:48.081263Z","shell.execute_reply":"2022-06-27T03:30:58.403530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}