{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center>\n<img src=\"https://hubmapconsortium.org/wp-content/uploads/2019/01/HuBMAP-Retina-Logo-Color.png\">\n</center>","metadata":{}},{"cell_type":"markdown","source":"## 00. Imports","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\n\nimport plotly.express as px\n\nfrom skimage import color\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nfrom typing import Tuple, List, Union","metadata":{"execution":{"iopub.status.busy":"2022-07-15T13:19:22.806841Z","iopub.execute_input":"2022-07-15T13:19:22.807960Z","iopub.status.idle":"2022-07-15T13:19:22.941186Z","shell.execute_reply.started":"2022-07-15T13:19:22.807922Z","shell.execute_reply":"2022-07-15T13:19:22.939727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 01. Data loading","metadata":{}},{"cell_type":"code","source":"DATASET_FOLDER = \"/kaggle/input/hubmap-organ-segmentation\"","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:25.837371Z","iopub.execute_input":"2022-07-15T12:26:25.837899Z","iopub.status.idle":"2022-07-15T12:26:25.843860Z","shell.execute_reply.started":"2022-07-15T12:26:25.837852Z","shell.execute_reply":"2022-07-15T12:26:25.842403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -la $DATASET_FOLDER","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:26.036598Z","iopub.execute_input":"2022-07-15T12:26:26.037071Z","iopub.status.idle":"2022-07-15T12:26:26.842153Z","shell.execute_reply.started":"2022-07-15T12:26:26.037037Z","shell.execute_reply":"2022-07-15T12:26:26.840215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = os.path.join(DATASET_FOLDER, \"train.csv\")\ntrain_images_path = os.path.join(DATASET_FOLDER, \"train_images\")\ntrain_annotations_path = os.path.join(DATASET_FOLDER, \"train_annotations\")\ntest_csv_path = os.path.join(DATASET_FOLDER, \"test.csv\")\ntest_images_path = os.path.join(DATASET_FOLDER, \"test_images\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:26.844989Z","iopub.execute_input":"2022-07-15T12:26:26.845442Z","iopub.status.idle":"2022-07-15T12:26:26.852337Z","shell.execute_reply.started":"2022-07-15T12:26:26.845406Z","shell.execute_reply":"2022-07-15T12:26:26.851330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_images_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:26.990663Z","iopub.execute_input":"2022-07-15T12:26:26.991151Z","iopub.status.idle":"2022-07-15T12:26:26.997854Z","shell.execute_reply.started":"2022-07-15T12:26:26.991110Z","shell.execute_reply":"2022-07-15T12:26:26.996686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path)\ntest_df = pd.read_csv(test_csv_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:27.041841Z","iopub.execute_input":"2022-07-15T12:26:27.042195Z","iopub.status.idle":"2022-07-15T12:26:27.378062Z","shell.execute_reply.started":"2022-07-15T12:26:27.042167Z","shell.execute_reply":"2022-07-15T12:26:27.376683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:27.379920Z","iopub.execute_input":"2022-07-15T12:26:27.380448Z","iopub.status.idle":"2022-07-15T12:26:27.424938Z","shell.execute_reply.started":"2022-07-15T12:26:27.380416Z","shell.execute_reply":"2022-07-15T12:26:27.423562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:27.428219Z","iopub.execute_input":"2022-07-15T12:26:27.428562Z","iopub.status.idle":"2022-07-15T12:26:27.456019Z","shell.execute_reply.started":"2022-07-15T12:26:27.428532Z","shell.execute_reply":"2022-07-15T12:26:27.454462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(train_images_path))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:27.851798Z","iopub.execute_input":"2022-07-15T12:26:27.852224Z","iopub.status.idle":"2022-07-15T12:26:27.896100Z","shell.execute_reply.started":"2022-07-15T12:26:27.852189Z","shell.execute_reply":"2022-07-15T12:26:27.895294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:28.336009Z","iopub.execute_input":"2022-07-15T12:26:28.337054Z","iopub.status.idle":"2022-07-15T12:26:28.350240Z","shell.execute_reply.started":"2022-07-15T12:26:28.337004Z","shell.execute_reply":"2022-07-15T12:26:28.349256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(test_images_path))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:28.593119Z","iopub.execute_input":"2022-07-15T12:26:28.594282Z","iopub.status.idle":"2022-07-15T12:26:28.604562Z","shell.execute_reply.started":"2022-07-15T12:26:28.594239Z","shell.execute_reply":"2022-07-15T12:26:28.603255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ID_COLUMN = \"id\"\nORGAN_COLUMN = \"organ\"\nDATA_SOURCE_COLUMN = \"data_source\"\nIMG_HEIGH_COLUMN = \"img_height\"\nIMG_WIDTH_COLUMN = \"img_width\"\nPIXEL_SIZE_COLUMN = \"pixel_size\"\nTISSUE_THICKNESS = \"tissue_thickness\"\nRLE_COLUMN = \"rle\"\nAGE_COLUMN = \"age\"\nSEX_COLUMN = \"sex\"","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:29.049179Z","iopub.execute_input":"2022-07-15T12:26:29.049566Z","iopub.status.idle":"2022-07-15T12:26:29.055883Z","shell.execute_reply.started":"2022-07-15T12:26:29.049534Z","shell.execute_reply":"2022-07-15T12:26:29.054803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 02. Utils","metadata":{}},{"cell_type":"code","source":"def plot_group_by_count(df: pd.DataFrame, column_name: str, plot_title: str) -> None:\n    temp_df = df.groupby([column_name])[column_name].count().reset_index(name='counts')\n    fig = px.pie(temp_df, values=\"counts\", names=column_name, title=plot_title)\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:29.522759Z","iopub.execute_input":"2022-07-15T12:26:29.523159Z","iopub.status.idle":"2022-07-15T12:26:29.529439Z","shell.execute_reply.started":"2022-07-15T12:26:29.523119Z","shell.execute_reply":"2022-07-15T12:26:29.528564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/paulorzp/rle-functions-run-length-encode-decode\n\ndef mask2rle(img: np.ndarray) -> str:\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n \ndef rle2mask(mask_rle: str, shape: Tuple[int, int]) -> np.ndarray:\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:29.899455Z","iopub.execute_input":"2022-07-15T12:26:29.899867Z","iopub.status.idle":"2022-07-15T12:26:29.912473Z","shell.execute_reply.started":"2022-07-15T12:26:29.899837Z","shell.execute_reply":"2022-07-15T12:26:29.910416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 03. Data distribution","metadata":{}},{"cell_type":"code","source":"plot_group_by_count(df=train_df, column_name=DATA_SOURCE_COLUMN, plot_title=DATA_SOURCE_COLUMN)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:31.184192Z","iopub.execute_input":"2022-07-15T12:26:31.184677Z","iopub.status.idle":"2022-07-15T12:26:32.241888Z","shell.execute_reply.started":"2022-07-15T12:26:31.184634Z","shell.execute_reply":"2022-07-15T12:26:32.240480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_group_by_count(df=train_df, column_name=ORGAN_COLUMN, plot_title=ORGAN_COLUMN)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:32.243815Z","iopub.execute_input":"2022-07-15T12:26:32.244156Z","iopub.status.idle":"2022-07-15T12:26:32.303488Z","shell.execute_reply.started":"2022-07-15T12:26:32.244125Z","shell.execute_reply":"2022-07-15T12:26:32.302172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_group_by_count(df=train_df, column_name=SEX_COLUMN, plot_title=SEX_COLUMN)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:32.451895Z","iopub.execute_input":"2022-07-15T12:26:32.452363Z","iopub.status.idle":"2022-07-15T12:26:32.512364Z","shell.execute_reply.started":"2022-07-15T12:26:32.452326Z","shell.execute_reply":"2022-07-15T12:26:32.511264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 04. Per organ image examples","metadata":{}},{"cell_type":"code","source":"def data_frame_contains_columns(df: pd.DataFrame, columns: List[str]) -> bool:\n    return set(columns).issubset(df.columns)\n\ndef assert_data_frame_contains_columns(df: pd.DataFrame, columns: List[str]) -> None:\n    if not data_frame_contains_columns(df=df, columns=columns):\n        raise Exception(f\"Given df does not contain required columns. Given columns: {list(test_df.columns)}. Required columns: {columns}.\")\n\ndef parse_image_paths(image_ids: List[Union[int, str]], image_directory_path: str = \"/kaggle/input/hubmap-organ-segmentation/train_images\", image_extension: str = \"tiff\") -> str:\n    return [\n        os.path.join(image_directory_path, f\"{image_id}.{image_extension}\")\n        for image_id\n        in image_ids\n    ]\n\ndef load_images(image_paths: List[str]) -> List[np.ndarray]:\n    return [\n        plt.imread(image_path)\n        for image_path\n        in image_paths\n    ]\n\ndef load_images_from_df(df: pd.DataFrame, image_directory_path: str = \"/kaggle/input/hubmap-organ-segmentation/train_images\") -> List[np.ndarray]:\n    assert_data_frame_contains_columns(df=df, columns=[ID_COLUMN])\n    image_ids = df[ID_COLUMN].tolist()\n    image_paths = parse_image_paths(image_ids=image_ids, image_directory_path=image_directory_path)\n    return load_images(image_paths=image_paths)\n    \n\ndef load_masks_from_df(df: pd.DataFrame) -> List[np.ndarray]:\n    assert_data_frame_contains_columns(df=df, columns=[IMG_HEIGH_COLUMN, IMG_WIDTH_COLUMN, RLE_COLUMN])\n    mask_rles = df[RLE_COLUMN].tolist()\n    img_widths = df[IMG_WIDTH_COLUMN].tolist()\n    img_heights = df[IMG_HEIGH_COLUMN].tolist()\n    return [\n        rle2mask(mask_rle=mask_rle, shape=(img_width, img_height))\n        for mask_rle, img_width, img_height\n        in zip(mask_rles, img_widths, img_heights)\n    ]\n    \ndef load_annotated_images_from_df(df: pd.DataFrame, image_directory_path: str = \"/kaggle/input/hubmap-organ-segmentation/train_images\") -> List[np.ndarray]:\n    images = load_images_from_df(df=df, image_directory_path=image_directory_path)\n    masks = load_masks_from_df(df)\n    return [\n        color.label2rgb(mask, image, bg_label=0, bg_color=(1.,1.,1.), alpha=0.25)\n        for image, mask\n        in zip(images, masks)\n    ]\n    \ndef plot_image_grid(images: List[np.ndarray], grid_shape: Tuple[int, int] = (4, 4), tile_size: int = 6) -> None:\n    plt.figure(figsize=(grid_shape[0] * tile_size, grid_shape[1] * tile_size))\n\n    for i in range(grid_shape[0] * grid_shape[1]):\n        plt.subplot(grid_shape[0], grid_shape[1], i + 1)\n        plt.imshow(images[i])\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T14:06:42.445504Z","iopub.execute_input":"2022-07-15T14:06:42.446064Z","iopub.status.idle":"2022-07-15T14:06:42.475267Z","shell.execute_reply.started":"2022-07-15T14:06:42.446025Z","shell.execute_reply":"2022-07-15T14:06:42.474408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.01. Kidney train image examples","metadata":{}},{"cell_type":"code","source":"kidney_batch_df = train_df[train_df[ORGAN_COLUMN] == \"kidney\"].head(3 * 3)\nkidney_images = load_annotated_images_from_df(df=kidney_batch_df)\nplot_image_grid(images=kidney_images, grid_shape=(3, 3))\ndel kidney_batch_df\ndel kidney_images","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:26:35.886683Z","iopub.execute_input":"2022-07-15T12:26:35.887788Z","iopub.status.idle":"2022-07-15T12:27:27.256899Z","shell.execute_reply.started":"2022-07-15T12:26:35.887748Z","shell.execute_reply":"2022-07-15T12:27:27.255133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.02. Prostate train image examples","metadata":{}},{"cell_type":"code","source":"prostate_batch_df = train_df[train_df[ORGAN_COLUMN] == \"prostate\"].head(3 * 3)\nprostate_images = load_annotated_images_from_df(df=prostate_batch_df)\nplot_image_grid(images=prostate_images, grid_shape=(3, 3))\ndel prostate_batch_df\ndel prostate_images","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:27:27.259781Z","iopub.execute_input":"2022-07-15T12:27:27.260148Z","iopub.status.idle":"2022-07-15T12:28:15.844775Z","shell.execute_reply.started":"2022-07-15T12:27:27.260116Z","shell.execute_reply":"2022-07-15T12:28:15.842996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.03. Largeintestine train image examples","metadata":{}},{"cell_type":"code","source":"largeintestine_batch_df = train_df[train_df[ORGAN_COLUMN] == \"largeintestine\"].head(3 * 3)\nlargeintestine_images = load_annotated_images_from_df(df=largeintestine_batch_df)\nplot_image_grid(images=largeintestine_images, grid_shape=(3, 3))\ndel largeintestine_batch_df\ndel largeintestine_images","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:28:15.846511Z","iopub.execute_input":"2022-07-15T12:28:15.846914Z","iopub.status.idle":"2022-07-15T12:29:05.450831Z","shell.execute_reply.started":"2022-07-15T12:28:15.846864Z","shell.execute_reply":"2022-07-15T12:29:05.449341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.04. Spleen train image examples","metadata":{}},{"cell_type":"code","source":"spleen_batch_df = train_df[train_df[ORGAN_COLUMN] == \"spleen\"].head(3 * 3)\nspleen_images = load_annotated_images_from_df(df=spleen_batch_df)\nplot_image_grid(images=spleen_images, grid_shape=(3, 3))\ndel spleen_batch_df\ndel spleen_images","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:29:05.453414Z","iopub.execute_input":"2022-07-15T12:29:05.453839Z","iopub.status.idle":"2022-07-15T12:29:55.452426Z","shell.execute_reply.started":"2022-07-15T12:29:05.453803Z","shell.execute_reply":"2022-07-15T12:29:55.450852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.05. Lung train image examples","metadata":{}},{"cell_type":"code","source":"lung_batch_df = train_df[train_df[ORGAN_COLUMN] == \"lung\"].head(3 * 3)\nlung_images = load_annotated_images_from_df(df=lung_batch_df)\nplot_image_grid(images=lung_images, grid_shape=(3, 3))\ndel lung_batch_df\ndel lung_images","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:29:55.454539Z","iopub.execute_input":"2022-07-15T12:29:55.456589Z","iopub.status.idle":"2022-07-15T12:30:44.984409Z","shell.execute_reply.started":"2022-07-15T12:29:55.456510Z","shell.execute_reply":"2022-07-15T12:30:44.983437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 04.06. Test example","metadata":{}},{"cell_type":"code","source":"test_images = load_images_from_df(df=test_df, image_directory_path=test_images_path)\nplot_image_grid(images=test_images, grid_shape=(1, 1))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T12:34:57.873758Z","iopub.execute_input":"2022-07-15T12:34:57.874225Z","iopub.status.idle":"2022-07-15T12:34:58.765206Z","shell.execute_reply.started":"2022-07-15T12:34:57.874187Z","shell.execute_reply":"2022-07-15T12:34:58.763598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 05. Export masks","metadata":{}},{"cell_type":"code","source":"LABELS = list(train_df[ORGAN_COLUMN].unique())\nLABELS","metadata":{"execution":{"iopub.status.busy":"2022-07-15T13:15:31.603857Z","iopub.execute_input":"2022-07-15T13:15:31.605237Z","iopub.status.idle":"2022-07-15T13:15:31.616632Z","shell.execute_reply.started":"2022-07-15T13:15:31.605176Z","shell.execute_reply":"2022-07-15T13:15:31.615119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ORIGINAL_BINARY_MASKS_DATASET_NAME = \"original_binary_masks\"\nORIGINAL_CLASS_MASKS_DATASET_NAME = \"original_class_masks\"\n\nORIGINAL_BINARY_MASKS_DATASET_PATH = os.path.join(\".\", ORIGINAL_BINARY_MASKS_DATASET_NAME)\nORIGINAL_CLASS_MASKS_DATASET_PATH = os.path.join(\".\", ORIGINAL_CLASS_MASKS_DATASET_NAME)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T13:43:06.202538Z","iopub.execute_input":"2022-07-15T13:43:06.203161Z","iopub.status.idle":"2022-07-15T13:43:06.210354Z","shell.execute_reply.started":"2022-07-15T13:43:06.203116Z","shell.execute_reply":"2022-07-15T13:43:06.209417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir -p $ORIGINAL_BINARY_MASKS_DATASET_PATH\n! mkdir -p $ORIGINAL_CLASS_MASKS_DATASET_NAME\n\ntrain_masks = load_masks_from_df(df=train_df)\n\nimage_ids = train_df[ID_COLUMN].tolist()\nimage_organs = train_df[ORGAN_COLUMN].tolist()\n\noriginal_binary_mask_paths = parse_image_paths(image_ids=image_ids, image_directory_path=ORIGINAL_BINARY_MASKS_DATASET_PATH, image_extension=\"png\")\noriginal_class_mask_paths = parse_image_paths(image_ids=image_ids, image_directory_path=ORIGINAL_CLASS_MASKS_DATASET_NAME, image_extension=\"png\")\n\nentries = zip(train_masks, original_binary_mask_paths, original_class_mask_paths, image_organs)\nfor binary_mask, original_binary_mask_path, original_class_mask_path, organ in tqdm(entries, total=len(train_df)):\n    Image.fromarray(binary_mask).save(original_binary_mask_path)\n    class_mask = binary_mask * (LABELS.index(organ) + 1)\n    Image.fromarray(class_mask).save(original_class_mask_path)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T14:08:35.072027Z","iopub.execute_input":"2022-07-15T14:08:35.072557Z","iopub.status.idle":"2022-07-15T14:10:52.852584Z","shell.execute_reply.started":"2022-07-15T14:08:35.072518Z","shell.execute_reply":"2022-07-15T14:10:52.850787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Work still in progress\n---\n**Please upvote**","metadata":{}}]}