{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52279,"databundleVersionId":5822112,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import cv2\nimport pickle\nimport json, math\nimport numpy as np\nimport random, time\nimport pandas as pd\nfrom tqdm import tqdm\nimport seaborn as sns\nimport shutil, sys, os, gc\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T00:02:05.397928Z","iopub.execute_input":"2024-05-11T00:02:05.398356Z","iopub.status.idle":"2024-05-11T00:02:06.937586Z","shell.execute_reply.started":"2024-05-11T00:02:05.398330Z","shell.execute_reply":"2024-05-11T00:02:06.935896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"code","source":"# to remove a file or a folder in kaggle directory\ndef delete_file_or_directory(path):\n    \"\"\"\n    Delete a file or directory if it exists.\n\n    Parameters:\n        path (str): Path to the file or directory.\n    \"\"\"\n    if os.path.exists(path):\n        if os.path.isfile(path):\n            os.remove(path)\n            print(f\"{path} has been deleted.\")\n        elif os.path.isdir(path):\n            shutil.rmtree(path)\n            print(f\"{path} and its contents have been deleted.\")\n    else:\n        print(f\"{path} does not exist.\")\n        \ndelete_file_or_directory(\"/kaggle/working/HuPMap\")","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:07:54.321676Z","iopub.execute_input":"2024-05-11T00:07:54.322099Z","iopub.status.idle":"2024-05-11T00:07:54.332091Z","shell.execute_reply.started":"2024-05-11T00:07:54.322067Z","shell.execute_reply":"2024-05-11T00:07:54.330132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To read 'tif' images\ndef image_from_tif(df):\n    image_fpath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\"\n    images = [{'id': row['id'], 'img': cv2.imread(image_fpath + row['id'] + \".tif\")[:,:,::-1]} for _, row in tqdm(df.iterrows(), total=df.shape[0])]\n    \n    return np.array(images)\n\n# To read 'polygons.json' and return an image\ndef mask_from_json(data, image_shape=(512, 512, 3)):\n    image = np.zeros(image_shape, dtype=np.uint8)\n    \n    for annotation in data['annotations']:\n        coordinates = np.array(annotation['coordinates'])\n        \n        if annotation['type'] == 'blood_vessel':\n            color = (255, 0, 0)  # red for glomerulus\n            cv2.fillPoly(image, [coordinates], color=color)\n        if annotation['type'] == 'glomerulus':\n            color = (0, 255, 0)  # Green for blood vessel\n            cv2.fillPoly(image, [coordinates], color=color)\n        if annotation['type'] == 'unsure':\n            color = (0, 0, 255)  # blue for unsure\n            cv2.fillPoly(image, [coordinates], color=color)\n    \n    return image\n\n# To read 'polygons.json' and return an image\ndef masks_info_df(annots):\n    infos = []\n    \n    for annot in annots:\n        info = {}\n        info['id'] = annot['id']\n        info['blood_vessel'] = 0\n        info['glomerulus'] = 0\n        info['unsure'] = 0\n    \n        for annotation in annot['annotations']:\n            info[annotation['type']] = info[annotation['type']] + 1\n            \n        infos.append(info)\n    \n    return pd.DataFrame(infos)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:50:16.321212Z","iopub.execute_input":"2024-05-11T00:50:16.323088Z","iopub.status.idle":"2024-05-11T00:50:16.341022Z","shell.execute_reply.started":"2024-05-11T00:50:16.322999Z","shell.execute_reply":"2024-05-11T00:50:16.338018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Building a dataframe","metadata":{}},{"cell_type":"code","source":"tile_fpath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\"\nwsi_fpath = \"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:00.101754Z","iopub.execute_input":"2024-05-11T00:33:00.102247Z","iopub.status.idle":"2024-05-11T00:33:00.108320Z","shell.execute_reply.started":"2024-05-11T00:33:00.102211Z","shell.execute_reply":"2024-05-11T00:33:00.106862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_df = pd.read_csv(tile_fpath)\ntile_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:00.302016Z","iopub.execute_input":"2024-05-11T00:33:00.302590Z","iopub.status.idle":"2024-05-11T00:33:00.330001Z","shell.execute_reply.started":"2024-05-11T00:33:00.302548Z","shell.execute_reply":"2024-05-11T00:33:00.328477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_df = pd.read_csv(wsi_fpath)\nwsi_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:00.478120Z","iopub.execute_input":"2024-05-11T00:33:00.478539Z","iopub.status.idle":"2024-05-11T00:33:00.496849Z","shell.execute_reply.started":"2024-05-11T00:33:00.478508Z","shell.execute_reply":"2024-05-11T00:33:00.494369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Joining tile_df, and wsi_df on 'source_wsi'\nkidney_df = pd.merge(tile_df, wsi_df, on='source_wsi', how='left')\nkidney_df['dataset_wsi'] = kidney_df['dataset'].astype(str) + kidney_df['source_wsi'].astype(str)\nprint(kidney_df.shape)\nkidney_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:00.659358Z","iopub.execute_input":"2024-05-11T00:33:00.660108Z","iopub.status.idle":"2024-05-11T00:33:00.701618Z","shell.execute_reply.started":"2024-05-11T00:33:00.660051Z","shell.execute_reply":"2024-05-11T00:33:00.697790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.1. Reading and saving masks","metadata":{}},{"cell_type":"code","source":"# Initialize an empty list to store the parsed JSON objects\nannots = []\n\n# Open the JSONL file\nwith open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r') as jsonl_file:\n    # Iterate over each line in the file\n    for line in jsonl_file:\n        # Parse each line as JSON and append it to the list\n        annots.append(json.loads(line))","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:02:06.939593Z","iopub.execute_input":"2024-05-11T00:02:06.940075Z","iopub.status.idle":"2024-05-11T00:02:10.881935Z","shell.execute_reply.started":"2024-05-11T00:02:06.940020Z","shell.execute_reply":"2024-05-11T00:02:10.880733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masks = []\n\nfor annot in tqdm(annots, total=len(annots)):\n    mask = {}\n    mask['id'] = annot['id']\n    mask['img'] = mask_from_json(annot)\n    masks.append(mask)\n    \nmasks = np.array(masks)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:05:44.980881Z","iopub.execute_input":"2024-05-11T00:05:44.982217Z","iopub.status.idle":"2024-05-11T00:05:46.298022Z","shell.execute_reply.started":"2024-05-11T00:05:44.982174Z","shell.execute_reply":"2024-05-11T00:05:46.296373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directory to save the masks\ndirectory = \"HuPMap/masks/\"\n\n# Create the directory if it doesn't exist\nif not os.path.exists(directory):\n    os.makedirs(directory)\n    \n# Iterate over the array of objects\nfor obj in masks:\n    # Extract ID and image array\n    id_ = obj['id']\n    img_array = obj['img']\n    \n    # Save the image array as a numpy file with the ID as the filename\n    np.save(f\"{directory}{id_}.npy\", img_array)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:08:13.945453Z","iopub.execute_input":"2024-05-11T00:08:13.945924Z","iopub.status.idle":"2024-05-11T00:08:15.529155Z","shell.execute_reply.started":"2024-05-11T00:08:13.945886Z","shell.execute_reply":"2024-05-11T00:08:15.527314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.2. Building masks dataframe\n\nand merge it with `kidney_df` for advanced needs.","metadata":{}},{"cell_type":"code","source":"# intialising our masks df\nmask_df = masks_info_df(annots)\nmask_df['annotated'] = 1\nprint(mask_df.shape)\nmask_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:07.682946Z","iopub.execute_input":"2024-05-11T00:33:07.683410Z","iopub.status.idle":"2024-05-11T00:33:07.716863Z","shell.execute_reply.started":"2024-05-11T00:33:07.683383Z","shell.execute_reply":"2024-05-11T00:33:07.714790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Joining tile_df, and wsi_df on 'source_wsi'\nkidney_df = pd.merge(kidney_df, mask_df, on='id', how='left')\nkidney_df['annotated'] = kidney_df['annotated'].fillna(0)\nprint(kidney_df.shape)\nkidney_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:33:07.861563Z","iopub.execute_input":"2024-05-11T00:33:07.862174Z","iopub.status.idle":"2024-05-11T00:33:07.901086Z","shell.execute_reply.started":"2024-05-11T00:33:07.862133Z","shell.execute_reply":"2024-05-11T00:33:07.899565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ensuring everything is fine\nprint(kidney_df[kidney_df['annotated'] == 1].shape)\nprint(kidney_df[(kidney_df['annotated'] == 1) & (kidney_df['blood_vessel'] != 0)].shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:34:39.929953Z","iopub.execute_input":"2024-05-11T00:34:39.930364Z","iopub.status.idle":"2024-05-11T00:34:39.941766Z","shell.execute_reply.started":"2024-05-11T00:34:39.930334Z","shell.execute_reply":"2024-05-11T00:34:39.940275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving dataframe\nkidney_df.to_csv('HuPMap/kidney_tiles.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:52:37.415306Z","iopub.execute_input":"2024-05-11T00:52:37.419503Z","iopub.status.idle":"2024-05-11T00:52:37.502900Z","shell.execute_reply.started":"2024-05-11T00:52:37.419413Z","shell.execute_reply":"2024-05-11T00:52:37.500394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.3. Reading and saving images","metadata":{}},{"cell_type":"code","source":"df = kidney_df[kidney_df['annotated'] == 1]\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:35:21.439141Z","iopub.execute_input":"2024-05-11T00:35:21.439551Z","iopub.status.idle":"2024-05-11T00:35:21.465523Z","shell.execute_reply.started":"2024-05-11T00:35:21.439517Z","shell.execute_reply":"2024-05-11T00:35:21.464672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading images\nimages = image_from_tif(df)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:50:43.251357Z","iopub.execute_input":"2024-05-11T00:50:43.251827Z","iopub.status.idle":"2024-05-11T00:51:22.530355Z","shell.execute_reply.started":"2024-05-11T00:50:43.251779Z","shell.execute_reply":"2024-05-11T00:51:22.528621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directory to save the masks\ndirectory = \"HuPMap/images/\"\n\n# Create the directory if it doesn't exist\nif not os.path.exists(directory):\n    os.makedirs(directory)\n    \n# Iterate over the array of objects\nfor obj in images:\n    # Extract ID and image array\n    id_ = obj['id']\n    img_array = obj['img']\n    \n    # Save the image array as a numpy file with the ID as the filename\n    np.save(f\"{directory}{id_}.npy\", img_array)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:51:24.645799Z","iopub.execute_input":"2024-05-11T00:51:24.646251Z","iopub.status.idle":"2024-05-11T00:51:55.105688Z","shell.execute_reply.started":"2024-05-11T00:51:24.646214Z","shell.execute_reply":"2024-05-11T00:51:55.104129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Showing how to load data","metadata":{}},{"cell_type":"markdown","source":"## loading the dataframe","metadata":{}},{"cell_type":"code","source":"tiles_fpath = \"/kaggle/working/HuPMap/kidney_tiles.csv\"\nkidney_df = pd.read_csv(tiles_fpath)\nprint(kidney_df.shape)\nkidney_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:55:34.243438Z","iopub.execute_input":"2024-05-11T00:55:34.243946Z","iopub.status.idle":"2024-05-11T00:55:34.290538Z","shell.execute_reply.started":"2024-05-11T00:55:34.243911Z","shell.execute_reply":"2024-05-11T00:55:34.289126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting our annotated data only\nkidney_df = kidney_df[kidney_df['annotated'] == 1]\nprint(kidney_df.shape)\nkidney_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T00:56:10.039628Z","iopub.execute_input":"2024-05-11T00:56:10.040212Z","iopub.status.idle":"2024-05-11T00:56:10.074283Z","shell.execute_reply.started":"2024-05-11T00:56:10.040174Z","shell.execute_reply":"2024-05-11T00:56:10.072155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## loading image and mask","metadata":{}},{"cell_type":"markdown","source":"Red color refers to `blood_vessels` channel `1`<br/>\nGreen color refers to `glomerulus` channel `2`<br/>\nBlue color refers to `unsure` channel `3`, where it can be `blood_vessels` or `glomerulus`","metadata":{}},{"cell_type":"code","source":"\"\"\" channel param when one is specified the mask will contain only 'blood_vessels' type,\nwhen 2 is specified 'unsure' type will be added to 'blood_vessels' type,\nany other option the mask will be returned with the 3 channels\n\"\"\"\ndef read_mask(path, channel=2):\n    # loading mask\n    mask = np.load(path)\n    \n    # Select the specified channels\n    if channel == 1:\n        mask = mask[:, :, 0]\n    elif channel == 2:\n        # Sum the pixel values of selected channels along the last axis (channel axis)\n        selected_channels = mask[:, :, [0, 2]]\n        mask = np.sum(selected_channels, axis=2)\n    else:\n        pass\n    \n    # expanding dimension if needed\n    if len(mask.shape) != 3:\n        mask = np.expand_dims(mask, axis=-1)\n        mask = np.where(mask > 0, 1, 0).astype(np.uint8)\n        \n    return mask","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:30:48.402470Z","iopub.execute_input":"2024-05-11T01:30:48.402915Z","iopub.status.idle":"2024-05-11T01:30:48.414094Z","shell.execute_reply.started":"2024-05-11T01:30:48.402881Z","shell.execute_reply":"2024-05-11T01:30:48.411169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_kidney_tiles(df, read_mask=read_mask):\n    fpath = \"/kaggle/working/HuPMap\"\n    X = np.array([np.load(f\"{fpath}/images/{row['id']}.npy\") for i, row in tqdm(df.iterrows(), total=df.shape[0], desc=\"loading images\")])\n    y = np.array([read_mask(f\"{fpath}/masks/{row['id']}.npy\") for i, row in tqdm(df.iterrows(), total=df.shape[0], desc=\"loading masks\")])\n    \n    return X, y","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:30:48.417193Z","iopub.execute_input":"2024-05-11T01:30:48.417634Z","iopub.status.idle":"2024-05-11T01:30:48.446562Z","shell.execute_reply.started":"2024-05-11T01:30:48.417596Z","shell.execute_reply":"2024-05-11T01:30:48.444640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 522\nx, y = load_kidney_tiles(kidney_df.iloc[idx:idx+1])\nprint(kidney_df.iloc[idx])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:30:48.448548Z","iopub.execute_input":"2024-05-11T01:30:48.449081Z","iopub.status.idle":"2024-05-11T01:30:48.512019Z","shell.execute_reply.started":"2024-05-11T01:30:48.449011Z","shell.execute_reply":"2024-05-11T01:30:48.509769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(x[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:30:48.687368Z","iopub.execute_input":"2024-05-11T01:30:48.687816Z","iopub.status.idle":"2024-05-11T01:30:49.119440Z","shell.execute_reply.started":"2024-05-11T01:30:48.687780Z","shell.execute_reply":"2024-05-11T01:30:49.117952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(y[0])\nprint(y[0].shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:30:49.121594Z","iopub.execute_input":"2024-05-11T01:30:49.121960Z","iopub.status.idle":"2024-05-11T01:30:49.469358Z","shell.execute_reply.started":"2024-05-11T01:30:49.121928Z","shell.execute_reply":"2024-05-11T01:30:49.466125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-11T01:31:00.425559Z","iopub.execute_input":"2024-05-11T01:31:00.426072Z","iopub.status.idle":"2024-05-11T01:31:00.444795Z","shell.execute_reply.started":"2024-05-11T01:31:00.426007Z","shell.execute_reply":"2024-05-11T01:31:00.442962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}