{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Requirements","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport glob\nimport matplotlib\nimport numpy as np \nimport pandas as pd\nfrom tqdm import tqdm\nimport tifffile as tiff \nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:13:55.809008Z","iopub.execute_input":"2022-07-05T20:13:55.809696Z","iopub.status.idle":"2022-07-05T20:13:55.856939Z","shell.execute_reply.started":"2022-07-05T20:13:55.809613Z","shell.execute_reply":"2022-07-05T20:13:55.856127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting up Wandb","metadata":{}},{"cell_type":"code","source":"%%capture\n! pip install wandb --upgrade","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:27.161419Z","iopub.execute_input":"2022-07-05T20:14:27.161956Z","iopub.status.idle":"2022-07-05T20:14:38.153545Z","shell.execute_reply.started":"2022-07-05T20:14:27.161916Z","shell.execute_reply":"2022-07-05T20:14:38.152203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:43.595625Z","iopub.execute_input":"2022-07-05T20:14:43.596041Z","iopub.status.idle":"2022-07-05T20:14:45.761113Z","shell.execute_reply.started":"2022-07-05T20:14:43.596009Z","shell.execute_reply":"2022-07-05T20:14:45.759363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Directory Path","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = '../input/hubmap-organ-segmentation/train_images/'","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:48.925286Z","iopub.execute_input":"2022-07-05T20:14:48.925731Z","iopub.status.idle":"2022-07-05T20:14:48.932101Z","shell.execute_reply.started":"2022-07-05T20:14:48.925683Z","shell.execute_reply":"2022-07-05T20:14:48.930456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset Exploration","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/hubmap-organ-segmentation/train.csv\")\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:51.760607Z","iopub.execute_input":"2022-07-05T20:14:51.761060Z","iopub.status.idle":"2022-07-05T20:14:51.954064Z","shell.execute_reply.started":"2022-07-05T20:14:51.761028Z","shell.execute_reply":"2022-07-05T20:14:51.952963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Function to label the bar graph","metadata":{}},{"cell_type":"code","source":"def autolabel(rects):\n    for idx,rect in enumerate(bar_plot):\n        height = rect.get_height()\n        if type(x[idx]) == int:\n          ax.text(rect.get_x() + rect.get_width()/2., 1.0*height,\n                  [x[idx], y[idx]],\n                  ha='center', va='bottom', rotation=90)\n        else:\n          ax.text(rect.get_x() + rect.get_width()/2., 1.0*height,\n                  [x[idx], y[idx]],\n                  ha='center', va='bottom', rotation=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:56.000376Z","iopub.execute_input":"2022-07-05T20:14:56.000814Z","iopub.status.idle":"2022-07-05T20:14:56.009927Z","shell.execute_reply.started":"2022-07-05T20:14:56.000781Z","shell.execute_reply":"2022-07-05T20:14:56.008755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Organs Distribution","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\n\nx = list(train_df['organ'].unique())\ny = list(train_df['organ'].value_counts(sort=False))\n\nbar_plot = plt.bar(x, y)\nautolabel(bar_plot)\nplt.xlabel('organ')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:14:58.940581Z","iopub.execute_input":"2022-07-05T20:14:58.941049Z","iopub.status.idle":"2022-07-05T20:14:59.171536Z","shell.execute_reply.started":"2022-07-05T20:14:58.941013Z","shell.execute_reply":"2022-07-05T20:14:59.170239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Age Distribution","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(15, 5))\n\nx = list(map(int, train_df['age'].unique()))\ny = list(train_df['age'].value_counts(sort=False))\n\nbar_plot = plt.bar(x, y)\nautolabel(bar_plot)\nplt.xlabel('age')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:01.320146Z","iopub.execute_input":"2022-07-05T20:15:01.320572Z","iopub.status.idle":"2022-07-05T20:15:01.630625Z","shell.execute_reply.started":"2022-07-05T20:15:01.320541Z","shell.execute_reply":"2022-07-05T20:15:01.629217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(10, 5))\n\nx = list(train_df['sex'].unique())\ny = list(train_df['sex'].value_counts(sort=False))\n\nbar_plot = plt.bar(x, y)\nautolabel(bar_plot)\nplt.xlabel('sex')\nplt.ylabel('count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:07.060868Z","iopub.execute_input":"2022-07-05T20:15:07.061932Z","iopub.status.idle":"2022-07-05T20:15:07.238279Z","shell.execute_reply.started":"2022-07-05T20:15:07.061886Z","shell.execute_reply":"2022-07-05T20:15:07.236826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Original Image","metadata":{}},{"cell_type":"code","source":"image_id_1 = 10044\nimage_1 = tiff.imread(TRAIN_PATH + str(image_id_1) + \".tiff\")\nprint(image_1.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:10.000962Z","iopub.execute_input":"2022-07-05T20:15:10.001403Z","iopub.status.idle":"2022-07-05T20:15:10.034365Z","shell.execute_reply.started":"2022-07-05T20:15:10.001370Z","shell.execute_reply":"2022-07-05T20:15:10.032818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nplt.imshow(image_1)\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:12.060586Z","iopub.execute_input":"2022-07-05T20:15:12.061028Z","iopub.status.idle":"2022-07-05T20:15:13.421987Z","shell.execute_reply.started":"2022-07-05T20:15:12.060994Z","shell.execute_reply":"2022-07-05T20:15:13.421126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mask to RLE & RLE to Mask","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/paulorzp/rle-functions-run-length-encode-decode\ndef mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle2mask(mask_rle, shape=(1600,256)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:16.760417Z","iopub.execute_input":"2022-07-05T20:15:16.760802Z","iopub.status.idle":"2022-07-05T20:15:16.774267Z","shell.execute_reply.started":"2022-07-05T20:15:16.760772Z","shell.execute_reply":"2022-07-05T20:15:16.772677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Masked Image from RLE","metadata":{}},{"cell_type":"code","source":"mask_1 = rle2mask(train_df[\"rle\"][0], (image_1.shape[1], image_1.shape[0]))\nmask_1.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:19.620532Z","iopub.execute_input":"2022-07-05T20:15:19.620940Z","iopub.status.idle":"2022-07-05T20:15:19.638866Z","shell.execute_reply.started":"2022-07-05T20:15:19.620907Z","shell.execute_reply":"2022-07-05T20:15:19.637793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.imshow(mask_1, cmap='coolwarm', alpha=0.5)\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:21.745705Z","iopub.execute_input":"2022-07-05T20:15:21.746212Z","iopub.status.idle":"2022-07-05T20:15:22.992744Z","shell.execute_reply.started":"2022-07-05T20:15:21.746152Z","shell.execute_reply":"2022-07-05T20:15:22.991505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combining Mask Image and Original Image","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.imshow(image_1)\nplt.imshow(mask_1, cmap='coolwarm', alpha=0.5)\nplt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:26.082712Z","iopub.execute_input":"2022-07-05T20:15:26.083203Z","iopub.status.idle":"2022-07-05T20:15:28.473494Z","shell.execute_reply.started":"2022-07-05T20:15:26.083148Z","shell.execute_reply":"2022-07-05T20:15:28.472362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adding data to wandb artifacts","metadata":{}},{"cell_type":"code","source":"image_files = sorted(glob.glob(TRAIN_PATH+ \"*\"))\nimage_file_df = pd.DataFrame(image_files, columns=['file_name'])\ntrain_df[\"img_height\"] = train_df[\"img_height\"].replace([train_df[\"img_height\"].values],'512')\ntrain_df[\"img_width\"] = train_df[\"img_width\"].replace([train_df[\"img_width\"].values],'512')\ntrain_data = pd.concat([train_df, image_file_df], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:15:49.245790Z","iopub.execute_input":"2022-07-05T20:15:49.246708Z","iopub.status.idle":"2022-07-05T20:15:49.276027Z","shell.execute_reply.started":"2022-07-05T20:15:49.246647Z","shell.execute_reply":"2022-07-05T20:15:49.274633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"run = wandb.init(project='HuBMAP-HPA', entity='cosmo3769')\n\ndata_artifact = wandb.Artifact(name='train', type='RLE-TO-MASK dataset')\ndata_table = wandb.Table(columns=['image_id', \n                               'image', \n                               'mask', \n                               'masked image', \n                               'organ', \n                               'data source', \n                               'image_height', \n                               'image_width', \n                               'pixel size', \n                               'tissue thickness',\n                               'rle',\n                               'age',\n                               'sex'\n                               ])\n\nresized_size = (512, 512)\n\nfor i, df in tqdm(train_data.iterrows()):\n\n        img = tiff.imread(df.file_name)\n        mask = rle2mask(df.rle, (img.shape[1], img.shape[0]))\n        resized_image = cv2.resize(img, resized_size)\n        resized_mask = cv2.resize(mask, resized_size)\n        resized_rle = mask2rle(resized_mask)        \n        \n        plt.figure(figsize=(10,10))\n        plt.axis(\"off\")\n        plt.imshow(resized_image)\n        plt.imshow(resized_mask, cmap='coolwarm', alpha=0.5)\n        plt.savefig(str(df.id) + \"_masked.jpg\")\n        plt.close()\n\n        data_table.add_data(\n            df.id,\n            wandb.Image(resized_image), \n            wandb.Image(resized_mask),\n            wandb.Image(cv2.cvtColor(cv2.imread(str(df.id) + \"_masked.jpg\"), cv2.COLOR_BGR2RGB)),\n            df.organ,\n            df.data_source,\n            df.img_height,\n            df.img_width,\n            df.pixel_size,\n            df.tissue_thickness,\n            resized_rle,\n            df.age,\n            df.sex\n        )\n    \ndata_artifact.add(data_table, 'train-RLE-TO-MASK')\nrun.log_artifact(data_artifact)\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T20:17:09.530778Z","iopub.execute_input":"2022-07-05T20:17:09.531317Z","iopub.status.idle":"2022-07-05T20:22:07.371299Z","shell.execute_reply.started":"2022-07-05T20:17:09.531278Z","shell.execute_reply":"2022-07-05T20:22:07.369458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# [Visualize the image and the masked image at wandb](https://wandb.ai/cosmo3769/HuBMAP-HPA/artifacts/RLE-TO-MASK%20dataset/train/v0/files/train-RLE-TO-MASK.table.json)\n\nFollow the link above to find an interactive visualization.","metadata":{}},{"cell_type":"markdown","source":"### Work in Progress .....","metadata":{}}]}