{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center>\n<img src=\"https://hubmapconsortium.org/wp-content/uploads/2019/01/HuBMAP-Retina-Logo-Color.png\">\n</center>","metadata":{"execution":{"iopub.status.busy":"2022-10-10T22:07:25.985177Z","iopub.execute_input":"2022-10-10T22:07:25.985774Z","iopub.status.idle":"2022-10-10T22:07:27.103543Z","shell.execute_reply.started":"2022-10-10T22:07:25.985723Z","shell.execute_reply":"2022-10-10T22:07:27.102083Z"}}},{"cell_type":"markdown","source":"\n# HUBMAP EDA + Masks vizualisation \n\n\nSo i have learned now to **read the .tiff** files\n\nNow i wanna **learn how to see the masks** and understand what goes behind making and plotting these MASKS\n\nbut before that, I wanna see what do we have in dataset:)\n\n***If you find my notebooks helpful, you can please leave an upvote :)***","metadata":{}},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"# lets start by importing the libraries i would be needing :)\n\nimport cv2 \nimport tifffile\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:17:42.572956Z","iopub.execute_input":"2022-10-23T19:17:42.573428Z","iopub.status.idle":"2022-10-23T19:17:42.952812Z","shell.execute_reply.started":"2022-10-23T19:17:42.573334Z","shell.execute_reply":"2022-10-23T19:17:42.951421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:17:42.954710Z","iopub.execute_input":"2022-10-23T19:17:42.955106Z","iopub.status.idle":"2022-10-23T19:17:43.377537Z","shell.execute_reply.started":"2022-10-23T19:17:42.955070Z","shell.execute_reply":"2022-10-23T19:17:43.376289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so i wanna see what all organs do we have and their value counts\ndf['organ'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:17:43.378732Z","iopub.execute_input":"2022-10-23T19:17:43.379042Z","iopub.status.idle":"2022-10-23T19:17:43.393302Z","shell.execute_reply.started":"2022-10-23T19:17:43.379014Z","shell.execute_reply":"2022-10-23T19:17:43.391882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_list = ['10044', '10392']\n#2 samples \ninput_dir = '../input/hubmap-organ-segmentation/train_images'\noutput_dir = '.'\nprint(df['id'].dtype)# converting this int to Str\ndf['id'] = df['id'].astype(str)\n# declaring CMAP'S\ncmaps = ['coolwarm_r','bone_r']","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:21.115304Z","iopub.execute_input":"2022-10-23T19:19:21.115829Z","iopub.status.idle":"2022-10-23T19:19:21.126745Z","shell.execute_reply.started":"2022-10-23T19:19:21.115791Z","shell.execute_reply":"2022-10-23T19:19:21.125307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(cmaps) == len(image_list)) # this should be true :-) ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:26.943842Z","iopub.execute_input":"2022-10-23T19:19:26.944694Z","iopub.status.idle":"2022-10-23T19:19:26.951361Z","shell.execute_reply.started":"2022-10-23T19:19:26.944640Z","shell.execute_reply":"2022-10-23T19:19:26.950214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resizer(im_name,scale):\n    '''\n    The resizer takes in 2 args:\n    im_name- name of the image\n    scale- percentage by which the image has to be reduced\n    '''\n    image_path = os.path.join(input_dir, im_name +'.tiff')\n    im_read = tifffile.imread(image_path)\n    width = int(im_read.shape[1] * scale / 100)\n    height = int(im_read.shape[0] * scale / 100)\n    dim = (width, height)\n    print('File name: {}, original size: {}, resized to: {}'.format(im_name , (im_read.shape[0], im_read.shape[1]), (width, height)))\n    resized = cv2.resize(im_read, dim, interpolation=cv2.INTER_AREA)\n    image_path = os.path.join(output_dir, ('r_' + im_name +'.tiff'))\n    tifffile.imwrite(image_path, resized)    ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:31.127761Z","iopub.execute_input":"2022-10-23T19:19:31.128265Z","iopub.status.idle":"2022-10-23T19:19:31.137706Z","shell.execute_reply.started":"2022-10-23T19:19:31.128196Z","shell.execute_reply":"2022-10-23T19:19:31.136678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in image_list:\n    resizer(i,5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:33.713073Z","iopub.execute_input":"2022-10-23T19:19:33.713554Z","iopub.status.idle":"2022-10-23T19:19:34.532094Z","shell.execute_reply.started":"2022-10-23T19:19:33.713516Z","shell.execute_reply":"2022-10-23T19:19:34.530831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle2mask(rle,shape):\n    ## to see a complete code breakdown, REFER version 3.\n    s = rle.split()\n    # the \"s\" here is of dtype('<U7') hence we convert it to \"int\"\n    # very very important step \n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T\n\ndef resize_mask(im_name,scale):\n    '''\n    reads RLE encodings from the df\n    converts to masks and resizes it to a scaling_percentage of original size\n    '''\n    im_read = tifffile.imread(os.path.join(input_dir, im_name +'.tiff'))\n    mask_rle = df[df[\"id\"] == im_name][\"rle\"].values[0]\n    mask = rle2mask(df[df[\"id\"] == im_name][\"rle\"].values[0], (im_read.shape[1], im_read.shape[0]))*255\n    width = int(im_read.shape[1] * scale / 100)\n    height = int(im_read.shape[0] * scale / 100)\n    dim = (width, height)\n    print('File name: {}, original size: {}, resized to: {}'.format(im_name, (im_read.shape[0], im_read.shape[1]), (width, height)))\n    resized = cv2.resize(mask, dim, interpolation=cv2.INTER_AREA)\n    image_path = os.path.join(output_dir, (im_name + '.tiff'))\n    tifffile.imwrite(image_path, resized)    ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:37.013709Z","iopub.execute_input":"2022-10-23T19:19:37.014169Z","iopub.status.idle":"2022-10-23T19:19:37.027468Z","shell.execute_reply.started":"2022-10-23T19:19:37.014132Z","shell.execute_reply":"2022-10-23T19:19:37.026530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for im in image_list:\n    print(im)\n    resize_mask(im, 5)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:40.527748Z","iopub.execute_input":"2022-10-23T19:19:40.528153Z","iopub.status.idle":"2022-10-23T19:19:40.610946Z","shell.execute_reply.started":"2022-10-23T19:19:40.528120Z","shell.execute_reply":"2022-10-23T19:19:40.609824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(output_dir)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:43.998104Z","iopub.execute_input":"2022-10-23T19:19:43.998544Z","iopub.status.idle":"2022-10-23T19:19:44.005366Z","shell.execute_reply.started":"2022-10-23T19:19:43.998509Z","shell.execute_reply":"2022-10-23T19:19:44.004503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now lets plot some Masks","metadata":{}},{"cell_type":"code","source":"def show_image(image_id,cmaps):\n    fig, ax = plt.subplots(nrows=2, ncols=1, figsize=(16, 32))\n    image_path = os.path.join(output_dir, 'r_{}.tiff'.format(image_id))\n    mask_path = os.path.join(output_dir, '{}.tiff'.format(image_id))    \n    image = tifffile.imread(image_path)\n    mask = tifffile.imread(mask_path)\n    hybr = image[:, :,0]/2 + mask[:, :]\n\n    ax[0].imshow(image)\n    ax[0].axis('off')\n    ax[0].set_title('IMAGE')\n    ax[1].imshow(hybr,cmap=cmaps)\n    ax[1].axis('off')\n    ax[1].set_title('MASK ON IMAGE')\n    plt.show()    ","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:46.727647Z","iopub.execute_input":"2022-10-23T19:19:46.728087Z","iopub.status.idle":"2022-10-23T19:19:46.736983Z","shell.execute_reply.started":"2022-10-23T19:19:46.728050Z","shell.execute_reply":"2022-10-23T19:19:46.735776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,j in zip(image_list,cmaps):\n    show_image(i,j)","metadata":{"execution":{"iopub.status.busy":"2022-10-23T19:19:48.988915Z","iopub.execute_input":"2022-10-23T19:19:48.989614Z","iopub.status.idle":"2022-10-23T19:19:49.896864Z","shell.execute_reply.started":"2022-10-23T19:19:48.989575Z","shell.execute_reply":"2022-10-23T19:19:49.895582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NEXT TOPIC\n\nI know how to see masks, but i still **dont know how to make em useful** for modelling or maybe some terms i saw like '**TILING'& \"STAINING**\", etc. so next notebook i will learn how to use these masks :) ","metadata":{}},{"cell_type":"markdown","source":"# <center> learnt something cool !:o) Do leave an Upvote </center>","metadata":{}},{"cell_type":"markdown","source":"REFERENCES-\nI used previous years Yaroslav Isaienkov & The Devastator notebook for inspiration of this notebook :)","metadata":{}},{"cell_type":"markdown","source":"### see you next time folks :)","metadata":{}},{"cell_type":"markdown","source":"# ","metadata":{}}]}