{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![Segment multi-organ functional tissue units](https://storage.googleapis.com/kaggle-competitions/kaggle/34547/logos/header.png?t=2022-02-15-22-37-27) > Segment multi-organ functional tissue units","metadata":{}},{"cell_type":"markdown","source":"## Please upvote if you find it useful! Thanks","metadata":{}},{"cell_type":"markdown","source":"## **Elementary Data Exploration**","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\nimport tifffile\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport plotly.graph_objects as go\nfrom plotly import tools\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:19.247766Z","iopub.execute_input":"2022-06-22T23:10:19.248226Z","iopub.status.idle":"2022-06-22T23:10:20.500036Z","shell.execute_reply.started":"2022-06-22T23:10:19.24813Z","shell.execute_reply":"2022-06-22T23:10:20.498942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '../input/hubmap-organ-segmentation'\nTRAIN_PATH = os.path.join(BASE_PATH, 'train_images')\nTEST_PATH = os.path.join(BASE_PATH, 'test_images')\n\n\nprint(f'Total number of train images = {len(os.listdir(TRAIN_PATH))}')\nprint(f'Total number of test images = {len(os.listdir(TEST_PATH))}')","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:20.501801Z","iopub.execute_input":"2022-06-22T23:10:20.502125Z","iopub.status.idle":"2022-06-22T23:10:20.549861Z","shell.execute_reply.started":"2022-06-22T23:10:20.502094Z","shell.execute_reply":"2022-06-22T23:10:20.548772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train masks**","metadata":{}},{"cell_type":"markdown","source":"In addition to the individual IDs for each picture, train.csv also includes an RLE-encoded version of the mask for the image's objects.","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(os.path.join(BASE_PATH, 'train.csv'))\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:20.553742Z","iopub.execute_input":"2022-06-22T23:10:20.554082Z","iopub.status.idle":"2022-06-22T23:10:20.937638Z","shell.execute_reply.started":"2022-06-22T23:10:20.554053Z","shell.execute_reply":"2022-06-22T23:10:20.936867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Submission df**","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv(os.path.join(BASE_PATH, 'sample_submission.csv'))\n\ndf_sub","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:20.939821Z","iopub.execute_input":"2022-06-22T23:10:20.940279Z","iopub.status.idle":"2022-06-22T23:10:20.953226Z","shell.execute_reply.started":"2022-06-22T23:10:20.94025Z","shell.execute_reply":"2022-06-22T23:10:20.952458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility Functions","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/paulorzp/rle-functions-run-lenght-encode-decode\ndef rle2mask(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n\n    starts, lengths = [\n        np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])\n    ]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape).T\n\n\ndef read_image(image_id, scale=None, verbose=1):\n    image = tifffile.imread(\n        os.path.join(BASE_PATH, f\"train_images/{image_id}.tiff\")\n    )\n    \n    if len(image.shape) == 5:\n        image = image.squeeze().transpose(1, 2, 0)\n        \n    mask = rle2mask(\n        df_train[df_train['id'] == image_id]['rle'].values[0], \n        (image.shape[1], image.shape[0])\n                    )\n    if verbose:\n        print(f\"[{image_id}] Image shape: {image.shape}\")\n        print(f\"[{image_id}] Mask shape: {mask.shape}\")\n    \n    if scale:\n        new_size = (image.shape[1] // scale, image.shape[0] // scale)\n        image = cv2.resize(image, new_size)\n        mask = cv2.resize(mask, new_size)\n        \n        if verbose:\n            print(f\"[{image_id}] Resized Image shape: {image.shape}\")\n            print(f\"[{image_id}] Resized Mask shape: {mask.shape}\")\n        \n    return image, mask\n\n\ndef read_test_image(image_id, scale=None, verbose=1):\n    image = tifffile.imread(\n        os.path.join(BASE_PATH, f\"test/{image_id}.tiff\")\n    )\n    if len(image.shape) == 5:\n        image = image.squeeze().transpose(1, 2, 0)\n    \n    if verbose:\n        print(f\"[{image_id}] Image shape: {image.shape}\")\n    \n    if scale:\n        new_size = (image.shape[1] // scale, image.shape[0] // scale)\n        image = cv2.resize(image, new_size)\n        \n        if verbose:\n            print(f\"[{image_id}] Resized Image shape: {image.shape}\")\n        \n    return image\n\n\ndef plot_image_and_mask(image, mask, image_id):\n    plt.figure(figsize=(16, 10))\n    \n    plt.subplot(1, 3, 1)\n    plt.imshow(image)\n    plt.title(f\"Image {image_id}\", fontsize=18)\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(image)\n    plt.imshow(mask, cmap=\"hot\", alpha=0.5)\n    plt.title(f\"Image {image_id} + mask\", fontsize=18)    \n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(mask, cmap=\"hot\")\n    plt.title(f\"Mask\", fontsize=18)    \n    \n    plt.show()\n    \n    \ndef plot_grid_image_with_mask(image, mask):\n    plt.figure(figsize=(16, 16))\n    \n    w_len = image.shape[0]\n    h_len = image.shape[1]\n    \n    min_len = min(w_len, h_len)\n    w_start = (w_len - min_len) // 2\n    h_start = (h_len - min_len) // 2\n    \n    plt.imshow(image[w_start : w_start + min_len, h_start : h_start + min_len])\n    plt.imshow(\n        mask[w_start : w_start + min_len, h_start : h_start + min_len], cmap=\"hot\", alpha=0.5,\n    )\n    plt.axis(\"off\")\n            \n    plt.show()\n    \n\ndef plot_slice_image_and_mask(image, mask, start_h, end_h, start_w, end_w):\n    plt.figure(figsize=(16, 5))\n    \n    sub_image = image[start_h:end_h, start_w:end_w, :]\n    sub_mask = mask[start_h:end_h, start_w:end_w]\n    \n    plt.subplot(1, 3, 1)\n    plt.imshow(sub_image)\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 2)\n    plt.imshow(sub_image)\n    plt.imshow(sub_mask, cmap=\"hot\", alpha=0.5)\n    plt.axis(\"off\")\n    \n    plt.subplot(1, 3, 3)\n    plt.imshow(sub_mask, cmap=\"hot\")\n    plt.axis(\"off\")\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:20.954413Z","iopub.execute_input":"2022-06-22T23:10:20.954889Z","iopub.status.idle":"2022-06-22T23:10:20.982657Z","shell.execute_reply.started":"2022-06-22T23:10:20.954851Z","shell.execute_reply":"2022-06-22T23:10:20.981716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizations of Images and Masks","metadata":{}},{"cell_type":"code","source":"image_ids, images, masks = [[] for _ in range(3)]\n\nfor item in tqdm(range(4)):\n    idx = np.random.randint(0, high=len(df_train), size=(1,))[0]\n    image_id = df_train.iloc[idx].id\n    image, mask = read_image(image_id)\n    \n    images.append(image)\n    masks.append(mask)\n    \n    image_ids.append(image_id)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:20.983754Z","iopub.execute_input":"2022-06-22T23:10:20.984612Z","iopub.status.idle":"2022-06-22T23:10:23.022116Z","shell.execute_reply.started":"2022-06-22T23:10:20.984571Z","shell.execute_reply":"2022-06-22T23:10:23.020944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 16))\nfor ind, (image_id, image) in enumerate(zip(image_ids, images)):\n    plt.subplot(2, 2, ind + 1)\n    plt.imshow(image)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:23.024039Z","iopub.execute_input":"2022-06-22T23:10:23.024825Z","iopub.status.idle":"2022-06-22T23:10:27.584326Z","shell.execute_reply.started":"2022-06-22T23:10:23.024776Z","shell.execute_reply":"2022-06-22T23:10:27.582764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 16))\nfor ind, (image_id, image, mask) in enumerate(zip(image_ids, images, masks)):\n    plt.subplot(2, 2, ind + 1)\n    plt.imshow(image)\n    plt.imshow(mask, cmap=\"hot\", alpha=0.5)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:27.585594Z","iopub.execute_input":"2022-06-22T23:10:27.585961Z","iopub.status.idle":"2022-06-22T23:10:36.043104Z","shell.execute_reply.started":"2022-06-22T23:10:27.585926Z","shell.execute_reply":"2022-06-22T23:10:36.040858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Metadata Analysis**\n\nI learned this from [here](https://www.kaggle.com/code/rhtsingh/hubmap-hpa-exploratory-data-analysis).","metadata":{}},{"cell_type":"code","source":"def bar_hor(df, col, title, color, w=None, h=None, lm=0, limit=100, return_trace=False, rev=False, xlb = False):\n    cnt_srs = df[col].value_counts()\n    yy = cnt_srs.head(limit).index[::-1] \n    xx = cnt_srs.head(limit).values[::-1] \n    if rev:\n        yy = cnt_srs.tail(limit).index[::-1] \n        xx = cnt_srs.tail(limit).values[::-1] \n    if xlb:\n        trace = go.Bar(y=xlb, x=xx, orientation = 'h', marker=dict(color=color))\n    else:\n        trace = go.Bar(y=yy, x=xx, orientation = 'h', marker=dict(color=color))\n    if return_trace:\n        return trace \n    layout = dict(title=title, margin=dict(l=lm), width=w, height=h)\n    data = [trace]\n    fig = go.Figure(data=data, layout=layout)\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.044653Z","iopub.execute_input":"2022-06-22T23:10:36.045844Z","iopub.status.idle":"2022-06-22T23:10:36.059454Z","shell.execute_reply.started":"2022-06-22T23:10:36.045799Z","shell.execute_reply":"2022-06-22T23:10:36.058102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bar_hor(df_train, 'data_source', 'Data Source Distribution', [\"#F1948A\", '#ABEBC6'], h=350, w=600, lm=200, xlb = ['Source : HPA','Target : HuBMAP'])","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.06141Z","iopub.execute_input":"2022-06-22T23:10:36.061822Z","iopub.status.idle":"2022-06-22T23:10:36.215275Z","shell.execute_reply.started":"2022-06-22T23:10:36.061788Z","shell.execute_reply":"2022-06-22T23:10:36.213909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gp(df, col, title):\n    df1 = df[df[\"data_source\"] == \"HPA\"]\n    df0 = df[df[\"data_source\"] == \"HuBMAP\"]\n    a1 = df1[col].value_counts()\n    b1 = df0[col].value_counts()\n    \n    total = dict(df[col].value_counts())\n    x0 = a1.index\n    x1 = b1.index\n    \n    y0 = [float(x)*100 / total[x0[i]] for i,x in enumerate(a1.values)]\n    y1 = [float(x)*100 / total[x1[i]] for i,x in enumerate(b1.values)]\n\n    trace1 = go.Bar(x=a1.index, y=y0, name='Target : 1', marker=dict(color=\"#96D38C\"))\n    trace2 = go.Bar(x=b1.index, y=y1, name='Target : 0', marker=dict(color=\"#FEBFB3\"))\n    return trace1, trace2 ","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.218622Z","iopub.execute_input":"2022-06-22T23:10:36.218976Z","iopub.status.idle":"2022-06-22T23:10:36.229386Z","shell.execute_reply.started":"2022-06-22T23:10:36.218945Z","shell.execute_reply":"2022-06-22T23:10:36.228136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr0 = bar_hor(df_train, \"sex\", \"Distribution of Gender Variable\" ,\"#f975ae\", w=700, lm=100, return_trace=True)\ntr1, tr2 = gp(df_train, 'sex', 'Distribution of Source with Gender')\n\nfig = tools.make_subplots(rows=1, cols=3, print_grid=False, subplot_titles = [\"Gender Distribution\" , \"Gender, Data Source=HMAP\" ,\"Gender, Data Source=HuBMAP\"])\nfig.append_trace(tr0, 1, 1);\nfig.append_trace(tr1, 1, 2);\nfig.append_trace(tr2, 1, 3);\nfig['layout'].update(height=350, showlegend=False, margin=dict(l=50));\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.231005Z","iopub.execute_input":"2022-06-22T23:10:36.231769Z","iopub.status.idle":"2022-06-22T23:10:36.456859Z","shell.execute_reply.started":"2022-06-22T23:10:36.231724Z","shell.execute_reply":"2022-06-22T23:10:36.455698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bar_hor(df_train, \"organ\", \"Organ Type Distribution\" ,\"#f975ae\", w=700, lm=100, return_trace= False)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.458155Z","iopub.execute_input":"2022-06-22T23:10:36.458492Z","iopub.status.idle":"2022-06-22T23:10:36.470645Z","shell.execute_reply.started":"2022-06-22T23:10:36.458464Z","shell.execute_reply":"2022-06-22T23:10:36.469723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,5))\nplt.title(\"Distribution of Age\")\nax = sns.distplot((df_train[\"age\"]))","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.472127Z","iopub.execute_input":"2022-06-22T23:10:36.472762Z","iopub.status.idle":"2022-06-22T23:10:36.720966Z","shell.execute_reply.started":"2022-06-22T23:10:36.472721Z","shell.execute_reply":"2022-06-22T23:10:36.720121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,5));\nplt.style.use('default')\nsns.kdeplot(df_train.loc[df_train['sex'] == \"Male\", 'age'], label = 'Gender == Male')\nsns.kdeplot(df_train.loc[df_train['sex'] == \"Female\", 'age'], label = 'Gender == Female')\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages'); plt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:10:36.722334Z","iopub.execute_input":"2022-06-22T23:10:36.722905Z","iopub.status.idle":"2022-06-22T23:10:36.976652Z","shell.execute_reply.started":"2022-06-22T23:10:36.72287Z","shell.execute_reply":"2022-06-22T23:10:36.975289Z"},"trusted":true},"execution_count":null,"outputs":[]}]}