{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-02T02:23:40.672179Z","iopub.execute_input":"2022-09-02T02:23:40.672735Z","iopub.status.idle":"2022-09-02T02:23:40.679389Z","shell.execute_reply.started":"2022-09-02T02:23:40.672695Z","shell.execute_reply":"2022-09-02T02:23:40.677859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qq zarr\nimport cv2, zarr, gc\nimport matplotlib.pyplot as plt, numpy as np, pandas as pd\nfrom pathlib import Path\ngc.enable()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:53:52.203813Z","iopub.execute_input":"2022-09-02T01:53:52.204174Z","iopub.status.idle":"2022-09-02T01:54:08.448062Z","shell.execute_reply.started":"2022-09-02T01:53:52.204141Z","shell.execute_reply":"2022-09-02T01:54:08.446999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from geojson import Point, Feature, FeatureCollection, dump\nimport shapely.wkt\nimport shapely.geometry\nfrom shapely.geometry import MultiPolygon, Polygon","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:54:08.450131Z","iopub.execute_input":"2022-09-02T01:54:08.450594Z","iopub.status.idle":"2022-09-02T01:54:08.605386Z","shell.execute_reply.started":"2022-09-02T01:54:08.450540Z","shell.execute_reply":"2022-09-02T01:54:08.604136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:54:08.607416Z","iopub.execute_input":"2022-09-02T01:54:08.607735Z","iopub.status.idle":"2022-09-02T01:54:08.613195Z","shell.execute_reply.started":"2022-09-02T01:54:08.607697Z","shell.execute_reply":"2022-09-02T01:54:08.612026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle2mask(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [\n        np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])\n    ]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo : hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:54:08.614567Z","iopub.execute_input":"2022-09-02T01:54:08.614954Z","iopub.status.idle":"2022-09-02T01:54:08.624328Z","shell.execute_reply.started":"2022-09-02T01:54:08.614918Z","shell.execute_reply":"2022-09-02T01:54:08.622860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale = 2\nresize=512 # For pdf\n\n# Input \npath = Path('../input/hubmap-organ-segmentation/train_annotations')\n\ndf_info = pd.read_csv(\"../input/hubmap-organ-segmentation/train.csv\")\ngrp_anatomy = zarr.open_group('../input/hubmap2022-zarr/images_scale2')\n\n# PDF Setting\nweight_dict = {\n    'fbr': 0.01,          # Background Weights\n    'cortex_value': 0.75,  # Cortex Weights\n}\n\n# Output\nroot = zarr.group(f'/kaggle/working/masks_scale{scale}')\ng_msk, g_pdf = root.create_groups('labels', 'pdfs', overwrite=True)","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:57:58.156701Z","iopub.execute_input":"2022-09-02T01:57:58.157074Z","iopub.status.idle":"2022-09-02T01:57:58.335380Z","shell.execute_reply.started":"2022-09-02T01:57:58.157039Z","shell.execute_reply":"2022-09-02T01:57:58.334410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def polygons_to_mask(polygons):\n    img_mask = np.zeros(im_size, np.uint8)\n    if not polygons:\n        return img_mask\n    int_coords = lambda x: np.array(x).round().astype(np.int64)#int32\n    exteriors = [int_coords(poly.exterior.coords) for poly in polygons]\n    interiors = [int_coords(pi.coords) for poly in polygons\n                 for pi in poly.interiors]\n    cv2.fillPoly(img_mask, exteriors, 1)\n    cv2.fillPoly(img_mask, interiors, 0)\n    return img_mask","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:58:03.271392Z","iopub.execute_input":"2022-09-02T01:58:03.272925Z","iopub.status.idle":"2022-09-02T01:58:03.280913Z","shell.execute_reply.started":"2022-09-02T01:58:03.272861Z","shell.execute_reply":"2022-09-02T01:58:03.279306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_info\nb = df.img_height.tolist()\na = df.img_width.tolist()\nnames = df.id.astype(str).tolist()\nlist_name = []\nfor name in names:\n    name = name[:9]\n    list_name.append(name)\nshapes = []\nfor x,y in zip(b,a):\n    c = (x,y)\n    shapes.append(c)\n\ndic_shape2 = dict(zip(list_name,shapes))","metadata":{"execution":{"iopub.status.busy":"2022-09-02T01:59:40.295518Z","iopub.execute_input":"2022-09-02T01:59:40.296054Z","iopub.status.idle":"2022-09-02T01:59:40.304206Z","shell.execute_reply.started":"2022-09-02T01:59:40.296020Z","shell.execute_reply":"2022-09-02T01:59:40.303139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for _, row in df_info.iterrows():\n    \n    print(row)\n\n    idx = str(row.id)\n    print(idx)\n    \n\n    with open(path/(idx+\".json\")) as f:\n        shapes = json.load(f)\n    \n    all_pol = []\n\n    for shape in shapes:\n#         shape = shape['geometry']['coordinates']\n#         if len(shape)>1:\n#             shape =shape[0]\n        shape_numpy = np.array(shape,dtype=object)\n        shape_numpy = shape_numpy.squeeze()\n\n        pol = Polygon(shape_numpy)\n        all_pol.append(pol)\n    all_polygons = MultiPolygon(all_pol)\n    if not all_polygons.is_valid:\n        all_polygons = all_polygons.buffer(0)\n    # Sometimes buffer() converts a simple Multipolygon to just a Polygon,\n    # need to keep it a Multi throughout\n    if all_polygons.type == 'Polygon':\n        all_polygons = MultiPolygon([all_polygons])\n    im_size = dic_shape2[idx]\n    msk = polygons_to_mask(all_polygons)\n    \n    \n    # Plot\n    fig, ax = plt.subplots(ncols=2, figsize=(15,15))\n    resize_w = int((msk.shape[1]/msk.shape[0])*resize)\n    ax[0].imshow(cv2.resize(msk, dsize=(resize_w, resize)))\n    ax[0].set_title('Mask')\n    ax[0].set_axis_off()\n    \n    anatomy = grp_anatomy[idx][:]\n    \n    if scale:\n        new_size = (msk.shape[1] // scale, msk.shape[0] // scale)\n        print('Scaling to', new_size)\n        msk = cv2.resize(msk, new_size)\n        anatomy = cv2.resize(anatomy, new_size)\n        \n    anatomy = anatomy.astype('float16')     \n    anatomy[anatomy==0] = weight_dict['fbr']\n    anatomy[anatomy==1] = weight_dict['cortex_value']\n    anatomy[msk>0] = 1\n    \n    print('Saving msk')\n    g_msk[idx] = msk\n    del msk\n    gc.collect()\n        \n    if resize:\n        print('Resizing PDF')\n        if anatomy.shape[0]>resize:\n            resize_w = int((anatomy.shape[1]/anatomy.shape[0])*resize)\n            anatomy = cv2.resize(anatomy[:].astype('float32'), dsize=(resize_w, resize))\n            \n    ax[1].imshow(anatomy)\n    ax[1].set_title('Probability density function for sampling')\n    ax[1].set_axis_off() \n            \n    print('Saving pdf cumsum')\n    g_pdf[idx] = np.cumsum(anatomy/np.sum(anatomy)) \n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-09-02T02:14:00.120436Z","iopub.execute_input":"2022-09-02T02:14:00.120969Z","iopub.status.idle":"2022-09-02T02:18:44.933288Z","shell.execute_reply.started":"2022-09-02T02:14:00.120935Z","shell.execute_reply":"2022-09-02T02:18:44.932187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}