{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport random\ncolor_pal = plt.rcParams['axes.prop_cycle'].by_key()['color']\nplt.style.use('ggplot')\nimport tifffile\nimport cv2\nimport os\nfrom PIL import Image\nDATASET_IMAGES = \"../input/hubmap-organ-segmentation/train_images/\"\nDATASET_MASKS = \"../input/hacking-the-human-body-annotation-masks/train_masks\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-04T02:51:51.639345Z","iopub.execute_input":"2022-07-04T02:51:51.640102Z","iopub.status.idle":"2022-07-04T02:51:53.338410Z","shell.execute_reply.started":"2022-07-04T02:51:51.639975Z","shell.execute_reply":"2022-07-04T02:51:53.336705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"id - The image ID.\n\norgan - The organ that the biopsy sample was taken from.\n\n\ndata_source - Whether the image was provided by HuBMAP or HPA.\nimg_height - The height of the image in pixels.\n\nimg_width - The width of the image in pixels.\n\npixel_size - The height/width of a single pixel from this image in micrometers. All HPA images have a pixel size of 0.4\nµm. For HuBMAP imagery the pixel size is 0.5 µm for kidney, 0.2290 µm for large intestine, 0.7562 µm for lung, 0.4945 µm for spleen, and 6.263 µm for prostate.\n\ntissue_thickness - The thickness of the biopsy sample in micrometers. All HPA images have a thickness of 4 µm. The HuBMAP samples have tissue slice thicknesses 10 µm for kidney, 8 µm for large intestine, 4 µm for spleen, 5 µm for lung, and 5 µm for prostate.\n\nrle - The target column. A run length encoded copy of the annotations. Provided for the training set only.\n\nage - The patient's age in years. Provided for the training set only.\n\nsex - The sex of the patient. Provided for the training set only.\n\nsample_submission.csv\n\nid - The image ID.\n\nrle - A run length encoded mask of the FTUs in the image.","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv('../input/hubmap-organ-segmentation/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T02:53:04.608250Z","iopub.execute_input":"2022-07-04T02:53:04.608663Z","iopub.status.idle":"2022-07-04T02:53:04.795721Z","shell.execute_reply.started":"2022-07-04T02:53:04.608631Z","shell.execute_reply":"2022-07-04T02:53:04.794517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Unique values feature-wise\nfor col in df:\n    print(col,len(df[col].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:45:24.310375Z","iopub.execute_input":"2022-07-03T21:45:24.310736Z","iopub.status.idle":"2022-07-03T21:45:24.360568Z","shell.execute_reply.started":"2022-07-03T21:45:24.310708Z","shell.execute_reply":"2022-07-03T21:45:24.359280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols= ['organ','data_source','sex','pixel_size','tissue_thickness']\nnum_cols = [i for i in df.columns if i not in cat_cols and i not in ['id','rle']]","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:45:45.427228Z","iopub.execute_input":"2022-07-03T21:45:45.427536Z","iopub.status.idle":"2022-07-03T21:45:45.431774Z","shell.execute_reply.started":"2022-07-03T21:45:45.427514Z","shell.execute_reply":"2022-07-03T21:45:45.430874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initial analysis","metadata":{}},{"cell_type":"markdown","source":"**Categorical Columns**","metadata":{}},{"cell_type":"code","source":"fig,axes = plt.subplots(3,2,figsize=(12,12))\nfig.suptitle('Categorical Distribution')\naxes=axes.ravel()\nfor j,i in enumerate(cat_cols):\n    ax=sns.countplot(data=df,x=i,ax=axes[j])\n    for p in ax.patches:\n        percentage = '{:.1f}%'.format(100 * p.get_height()/len(df))\n        x = p.get_x() + p.get_width()\n        y = p.get_height()\n        ax.annotate(percentage, (x, y),ha='center')\naxes[j+1].remove()\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:45:46.939753Z","iopub.execute_input":"2022-07-03T21:45:46.940195Z","iopub.status.idle":"2022-07-03T21:45:47.594851Z","shell.execute_reply.started":"2022-07-03T21:45:46.940171Z","shell.execute_reply":"2022-07-03T21:45:47.593934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Numerical features**","metadata":{}},{"cell_type":"code","source":"fig,axes = plt.subplots(2,2,figsize=(10,10))\nfig.suptitle('Numerical Distribution Height and Width')\naxes=axes.ravel()\nfor j,i in enumerate(num_cols):\n    rgb = [random.random(), random.random(), random.random()]\n    ax=sns.kdeplot(data=df,x=i,ax=axes[j],color=rgb,fill=True)\n\naxes[j+1].remove()\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:45:50.750763Z","iopub.execute_input":"2022-07-03T21:45:50.751244Z","iopub.status.idle":"2022-07-03T21:45:51.244201Z","shell.execute_reply.started":"2022-07-03T21:45:50.751214Z","shell.execute_reply":"2022-07-03T21:45:51.243269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = tifffile.imread('../input/hubmap-organ-segmentation/train_images/24782.tiff')\nplt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T02:53:11.334426Z","iopub.execute_input":"2022-07-04T02:53:11.334849Z","iopub.status.idle":"2022-07-04T02:53:13.156636Z","shell.execute_reply.started":"2022-07-04T02:53:11.334812Z","shell.execute_reply":"2022-07-04T02:53:13.155402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading masks","metadata":{}},{"cell_type":"code","source":"def mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle2mask(mask_rle, shape=(3000,3000)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","metadata":{"execution":{"iopub.status.busy":"2022-07-04T00:12:28.111081Z","iopub.execute_input":"2022-07-04T00:12:28.111442Z","iopub.status.idle":"2022-07-04T00:12:28.121313Z","shell.execute_reply.started":"2022-07-04T00:12:28.111412Z","shell.execute_reply":"2022-07-04T00:12:28.120564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Export Masks","metadata":{}},{"cell_type":"code","source":"\ndef rle_decode(mask_rle: str, img_shape: tuple = None) -> np.ndarray:\n    seq = mask_rle.split()\n    starts = np.array(list(map(int, seq[0::2])))\n    lengths = np.array(list(map(int, seq[1::2])))\n    assert len(starts) == len(lengths)\n    ends = starts + lengths\n    img = np.zeros((np.product(img_shape),), dtype=np.uint8)\n    for begin, end in zip(starts, ends):\n        img[begin:end] = 1\n    return img.reshape(img_shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T02:53:23.329141Z","iopub.execute_input":"2022-07-04T02:53:23.329555Z","iopub.status.idle":"2022-07-04T02:53:23.338033Z","shell.execute_reply.started":"2022-07-04T02:53:23.329515Z","shell.execute_reply":"2022-07-04T02:53:23.336817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir -p ./train_masks\n\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nfor _, row in tqdm(df.iterrows(), total=len(df)):\n    mask = rle_decode(row['rle'], img_shape=(row[\"img_height\"], row[\"img_width\"]))\n    segm_path = os.path.join(\"train_masks\", f\"{row['id']}.png\")\n    Image.fromarray(mask.T).save(segm_path)\n    # plt.imsave(segm_path, mask.T)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T02:53:25.254529Z","iopub.execute_input":"2022-07-04T02:53:25.254918Z","iopub.status.idle":"2022-07-04T02:54:30.692255Z","shell.execute_reply.started":"2022-07-04T02:53:25.254887Z","shell.execute_reply":"2022-07-04T02:54:30.690563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (7,7))\nimage = tifffile.imread('../input/hubmap-organ-segmentation/train_images/24782.tiff')\nmask='train_masks/24782.png'\nplt.imshow(image)\nplt.imshow(Image.open(mask),alpha=0.25)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T00:36:10.698376Z","iopub.execute_input":"2022-07-04T00:36:10.698836Z","iopub.status.idle":"2022-07-04T00:36:12.859841Z","shell.execute_reply.started":"2022-07-04T00:36:10.698792Z","shell.execute_reply":"2022-07-04T00:36:12.858979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Plotting and making tiles**","metadata":{}},{"cell_type":"code","source":"# Function taken from https://www.kaggle.com/code/jirkaborovec/ftus-segm-decompose-large-images-to-tiles\ndef tile_image(p_img, folder, size: int = 1024) -> list:\n    w = h = size\n    im = np.array(Image.open(p_img))\n    # https://stackoverflow.com/a/47581978/4521646\n    tiles = [im[i:(i + h), j:(j + w), ...] for i in range(0, im.shape[0], h) for j in range(0, im.shape[1], w)]\n    idxs = [(i, (i + h), j, (j + w)) for i in range(0, im.shape[0], h) for j in range(0, im.shape[1], w)]\n    name, _ = os.path.splitext(os.path.basename(p_img))\n    files = []\n    for k, tile in enumerate(tiles):\n        if tile.shape[:2] != (h, w):\n            tile_ = tile\n            tile = np.zeros_like(tiles[0])\n            tile[:tile_.shape[0], :tile_.shape[1], ...] = tile_\n        p_img = os.path.join(folder, f\"{name}_{k:03}.png\")\n        Image.fromarray(tile).save(p_img)\n        files.append(p_img)\n    return files, idxs","metadata":{"execution":{"iopub.status.busy":"2022-07-04T03:02:33.325400Z","iopub.execute_input":"2022-07-04T03:02:33.325793Z","iopub.status.idle":"2022-07-04T03:02:33.339608Z","shell.execute_reply.started":"2022-07-04T03:02:33.325760Z","shell.execute_reply":"2022-07-04T03:02:33.338251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/temp/images\n!mkdir -p /kaggle/temp/masks\n\ntiles_img, _ = tile_image(os.path.join(DATASET_IMAGES, \"12233.tiff\"), \"/kaggle/temp/images\", size=512)\ntiles_seg, idxs = tile_image(os.path.join('./train_masks', \"12233.png\"), \"/kaggle/temp/masks\", size=512)\n\n!ls -lh /kaggle/temp/images\n!ls -lh /kaggle/temp/masks","metadata":{"execution":{"iopub.status.busy":"2022-07-04T03:02:35.480782Z","iopub.execute_input":"2022-07-04T03:02:35.481588Z","iopub.status.idle":"2022-07-04T03:02:43.790605Z","shell.execute_reply.started":"2022-07-04T03:02:35.481549Z","shell.execute_reply":"2022-07-04T03:02:43.789392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage import color\n\nnb_tiles_sqrt = int(np.sqrt(len(tiles_img)))\n\nfig, axes = plt.subplots(nrows=nb_tiles_sqrt, ncols=nb_tiles_sqrt, figsize=(9, 9))\nfor i, (p_img, p_seg) in enumerate(zip(tiles_img, tiles_seg)):\n    img = plt.imread(p_img)\n    mask = np.array(Image.open(p_seg))\n    axes[i // nb_tiles_sqrt, i % nb_tiles_sqrt].imshow(color.label2rgb(mask, img, bg_label=0, bg_color=(1.,1.,1.), alpha=0.25))\n    axes[i // nb_tiles_sqrt, i % nb_tiles_sqrt].set_axis_off()\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T03:13:50.386547Z","iopub.execute_input":"2022-07-04T03:13:50.386979Z","iopub.status.idle":"2022-07-04T03:13:55.821063Z","shell.execute_reply.started":"2022-07-04T03:13:50.386946Z","shell.execute_reply":"2022-07-04T03:13:55.820026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}