{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Credits**\n* https://www.kaggle.com/xhlulu/hubmap-break-down-images-into-512x512-tiles\n* https://www.kaggle.com/iafoss/256x256-images\n* https://www.kaggle.com/lifa08/run-length-encode-and-decode\n* https://www.kaggle.com/pestipeti/decoding-rle-masks"},{"metadata":{},"cell_type":"markdown","source":"**What is RLE?**\nRun-length encoding (RLE) is a form of lossless data compression in which runs of data (sequences in which the same data value occurs in many consecutive data elements) are stored as a single data value and count, rather than as the original run. For Eg:  [0,0,0,0,0,1,1,1,1,0] => [5,0,4,1,1,0]\n\n**What is mask?**\nmask is used to focus on particular area. In a mask 0 means background 1 means mask. So when you apply a mask on an image you can clearly see pixels where the value of mask is 0. In this case you can clearly see glomeruli  when we apply mask on the image.\n\n**Why we are creating tiles instead of using the whole image?**\nThe size of these tiff images is more than 4GB. In fact you wont be able to open these images in OpenCV. Loading these images into the ram and then process them without crashing your notebook is not possible. So we are dividing these images into the tiles."},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd \nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport cv2\nimport tifffile as tiff\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"MASKS = '../input/hubmap-kidney-segmentation/train.csv'\nTrain_Data = '../input/hubmap-kidney-segmentation/train'\nTest_Data = '../input/hubmap-kidney-segmentation/test'\ntest = '/kaggle/input/hubmap-kidney-segmentation/sample_submission.csv'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(MASKS)\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Get_ImageId(file_name):\n    df = pd.read_csv(file_name)\n    ImageID = [id for id in df['id']]\n    return ImageID\ntrain_ImageID = Get_ImageId(MASKS)\ntest_ImageID = Get_ImageId(test)\nprint(f\"Train_Image_ID = {train_ImageID} \\nTest_Image_ID = {test_ImageID}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets Checkout the size of all the images\nprint(\"Training data sets \\n\")\nfor id_im in train_ImageID:\n    image = tiff.imread('/kaggle/input/hubmap-kidney-segmentation/train/{}.tiff'.format(id_im))\n    print(str(image.shape) + '  ' + id_im)\n    \nprint(\"Testing data set \\n\")\nfor id_im in test_ImageID:\n    image = tiff.imread('/kaggle/input/hubmap-kidney-segmentation/test/{}.tiff'.format(id_im))\n    print(str(image.shape) + ' ' + id_im)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import rasterio\nfrom rasterio.plot import show\n\nimg_id = \"2f6ecfcdf\"\nimg_2f6ecfcdf = rasterio.open(f\"/kaggle/input/hubmap-kidney-segmentation/train/{img_id}.tiff\")\nfig, ax = plt.subplots(1, 1, figsize=(50, 50))\nshow(img_2f6ecfcdf, ax=ax)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_2f6ecfcdf.closed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rle2mask(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image = tiff.imread('/kaggle/input/hubmap-kidney-segmentation/train/{}.tiff'.format(\"2f6ecfcdf\"))\nimage.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mask = rle2mask(train_df.iloc[0, 1], (image.shape[1], image.shape[0]))\nmask.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"masked_image = cv2.bitwise_and(image, image, mask=mask)\nmasked_image.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(20,10))\nplt.imshow(image[20000:25000, 12000:17000, :])\nplt.imshow(masked_image[20000:25000, 12000:17000], cmap='jet', alpha=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Generate_Tiles(ImageID, path):\n    tile_size = 256 # size of a tile will be 512 * 512\n    df = pd.read_csv(f\"{path}.csv\") # Read csv to get rle str\n    os.makedirs('train_tiles/images', exist_ok=True) # create folders to save these images\n    os.makedirs('train_tiles/masks', exist_ok=True)\n    for idx in ImageID: \n        img = tiff.imread(f\"{path}/{idx}.tiff\")\n        img = np.squeeze(img) # Squeeze the dimension to remove (1,1)\n        mask = rle2mask(df.loc[train_df.id == idx].iloc[0, 1], (img.shape[1], img.shape[0])) # Generate mask\n        x_max, y_max = img.shape[:2] \n        for x0 in range(0, x_max, tile_size):\n            x1 = min(x_max, x0 + tile_size) # to make sure that x0 + tile_size  !> x_max \n            for y0 in range(0, y_max, tile_size):\n                y1 = min(y_max, y0+tile_size) # to make sure that y0 + tile_size  !> y_max \n                \n                img_tile = img[x0:x1, y0:y1]\n                mask_tile = mask[x0:x1, y0:y1]\n                img_tile_path = f\"train_tiles/images/{idx}_{x0}-{x1}x_{y0}-{y1}y.png\"\n                mask_tile_path = f\"train_tiles/masks/{idx}_{x0}-{x1}x_{y0}-{y1}y.png\"\n                cv2.imwrite(img_tile_path, cv2.cvtColor(img_tile, cv2.COLOR_RGB2BGR)) # Save the file \n                cv2.imwrite(mask_tile_path, mask_tile)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Generate_Tiles(train_ImageID, Train_Data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}