{"cells":[{"metadata":{"papermill":{"duration":0.013267,"end_time":"2020-11-17T20:07:28.244469","exception":false,"start_time":"2020-11-17T20:07:28.231202","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## References\n\n* This [notebook](https://www.kaggle.com/pestipeti/decoding-rle-masks) by Peter shows how to load the images using `skimage.io`"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2020-11-17T20:07:28.272083Z","iopub.status.busy":"2020-11-17T20:07:28.271238Z","iopub.status.idle":"2020-11-17T20:07:29.006044Z","shell.execute_reply":"2020-11-17T20:07:29.006851Z"},"papermill":{"duration":0.752148,"end_time":"2020-11-17T20:07:29.007043","exception":false,"start_time":"2020-11-17T20:07:28.254895","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"import os\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport skimage.io\nfrom tqdm.notebook import tqdm","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.011212,"end_time":"2020-11-17T20:07:29.029712","exception":false,"start_time":"2020-11-17T20:07:29.0185","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Variables"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2020-11-17T20:07:29.057077Z","iopub.status.busy":"2020-11-17T20:07:29.056216Z","iopub.status.idle":"2020-11-17T20:07:29.059894Z","shell.execute_reply":"2020-11-17T20:07:29.05897Z"},"papermill":{"duration":0.019926,"end_time":"2020-11-17T20:07:29.060036","exception":false,"start_time":"2020-11-17T20:07:29.04011","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"data_dir = '/kaggle/input/hubmap-kidney-segmentation'\nsplit = 'train' # Change this to use test\ntile_size = 512\next = 'png' # Change to jpg for smaller files","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.010156,"end_time":"2020-11-17T20:07:29.081775","exception":false,"start_time":"2020-11-17T20:07:29.071619","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Helper function"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-17T20:07:29.119968Z","iopub.status.busy":"2020-11-17T20:07:29.118965Z","iopub.status.idle":"2020-11-17T20:07:29.122808Z","shell.execute_reply":"2020-11-17T20:07:29.122181Z"},"papermill":{"duration":0.030732,"end_time":"2020-11-17T20:07:29.12294","exception":false,"start_time":"2020-11-17T20:07:29.092208","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# https://www.kaggle.com/paulorzp/rle-functions-run-lenght-encode-decode\ndef mask2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels= img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n \ndef rle2mask(mask_rle, shape=(1600,256)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (width,height) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.010306,"end_time":"2020-11-17T20:07:29.144245","exception":false,"start_time":"2020-11-17T20:07:29.133939","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Load the CSVs"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-17T20:07:29.174444Z","iopub.status.busy":"2020-11-17T20:07:29.173453Z","iopub.status.idle":"2020-11-17T20:07:29.291639Z","shell.execute_reply":"2020-11-17T20:07:29.290937Z"},"papermill":{"duration":0.136673,"end_time":"2020-11-17T20:07:29.291779","exception":false,"start_time":"2020-11-17T20:07:29.155106","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"train_df = pd.read_csv(f'{data_dir}/train.csv')\nsub_df = pd.read_csv(f'{data_dir}/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.010279,"end_time":"2020-11-17T20:07:29.312965","exception":false,"start_time":"2020-11-17T20:07:29.302686","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Break down all images"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-17T20:07:29.351164Z","iopub.status.busy":"2020-11-17T20:07:29.350243Z","iopub.status.idle":"2020-11-17T20:18:23.369328Z","shell.execute_reply":"2020-11-17T20:18:23.367438Z"},"papermill":{"duration":654.044974,"end_time":"2020-11-17T20:18:23.369472","exception":false,"start_time":"2020-11-17T20:07:29.324498","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# Those folders will store our images\nos.makedirs(f'{split}_tiles/images', exist_ok=True)\nos.makedirs(f'{split}_tiles/masks', exist_ok=True)\n\n# This list will contain information about all our images\nmeta_ls = []\n\n# Choose a dataframe based on the split\nif split == 'train':\n    df = train_df\nelse:\n    df = sub_df\n\n# The break down starts here\nfor ix in range(df.shape[0]):\n    img_id = df.id[ix]\n    path = f\"{data_dir}/{split}/{img_id}.tiff\"\n    img = skimage.io.imread(path).squeeze()\n    mask = rle2mask(df.encoding[ix], shape=img.shape[1::-1])\n\n    x_max, y_max = img.shape[:2]\n\n    for x0 in tqdm(range(0, x_max, tile_size)):\n        x1 = min(x_max, x0 + tile_size)\n        for y0 in range(0, y_max, tile_size):\n            y1 = min(y_max, y0 + tile_size)\n\n            img_tile = img[x0:x1, y0:y1]\n            mask_tile = mask[x0:x1, y0:y1]\n\n            img_tile_path = f\"{split}_tiles/images/{img_id}_{x0}-{x1}x_{y0}-{y1}y.{ext}\"\n            mask_tile_path = f\"{split}_tiles/masks/{img_id}_{x0}-{x1}x_{y0}-{y1}y.png\"\n\n            cv2.imwrite(img_tile_path, cv2.cvtColor(img_tile, cv2.COLOR_RGB2BGR))\n            cv2.imwrite(mask_tile_path, mask_tile)\n\n            meta_ls.append([\n                img_id, x0, x1, y0, y1, img_tile.min(), img_tile.max(), \n                mask_tile.max(), img_tile_path, mask_tile_path\n            ])","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2020-11-17T20:18:23.423362Z","iopub.status.busy":"2020-11-17T20:18:23.422362Z","iopub.status.idle":"2020-11-17T20:18:24.481266Z","shell.execute_reply":"2020-11-17T20:18:24.481944Z"},"papermill":{"duration":1.087298,"end_time":"2020-11-17T20:18:24.482101","exception":false,"start_time":"2020-11-17T20:18:23.394803","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"meta_df = pd.DataFrame(meta_ls, columns=['image_id', 'x0', 'x1', 'y0', 'y1', 'min_pixel_value', 'max_pixel_value', 'max_mask_value', 'image_tile_path', 'mask_tile_path'])\nmeta_df.to_csv(f'{split}_metadata.csv', index=False)\nmeta_df.head()","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":0.019269,"end_time":"2020-11-17T20:18:24.521065","exception":false,"start_time":"2020-11-17T20:18:24.501796","status":"completed"},"tags":[]},"cell_type":"markdown","source":"## Convert to tar"},{"metadata":{"execution":{"iopub.execute_input":"2020-11-17T20:18:24.567889Z","iopub.status.busy":"2020-11-17T20:18:24.567132Z","iopub.status.idle":"2020-11-17T20:20:52.892887Z","shell.execute_reply":"2020-11-17T20:20:52.893731Z"},"papermill":{"duration":148.352976,"end_time":"2020-11-17T20:20:52.893924","exception":false,"start_time":"2020-11-17T20:18:24.540948","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"%%time\n# c: create, q: quiet, f: file\n!tar -cf train_tiles.tar train_tiles --remove-files","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}