{"cells":[{"metadata":{"_uuid":"3b6847e0218ce495caee116aeb1e30ce64b0d57b"},"cell_type":"markdown","source":"# Overview\nGoing from RLE in a data.frame to segmentation image is a somewhat expensive step and so we can make training easier if we preprocess the RLE data and generate masks that we can load in."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.util.montage import montage2d as montage\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\nship_dir = '../input'\ntrain_image_dir = os.path.join(ship_dir, 'train')\ntest_image_dir = os.path.join(ship_dir, 'test')\nimport gc; gc.enable() # memory is tight\n\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef masks_as_image(in_mask_list):\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768), dtype = np.int16)\n    #if isinstance(in_mask_list, list):\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    return np.expand_dims(all_masks, -1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"masks = pd.read_csv(os.path.join('../input/',\n                                 'train_ship_segmentations.csv'))\nprint(masks.shape[0], 'masks found')\nprint(masks['ImageId'].value_counts().shape[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"77e252b4a4ba1a3b23d5d45ea71e861bf5cf8905"},"cell_type":"markdown","source":"# Prepare the dask-processing code\nHere we have the code to run the preprocessing and packaging in a more efficient distributed manner"},{"metadata":{"trusted":true,"_uuid":"700a178caead91c4dc6920c289a8487c601c3378"},"cell_type":"code","source":"import dask.array as da\nimport dask\nimport dask.diagnostics as diag\nfrom multiprocessing.pool import ThreadPool\nimport h5py\nfrom bokeh.io import output_notebook\nfrom bokeh.resources import CDN\noutput_notebook(CDN, hide_banner=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"10f01617d3222dcf7e3bb32dceef9d9553765adc"},"cell_type":"code","source":"all_batches = list(masks.groupby('ImageId'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9fbfea0966faa74950f4d709ab89e48ef387c25"},"cell_type":"code","source":"all_ids = pd.DataFrame({'ImageId': [id for id, _ in all_batches]})\nall_ids.to_csv('image_ids.csv', index=False)\nall_ids.sample(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ff9691ba9eee4eaeb33b0d7a2f896d71e5c11f66"},"cell_type":"code","source":"def dask_read_seg(in_batches, max_items = None):\n    d_mask_fun = dask.delayed(masks_as_image)\n    if max_items is None:\n        max_items = len(in_batches)\n    lazy_images = [d_mask_fun(c_masks['EncodedPixels'].values) \n                   for _, (_, c_masks) in zip(range(max_items), in_batches)\n                  ]     # Lazily evaluate on each group\n    s_img = lazy_images[0].compute()\n    arrays = [da.from_delayed(lazy_image,           # Construct a small Dask array\n                              dtype=s_img.dtype,   # for every lazy value\n                              shape=s_img.shape)\n              for lazy_image in lazy_images]\n\n    return da.stack(arrays, axis=0)                # Stack all small Dask arrays into one","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d19ad86e7ac6202b6caad7b3e5fee6a65feec832"},"cell_type":"code","source":"tiny_img_ds = dask_read_seg(all_batches, 20)\ntiny_img_ds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f315d1737690c401446085b09a4020f9fafde61b"},"cell_type":"code","source":"with diag.ProgressBar(), diag.Profiler() as prof, diag.ResourceProfiler(0.5) as rprof:\n    with dask.config.set(pool=ThreadPool(4)):\n        tiny_img_ds.to_hdf5('tiny_segmentions.h5', '/image', compression = 'lzf')\n!ls -lh *.h5","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ae39bd0b06517e6563c4e3131ab3b59d70aacbfb"},"cell_type":"markdown","source":"# Now Package Everything\ninstead of just using a small portion of the dataset we export all the results."},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ae93b4b6bac72b071d14c50f5a8582e4c0ba1e9f"},"cell_type":"code","source":"# larger chunks are more efficient for writing/compressing and make the paralellization more efficient\nlarger_chunker = lambda x: x.rechunk({0: x.shape[0]//400, 1: -1, 2: -1, 3: -1})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aec39f25a88ba77050bc497e36a5334bcd422db2"},"cell_type":"code","source":"all_img_ds = larger_chunker(dask_read_seg(all_batches))\nall_img_ds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"38041b0c9d4b5336ea6e2bfa621fb4bb6a9f415e"},"cell_type":"code","source":"with diag.ProgressBar(), diag.Profiler() as prof, diag.ResourceProfiler(0.5) as rprof:\n    with dask.config.set(pool=ThreadPool(4)):\n        all_img_ds.to_hdf5('segmentions.h5', '/image', compression = 'lzf')\n!ls -lh *.h5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"22312f2bbb649e175133085502c77082ac7c2829"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}