{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Imports\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom skimage.data import imread\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nimport math\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# Dataset\nimport os\nprint(os.listdir('../input'))\nprint(os.listdir('../input/test_v2')[:5])\nprint(os.listdir('../input/train_v2')[:5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b26afe08b54312ad2a9f42fa000410a257dc6548"},"cell_type":"code","source":"# Peak at the training data, EncodedPixels format\ntrain = pd.read_csv('../input/train_ship_segmentations_v2.csv')\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19cb6aee8cb05b638c254d702ccde2c71cc8a5d2"},"cell_type":"code","source":"def subplot_of_images(samples, n_cols=5, image_size_inches=20):\n    '''Quickly plot images to matplotlib subplots'''\n    n_rows = int(math.ceil(len(samples) / float(n_cols)))\n\n    # Create matplotlib subplots\n    fig, ax = plt.subplots(n_rows, n_cols, sharex='col', sharey='row')\n    fig.set_size_inches(image_size_inches, image_size_inches)\n\n    # Set the images to subplots\n    for i, imgid in enumerate(samples.ImageId):\n        col = i % n_cols\n        row = i // n_cols\n\n        path = Path('../input/train_v2') / '{}'.format(imgid)\n        img = imread(path)\n\n        ax[row, col].imshow(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67bd719cfdd47012afd7b02b4114ced7096b9400"},"cell_type":"code","source":"# Plot the images with ships\nn_sample = 16\nsample = train[~train.EncodedPixels.isna()].sample(n_sample)\nsubplot_of_images(sample, n_cols=4, image_size_inches=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e91fa78096fefa4232cf1bf062cc9f3eecbb4814"},"cell_type":"code","source":"# Plot the images without ships\nn_sample = 16\nsample = train[train.EncodedPixels.isna()].sample(n_sample)\nsubplot_of_images(sample, n_cols=4, image_size_inches=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a2c088b632e94b38fa1d2f4a4dc66352d82b40a"},"cell_type":"code","source":"# Histogram of training data with/without ships\nships = train[~train.EncodedPixels.isna()].ImageId.unique()\nnoships = train[train.EncodedPixels.isna()].ImageId.unique()\n\nplt.bar(['Ships', 'No Ships'], [len(ships), len(noships)])\nplt.ylabel('Number of Images');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e00bcf4db21abd3c7546822e72b54b9534b502d9"},"cell_type":"code","source":"# Decode run-length encoding to rectangular B&W image (mask)\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7afb57c80596e78884f2816b52205cde2e5e1ea"},"cell_type":"code","source":"# View a few of the masks from the training data, which should highlight tanker outlines\nmasks = pd.read_csv('../input/train_ship_segmentations_v2.csv')\nmasks.head(30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11ad33b9a251a37ab6bdc9d059e084107bb149ec"},"cell_type":"code","source":"def plot_image_and_rle_mask(img, img_masks):\n    '''Quickly plot image along side the run-length encoding mask(s)'''\n\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768))\n    for mask in img_masks:\n        all_masks += rle_decode(mask)\n\n    fig, axarr = plt.subplots(1, 3, figsize=(15, 40))\n    axarr[0].axis('off')\n    axarr[1].axis('off')\n    axarr[2].axis('off')\n    axarr[0].imshow(img)\n    axarr[1].imshow(all_masks, cmap='gray')\n    axarr[2].imshow(img)\n    axarr[2].imshow(all_masks, alpha=0.4)\n    plt.tight_layout(h_pad=0.1, w_pad=0.1)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ddfe466f2fba14dfc9a58c11982310d3e64ed4a"},"cell_type":"code","source":"# Compare a few images with tanker ships, the mask, and both overlaid\nimg_ids = []\nimg_ids.append('00113a75c.jpg')\nimg_ids.append('00b0fa633.jpg')\n\nfor img_id in img_ids:\n    img = imread('../input/train_v2/' + img_id)\n    img_masks = masks.loc[masks['ImageId'] == img_id, 'EncodedPixels'].tolist()\n    plot_image_and_rle_mask(img, img_masks)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}