{"cells":[{"metadata":{"_uuid":"e009bea549eb5745a2cab516db643e695f0172bc"},"cell_type":"markdown","source":""},{"metadata":{"trusted":true,"_uuid":"0d96027ae6ffc4c57ee3724675eee73664bcbbf1","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom skimage.segmentation import mark_boundaries\nship_dir = '../input'\ntrain_image_dir = os.path.join(ship_dir, 'train_v2')\ntest_image_dir = os.path.join(ship_dir, 'test_v2')\n##import gc; gc.enable() # memory is tight\nfrom skimage.util.montage import montage2d as montage\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\n\n## what does this do??\n# ref: https://www.kaggle.com/kmader/baseline-u-net-model-part-1#\nfrom skimage.morphology import label\ndef multi_rle_encode(img):\n    labels = label(img[:, :, 0])\n    return [rle_encode(labels==k) for k in np.unique(labels[labels>0])]\n\n## file is encoded so you need to use the run-length-encode-and-decode algorithm to get the ships masks \n## mapped onto the images\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef masks_as_image(in_mask_list):\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768), dtype = np.int16)\n    #if isinstance(in_mask_list, list):\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    return np.expand_dims(all_masks, -1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1565681ac4395087ef5a8735f7317f2a0f1e942"},"cell_type":"code","source":"## read in the encoded pixels file, then print the file size.  masks.shape[0] give you the size of the y-axis (number of rows)\n## and shape[1] give you the x-axis (number of columns=2)\n## masks['ImageId'].value_counts().shape[0] tells you how many unique images there are.  It's counting the unique values\n## in the ImageId column\nmasks = pd.read_csv(os.path.join('../input/',\n                                 'train_ship_segmentations_v2.csv'))\nprint(masks.shape[0], 'masks found')\nprint(masks['ImageId'].value_counts().shape[0]) ## there are duplicate images\nmasks.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1b42405cdf225de2aebfa4b1433c67ca73f39bf","scrolled":true},"cell_type":"code","source":"# ref: https://www.kaggle.com/kmader/baseline-u-net-model-part-1#\n## Make sure the encode-decode code is working correctly\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize = (10, 5))\n# get one image's encoded pixels to test\nrle_0 = masks.query('ImageId==\"00021ddc3.jpg\"')['EncodedPixels']\n# feed the encoded pixels to function that converts it to image\nimg_0 = masks_as_image(rle_0)\n# show the image\nax1.imshow(img_0[:, :, 0])\nax1.set_title('Image$_0$')\n# take the converted image and turn it back to encoded pixels\nrle_1 = multi_rle_encode(img_0)\n# take the converted encoded pixels and convert to image again\nimg_1 = masks_as_image(rle_1)\n# show the image\nax2.imshow(img_1[:, :, 0])\nax2.set_title('Image$_1$')\n# print the length of the 2 image masks to show the encoded and decoded lengths\nprint('Check Decoding->Encoding',\n      'RLE_0:', len(rle_0), '->',\n      'RLE_1:', len(rle_1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65ee140913927d7fae92451523d850c7b73867ee"},"cell_type":"code","source":"## create the test data set, will assign the validation set in model.fit(), already have test data set for after model is trained\n# create a column on the data set that says if image has at least one ship or not\nmasks['ships'] = masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nmasks.head()\n# only keep records with ships for initial training\nmasks_wships = masks.dropna()\nprint(masks_wships.shape[0], 'Num records')\nprint(masks_wships['ImageId'].value_counts().shape[0]) ## there are duplicate images\nmasks_wships.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4665a0a13c53a968876158fcc86322f767e4978b"},"cell_type":"code","source":"## ref: https://www.kaggle.com/kmader/baseline-u-net-model-part-1#\nBATCH_SIZE=48\nIMG_SCALING=(3,3)\ndef make_image_gen(in_df, batch_size = BATCH_SIZE):\n    all_batches = list(in_df.groupby('ImageId'))\n    out_rgb = []\n    out_mask = []\n    while True:\n        np.random.shuffle(all_batches)\n        for c_img_id, c_masks in all_batches:\n            rgb_path = os.path.join(train_image_dir, c_img_id)\n            c_img = imread(rgb_path)\n            # c_mask = np.expand_dims(masks_as_image(c_masks['EncodedPixels'].values), -1)\n            c_mask = masks_as_image(c_masks['EncodedPixels'].values)\n            if IMG_SCALING is not None:\n                c_img = c_img[::IMG_SCALING[0], ::IMG_SCALING[1]]\n                c_mask = c_mask[::IMG_SCALING[0], ::IMG_SCALING[1]]\n            out_rgb += [c_img]\n            out_mask += [c_mask]\n            if len(out_rgb)>=batch_size:\n                yield np.stack(out_rgb, 0)/255.0, np.stack(out_mask, 0)\n                out_rgb, out_mask=[], []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8fde959ac2e49f3ac925f660857ef7cc66da4fbc"},"cell_type":"code","source":"## ref: https://www.kaggle.com/kmader/baseline-u-net-model-part-1#\ntrain_gen = make_image_gen(masks_wships)\ntrain_x, train_y = next(train_gen)\nprint('x', train_x.shape, train_x.min(), train_x.max())\nprint('y', train_y.shape, train_y.min(), train_y.max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"956c6ce71f1eaf384e147f4cd28035fc46d5cf97"},"cell_type":"code","source":"## ref: https://www.kaggle.com/kmader/baseline-u-net-model-part-1#\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (30, 10))\nbatch_rgb = montage_rgb(train_x)\nbatch_seg = montage(train_y[:, :, :, 0])\nax1.imshow(batch_rgb)\nax1.set_title('Images')\nax2.imshow(batch_seg)\nax2.set_title('Segmentations')\nax3.imshow(mark_boundaries(batch_rgb, \n                           batch_seg.astype(int)))\nax3.set_title('Outlined Ships')\nfig.savefig('overview.png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5d94ec928e20e4f25f43d804813f6f5f0e77032"},"cell_type":"code","source":"## Build the model\nfrom keras.applications import ResNet50\nfrom keras.models import Model\nfrom keras.layers import Dense\nfrom keras.layers import BatchNormalization\nfrom keras.layers import Flatten\nfrom keras.layers import Dropout\nimport pickle","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}