{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom PIL import Image\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"2da3ae70475c75a6b9784c1940f135401c25bd6b"},"cell_type":"code","source":"#load dataset\ndata=pd.read_csv('../input/train_ship_segmentations.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"9fd3a653c8e451286cec698cf2f698907667852f"},"cell_type":"code","source":"import os\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"52885fe64745566d57e24d515dd91e0a0e16bba1"},"cell_type":"code","source":"PATH='../input/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1af9911a7afcd068f3ac25f6a299c4cb286e9b5a","collapsed":true},"cell_type":"code","source":"#get the train and test images\ntrain_imgs=os.listdir(PATH+'train')\ntest_imgs=os.listdir(PATH+'test')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2712bc463142368f4d1d69b980339f391bd4b6e5","collapsed":true},"cell_type":"code","source":"#lets peek\ntrain_imgs[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96cfeb548af4db1dbf8b2d78c131580493a5de54","collapsed":true},"cell_type":"code","source":"#lets peek\ntest_imgs[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ba0b8c983b795480d690a6e915d33a95b4ae0eae"},"cell_type":"code","source":"#ffunction to show images\ndef show_img(PATH):\n    plt.figure(figsize=(10,7))\n    img=plt.imread(PATH)\n    plt.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4abca7da2086f69a097c6021b14d4e2394dc828a","collapsed":true},"cell_type":"code","source":"#lets look at some training samples\nfor i in train_imgs[:5]:\n    show_img(PATH+'train/'+i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"b8aa47de9d1d0d86b4d1bb73ff84d9264c4dffef","collapsed":true},"cell_type":"code","source":"#lets look at some testing samples\nfor i in test_imgs[:5]:\n    show_img(PATH+'test/'+i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0024ddc5d73cb8b2590eb815e5a77a56857cda70"},"cell_type":"code","source":"#a peek at data\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"10eb47353f2f9b6a477ebd77d52063e2b28016a5","collapsed":true},"cell_type":"code","source":"#lets look at some samples from dataset\nfor i in data['ImageId'].head():\n    show_img(PATH+'train/'+i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f2edccd7a17488bb276602eb6244ebc2cb5f69b","collapsed":true},"cell_type":"code","source":"#make path for images\nmake_path=lambda x: PATH+'train/'+x\ndata['ImagePath']=make_path(data['ImageId'].values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d725584f16cdd88af96ccee252c87c25853f1810","collapsed":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f520a3ce8cffb6d66b6830b283828620ad46d7f","collapsed":true},"cell_type":"code","source":"#check for any missing values\ndata.isnull().sum()/data.shape[0]*100.0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"345136234bce31831bc62b8b90e7b29d314936b9"},"cell_type":"markdown","source":"57%  segmented values are missings .\n"},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ddb83a02afc8981d8f9275ac0e26c8b39ecb7591"},"cell_type":"code","source":"# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f9f5915895aa854ead13da316b6f55feec3bed63"},"cell_type":"code","source":"def rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd78319aca0051b0b76c6da21526f77359c7bcf1","collapsed":true},"cell_type":"code","source":"data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ef9fc6fcc92ad2bac5c845e0ab1c6925d97735d","collapsed":true},"cell_type":"code","source":"data['ImageId'].value_counts().shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6eeb3d217a48d161ec602cea0496578b0489ca8c","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"30b0b473f53310de0d509df6eeb6bd1101220058"},"cell_type":"code","source":"from skimage.segmentation import mark_boundaries\nfrom skimage.util.montage import montage2d as montage\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"8a2a8aae1d2b4e3b9dbc62075692e665f3f579c0"},"cell_type":"code","source":"from skimage.morphology import label\ndef multi_rle_encode(img):\n    labels = label(img[:, :, 0])\n    return [rle_encode(labels==k) for k in np.unique(labels[labels>0])]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"81015615835172967e2eca4c1778cde35906e3b1"},"cell_type":"code","source":"def masks_as_image(in_mask_list):\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768), dtype = np.int16)\n    #if isinstance(in_mask_list, list):\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    return np.expand_dims(all_masks, -1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"546060d60ebc3bac8413f4cefe274abcb56ad162"},"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize = (10, 5))\nrle_0 = data.query('ImageId==\"000155de5.jpg\"')['EncodedPixels']\nimg_0 = masks_as_image(rle_0)\nax1.imshow(img_0[:, :, 0])\nax1.set_title('Image$_0$')\nrle_1 = multi_rle_encode(img_0)\nimg_1 = masks_as_image(rle_1)\nax2.imshow(img_1[:, :, 0])\nax2.set_title('Image$_1$')\nprint('Check Decoding->Encoding',\n      'RLE_0:', len(rle_0), '->',\n      'RLE_1:', len(rle_1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"e200967b75f2d8845573352f68d1fc6dd675e98f"},"cell_type":"code","source":"data['ships'] = data['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nunique_img_ids = data.groupby('ImageId').agg({'ships': 'sum'}).reset_index()\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids['has_ship_vec'] = unique_img_ids['has_ship'].map(lambda x: [x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7187d4c63f902233ea53e32dcf49d4389171ba46"},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"decf674d37360e8b42d7fb6d4ef71972eecb9002"},"cell_type":"code","source":"unique_img_ids.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d5aca5cc9b835bce76e8920d48ce1f5831b67d9"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_ids, valid_ids = train_test_split(unique_img_ids, \n                 test_size = 0.3, \n                 stratify = unique_img_ids['ships'])\ntrain_df = pd.merge(data, train_ids)\nvalid_df = pd.merge(data, valid_ids)\nprint(train_df.shape[0], 'training masks')\nprint(valid_df.shape[0], 'validation masks')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5119fd375fa6fb047a8d577ef129b5fa18a1b735"},"cell_type":"code","source":"train_ids.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2255f348eec6f5b1c40adf9bbe709223f5e09a3"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"c7ad14aba9f57d9f5ebdbfa7972824d777ec8e0c"},"cell_type":"code","source":"BATCH_SIZE = 32\nEDGE_CROP = 16\nGAUSSIAN_NOISE = 0.1\nUPSAMPLE_MODE = 'SIMPLE'\n# downsampling inside the network\nNET_SCALING = (1, 1)\n# downsampling in preprocessing\nIMG_SCALING = (2, 2)\n# number of validation images to use\nVALID_IMG_COUNT = 600\n# maximum number of steps_per_epoch in training\nMAX_TRAIN_STEPS = 150\nAUGMENT_BRIGHTNESS = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c4a4496bcffe6563a6354f8ef30306b66e776b5"},"cell_type":"code","source":"def make_image_gen(in_df, batch_size = BATCH_SIZE):\n    all_batches = list(in_df.groupby('ImageId'))\n    out_rgb = []\n    out_mask = []\n    while True:\n        np.random.shuffle(all_batches)\n        for c_img_id, c_masks in all_batches:\n            rgb_path = os.path.join(train_image_dir, c_img_id)\n            c_img = imread(rgb_path)\n            c_mask = masks_as_image(c_masks['EncodedPixels'].values)\n            if IMG_SCALING is not None:\n                c_img = c_img[::IMG_SCALING[0], ::IMG_SCALING[1]]\n                c_mask = c_mask[::IMG_SCALING[0], ::IMG_SCALING[1]]\n            out_rgb += [c_img]\n            out_mask += [c_mask]\n            if len(out_rgb)>=batch_size:\n                yield np.stack(out_rgb, 0)/255.0, np.stack(out_mask, 0)\n                out_rgb, out_mask=[], []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5187b8557fd6d2817e4ec72f5faae3ff69d29dd"},"cell_type":"code","source":"train_df['grouped_ship_count'] = train_df['ships'].map(lambda x: (x+2)//3)\nbalanced_train_df = train_df.groupby('grouped_ship_count').apply(lambda x: x.sample(1500))\nbalanced_train_df['ships'].hist()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e10ec884e9d501ba08331efaecb1a2a91ff1f25"},"cell_type":"code","source":"balanced_train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54709fc6d901c3e2bebe4bff46d1eff521022cb2"},"cell_type":"code","source":"balanced_train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"db8f09eba65b0002b3b370a94fa13f06fc84c097"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}