{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nimport gc\ngc.enable()\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true,"collapsed":true},"cell_type":"code","source":"from keras import optimizers, losses\nfrom keras.layers import *\nfrom keras.models import Model\nfrom keras.backend import int_shape\nfrom keras.utils import to_categorical\nfrom keras import backend as K\nfrom keras.applications.densenet import DenseNet169\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom skimage import io","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da9b6d017a6b5f36585f5ccb68b4672828b6e05c","collapsed":true},"cell_type":"code","source":"masks = pd.read_csv(os.path.join('../input/',\n                                 'train_ship_segmentations.csv'))\nprint(masks.shape[0], 'masks found')\nprint(masks['ImageId'].value_counts().shape[0])\nmasks['path'] = masks['ImageId'].map(lambda x: os.path.join('../input/train/', x))\nmasks.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01f608ef5dd3bf2462da10ef613adf4db7875fc9","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nmasks['ships'] = masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nunique_img_ids = masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index()\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids['has_ship_vec'] = unique_img_ids['has_ship'].map(lambda x: [x])\nmasks.drop(['ships'], axis=1, inplace=True)\ntrain_ids, valid_ids = train_test_split(unique_img_ids, \n                 test_size = 0.3, \n                 stratify = unique_img_ids['ships'])\ntrain_df = pd.merge(masks, train_ids)\nvalid_df = pd.merge(masks, valid_ids)\nprint(train_df.shape[0], 'training masks')\nprint(valid_df.shape[0], 'validation masks')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"704b9603ed1fe44ccadc672b4c75445f172081cf"},"cell_type":"code","source":"train_df = train_df.sample(min(15000, train_df.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfc6b9b60aff2ed982288b9a735706defc500849","collapsed":true},"cell_type":"code","source":"train_df[['ships', 'has_ship']].hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57c8c4e6f2170d8ab665cbe16bc3d765728f76ff","collapsed":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"721758dd98fd6e6a3e4c08fd76a55d81a8327e4c","collapsed":true},"cell_type":"code","source":"train,test = train_test_split(train_df, test_size=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd2efa0940bc1adf9ac9b0af6573be97db1423bb","collapsed":true},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nidg_train = ImageDataGenerator(rescale=1. / 255,\n                               shear_range=0.2,\n                               zoom_range=0.2,\n                               horizontal_flip=True)\n\nidg_test = ImageDataGenerator(rescale=1. / 255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"df5571a38183add6f011970296f59de662d94686"},"cell_type":"code","source":"def flow_from_dataframe(img_data_gen, in_df, path_col, y_col, **dflow_args):\n    base_dir = os.path.dirname(in_df[path_col].values[0])\n    print('## Ignore next message from keras, values are replaced anyways')\n    df_gen = img_data_gen.flow_from_directory(base_dir, \n                                     class_mode = 'sparse',\n                                    **dflow_args)\n    df_gen.filenames = in_df[path_col].values\n    df_gen.classes = np.stack(in_df[y_col].values)\n    df_gen.samples = in_df.shape[0]\n    df_gen.n = in_df.shape[0]\n    df_gen._set_index_array()\n    df_gen.directory = '' # since we have the full path\n    print('Reinserting dataframe: {} images'.format(in_df.shape[0]))\n    return df_gen","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19f145eb50db184a2cdd68fa5a4a4aebf21928fd","collapsed":true},"cell_type":"code","source":"train_images = flow_from_dataframe(idg_train, train, 'path', 'has_ship', batch_size=128, target_size=(256, 256))\ntest_images = flow_from_dataframe(idg_train, test, 'path', 'has_ship', batch_size=128, target_size=(256, 256))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9dc3d8436e1329d41f62015d514349371daadf8","collapsed":true},"cell_type":"code","source":"inputs = Input(shape=(256, 256, 3))\nbase_model = DenseNet169(input_tensor=inputs, weights='imagenet', include_top=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44e16db6665386529977f4ba66a3d8516a271b92","collapsed":true},"cell_type":"code","source":"x = base_model.output\nx = GlobalMaxPooling2D()(x)\nx = Dense(128, activation='relu')(x)\nx = Dense(1, activation='sigmoid')(x)\n\nmodel = Model(inputs=inputs, output=x)\n\nfor layer in base_model.layers:\n    layer.trainable = False\nmodel.compile(loss=losses.binary_crossentropy, optimizer=optimizers.Adam(), metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9fbf64021a690d8e2e5827bcb2334159a891f1b","collapsed":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24b50a9905d8f14b7121600a9cf28e10b6a287fd","collapsed":true},"cell_type":"code","source":"result = model.fit_generator(train_images, steps_per_epoch=12000//128, epochs=10, validation_data=test_images, validation_steps=3000//128)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8261181ab3573d5e7c289fababa75000119136fa","collapsed":true},"cell_type":"code","source":"model.save('ship_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"babc6d497442c90285b79afba73b7b61165fee02","collapsed":true},"cell_type":"code","source":"test_paths = os.listdir('../input/test/')\nprint(len(test_paths), 'test images found')\nsubmission_df = pd.read_csv('../input/sample_submission.csv')\nsubmission_df['path'] = submission_df['ImageId'].map(lambda x: os.path.join('../input/test/', x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91dd4459e0438656191b13c14ef73240c31b6520","collapsed":true},"cell_type":"code","source":"test_gen = flow_from_dataframe(idg_test, \n                               submission_df, \n                             path_col = 'path',\n                            y_col = 'ImageId', \n                            target_size = (256, 256),\n                             color_mode = 'rgb',\n                            batch_size = 128, \n                              shuffle = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"74b018ab77e3d36a6d7b3b90a7046866ab08475c"},"cell_type":"code","source":"def multi_rle_encode(img):\n    labels = label(img[:, :, 0])\n    return [rle_encode(labels==k) for k in np.unique(labels[labels>0])]\n\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef masks_as_image(in_mask_list):\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768), dtype = np.int16)\n    #if isinstance(in_mask_list, list):\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    return np.expand_dims(all_masks, -1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c530195b0f891f3f28cfc966238ec79b4eb25f22","collapsed":true},"cell_type":"code","source":"masks['ships'] = masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nunique_img_ids = masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index()\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids['has_ship_vec'] = unique_img_ids['has_ship'].map(lambda x: [x])\n# some files are too small/corrupt\nunique_img_ids['file_size_kb'] = unique_img_ids['ImageId'].map(lambda c_img_id: \n                                                               os.stat(os.path.join('../input/train/', \n                                                                                    c_img_id)).st_size/1024)\nunique_img_ids = unique_img_ids[unique_img_ids['file_size_kb']>50] # keep only 50kb files\nunique_img_ids['file_size_kb'].hist()\nmasks.drop(['ships'], axis=1, inplace=True)\nunique_img_ids.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1272c1c80e514eddf78128bfa68faf49d9ae950","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_ids, valid_ids = train_test_split(unique_img_ids, \n                 test_size = 0.3, \n                 stratify = unique_img_ids['ships'])\ntrain_df = pd.merge(masks, train_ids)\nvalid_df = pd.merge(masks, valid_ids)\nprint(train_df.shape[0], 'training masks')\nprint(valid_df.shape[0], 'validation masks')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9527cb53b257d9b44c2ad9a13d77c1ece22b93e7","collapsed":true},"cell_type":"code","source":"train_df['grouped_ship_count'] = train_df['ships'].map(lambda x: (x+1)//2).clip(0, 7)\ndef sample_ships(in_df, base_rep_val=1500):\n    if in_df['ships'].values[0]==0:\n        return in_df.sample(base_rep_val//3) # even more strongly undersample no ships\n    else:\n        return in_df.sample(base_rep_val, replace=(in_df.shape[0]<base_rep_val))\n    \nbalanced_train_df = train_df.groupby('grouped_ship_count').apply(sample_ships)\nbalanced_train_df['ships'].hist(bins=np.arange(10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b6bd0730852a89501a6b36b7a38d6a93c905f4c","collapsed":true},"cell_type":"code","source":"def make_image_gen(in_df, batch_size = 128):\n    all_batches = list(in_df.groupby('ImageId'))\n    out_rgb = []\n    out_mask = []\n    while True:\n        np.random.shuffle(all_batches)\n        for c_img_id, c_masks in all_batches:\n            rgb_path = os.path.join('../input/train/', c_img_id)\n            c_img = io.imread(rgb_path)\n            c_mask = masks_as_image(c_masks['EncodedPixels'].values)\n            if (1,1) is not None:\n                c_img = c_img[::1, ::1]\n                c_mask = c_mask[::1, ::1]\n            out_rgb += [c_img]\n            out_mask += [c_mask]\n            if len(out_rgb)>=batch_size:\n                yield np.stack(out_rgb, 0)/255.0, np.stack(out_mask, 0)\n                out_rgb, out_mask=[], []","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f8ed855b634131cbe36acc41d9264016decc2dd","collapsed":true},"cell_type":"code","source":"train_gen = make_image_gen(balanced_train_df)\ntrain_x, train_y = next(train_gen)\nprint('x', train_x.shape, train_x.min(), train_x.max())\nprint('y', train_y.shape, train_y.min(), train_y.max())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed2e9ae70806d4b182ac7ed4da4424c9037a4281","collapsed":true},"cell_type":"code","source":"valid_x, valid_y = next(make_image_gen(valid_df, 400))\nprint(valid_x.shape, valid_y.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"98d951449801354c85ba76825bc366a9955603c8"},"cell_type":"code","source":"dg_args = dict(featurewise_center = False, \n                  samplewise_center = False,\n                  rotation_range = 15, \n                  width_shift_range = 0.1, \n                  height_shift_range = 0.1, \n                  shear_range = 0.01,\n                  zoom_range = [0.9, 1.25],  \n                  horizontal_flip = True, \n                  vertical_flip = True,\n                  fill_mode = 'reflect',\n                   data_format = 'channels_last')\nimage_gen = ImageDataGenerator(**dg_args)\n\nlabel_gen = ImageDataGenerator(**dg_args)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f267416fb4f1705ec1ebda07b01d1bf9c43f9e53"},"cell_type":"code","source":"def create_aug_gen(in_gen, seed = None):\n    np.random.seed(seed if seed is not None else np.random.choice(range(9999)))\n    for in_x, in_y in in_gen:\n        seed = np.random.choice(range(9999))\n        # keep the seeds syncronized otherwise the augmentation to the images is different from the masks\n        g_x = image_gen.flow(255*in_x, \n                             batch_size = in_x.shape[0], \n                             seed = seed, \n                             shuffle=True)\n        g_y = label_gen.flow(in_y, \n                             batch_size = in_x.shape[0], \n                             seed = seed, \n                             shuffle=True)\n\n        yield next(g_x)/255.0, next(g_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f82fe87309eae60730b187a50bd74a0fbfaec90c","collapsed":true},"cell_type":"code","source":"from keras import layers,models\ndef upsample(filters, kernel_size, strides, padding):\n    return layers.UpSampling2D(strides)\n\ninput_img = layers.Input((768, 768, 3), name = 'RGB_Input')\n\npp_in_layer = layers.BatchNormalization()(input_img)\n\nc1 = layers.Conv2D(8, (3, 3), activation='relu', padding='same') (pp_in_layer)\nc1 = layers.Conv2D(8, (3, 3), activation='relu', padding='same') (c1)\np1 = layers.MaxPooling2D((2, 2)) (c1)\n\nc2 = layers.Conv2D(16, (3, 3), activation='relu', padding='same') (p1)\nc2 = layers.Conv2D(16, (3, 3), activation='relu', padding='same') (c2)\np2 = layers.MaxPooling2D((2, 2)) (c2)\n\nc3 = layers.Conv2D(32, (3, 3), activation='relu', padding='same') (p2)\nc3 = layers.Conv2D(32, (3, 3), activation='relu', padding='same') (c3)\np3 = layers.MaxPooling2D((2, 2)) (c3)\n\nc4 = layers.Conv2D(64, (3, 3), activation='relu', padding='same') (p3)\nc4 = layers.Conv2D(64, (3, 3), activation='relu', padding='same') (c4)\np4 = layers.MaxPooling2D(pool_size=(2, 2)) (c4)\n\n\nc5 = layers.Conv2D(128, (3, 3), activation='relu', padding='same') (p4)\nc5 = layers.Conv2D(128, (3, 3), activation='relu', padding='same') (c5)\n\nu6 = upsample(64, (2, 2), strides=(2, 2), padding='same') (c5)\nu6 = layers.concatenate([u6, c4])\nc6 = layers.Conv2D(64, (3, 3), activation='relu', padding='same') (u6)\nc6 = layers.Conv2D(64, (3, 3), activation='relu', padding='same') (c6)\n\nu7 = upsample(32, (2, 2), strides=(2, 2), padding='same') (c6)\nu7 = layers.concatenate([u7, c3])\nc7 = layers.Conv2D(32, (3, 3), activation='relu', padding='same') (u7)\nc7 = layers.Conv2D(32, (3, 3), activation='relu', padding='same') (c7)\n\nu8 = upsample(16, (2, 2), strides=(2, 2), padding='same') (c7)\nu8 = layers.concatenate([u8, c2])\nc8 = layers.Conv2D(16, (3, 3), activation='relu', padding='same') (u8)\nc8 = layers.Conv2D(16, (3, 3), activation='relu', padding='same') (c8)\n\nu9 = upsample(8, (2, 2), strides=(2, 2), padding='same') (c8)\nu9 = layers.concatenate([u9, c1], axis=3)\nc9 = layers.Conv2D(8, (3, 3), activation='relu', padding='same') (u9)\nc9 = layers.Conv2D(8, (3, 3), activation='relu', padding='same') (c9)\n\nd = layers.Conv2D(1, (1, 1), activation='sigmoid') (c9)\n\nseg_model = models.Model(inputs=[input_img], outputs=[d])\nseg_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"79cbe75de44df5bb53ed09c796e616b9cafe3d5e"},"cell_type":"code","source":"import keras.backend as K\nfrom keras.optimizers import Adam\nfrom keras.losses import binary_crossentropy\ndef dice_coef(y_true, y_pred, smooth=1):\n    intersection = K.sum(y_true * y_pred, axis=[1,2,3])\n    union = K.sum(y_true, axis=[1,2,3]) + K.sum(y_pred, axis=[1,2,3])\n    return K.mean( (2. * intersection + smooth) / (union + smooth), axis=0)\ndef dice_p_bce(in_gt, in_pred):\n    return 1e-3*binary_crossentropy(in_gt, in_pred) - dice_coef(in_gt, in_pred)\ndef true_positive_rate(y_true, y_pred):\n    return K.sum(K.flatten(y_true)*K.flatten(K.round(y_pred)))/K.sum(y_true)\nseg_model.compile(optimizer=Adam(1e-4, decay=1e-6), loss=dice_p_bce, metrics=[dice_coef, 'binary_accuracy', true_positive_rate])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"15225672198f07beabb94061cc20882412804a4e","collapsed":true},"cell_type":"code","source":"step_count = min(200, balanced_train_df.shape[0]//128)\naug_gen = create_aug_gen(make_image_gen(balanced_train_df))\nseg_model.fit_generator(aug_gen, \n                             steps_per_epoch=step_count, \n                             epochs=5, \n                             validation_data=(valid_x, valid_y),\n                            workers=1 # the generator is not very thread safe\n                                       )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a3f480d68203e8656ea08233e945fc5b4fb3784","collapsed":true},"cell_type":"code","source":"seg_model.save('seg_model.h5')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}