{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.util import montage as montage\nmontage_rgb = lambda x: np.stack([montage(x[:, :, :, i]) for i in range(x.shape[3])], -1)\nship_dir = '../input'\ntrain_image_dir = os.path.join(ship_dir, '../input/airbus-ship-detection/train_v2')\ntest_image_dir = os.path.join(ship_dir, '../input/airbus-ship-detection/test_v2')\nimport gc; gc.enable() # memory is tight\n\nfrom skimage.morphology import label\n\n## dekodiranje slika i maski\n\ndef multi_rle_encode(img):\n    labels = label(img[:, :, 0])\n    return [rle_encode(labels==k) for k in np.unique(labels[labels>0])]\n\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef masks_as_image(in_mask_list):\n    # Take the individual ship masks and create a single mask array for all ships\n    all_masks = np.zeros((768, 768), dtype = np.int16)\n    #if isinstance(in_mask_list, list):\n    for mask in in_mask_list:\n        if isinstance(mask, str):\n            all_masks += rle_decode(mask)\n    return np.expand_dims(all_masks, -1)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T16:56:08.568468Z","iopub.execute_input":"2022-07-01T16:56:08.568950Z","iopub.status.idle":"2022-07-01T16:56:10.176402Z","shell.execute_reply.started":"2022-07-01T16:56:08.568791Z","shell.execute_reply":"2022-07-01T16:56:10.175335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('training data:')\n\ntrain = os.listdir('../input/airbus-ship-detection/train_v2')\nprint(len(train), 'training images')\nprint('-' * 80)\n\ntrain_masks = pd.read_csv('../input/airbus-ship-detection/train_ship_segmentations_v2.csv')\nprint(train_masks.shape[0], 'training masks')\nprint('-' * 80)\n\nprint('test data:')\n\ntest = os.listdir('../input/airbus-ship-detection/test_v2')\nprint(len(test), 'test images')\nprint('-' * 80)\n\ntest_masks = pd.read_csv('../input/airbus-ship-detection/sample_submission_v2.csv')\nprint(len(test_masks), 'test masks')\nprint('-' * 80)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T16:56:10.178139Z","iopub.execute_input":"2022-07-01T16:56:10.178489Z","iopub.status.idle":"2022-07-01T16:56:17.741222Z","shell.execute_reply.started":"2022-07-01T16:56:10.178443Z","shell.execute_reply":"2022-07-01T16:56:17.740339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_masks['ships'] = train_masks['EncodedPixels'].map(lambda c_row: 1 if isinstance(c_row, str) else 0)\nunique_img_ids = train_masks.groupby('ImageId').agg({'ships': 'sum'}).reset_index()\nunique_img_ids['has_ship'] = unique_img_ids['ships'].map(lambda x: 1.0 if x>0 else 0.0)\nunique_img_ids['has_ship_vec'] = unique_img_ids['has_ship'].map(lambda x: [x])\n# some files are too small/corrupt\nunique_img_ids['file_size_kb'] = unique_img_ids['ImageId'].map(lambda c_img_id: \n                                                               os.stat(os.path.join(train_image_dir, \n                                                                                    c_img_id)).st_size/1024)\nunique_img_ids = unique_img_ids[unique_img_ids['file_size_kb']>50] # keep only 50kb files\nunique_img_ids['file_size_kb'].hist()\ntrain_masks.drop(['ships'], axis=1, inplace=True)\nunique_img_ids.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T16:56:17.742446Z","iopub.execute_input":"2022-07-01T16:56:17.742755Z","iopub.status.idle":"2022-07-01T17:06:18.735392Z","shell.execute_reply.started":"2022-07-01T16:56:17.742723Z","shell.execute_reply":"2022-07-01T17:06:18.734576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_img_ids['ships'].hist(bins=unique_img_ids['ships'].max())","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:18.737435Z","iopub.execute_input":"2022-07-01T17:06:18.737663Z","iopub.status.idle":"2022-07-01T17:06:18.947149Z","shell.execute_reply.started":"2022-07-01T17:06:18.737635Z","shell.execute_reply":"2022-07-01T17:06:18.946191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df = pd.merge(train_masks, unique_img_ids)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:18.948289Z","iopub.execute_input":"2022-07-01T17:06:18.948510Z","iopub.status.idle":"2022-07-01T17:06:18.953871Z","shell.execute_reply.started":"2022-07-01T17:06:18.948481Z","shell.execute_reply":"2022-07-01T17:06:18.952943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_df['grouped_ship_count'] = train_df['ships'].map(lambda x: (x+1)//2).clip(0, 7)\n#def sample_ships(in_df, base_rep_val=1500):\n#   if in_df['ships'].values[0]==0:\n#        return in_df.sample(base_rep_val//3) # even more strongly undersample no ships\n#    else:\n#        return in_df.sample(base_rep_val, replace=(in_df.shape[0]<base_rep_val))\n    \n#balanced_train_df = train_df.groupby('grouped_ship_count').apply(sample_ships)\n#balanced_train_df['ships'].hist(bins=np.arange(10))","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:18.955199Z","iopub.execute_input":"2022-07-01T17:06:18.955673Z","iopub.status.idle":"2022-07-01T17:06:18.963726Z","shell.execute_reply.started":"2022-07-01T17:06:18.955629Z","shell.execute_reply":"2022-07-01T17:06:18.963184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLES_PER_GROUP = 2000\nbalanced_train_df = unique_img_ids.groupby('ships').apply(lambda x: x.sample(SAMPLES_PER_GROUP) if len(x) > SAMPLES_PER_GROUP else x)\nbalanced_train_df['ships'].hist(bins=balanced_train_df['ships'].max()+1)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:18.964695Z","iopub.execute_input":"2022-07-01T17:06:18.965344Z","iopub.status.idle":"2022-07-01T17:06:19.238497Z","shell.execute_reply.started":"2022-07-01T17:06:18.965313Z","shell.execute_reply":"2022-07-01T17:06:19.237553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(balanced_train_df.shape[0], 'masks')","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:19.239512Z","iopub.execute_input":"2022-07-01T17:06:19.239724Z","iopub.status.idle":"2022-07-01T17:06:19.245683Z","shell.execute_reply.started":"2022-07-01T17:06:19.239691Z","shell.execute_reply":"2022-07-01T17:06:19.244712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.merge(train_masks, balanced_train_df)\ntrain_df['ships'].hist(bins=train_df['ships'].max()+1)\nprint(train_df.shape[0], 'training masks')","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:06:19.247113Z","iopub.execute_input":"2022-07-01T17:06:19.247421Z","iopub.status.idle":"2022-07-01T17:06:19.565183Z","shell.execute_reply.started":"2022-07-01T17:06:19.247378Z","shell.execute_reply":"2022-07-01T17:06:19.564208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 1\nIMG_SCALING = (1,1)\n\ndef make_image_gen(in_df, batch_size = BATCH_SIZE):\n    all_batches = list(in_df.groupby('ImageId'))\n    out_rgb = []\n    out_mask = []\n    while True:\n        np.random.shuffle(all_batches)\n        for c_img_id, c_masks in all_batches:\n            rgb_path = os.path.join(train_image_dir, c_img_id)\n            c_img = imread(rgb_path)\n            c_mask = masks_as_image(c_masks['EncodedPixels'].values)\n            if IMG_SCALING is not None:\n                c_img = c_img[::IMG_SCALING[0], ::IMG_SCALING[1]]\n                c_mask = c_mask[::IMG_SCALING[0], ::IMG_SCALING[1]]\n            out_rgb += [c_img]\n            out_mask += [c_mask]\n            if len(out_rgb)>=batch_size:\n                yield np.stack(out_rgb, 0)/255.0, np.stack(out_mask, 0)\n                out_rgb, out_mask=[], []\n                \ntrain_gen = make_image_gen(train_df)\ntrain_x, train_y = next(train_gen)\n\n#print('x', train_x.shape, train_x.min(), train_x.max(), train_x.dtype)\n#print('y', train_y.shape, train_y.min(), train_y.max(), train_y.dtype)","metadata":{"execution":{"iopub.status.busy":"2022-07-01T17:10:34.636483Z","iopub.execute_input":"2022-07-01T17:10:34.636752Z","iopub.status.idle":"2022-07-01T17:10:35.247869Z","shell.execute_reply.started":"2022-07-01T17:10:34.636724Z","shell.execute_reply":"2022-07-01T17:10:35.246996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"gc.collect()","metadata":{}},{"cell_type":"code","source":"os.makedirs('./train_images')\nos.makedirs('./train_masks')\n\nos.makedirs('./valid_images')\nos.makedirs('./valid_masks')\n\nos.makedirs('./test_images')\nos.makedirs('./test_masks')","metadata":{"execution":{"iopub.status.busy":"2022-06-11T21:00:14.608721Z","iopub.execute_input":"2022-06-11T21:00:14.608935Z","iopub.status.idle":"2022-06-11T21:00:14.613252Z","shell.execute_reply.started":"2022-06-11T21:00:14.608909Z","shell.execute_reply":"2022-06-11T21:00:14.612661Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_IMAGES_DIR = './train_images'\nTRAIN_MASKS_DIR = './train_masks'\n\nVALID_IMAGES_DIR = './valid_images'\nVALID_MASKS_DIR = './valid_masks'\n\nTEST_IMAGES_DIR = './test_images'\nTEST_MASKS_DIR = './test_masks'","metadata":{"execution":{"iopub.status.busy":"2022-06-11T21:00:14.614123Z","iopub.execute_input":"2022-06-11T21:00:14.614396Z","iopub.status.idle":"2022-06-11T21:00:14.625483Z","shell.execute_reply.started":"2022-06-11T21:00:14.614361Z","shell.execute_reply":"2022-06-11T21:00:14.624688Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import save_img\nfrom time import sleep\nfrom progressbar import progressbar\n\ni=0\nfor img in progressbar(train_gen):\n    \n    if i < 22500:\n        image=img[0]\n        image=image[0,:,:,:]\n        train_image_path=os.path.join(TRAIN_IMAGES_DIR, 'image{}.png'.format(i))\n        save_img(train_image_path, image)\n    \n        mask=img[1]\n        mask=mask[0,:,:,:]\n        mask=np.clip(mask, 0, 1)\n        train_mask_path=os.path.join(TRAIN_MASKS_DIR, 'image{}.png'.format(i))\n        save_img(train_mask_path, mask, scale=False)\n    \n    elif 22500 <= i < 25000:\n        image=img[0]\n        image=image[0,:,:,:]\n        valid_image_path=os.path.join(VALID_IMAGES_DIR, 'image{}.png'.format(i))\n        save_img(valid_image_path, image)\n    \n        mask=img[1]\n        mask=mask[0,:,:,:]\n        mask=np.clip(mask, 0, 1)\n        valid_mask_path=os.path.join(VALID_MASKS_DIR, 'image{}.png'.format(i))\n        save_img(valid_mask_path, mask, scale=False)\n        \n          \n    elif i == 25000:\n        break\n        \n    i+=1  ","metadata":{"execution":{"iopub.status.busy":"2022-06-11T21:08:32.105786Z","iopub.execute_input":"2022-06-11T21:08:32.106041Z","iopub.status.idle":"2022-06-11T21:08:42.181987Z","shell.execute_reply.started":"2022-06-11T21:08:32.106014Z","shell.execute_reply":"2022-06-11T21:08:42.181101Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}