{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We combine here the 2 best public models:\n\nThe U-Net segmentation from: https://www.kaggle.com/hmendonca/u-net-model-with-submission\n\nand reduce the false positives with the classification CNN: https://www.kaggle.com/kmader/transfer-learning-for-boat-or-no-boat (Thanks Kevin!)\n","metadata":{"_uuid":"1027d25b1beb614f2e1f4d0a916c38bc3dff8af0"}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nprint(os.listdir(\"../input/\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-08-16T22:28:36.276469Z","iopub.execute_input":"2023-08-16T22:28:36.277276Z","iopub.status.idle":"2023-08-16T22:28:36.283010Z","shell.execute_reply.started":"2023-08-16T22:28:36.277200Z","shell.execute_reply":"2023-08-16T22:28:36.282177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boat_df = pd.read_csv(\"../input/transfer-learning-for-boat-or-no-boat/submission.csv\")\nboat_df.head()","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-08-16T22:28:36.284421Z","iopub.execute_input":"2023-08-16T22:28:36.284723Z","iopub.status.idle":"2023-08-16T22:28:36.698648Z","shell.execute_reply.started":"2023-08-16T22:28:36.284664Z","shell.execute_reply":"2023-08-16T22:28:36.697676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"is_boat = boat_df.EncodedPixels.notnull()\nprint('Found {} boats'.format(is_boat.sum()))","metadata":{"_uuid":"ddbed4ead9e524eb4be8c0a9f0449d6e704e1c06","execution":{"iopub.status.busy":"2023-08-16T22:28:36.700127Z","iopub.execute_input":"2023-08-16T22:28:36.700481Z","iopub.status.idle":"2023-08-16T22:28:36.708091Z","shell.execute_reply.started":"2023-08-16T22:28:36.700408Z","shell.execute_reply":"2023-08-16T22:28:36.706885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import load_model\nfullres_model = load_model('../input/u-net-model-with-submission/fullres_model.h5')","metadata":{"_uuid":"67ee5a43524e0987275e47bcde59e28ca41aea41","execution":{"iopub.status.busy":"2023-08-16T22:28:36.709441Z","iopub.execute_input":"2023-08-16T22:28:36.709694Z","iopub.status.idle":"2023-08-16T22:28:38.131775Z","shell.execute_reply.started":"2023-08-16T22:28:36.709646Z","shell.execute_reply":"2023-08-16T22:28:38.129530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skimage.io import imread\nimport matplotlib.pyplot as plt\nfrom matplotlib.cm import get_cmap\nfrom skimage.segmentation import mark_boundaries\nfrom skimage.morphology import binary_opening, disk, label\nimport gc; gc.enable() # memory is tight\n\nship_dir = '../input/airbus-ship-detection/'\ntrain_image_dir = os.path.join(ship_dir, 'train_v2')\ntest_image_dir = os.path.join(ship_dir, 'test_v2')\n\ndef predict(img, path=test_image_dir):\n    c_img = imread(os.path.join(path, c_img_name))\n    c_img = np.expand_dims(c_img, 0) / 255.0\n    cur_seg = fullres_model.predict(c_img)[0]\n    cur_seg = binary_opening(cur_seg > 0.9, np.expand_dims(disk(2), -1))\n    return cur_seg, c_img\n\ndef multi_rle_encode(img, **kwargs):\n    '''\n    Encode connected regions as separated masks\n    '''\n    labels = label(img)\n    if img.ndim > 2:\n        return [rle_encode(np.sum(labels==k, axis=2), **kwargs) for k in np.unique(labels[labels>0])]\n    else:\n        return [rle_encode(labels==k, **kwargs) for k in np.unique(labels[labels>0])]\n\n# ref: https://www.kaggle.com/paulorzp/run-length-encode-and-decode\ndef rle_encode(img, min_max_threshold=1e-3, max_mean_threshold=None):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    if np.max(img) < min_max_threshold:\n        return '' ## no need to encode if it's all zeros\n    if max_mean_threshold and np.mean(img) > max_mean_threshold:\n        return '' ## ignore overfilled mask\n    pixels = img.T.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef pred_encode(img, **kwargs):\n    cur_seg, _ = predict(img)\n    cur_rles = multi_rle_encode(cur_seg, **kwargs)\n    return [[img, rle] for rle in cur_rles if rle is not None]","metadata":{"_uuid":"20e408bae7562b1eff5e1008d2644225ee2ce8f0","execution":{"iopub.status.busy":"2023-08-16T22:28:38.133276Z","iopub.execute_input":"2023-08-16T22:28:38.133575Z","iopub.status.idle":"2023-08-16T22:28:39.228204Z","shell.execute_reply.started":"2023-08-16T22:28:38.133523Z","shell.execute_reply":"2023-08-16T22:28:39.227304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm_notebook\n\nout_pred_rows = []\nfor c_img_name in tqdm_notebook(boat_df.ImageId[is_boat]):\n    out_pred_rows += pred_encode(c_img_name, min_max_threshold=1.0)","metadata":{"_uuid":"2610667e635db23c77f8ab0f4ffb8ccb7dd7ec82","execution":{"iopub.status.busy":"2023-08-16T22:28:39.229407Z","iopub.execute_input":"2023-08-16T22:28:39.229656Z","iopub.status.idle":"2023-08-16T22:38:28.280810Z","shell.execute_reply.started":"2023-08-16T22:28:39.229612Z","shell.execute_reply":"2023-08-16T22:38:28.279729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(out_pred_rows)\nsub.columns = ['ImageId', 'EncodedPixels']\nsub = sub[sub.EncodedPixels.notnull()]\nsub.head()","metadata":{"_uuid":"768bb63115c35ade27f3fc1673969fa19ba7603b","execution":{"iopub.status.busy":"2023-08-16T22:38:28.282525Z","iopub.execute_input":"2023-08-16T22:38:28.283262Z","iopub.status.idle":"2023-08-16T22:38:28.307693Z","shell.execute_reply.started":"2023-08-16T22:38:28.283190Z","shell.execute_reply":"2023-08-16T22:38:28.306712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_decode(mask_rle, shape=(768, 768)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape).T  # Needed to align to RLE direction\n\ndef masks_as_color(in_mask_list):\n    # Take the individual ship masks and create a color mask array for each ships\n    all_masks = np.zeros((768, 768), dtype = np.float)\n    scale = lambda x: (len(in_mask_list)+x+1) / (len(in_mask_list)*2) ## scale the heatmap image to shift \n    for i,mask in enumerate(in_mask_list):\n        if isinstance(mask, str):\n            all_masks[:,:] += scale(i) * rle_decode(mask)\n    return all_masks","metadata":{"_uuid":"b803a3b4b37bd6abe6f1c0a7e69b915c720d7524","execution":{"iopub.status.busy":"2023-08-16T22:38:28.309111Z","iopub.execute_input":"2023-08-16T22:38:28.309555Z","iopub.status.idle":"2023-08-16T22:38:28.321192Z","shell.execute_reply.started":"2023-08-16T22:38:28.309358Z","shell.execute_reply":"2023-08-16T22:38:28.319989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## let's see what we got\nTOP_PREDICTIONS = 6\nfig, m_axs = plt.subplots(TOP_PREDICTIONS, 3, figsize = (14, TOP_PREDICTIONS*4))\n[c_ax.axis('off') for c_ax in m_axs.flatten()]\n\ndef raw_prediction(img, path=test_image_dir):\n    c_img = imread(os.path.join(path, c_img_name))\n    c_img = np.expand_dims(c_img, 0)/255.0\n    cur_seg = fullres_model.predict(c_img)[0]\n    return cur_seg, c_img[0]\n\nfor (ax1, ax2, ax3), c_img_name in zip(m_axs, sub.ImageId.unique()[:TOP_PREDICTIONS]):\n    pred, c_img = raw_prediction(c_img_name)\n    ax1.imshow(c_img)\n    ax1.set_title('Image: ' + c_img_name)\n    ax2.imshow(pred[...,0], cmap=get_cmap('jet'))\n    ax2.set_title('Prediction')\n    ax3.imshow(masks_as_color(sub.query('ImageId==\\\"{}\\\"'.format(c_img_name))['EncodedPixels']))\n    ax3.set_title('Masks')","metadata":{"_uuid":"5dfa388dadc7154e68f61a3914c8fddaf15ddebd","execution":{"iopub.status.busy":"2023-08-16T22:38:45.922440Z","iopub.execute_input":"2023-08-16T22:38:45.922864Z","iopub.status.idle":"2023-08-16T22:38:49.791196Z","shell.execute_reply.started":"2023-08-16T22:38:45.922795Z","shell.execute_reply":"2023-08-16T22:38:49.790181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = pd.read_csv('../input/airbus-ship-detection/sample_submission_v2.csv')\nsub1 = pd.DataFrame(np.setdiff1d(sub1['ImageId'].unique(), sub['ImageId'].unique(), assume_unique=True), columns=['ImageId'])\nsub1['EncodedPixels'] = None\nprint(len(sub1), len(sub))\n\nsub = pd.concat([sub, sub1])\nprint(len(sub))\nsub.to_csv('blended_submission.csv', index=False)\nsub.head()","metadata":{"_uuid":"9e120e5bf35334c04fe8eb2e85e06b6a4ebc9cc6","execution":{"iopub.status.busy":"2023-08-16T22:38:31.792935Z","iopub.execute_input":"2023-08-16T22:38:31.793468Z","iopub.status.idle":"2023-08-16T22:38:33.899979Z","shell.execute_reply.started":"2023-08-16T22:38:31.793410Z","shell.execute_reply":"2023-08-16T22:38:33.898850Z"},"trusted":true},"execution_count":null,"outputs":[]}]}