{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"72483a9d-6dd0-994c-24eb-f35fb3604752"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"fa1b3347-9665-a26c-6cc5-52613e833101"},"outputs":[],"source":"from collections import defaultdict\nimport csv\nimport sys\n\nimport cv2\nfrom shapely.geometry import MultiPolygon, Polygon\nimport shapely.wkt\nimport shapely.affinity\nimport numpy as np\nimport tifffile as tiff\n\ncsv.field_size_limit(sys.maxsize);"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"af4f6d17-5958-c83d-7ad9-ef792ff2a256"},"outputs":[],"source":"IM_ID = '6120_2_2'\nPOLY_TYPE = '1'  # buildings\n\n# Load grid size\nx_max = y_min = None\nfor _im_id, _x, _y in csv.reader(open('../input/grid_sizes.csv')):\n    if _im_id == IM_ID:\n        x_max, y_min = float(_x), float(_y)\n        break\n\n# Load train poly with shapely\ntrain_polygons = None\nfor _im_id, _poly_type, _poly in csv.reader(open('../input/train_wkt_v4.csv')):\n    if _im_id == IM_ID and _poly_type == POLY_TYPE:\n        train_polygons = shapely.wkt.loads(_poly)\n        break\n\n# Read image with tiff\nim_rgb = tiff.imread('../input/three_band/{}.tif'.format(IM_ID)).transpose([1, 2, 0])\nim_size = im_rgb.shape[:2]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"92975c99-cc16-926a-9008-bf7c02690c31"},"outputs":[],"source":"im_size"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f1ef604e-8472-bd55-8ebe-24f7058c2b59"},"outputs":[],"source":"im_rgb.shape"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"27fa4842-2e06-f745-c457-2ef29b611cb3"},"outputs":[],"source":"def get_scalers():\n    h, w = im_size  # they are flipped so that mask_for_polygons works correctly\n    w_ = w * (w / (w + 1))\n    h_ = h * (h / (h + 1))\n    return w_ / x_max, h_ / y_min"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"67709ec8-c5e9-44df-fa1a-88726bf5c75a"},"outputs":[],"source":"x_scaler, y_scaler = get_scalers()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a5591107-4147-e537-4f2f-1015ed846570"},"outputs":[],"source":"train_polygons_scaled = shapely.affinity.scale(\n    train_polygons, xfact=x_scaler, yfact=y_scaler, origin=(0, 0, 0))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"93763f55-3965-15fb-1fe0-6e888e878ce6"},"outputs":[],"source":"def mask_for_polygons(polygons):\n    img_mask = np.zeros(im_size, np.uint8)\n    if not polygons:\n        return img_mask\n    int_coords = lambda x: np.array(x).round().astype(np.int32)\n    exteriors = [int_coords(poly.exterior.coords) for poly in polygons]\n    interiors = [int_coords(pi.coords) for poly in polygons\n                 for pi in poly.interiors]\n    cv2.fillPoly(img_mask, exteriors, 1)\n    cv2.fillPoly(img_mask, interiors, 0)\n    return img_mask"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b456546f-3831-be75-ab23-e22d2b8b5258"},"outputs":[],"source":"train_mask = mask_for_polygons(train_polygons_scaled)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3aeddf38-8aba-da39-b514-9a2815f54358"},"outputs":[],"source":"def scale_percentile(matrix):\n    w, h, d = matrix.shape\n    matrix = np.reshape(matrix, [w * h, d]).astype(np.float64)\n    # Get 2nd and 98th percentile\n    mins = np.percentile(matrix, 1, axis=0)\n    maxs = np.percentile(matrix, 99, axis=0) - mins\n    matrix = (matrix - mins[None, :]) / maxs[None, :]\n    matrix = np.reshape(matrix, [w, h, d])\n    matrix = matrix.clip(0, 1)\n    return matrix"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6ab04da7-c359-29ac-0c65-83c0906d3d1b"},"outputs":[],"source":"tiff.imshow(255 * scale_percentile(im_rgb[2900:3200,2000:2300]));"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"32675637-e1d1-c007-f792-9766b39cdd13"},"outputs":[],"source":"def show_mask(m):\n    # hack for nice display\n    tiff.imshow(255 * np.stack([m, m, m]));\n    \nshow_mask(train_mask[2900:3200,2000:2300])"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2e6e509e-3cbf-bcdc-3a06-16af50193027"},"outputs":[],"source":"xs = im_rgb.reshape(-1, 3).astype(np.float32)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f1049c08-e0fb-d7a5-9146-97e92b818168"},"outputs":[],"source":"xs.shape"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b7df30ad-291e-1c6c-5685-56f8e5c06d50"},"outputs":[],"source":"3348*3403"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"fea6b651-5c09-def6-6243-5b32df3db116"},"outputs":[],"source":"ys = train_mask.reshape(-1)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"004d89d5-6675-9acd-2cc5-3143bdcaeee0"},"outputs":[],"source":"ys.shape"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}