{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Average Masks\n\nGiven that almost everyone will be using largely translation invariant CNN architectures, it seems important that location within an image impacts the prior probability that a pixel is a part of a mask."},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rle_to_mask(rle_string, width, height):\n    '''\n    convert RLE(run length encoding) string to numpy array\n\n    Parameters:\n    rle_string (str): string of rle encoded mask\n    height (int): height of the mask\n    width (int): width of the mask\n\n    Returns:\n    numpy.array: numpy array of the mask\n    '''\n\n    rows, cols = height, width\n\n    if rle_string == -1:\n        return np.zeros((height, width))\n    else:\n        rle_numbers = [int(num_string) for num_string in rle_string.split(' ')]\n        rle_pairs = np.array(rle_numbers).reshape(-1, 2)\n        img = np.zeros(rows * cols, dtype=np.uint8)\n        for index, length in rle_pairs:\n            index -= 1\n            img[index:index + length] = 255\n        img = img.reshape(cols, rows)\n        img = img.T\n        return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/understanding_cloud_organization/train.csv')\ndf['Label'] = df['Image_Label'].apply(lambda x: x.split('_')[1])\nlabels = sorted(list(df['Label'].unique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for l in labels:\n    m = np.zeros((1400, 2100), dtype=np.float)\n    count = 0\n\n    for idx, row in df.iterrows():\n        if row['Label'] == l:\n            if not isinstance(row['EncodedPixels'], float):\n                rle = row['EncodedPixels']\n                mask = rle_to_mask(rle, 2100, 1400)\n                mask = np.clip(mask, 0, 1)\n                m += mask.astype(np.float)\n            count += 1\n\n    view = m / count\n\n    plt.figure()\n    plt.imshow(view)\n    plt.title(l)\n    \n    plt.figure()\n    plt.hist(view.flatten(), bins=100)\n    plt.title('Histogram of Values for {}'.format(l))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}