{"cells":[{"metadata":{"_uuid":"1370590e26df773aca12d6ccf8f4a47d053ca694"},"cell_type":"markdown","source":"Notebook is in progress\n\nDon't know how to collapse sections yet\n\n* [Functions](#functions)\n* [Visualizations](#visualizations)\n* [Stats](#stats)"},{"metadata":{"_uuid":"3e2d7a8509028cbbf0f3c1ef0deca1dbcf8809d0"},"cell_type":"markdown","source":"<a id='functions'></a>\n# Functions"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Importing required things\nimport os\nimport csv\nfrom tqdm import tqdm\nfrom collections import OrderedDict\nfrom collections import Counter\nimport regex\nimport itertools\nimport random\n\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee45eeb5ed414a676ad60a13dee2741be752f33e"},"cell_type":"code","source":"# file locations\ninput_filepath = '../input'\ntrain_csv_filepath = os.path.join(input_filepath, 'train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25cd2d7c5deb51bd7c186d446de0dabe57494d3c"},"cell_type":"code","source":"# labels list\n# order should be the same as number in dataset\nlabels = [\"Nucleoplasm\", \"Nuclear membrane\", \"Nucleoli\", \"Nucleoli fibrillar center\", \"Nuclear speckles\", \"Nuclear bodies\", \"Endoplasmic reticulum\", \"Golgi apparatus\", \n          \"Peroxisomes\", \"Endosomes\", \"Lysosomes\", \"Intermediate filaments\", \"Actin filaments\", \"Focal adhesion sites\", \"Microtubules\", \"Microtubule ends\", \"Cytokinetic bridge\", \n          \"Mitotic spindle\", \"Microtubule organizing center\", \"Centrosome\", \"Lipid droplets\", \"Plasma membrane\", \"Cell junctions\", \"Mitochondria\", \"Aggresome\", \"Cytosol\", \"Cytoplasmic bodies\", \"Rods & rings\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56a582e6697d5d830f99e2cde424c2b72a2c6a24"},"cell_type":"code","source":"def parse_indice(indice):\n    \"\"\"\n    Converts various types of indice lists into normal Python list of integers\n    \"\"\"\n    # case of space separated string, as in dataset\n    if isinstance(indice, str):\n        indice = indice.split(' ')\n        indice = [int(index) for index in indice]\n        \n    # scalar converted to single-element list\n    if not isinstance(indice, list):\n        indice = [indice]\n        \n    return indice","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"170d731a46c5da0383437a755a66669a1fec6067"},"cell_type":"code","source":"def get_labels(indice):\n    \"\"\"\n    Returns the labels of given indice. Indice may be give as space separated string (as in dataset), as python list or as single value\n    \"\"\"\n\n    ans = [labels[index] for index in parse_indice(indice)]\n    return ans  \n\n# def try_get_labels():\n#     arguments = [0, [0], '1 5']\n#     return [get_labels(arg) for arg in arguments]\n# try_get_labels()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d072ac5567dc15e34bceda474752aa56595c244"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c21afb7b5886696a405afb22da883bd4689c00d2"},"cell_type":"code","source":"def get_hots(indice):\n    \"\"\"\n    Returns 1-hot representation of a given indice. \n    Since it can be multiple indice, it is actually many-hot\n    \"\"\"\n    \n    # range(len(labels)): sequence of integers from zero to number of labels-1\n    # int(index in parse_indice(indice)): 1-hot computation\n    ans = np.asarray([int(index in parse_indice(indice)) for index in range(len(labels))])\n    return ans\n    \n# def try_get_hots():\n#     arguments = [0, [0], '1 5']\n#     return [get_hots(arg) for arg in arguments]\n\n# np.asarray(try_get_hots())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec9726fb24649e04c9af5f2aec8c209bd5e7f47a"},"cell_type":"code","source":"def read_train_set():\n    \"\"\"\n    Reads entire trainset into the list of dicts\n    Images are not readen\n    \"\"\"\n    ans = []\n    \n    # train set is guided by CSV file\n    with open(train_csv_filepath) as fp:\n        reader = csv.DictReader(fp, delimiter=',')\n        # reading all rows and appending extra keys\n        for row in reader:\n            row['Hots'] = get_hots(row['Target'])\n            row['Labels'] = get_labels(row['Target'])\n            row['Train'] = True\n            ans.append(row)\n    return ans\n\n# def try_read_train_set():\n    \n#     train_set = read_train_set();\n#     return train_set[0];\n\n# try_read_train_set()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cdbbc64b050443a34034180779d2473dc8e0d6ea"},"cell_type":"code","source":"def parse_filename(filename: str):\n    \"\"\"\n    Extracts sample id and \"color\" from filename\n    \"\"\"\n    filename = os.path.splitext(filename)[0]\n    Id, Color = filename.split('_')\n    return {'Id': Id, 'Color': Color}\n\n# def try_parse_filename():\n    \n#     args = ['00631ec8-bad9-11e8-b2b9-ac1f6b6435d0_red.png']\n#     return [parse_filename(filename) for filename in args]\n\n# try_parse_filename()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8295740d717410820a8c0b44b2f2eeaaba790f6b"},"cell_type":"code","source":"def read_test_set():\n    \"\"\"\n    Reads entire test set into list of dicts\n    Keys are the same as for dataset, except labels are not given\n    \"\"\"\n    \n    # test set in guided by present files \n    filenames = os.listdir(os.path.join(input_filepath, 'test'))\n    # unique ids\n    Ids = set([parse_filename(filename)['Id'] for filename in filenames])\n    # forming the same dicts as in train set\n    return [OrderedDict([('Id', id), ('Train', False)]) for id in Ids]\n\n# def try_read_test_set():\n#     return read_test_set()[0]\n\n# try_read_test_set()   \n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d97607fbb81a18f5ba20b4863da3756da579a21"},"cell_type":"code","source":"# dataset, including both train and test parts, one after another\ndataset0 = read_train_set() + read_test_set()\n# shuffling\nrandom.shuffle(dataset0)\n\n# dict to address samples by id\nby_id_index = {Sample['Id']: i for (i,Sample) in enumerate(dataset0)}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ab214b619222657a7bdb200ac67221ecc8665f5"},"cell_type":"code","source":"def dataset0_get_sample(indice_or_ids):\n    \"\"\"\n    Returns samples by given list of (ordinal) indice or string Ids\n    \"\"\"\n    if isinstance(indice_or_ids, list):\n        return [dataset0_get_sample(index) for index in indice_or_ids]\n\n    if isinstance(indice_or_ids, str):\n        indice_or_ids = by_id_index[indice_or_ids]\n\n    return dataset0[indice_or_ids]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e65f37f4037b6a664e9764cc5bfb7f27ac7ab1bc"},"cell_type":"code","source":"def dataset0_filter(Train: bool = True, Ids = None, Labels = None, AllLabels = False, LabelCount = None, Folds = None, FoldsCount: int = None):\n    \"\"\"\n    Creates filtering generator, which returns all samples in sequence, matching conditions\n\n    All specified conditions should be satisfied (logical AND)\n\n    :param Train: include train set (`True`, default), test set (`False`) or both (`None`)\n    :param Ids: include specified Id or Ids, can be regex; \n    default is `None`, which means include all\n    :param Labels: include specified labels; `None` means any labels (Default)\n    :param AllLabels: include if all label are present (True) or \n    any label is present (False, default)\n    :param LabelCount: number of labels in sample\n    :param Folds: numeric indice of folds to include; \n    dataset is slit into folds by hash code\n    :param FoldsCount: total number of folds to split dataset into\n    \"\"\"\n\n    for sample in dataset0:\n\n        # checking train or test set parameter\n        if Train is not None and sample['Train'] != Train:\n            continue\n\n        # checking ids regexes\n        if Ids is not None:\n            if not isinstance(Ids, list):\n                Ids = [Ids]\n\n            if not any(regex.match(sample['Id']) for regex in Ids):\n                continue\n\n        # checking labels filter\n        if Labels is not None:\n\n            if 'Labels' not in sample:\n                continue\n\n            if not isinstance(Labels, list):\n                Labels = [Labels]\n    \n            if AllLabels:\n                if not all(label in Labels for label in sample['Labels']):\n                    continue\n            else:\n                if not any(label in Labels for label in sample['Labels']):\n                    continue\n                \n                \n        if LabelCount is not None:\n            \n            \n            \n            if 'Labels' not in sample:\n                actual_count = 0\n            else:\n                actual_count = len(sample['Labels'])\n                \n            if actual_count != LabelCount:\n                continue\n\n        # checking folds parameters\n        # there are two of them: folds list and folds count, working together\n        # dataset is splitten into number of folds, specified by folds count\n        # and then only folds specified by list returned\n        if Folds is not None:\n\n            if not isinstance(Folds, list):\n                Folds = [Folds]\n\n            # fold is computed from hash\n            # by definition, hash should be random\n            h = hash(sample['Id'])\n            Fold = h % FoldsCount\n            \n            if Fold not in Folds:\n                continue\n\n        yield sample       \n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"093a79eed8a624597b57c84f45051d42143d3095"},"cell_type":"code","source":"#dataset0_get_sample(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1616fc3bb46bb68aa52e56110ea443180df85bc7"},"cell_type":"code","source":"#dataset0_get_sample('00631ec8-bad9-11e8-b2b9-ac1f6b6435d0')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"28c0b70f1bf373b532bd4d9285591ccceb94a8ad"},"cell_type":"code","source":"#dataset0_filter(Train = False).__next__()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"874dd14e20638fa31cb9f0ffb670710eb242d635"},"cell_type":"code","source":"#list(itertools.islice(dataset0_filter(Train = True),1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"470ccd907032abc1e03cf2ec03bc5f5843dc8a24"},"cell_type":"code","source":"#list(itertools.islice(dataset0_filter(Labels = \"Focal adhesion sites\"),3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"545dd87d271c9dadcbe1c0a21095e2f21977805f"},"cell_type":"code","source":"#list(itertools.islice(dataset0_filter(Folds = 0, FoldsCount = 3),1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a10d897e13324a1d785707faff0d8cf17ddfd85b"},"cell_type":"code","source":"#list(itertools.islice(dataset0_filter(Folds = 1, FoldsCount = 3),1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b0ec8a60c7c2225d5cda4406896df8b0f5af6b1"},"cell_type":"code","source":"#list(itertools.islice(dataset0_filter(Folds = 2, FoldsCount = 3),1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce8cda2a14ab087c320e6af8c69bbbde6b2dc8f2"},"cell_type":"code","source":"colors = ['red', 'green', 'blue', 'yellow']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc76fe7cf531a85c6fed4c35f381ffd1648670d1"},"cell_type":"code","source":"def get_filepath(sample, color):\n    \"\"\"\n    Computes path to image file, specified by given sample dict and color\n    \n    Image can loose data, so use it for visualization only\n    \"\"\"\n    filename = '%s_%s.png' % (sample['Id'], color)\n    if sample['Train']:\n        return os.path.join(input_filepath, 'train', filename)\n    else:\n        return os.path.join(input_filepath, 'test', filename)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f327a1b9ff25c611382b82955d191d8ec35dfc45"},"cell_type":"code","source":"def get_PIL_image(sample, color):\n    \"\"\"\n    Reads sample as PIL image\n    \n    Image retained pale to conserve data\n    \"\"\"\n    filepath = get_filepath(sample, color)\n    return Image.open(filepath).convert('RGB')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbb5cc4f97599645aa1aa4fe0538dd3dcda6b494"},"cell_type":"code","source":"def get_PIL_image_colored(sample, color):\n    \"\"\"\n    Reads sample as PIL images and colors it according to color suffix\n    \"\"\"\n    if 'red' == color:\n        matrix = (1, 0, 0, 0,\n              0, 0, 0, 0,\n              0, 0, 0, 0)\n    elif 'green' == color:\n        matrix = (0, 0, 0, 0,\n              1, 0, 0, 0,\n              0, 0, 0, 0)\n    elif 'blue' == color:\n        matrix = (0, 0, 0, 0,\n              0, 0, 0, 0,\n              1, 0, 0, 0)\n    elif 'yellow' == color:\n        matrix = (1, 0, 0, 0,\n              1, 0, 0, 0,\n              0, 0, 0, 0)\n    return get_PIL_image(sample, color).convert('RGB', matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dee9d010d1356a1dc11eb8c35af65bbe6d60b366"},"cell_type":"code","source":"#get_PIL_image_colored(dataset0_get_sample('0ba299c4-bbbc-11e8-b2ba-ac1f6b6435d0'), 'yellow')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad0b4fcf57c93ede38ff2dea427ca88346a88dcd"},"cell_type":"code","source":"def get_numpy_images(sample):\n    \"\"\"\n    Reads all sample images as numpy 4-channel array \n    Pixel values normalized to one\n    \"\"\"\n    images = [np.array(get_PIL_image_colored(sample, color)) for color in colors]\n    images_np = np.stack(images)\n    images_np = images_np / 255\n    return images_np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40824c86b15eab05ed56f866358c94ebf7c988ed"},"cell_type":"code","source":"def get_PIL_image_mixed(sample):\n    \"\"\"\n    Reads all sample images and mixes them into single multicolor image\n    \n    Image can loose data, so use it for visualization only\n    \"\"\"\n    images_np = get_numpy_images(sample) * 255\n    #images_np = np.sum(images_np, axis=0) / len(images)\n    images_np = np.sum(images_np, axis=0)\n    images_np = images_np.astype( np.uint8 )\n    ans = Image.fromarray(images_np)\n    return ans","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd9231d667c9d841e85b6d9d6c395002e3ff867c"},"cell_type":"code","source":"#get_PIL_image_mixed(dataset0_get_sample('0ba299c4-bbbc-11e8-b2ba-ac1f6b6435d0'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c7892efbce1a4e325ed6bc6e8b3788d76aa957f"},"cell_type":"code","source":"def create_subplots(rows, cols, scale_factor=5):\n    \"\"\"\n    Creates subplot with given number of rows and columns and a given scale\n    :param rows:\n    :param cols: number of columns\n    :param scale_factor: to set figure size on screen or browser\n    \"\"\"\n    fig, axes = plt.subplots(rows, cols,figsize=(cols*scale_factor,rows*scale_factor))\n    return fig, axes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35f3d607e9ce7ca8074d9e8f802babdcfef66c72"},"cell_type":"code","source":"def plot_func_image(pil_image_func):\n    plt.imshow(pil_image_func())\n    plt.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"343be5a501eb6aa8947771c975401a049493811a"},"cell_type":"code","source":"def plot_sample_title(sample):\n    if 'Labels' in sample:\n        title = ', '.join(sample['Labels'])\n    else:\n        title = 'No Labels (test set)'\n    plt.title(title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de655d203c2536e75592f1d55b74802fd2b9b30e"},"cell_type":"code","source":"# # testing\n# fig, axes = create_subplots(3,3)\n# for ax, sample in zip(itertools.chain(*axes), dataset0):\n#     plt.sca(ax)\n#     plot_func_image(lambda: get_PIL_image(sample, 'green'))\n#     plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"08a55c81d754b97b674b92240c5db824299dd448"},"cell_type":"code","source":"# # testing\n# fig, axes = create_subplots(3,3)\n# for ax, sample, color in zip(itertools.chain(*axes), dataset0, itertools.cycle(colors)):\n#     plt.sca(ax)\n#     plot_func_image(lambda: get_PIL_image_colored(sample, color))\n#     plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e0cf6edc57d5940f30902c89f3fd75eee73c452e"},"cell_type":"markdown","source":"<a id='visualizations'></a>\n# Visualizations"},{"metadata":{"trusted":true,"_uuid":"14f40e879a535edb79fd2741877d38c01b983689"},"cell_type":"code","source":"# Drawing some random images from entire dataset\nfig, axes = create_subplots(3,3)\nfor ax, sample in zip(itertools.chain(*axes), dataset0):\n    plt.sca(ax)\n    plot_func_image(lambda: get_PIL_image_mixed(sample))\n    plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e49178026431ccaf5ed87f709b822631f65a6a5"},"cell_type":"code","source":"# Drawing several random images per each label from train dataset \nncols = 5\nfig, axes = create_subplots(len(labels), ncols)\nfor row, label in enumerate(labels):\n    for col, sample in zip(range(ncols), dataset0_filter(Labels=label, LabelCount=1)):\n        plt.sca(axes[row,col])\n        plot_func_image(lambda: get_PIL_image_mixed(sample))\n        plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a956f93e78f858640276f14ed9e7aebf1bd9051"},"cell_type":"markdown","source":"Hmmm!! Some labels are not met alone!"},{"metadata":{"trusted":true,"_uuid":"7f0319548a51f471ae4f6400331898c013b0c11f"},"cell_type":"code","source":"# Drawing several random images per each label from train dataset \n# removed condition to have only one label and now displaying in grayscale of green part, which will be used in recognition\nncols = 3\ncolor = 'green'\nfig, axes = create_subplots(len(labels), ncols)\nfor row, label in enumerate(labels):\n    for col, sample in zip(range(ncols), dataset0_filter(Labels=label)):\n        plt.sca(axes[row,col])\n        plot_func_image(lambda: get_PIL_image(sample, color))\n        plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f423680b56aad5754af046467a93d78adea0643a"},"cell_type":"markdown","source":"<a id='stats'></a>\n# Stats"},{"metadata":{"trusted":true,"_uuid":"a424268294d2ad757ef098050ce018101b8de6c3"},"cell_type":"code","source":"# example: counting samples in train subset\nlen(list(dataset0_filter(Train = True)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f86659e9c7551172d5766ecda1ee1422ed2cbb63"},"cell_type":"code","source":"# numbers of samples in train and test sets\n# wrapping into dataframe\npd.DataFrame(\n    [(\n        'Train' if train else 'Test', \n        len(list(dataset0_filter(Train = train)))\n    ) for train in [True, False]],\n    columns=['Subset','Count']\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77da2d5fd1ee33c3b930cbd33e38bd989d08d20e"},"cell_type":"code","source":"# numbers of samples with given number of labels\n# we are specifying filter `Train=None` to include test set in counting\n# test samples contain no labels\npd.DataFrame(\n    [(\n        n, \n        len(list(dataset0_filter(Train=None, LabelCount=n)))\n    ) for n in range(len(labels)+1)],\n    columns=['# of Labels','Count']\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08d5b62221e8ed53a32477fdf68321180a8bb8f1"},"cell_type":"markdown","source":"Most of samples contain 1 or 2 labels, many samples contain 3 or 4 labels two samples contain 5 labels\nDrawing samples with 5 labels"},{"metadata":{"trusted":true,"_uuid":"48338ae033ab2cc5eabd1a6f88247b9419204e97"},"cell_type":"code","source":"fig, axes = create_subplots(2,1)\nfor ax, sample in zip(axes, dataset0_filter(LabelCount=5)):\n    plt.sca(ax)\n    plot_func_image(lambda: get_PIL_image_mixed(sample))\n    plot_sample_title(sample)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a9eaea28fbd3a4380685bb0d743678041cf19a9"},"cell_type":"markdown","source":"The name \"Focal adhesion sites\" may mean instrumental artefact"},{"metadata":{"_uuid":"205dd035c0bf86868666f7558a299cbd7a267245","trusted":true},"cell_type":"code","source":"# counting labels\nlabels = []\nfor sample in dataset0_filter():\n    for label in sample['Labels']:\n        labels.append(label)\ncounts = Counter(labels)\ndf = pd.DataFrame.from_dict(counts)\ndf.plot(kind='bar')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b2b9650e8ab47efcef1d1c2666cb43588f5b305"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}