{"cells":[{"metadata":{"_uuid":"16c5e9f6b0cc1e2be83d54adff6ec5ed16dbe0d3"},"cell_type":"markdown","source":"## Our goal\n\n* Understand the dataset"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport itertools as itt\nimport networkx as nx\nfrom sklearn.preprocessing import MinMaxScaler","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"323be0d0654cb17920cc5aa9e9af6aeffed88b96","trusted":true},"cell_type":"code","source":"import sys\nfor name, module in sorted(sys.modules.items()):\n    if name in ['numpy', 'pandas', 'seaborn', 'matplotlib', 'networkx', 'sklearn']:\n        if hasattr(module, '__version__'): \n            print(name, module.__version__)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_labels = pd.read_csv(\"../input/train.csv\")\ntrain_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"69a6b9c1584ebcf68b5aab128199f524daccd365"},"cell_type":"markdown","source":"How many samples do we have?"},{"metadata":{"_kg_hide-input":true,"_uuid":"abaa2a7225e485102ceae9bdda12f4cf835c492f","trusted":true},"cell_type":"code","source":"train_labels.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"882bba3d7f68f209051fea3899f046bd96dc2915"},"cell_type":"markdown","source":"## Helper code"},{"metadata":{"_kg_hide-input":true,"_uuid":"beaac70d0392fab7c5b1d49649a8c994310320e4","trusted":true},"cell_type":"code","source":"label_names = {\n    0:  \"Nucleoplasm\",  \n    1:  \"Nuclear membrane\",   \n    2:  \"Nucleoli\",   \n    3:  \"Nucleoli fibrillar center\",   \n    4:  \"Nuclear speckles\",\n    5:  \"Nuclear bodies\",   \n    6:  \"Endoplasmic reticulum\",   \n    7:  \"Golgi apparatus\",   \n    8:  \"Peroxisomes\",   \n    9:  \"Endosomes\",   \n    10:  \"Lysosomes\",   \n    11:  \"Intermediate filaments\",   \n    12:  \"Actin filaments\",   \n    13:  \"Focal adhesion sites\",   \n    14:  \"Microtubules\",   \n    15:  \"Microtubule ends\",   \n    16:  \"Cytokinetic bridge\",   \n    17:  \"Mitotic spindle\",   \n    18:  \"Microtubule organizing center\",   \n    19:  \"Centrosome\",   \n    20:  \"Lipid droplets\",   \n    21:  \"Plasma membrane\",   \n    22:  \"Cell junctions\",   \n    23:  \"Mitochondria\",   \n    24:  \"Aggresome\",   \n    25:  \"Cytosol\",   \n    26:  \"Cytoplasmic bodies\",   \n    27:  \"Rods & rings\"\n}\n\nreverse_train_labels = dict((v,k) for k,v in label_names.items())\n\ndef fill_targets(row):\n    row.Target = np.array(row.Target.split(\" \")).astype(np.int)\n    for num in row.Target:\n        name = label_names[int(num)]\n        row.loc[name] = 1\n    return row","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"707280dec8e4b10dc743b5473ba8d1c4492141ff"},"cell_type":"markdown","source":"## Which protein organelle localizations occur most often in images?"},{"metadata":{"_kg_hide-input":true,"_uuid":"13a1f4fd6b594cd4a225f6d79d966096d1c800ae","trusted":true},"cell_type":"code","source":"for key in label_names.keys():\n    train_labels[label_names[key]] = 0","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"3662893c91d0f74fe666300c214f67cfb03060c7","trusted":true},"cell_type":"code","source":"train_labels = train_labels.apply(fill_targets, axis=1)\ntrain_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"f516b3a1a297a167264b0f22803de231096c9719","scrolled":false,"trusted":true},"cell_type":"code","source":"target_counts = train_labels.drop([\"Id\", \"Target\"],axis=1).sum(axis=0).sort_values(ascending=False)\nplt.figure(figsize=(15,15))\nsns.barplot(y=target_counts.index.values, x=target_counts.values, order=target_counts.index)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"67adc10b191184b2e4bbbb5e3939a48277ad0896","trusted":true},"cell_type":"code","source":"target_counts.tail()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af190deeab9f05afcba339f31c7162d841e58b33"},"cell_type":"markdown","source":"### Take-Away\n\n* We can see that most common protein structures belong to coarse grained cellular components like the plasma membrane, the cytosol and the nucleus. \n* In contrast small components like the lipid droplets, peroxisomes, endosomes, lysosomes, microtubule ends, rods and rings are very seldom in our train data. For these classes the prediction will be very difficult as we have only a few examples that may not cover all variabilities and as our model probably will be confused during ins learning process by the major classes. Due to this confusion we will make less accurate predictions on the minor classes.\n* Consequently accuracy is not the right score here to measure your performance and validation strategy should be very fine. "},{"metadata":{"_uuid":"40bff26b2d170e9936c5e6ce4fbef12e7cb4f2ba"},"cell_type":"markdown","source":"## How many targets are most common?"},{"metadata":{"_kg_hide-input":true,"_uuid":"313f47ac92fa1ff241148af6282c200451edbea7","scrolled":true,"trusted":true},"cell_type":"code","source":"train_labels[\"number_of_targets\"] = train_labels.drop([\"Id\", \"Target\"],axis=1).sum(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b7b727a668a178276b664ece37b9bce62f50bc94","trusted":true},"cell_type":"code","source":"count_perc = train_labels.groupby(\"number_of_targets\").count()['Id']\ncount_perc","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"923c9a88edc91ea80584df83d3f14b419db2fd2c","trusted":true},"cell_type":"code","source":"plt.figure(figsize=(5,5))\nplt.pie(count_perc,\n        labels=[\"%d targets\" % x for x in count_perc.index],\n        autopct='%1.1f%%')\nplt.ylabel('');","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d6e2a72c6f544f4d03a7f0e273a5819982b58108"},"cell_type":"markdown","source":"### Take-away\n\n* Most train images only have 1 or two target labels.\n* More than 3 targets are very seldom!"},{"metadata":{"_uuid":"cb957f819453857d62a96a236febabcfb9ce5a22"},"cell_type":"markdown","source":"## Which targets are correlated?\n\nLet's see if we find some correlations between our targets. This way we may already see that some proteins often come together."},{"metadata":{"_uuid":"99f3865f72d35a5126551a360cc7d6eb07efef4a","trusted":true},"cell_type":"code","source":"def heatmap(C):\n    mask = np.zeros_like(C)\n    mask[np.triu_indices_from(mask)] = True\n    f, ax = plt.subplots(figsize=(11, 9))\n    hm = sns.heatmap(C,\n                mask=mask, cmap=sns.diverging_palette(220, 10, as_cmap=True),\n                vmax=.3,\n                center=0,\n                square=True,\n                linewidths=1,\n                cbar_kws={\"shrink\": .5})\n    hm.set_facecolor('w')\n    return hm","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"b3c0f5a77e34296416136ca2800b8623f94885d8","trusted":true},"cell_type":"code","source":"C = train_labels[train_labels.number_of_targets>1].drop(\n    [\"Id\", \"Target\", \"number_of_targets\"],axis=1\n).corr()\nheatmap(C);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c8083f38bf943a2d38c86544cf322ffe721d8a19"},"cell_type":"markdown","source":"### Take-away\n\n* We can see that many targets only have very slight correlations. \n* In contrast, endosomes and lysosomes often occur together and sometimes seem to be located at the endoplasmatic reticulum. \n* In addition we find that the mitotic spindle often comes together with the cytokinetic bridge. This makes sense as both are participants for cellular division. And in this process microtubules and thier ends are active and participate as well. Consequently we find a positive correlation between these targets."},{"metadata":{"_uuid":"851da464c3dcc86b4ad6686445b592a98ed01b6c"},"cell_type":"markdown","source":"To spot the correlations, let's select those with absolute value greater than 0.1"},{"metadata":{"_uuid":"31198d961a57d0db8e3cf98d46c060d97cd59b9c","trusted":true},"cell_type":"code","source":"heatmap(C[abs(C)> 0.1]);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"05cfeeef96420bd653c9b805fe1789aaaa48aecc"},"cell_type":"markdown","source":"Now we create a function to get the edges weighted the correlations above 0.1"},{"metadata":{"_uuid":"3629e922d9607fa2012b2d5ccd6e3d8666311325","trusted":true},"cell_type":"code","source":"import itertools as itt\n\ndef get_correlation_graph(C, threashold = 0.1):\n    return [(i, j, {'weight': abs(C.iloc[i, j])})\n             for i, j in itt.combinations(range(C.shape[0]), 2)\n            if abs(C.iloc[i, j]) >= threashold]\n\nimport networkx as nx","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aab332514dc223115aec8c6ea88acea049a2d57e"},"cell_type":"markdown","source":"Then, we create the graph:"},{"metadata":{"_uuid":"0b5992bc8e08e75d22149689355ff8e646325cc0","trusted":true},"cell_type":"code","source":"G = nx.Graph()\n\nedges = get_correlation_graph(C)\nG.add_edges_from(edges)\ngraph_pos = nx.spring_layout(G)\n\nplt.figure(figsize=(15,15))\nnx.draw(G, graph_pos, alpha=.4)\nlabels = {i : '\\n'.join(C.columns[i].split(' '))\n          for i in set([i for (i, j, k) in edges] + [j for (i, j, k) in edges])}\nnx.draw_networkx_labels(G, graph_pos, labels, font_size=16);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b7bc009eb7ec4b6d24e4e0951d5d799ac42d289"},"cell_type":"markdown","source":"## How are special and seldom targets grouped?"},{"metadata":{"_uuid":"6f0bf1de2bf7c2550cd7bb320a16ce0d79b25087"},"cell_type":"markdown","source":"### Lysosomes and endosomes\n\nLet's start with these high correlated features!"},{"metadata":{"_kg_hide-input":true,"_uuid":"86c06ed4c0dbf02c605e447a5ecc40373b35771a","trusted":true},"cell_type":"code","source":"def find_counts(special_target, labels):\n    counts = labels[labels[special_target] == 1].drop(\n        [\"Id\", \"Target\", \"number_of_targets\"],axis=1\n    ).sum(axis=0)\n    counts = counts[counts > 0]\n    counts = counts.sort_values()\n    return counts","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"e2af724b721a15ebd294771629fb270c9ad9f0ae","trusted":true},"cell_type":"code","source":"lyso_endo_counts = find_counts(\"Lysosomes\", train_labels)\n\nplt.figure(figsize=(10,3))\nsns.barplot(x=lyso_endo_counts.index.values, y=lyso_endo_counts.values, palette=\"Blues\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dc6dbcea143b60d0ef360e75d14b53b09d36e7be"},"cell_type":"markdown","source":"### Rods and rings"},{"metadata":{"_kg_hide-input":true,"_uuid":"5f13e4113cfb00ec8b2de6d01fb548e10b794279","trusted":true},"cell_type":"code","source":"rod_rings_counts = find_counts(\"Rods & rings\", train_labels)\nplt.figure(figsize=(15,3))\nsns.barplot(x=rod_rings_counts.index.values, y=rod_rings_counts.values, palette=\"Greens\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fea4a1a472bf60fdc1a5317a2e616c98b781f7ab"},"cell_type":"markdown","source":"### Peroxisomes"},{"metadata":{"_kg_hide-input":true,"_uuid":"11fef619f19eafc3657d693d05f7a50c786e0621","trusted":true},"cell_type":"code","source":"peroxi_counts = find_counts(\"Peroxisomes\", train_labels)\n\nplt.figure(figsize=(15,3))\nsns.barplot(x=peroxi_counts.index.values, y=peroxi_counts.values, palette=\"Reds\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"555447f455cac28ec0c84b09b89f32b657371470"},"cell_type":"markdown","source":"### Microtubule ends"},{"metadata":{"_kg_hide-input":true,"_uuid":"a543faa3994e6bd7b57fda135ea4b5bea40efed3","trusted":true},"cell_type":"code","source":"tubeends_counts = find_counts(\"Microtubule ends\", train_labels)\n\nplt.figure(figsize=(15,3))\nsns.barplot(x=tubeends_counts.index.values, y=tubeends_counts.values, palette=\"Purples\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a04ae9a48f60f02021c0dba920b2ac3d6d65b38"},"cell_type":"markdown","source":"### Nuclear speckles"},{"metadata":{"_kg_hide-input":true,"_uuid":"85d838037fc3310b9eef31dc4e6c64e1bc35efec","trusted":true},"cell_type":"code","source":"nuclear_speckles_counts = find_counts(\"Nuclear speckles\", train_labels)\n\nplt.figure(figsize=(15,3))\nsns.barplot(x=nuclear_speckles_counts.index.values, y=nuclear_speckles_counts.values, palette=\"Oranges\")\nplt.xticks(rotation=\"70\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ba2b973b09d2436f65e8775b214d377d46b29188"},"cell_type":"markdown","source":"### Take-away\n\n* We can see that even with very seldom targets we find some kind of grouping with other targets that reveal where the protein structure seems to be located. \n* For example, we can see that rods and rings have something to do with the nucleus whereas peroxisomes may be located in the nucleus as well as in the cytosol.\n* Perhaps this patterns might help to build a more robust model!  "},{"metadata":{"_uuid":"e2e49a6a50960b6ef0d06c0d5d3dfef174d6f777"},"cell_type":"markdown","source":"Taking back to the analysis of the correlation made above, let's construct a correlation based on the scaled value counts:"},{"metadata":{"_uuid":"27ef70366439ce25fc7f536e9cb4969cd222b2bd","trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nscaler = MinMaxScaler()\n\ndef find_counts_normalized(special_target, labels):\n    counts = labels[labels[special_target] == 1].drop(\n        [\"Id\", \"Target\", \"number_of_targets\"],axis=1\n    ).sum(axis=0)\n    counts = pd.DataFrame(scaler.fit_transform(counts.astype(float).values.reshape(-1,1)).reshape(-1),\n                          index=counts.index, columns=[special_target])\n    return counts","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45f5828835c642e4b6b019b7ff86302be18e4322","trusted":true},"cell_type":"code","source":"normed = pd.concat([find_counts_normalized(col, train_labels)\n           for col in sorted(train_labels.drop([\"Id\", \"Target\", \"number_of_targets\"], axis=1).columns)],\n          axis='columns',\n         sort=True)\nheatmap(normed.corr());","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f4913966ab4319c19fd550b1228273df11565134"},"cell_type":"markdown","source":"Using the same threashold of 0.1."},{"metadata":{"_uuid":"fc5f0ea054c9b18a160df16823cf242943a97906","trusted":true},"cell_type":"code","source":"img = normed.corr()\nheatmap(img[abs(img) > 0.1]);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ff702c5037345e384c071f9c27ab86779394e84","trusted":true},"cell_type":"code","source":"C = img[abs(img) > 0.1]\nG = nx.Graph()\n\nedges = get_correlation_graph(C)\nG.add_edges_from(edges)\ngraph_pos = nx.spring_layout(G)\n\nplt.figure(figsize=(15,15))\nnx.draw(G, graph_pos, alpha=.4)\nlabels = {i : '\\n'.join(C.columns[i].split(' '))\n          for i in set([i for (i, j, k) in edges] + [j for (i, j, k) in edges])}\nnx.draw_networkx_labels(G, graph_pos, labels, font_size=16);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1afb222b5d49b39e7379f82f2b177d430f7ef34"},"cell_type":"markdown","source":"It looks like we have got a [hair ball problem!](https://image.slidesharecdn.com/bokehdatashader-odscboston2016-160523145441/95/visualizing-a-billion-points-w-bokeh-datashader-16-638.jpg?cb=1464015395)\n\nThe graph made with a 0.15 threashold is more easy to understand and get some insight:"},{"metadata":{"_uuid":"b310f878e51920d8f447dcfe89e6cf0327020131","scrolled":false,"trusted":true},"cell_type":"code","source":"C = img\n\nheatmap(C[abs(img) > 0.15])\nplt.show()\n\nG = nx.Graph()\n\nedges = get_correlation_graph(C, 0.15)\nG.add_edges_from(edges)\ngraph_pos = nx.spring_layout(G)\n\nplt.figure(figsize=(15,15))\nnx.draw(G, graph_pos, alpha=.4)\nlabels = {i : '\\n'.join(C.columns[i].split(' '))\n          for i in set([i for (i, j, k) in edges] + [j for (i, j, k) in edges])}\nnx.draw_networkx_labels(G, graph_pos, labels, font_size=16)\nplt.show();","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d8b34a307d197919553524f50e1e2b6fcad6b82b"},"cell_type":"markdown","source":"Hey! Isn't it obvious that the most common places in the cell will spot Cytosol and Nucleoplasm? What happens if we drop them?"},{"metadata":{"_uuid":"54b8e15a1640380d5e3a4c046310040a4dfc546d","trusted":true},"cell_type":"code","source":"D = C.drop(['Cytosol', 'Nucleoplasm'], axis=0).drop(['Cytosol', 'Nucleoplasm'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0569583139306906488374b5fc56eaeab0d54490","trusted":true},"cell_type":"code","source":"heatmap(D[D > .15])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e27353ea0cb3d731fba17038a48bf0557b61410b","scrolled":false,"trusted":true},"cell_type":"code","source":"G = nx.Graph()\n\nedges = get_correlation_graph(D, 0.15)\nG.add_edges_from(edges)\ngraph_pos = nx.spring_layout(G)\n\nplt.figure(figsize=(15,15))\nnx.draw(G, graph_pos, alpha=.4)\nlabels = {i : '\\n'.join(D.columns[i].split(' '))\n          for i in set([i for (i, j, k) in edges] + [j for (i, j, k) in edges])}\nnx.draw_networkx_labels(G, graph_pos, labels, font_size=16)\nplt.show();","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4b9392a7e6e8b5fab18f177e160fb6284dc96f67"},"cell_type":"markdown","source":"That it! Look, the patterns are coherent with the simple correlation based graph and with the counting plots."},{"metadata":{"_uuid":"0f839aa60819e93361137261fc0510ddbe2a95fe","scrolled":false,"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,3))\nsns.barplot(x=tubeends_counts.index.values, y=tubeends_counts.values, palette=\"Purples\");\n\nplt.figure(figsize=(15,3))\nsns.barplot(x=rod_rings_counts.index.values, y=rod_rings_counts.values, palette=\"Greens\");","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57ed5eac7345537d4c8196145ab738eb6cfc8da9"},"cell_type":"markdown","source":"There is something wrong here: 'Rods & Rings' should not be associated with 'Microtubule ends'\n"},{"metadata":{"_uuid":"6e535750edb682126a57e3a0f4b75da24adfd081","scrolled":false,"trusted":true},"cell_type":"code","source":"train_labels[train_labels.number_of_targets == 1].drop(\n    ['Id', 'Target', 'number_of_targets'],\n    axis='columns'\n).sum(axis='rows')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2cf9fc6dab23cf6a3f46c278b2ae5c2646d96624","scrolled":true,"trusted":true},"cell_type":"code","source":"train_labels[(train_labels.number_of_targets == 1) & (train_labels['Rods & rings'] == 1) ]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0714e168ea347df2111124d0f341de3172bac00d"},"cell_type":"markdown","source":"## How do the images look like?\n\n"},{"metadata":{"_uuid":"4af4b12200fd62e7726ff8a481d7566e564947ad"},"cell_type":"markdown","source":"### Peek into the directory\n\nBefore we start loading images, let's have a look into the train directory to get an impression of what we can find there:"},{"metadata":{"_kg_hide-input":true,"_uuid":"1d11fa62aaa798cfcb8671b7aac0e4cd4fa18776","trusted":true},"cell_type":"code","source":"from os import listdir\n\nfiles = listdir(\"../input/train\")\nfor n in range(10):\n    print(files[n])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f8804380faaaeb428227dec5ee280f27c57baea"},"cell_type":"markdown","source":"Ah, ok, great! It seems that for one image id, there are different color channels present. Looking into the data description of this competition we can find that:\n\n* Each image is actually splitted into 4 different image files. \n* These 4 files correspond to 4 different filter:\n    * a **green** filter for the **target protein structure** of interest\n    * **blue** landmark filter for the **nucleus**\n    * **red** landmark filter for **microtubules**\n    * **yellow** landmark filter for the **endoplasmatic reticulum**\n* Each image is of size 512 x 512"},{"metadata":{"_uuid":"e29074f7ddba22fbdd2275fdc8782dc2dd5885c6"},"cell_type":"markdown","source":"Let's check if the number of files divided by 4 yields the number of target samples:"},{"metadata":{"_kg_hide-input":true,"_uuid":"8545dd9afec3e0cdc02873375ba81ff4f4658aab","trusted":true},"cell_type":"code","source":"len(files) / 4 == train_labels.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b89a5ae1b6e358d7100c79e5a2cdc5d77ff88cc2"},"cell_type":"markdown","source":"## How do images of specific targets look like?\n\nWhile looking at examples, we can build an batch loader:"},{"metadata":{"_uuid":"ce954da724c8a400c12080f8c3f986a537d4a4c4","trusted":true},"cell_type":"code","source":"train_path = \"../input/train/\"","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"c7e4ffb20ebaee046d888620d0cf016f0cebf2c4","trusted":true},"cell_type":"code","source":"def load_image(basepath, image_id):\n    images = np.zeros(shape=(4,512,512))\n    images[0,:,:] = plt.imread(basepath + image_id + \"_green\" + \".png\")\n    images[1,:,:] = plt.imread(basepath + image_id + \"_red\" + \".png\")\n    images[2,:,:] = plt.imread(basepath + image_id + \"_blue\" + \".png\")\n    images[3,:,:] = plt.imread(basepath + image_id + \"_yellow\" + \".png\")\n    return images\n\ndef make_image_row(image, subax, title):\n    subax[0].imshow(image[0], cmap=\"Greens\")\n    subax[1].imshow(image[1], cmap=\"Reds\")\n    subax[1].set_title(\"stained microtubules\")\n    subax[2].imshow(image[2], cmap=\"Blues\")\n    subax[2].set_title(\"stained nucleus\")\n    subax[3].imshow(image[3], cmap=\"Oranges\")\n    subax[3].set_title(\"stained endoplasmatic reticulum\")\n    subax[0].set_title(title)\n    return subax\n\ndef make_title(file_id):\n    file_targets = train_labels.loc[train_labels.Id==file_id, \"Target\"].values[0]\n    title = \" - \"\n    for n in file_targets:\n        title += label_names[n] + \" - \"\n    return title","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b884c598911c0f86818bb287bb433023cca15912"},"cell_type":"code","source":"a = load_image('../input/train/', 'e403806e-bbbf-11e8-b2bb-ac1f6b6435d0')\nnp.shape(a)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e250f69a4de3ef6bf81ab5c13ab0cb1e2aa1ba7b"},"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor i in range(4):\n    plt.subplot(2,2,i+1)\n    plt.imshow(a[i], cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2fae335ced0a39dafb7f3da136e2bf607cb010d6","scrolled":false,"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,15))\nplt.imshow(a[0], cmap='gray')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"739e4c4c34b29e72975316cfea312b23e200c1b7","trusted":true},"cell_type":"code","source":"a = []\na.append(plt.imread('../input/train/039085dc-bbaa-11e8-b2ba-ac1f6b6435d0_red.png'))\na.append(plt.imread('../input/train/039085dc-bbaa-11e8-b2ba-ac1f6b6435d0_green.png'))\na.append(plt.imread('../input/train/039085dc-bbaa-11e8-b2ba-ac1f6b6435d0_blue.png'))\na.append(plt.imread('../input/train/039085dc-bbaa-11e8-b2ba-ac1f6b6435d0_yellow.png'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa9c2319254e6d0116c46ea1f5c6e2e0b78205a5","trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,15))\ncmaps = [sns.dark_palette(\"red\", as_cmap=True),\n         sns.dark_palette(\"green\", as_cmap=True),\n         sns.dark_palette(\"blue\", as_cmap=True),\n         sns.dark_palette(\"yellow\", as_cmap=True)]\nfor i in range(4):\n    plt.subplot(2,2,i+1)\n    plt.imshow(a[i], cmap=cmaps[i])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0092e857cac65f199f80ab419e015b78a5292fe1","trusted":true},"cell_type":"code","source":"def to_rgba2(img):\n    r = np.transpose(np.vectorize(lambda x: (1,0,0,x))(img[0]))\n    g = np.transpose(np.vectorize(lambda x: (0,1,0,x))(img[1]))\n    b = np.transpose(np.vectorize(lambda x: (0,0,1,x))(img[2]))\n    y = np.transpose(np.vectorize(lambda x: (1,1,0,x))(img[3]))\n    return np.array([r,g,b,y])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"823fa08cbe57a882d592496de762a576c101beee","scrolled":true,"trusted":true},"cell_type":"code","source":"%time r = to_rgba2(a)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5650b8ea4ea7b1473a7673baad09a07d461f207e","scrolled":false,"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor i in range(4):\n    plt.imshow(r[i])","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"090d1b258d2641e747e58a49376ab3d32b3893b8","trusted":true},"cell_type":"code","source":"class TargetGroupIterator:\n    \n    def __init__(self, target_names, batch_size, basepath):\n        self.target_names = target_names\n        self.target_list = [reverse_train_labels[key] for key in target_names]\n        self.batch_shape = (batch_size, 4, 512, 512)\n        self.basepath = basepath\n    \n    def find_matching_data_entries(self):\n        train_labels[\"check_col\"] = train_labels.Target.apply(\n            lambda l: self.check_subset(l)\n        )\n        self.images_identifier = train_labels[train_labels.check_col==1].Id.values\n        train_labels.drop(\"check_col\", axis=1, inplace=True)\n    \n    def check_subset(self, targets):\n        return np.where(set(self.target_list).issuperset(set(targets)), 1, 0)\n    \n    def get_loader(self):\n        filenames = []\n        idx = 0\n        images = np.zeros(self.batch_shape)\n        for image_id in self.images_identifier:\n            images[idx,:,:,:] = load_image(self.basepath, image_id)\n            filenames.append(image_id)\n            idx += 1\n            if idx == self.batch_shape[0]:\n                yield filenames, images\n                filenames = []\n                images = np.zeros(self.batch_shape)\n                idx = 0\n        if idx > 0:\n            yield filenames, images\n            ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"61ba9ebe5b6bfbea57d24e43bc8f4a1c2638d5eb"},"cell_type":"markdown","source":"Let's try to visualize specific target groups. **In this example we will see images that contain the protein structures lysosomes or endosomes**. Set target values of your choice and the target group iterator will collect all images that are subset of your choice:"},{"metadata":{"_uuid":"445f454dc36c1f4056bc870e8810f219f1c560ea","trusted":true},"cell_type":"code","source":"your_choice = [\"Lysosomes\", \"Endosomes\"]\nyour_batch_size = 3","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"6dcd0c8f57651ba21b82420edd37f70f93953215","trusted":true},"cell_type":"code","source":"imageloader = TargetGroupIterator(your_choice, your_batch_size, train_path)\nimageloader.find_matching_data_entries()\niterator = imageloader.get_loader()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_uuid":"5f7c98a6e15c081fcbe54c4e678ea2ea5ae4fbd7","trusted":true},"cell_type":"code","source":"file_ids, images = next(iterator)\n\nfig, ax = plt.subplots(len(file_ids),4,figsize=(20,5*len(file_ids)))\nif ax.shape == (4,):\n    ax = ax.reshape(1,-1)\nfor n in range(len(file_ids)):\n    make_image_row(images[n], ax[n], make_title(file_ids[n]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae6e86e63662927bf09efd9b28ea5f1442b86018"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}