{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Human Protein Atlas - Single Cell Classification\n### Finding individual human cell differences in microscope images\n![image](https://storage.googleapis.com/kaggle-competitions/kaggle/23823/logos/header.png?t=2020-11-24-14-18-10)\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ROOT = \"../input/hpa-single-cell-image-classification/\"\nos.listdir(ROOT)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train =  pd.read_csv(ROOT+\"train.csv\")\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub =  pd.read_csv(ROOT+\"sample_submission.csv\")\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(ROOT+\"train/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"assert len(os.listdir(ROOT+\"train/\")) == (4 * train.shape[0])\nassert set([i.split(\"_\")[0] for i in os.listdir(ROOT+\"train/\")]) == set(train.ID.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(ROOT+\"test/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir(ROOT+\"train/\")[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"red = cv2.imread(ROOT+\"train/5e3a2e6a-bb9c-11e8-b2b9-ac1f6b6435d0_red.png\", cv2.IMREAD_UNCHANGED)\nyellow = cv2.imread(ROOT+\"train/5e3a2e6a-bb9c-11e8-b2b9-ac1f6b6435d0_yellow.png\", cv2.IMREAD_UNCHANGED)\nblue = cv2.imread(ROOT+\"train/5e3a2e6a-bb9c-11e8-b2b9-ac1f6b6435d0_blue.png\", cv2.IMREAD_UNCHANGED)\ngreen = cv2.imread(ROOT+\"train/5e3a2e6a-bb9c-11e8-b2b9-ac1f6b6435d0_green.png\", cv2.IMREAD_UNCHANGED)\nred.shape, yellow.shape, blue.shape, green.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**We have 4 images per training example. The important one is green one.\nOthers can be used if needed**"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(green, cmap='gray');\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(yellow, cmap='gray');\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(blue, cmap='gray');\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(red, cmap='gray');\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = cv2.merge((red, green, blue))  \nplt.imshow(img);\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.Label = train.Label.apply(lambda x: x.split(\"|\"))\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# There are total 18 labels\n\nlabels = {\n0: \"Nucleoplasm\",\n1: \"Nuclear membrane\",\n2: \"Nucleoli\",\n3: \"Nucleoli fibrillar center\",\n4: \"Nuclear speckles\",\n5: \"Nuclear bodies\",\n6: \"Endoplasmic reticulum\",\n7: \"Golgi apparatus\",\n8: \"Intermediate filaments\",\n9: \"Actin filaments\",\n10: \"Microtubules\",\n11: \"Mitotic spindle\",\n12: \"Centrosome\",\n13: \"Plasma membrane\",\n14: \"Mitochondria\",\n15: \"Aggresome\",\n16: \"Cytosol\",\n17: \"Vesicles and punctate cytosolic patterns\",\n18: \"Negative\",\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import itertools\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1,1, figsize=(14, 7))\nsns.countplot([labels[int(i)] for i in itertools.chain.from_iterable(train.Label)], axes=ax);\nplt.title(\"Distribution of labels in training data\")\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Co-occurrence Matrix"},{"metadata":{"trusted":true},"cell_type":"code","source":"u = pd.get_dummies(pd.DataFrame(train.Label.tolist()), prefix='', prefix_sep='').groupby(level=0, axis=1).sum()\nv = u.T.dot(u)\nv.values[(np.r_[:len(v)], ) * 2] = 0\nv = v.reindex([str(i) for i in range(1, 19)], axis=1)\nv = v.reindex([str(i) for i in range(1, 19)], axis=0)\nv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"co_mat = v\nfig, ax = plt.subplots(1, 1, figsize=(10, 8))\nsns.heatmap((v / np.sum(v, axis=0)).T, cbar=True, annot=False)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"counts = np.bincount(green.reshape(-1))\ncounts = counts / np.sum(counts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Given that we dont have pixel level labels for each class let's see how   \ncan we guess presence and absence of a class by just using raw image pixel values**"},{"metadata":{"trusted":true},"cell_type":"code","source":"import tqdm.auto as tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plotdist(k, color):\n    sample = train.sample(k)\n    count_list = []\n    for i in tqdm.tqdm(sample.ID, leave=False):\n        path = ROOT+\"train/\"+i+\"_\"+color+\".png\"\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        counts = np.bincount(img.reshape(-1), minlength=256)\n        counts = counts / np.sum(counts)\n        count_list.append(counts)\n    sample['counts'] = count_list\n\n    fig, ax =  plt.subplots(1,1, figsize=(12, 7))\n    for class_id in range(19):\n        cts = sample[sample.Label.apply(lambda x: str(class_id) in x)].counts.tolist()\n        stats = (np.array(cts).sum(axis=0) / len(cts))\n        if (not np.isnan(stats).any()):\n            # Not ploting 0 as it is outlier\n            plt.plot([i for i in range(1, 256)], stats[1:], label=\"class \"+str(class_id));\n    plt.legend()\n    plt.xlabel(\"Pixel value\")\n    plt.title(color+\" Pixel Value Distribution for images containing different classes\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotdist(500, 'green')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotdist(500, 'red')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotdist(500, 'blue')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plotdist(500, 'yellow')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plotimg(axes, ID):\n    red = cv2.imread(ROOT+\"train/\"+ID+\"_red.png\", cv2.IMREAD_UNCHANGED)\n    green = cv2.imread(ROOT+\"train/\"+ID+\"_green.png\", cv2.IMREAD_UNCHANGED)\n    blue = cv2.imread(ROOT+\"train/\"+ID+\"_blue.png\", cv2.IMREAD_UNCHANGED)\n    img = cv2.merge((red, green, blue))\n    axes.imshow(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for k in range(0, 19):\n    IDS = train[train.Label.apply(lambda x: str(k) in x)].sample(4).ID.tolist()\n    fig,axes = plt.subplots(1, 4, figsize=(16, 4))\n    for ID, ax in zip(IDS, axes):\n        plotimg(ax, ID)\n    fig.suptitle(labels[k] + \" samples\")\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"the notebook is still WIP but\n### do upvote if it helped :)"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}