{"cells":[{"metadata":{"_uuid":"1bc213b23f24303b3b78177de6f64bcbd813ce7d"},"cell_type":"markdown","source":"# Loading Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport seaborn as sns\nimport os\nfrom itertools import chain\nfrom collections import Counter\nsns.set_style(\"white\")\nsns.set_context(\"notebook\", font_scale=1.5, rc={\"lines.linewidth\": 2.5})\n\nfrom keras.preprocessing.image import load_img\nfrom matplotlib import pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5acfa320892a15d4454291d9555c3489c5410fcf"},"cell_type":"code","source":"os.listdir('../input/')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2b8b6a41eee1eac57e60d95feeca6a6c22eaf77a"},"cell_type":"markdown","source":"# The input directory contains the following files:\n1. train - This folder contains the training images of size 512x512 in PNG format.\n1. test -  This folder contains the test images of size 512x512 in PNG format.\n1. train.csv - This CSV file contains all the filenames and labels for the training set.\n1. sample_submission.csv - This file contains a sample submission file with two columns Id and Predicted. The predicted column contains all the predicted test classes.."},{"metadata":{"_uuid":"d2c110b67d4e4854eaf3b77b57748f24fd4401e9"},"cell_type":"markdown","source":"Let's load the train.csv file.\nIt contains two columns:\n1.     Id - the base filename of the sample. All ids consist of four files - blue, green, red, and yellow.\n1.     Target - in the training data, this represents the labels assigned to each sample.\n"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndef count_target(target_val):\n    return len(target_val.split(' '))\n\ndf['nclass'] = df.Target.map(count_target)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f7ce7aedf826a5a4bb3444be8c1d8664239022ee"},"cell_type":"markdown","source":"# Function for plotting the images"},{"metadata":{"trusted":true,"_uuid":"3f9e33539228ce58d14ebc6ed1930c22a28c8e81"},"cell_type":"code","source":"def plot_img(img_id):\n    filt_no = 0\n    plt.figure(figsize=(30,60))\n    plt.tight_layout()\n    for filt in img_filt:\n        img_path = train_img_path + img_id + filt\n        img = np.array(load_img(img_path))\n        plt.subplot(1,4,filt_no+1)\n        plt.imshow(img)\n        if i==0: plt.title(filt[1:-4]+' filter')\n        plt.axis('off')\n        filt_no += 1        ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39eb61003ca239ca018fa2050acedc024d429298"},"cell_type":"markdown","source":"# Label Count Distribribution"},{"metadata":{"trusted":true,"_uuid":"83eefd2ad8bad4fa33a27b6efd2062f424df63cf"},"cell_type":"code","source":"label_count = []\nfor i in range(df.nclass.min(),df.nclass.max()+1):\n    label_count.append(np.sum(df.nclass==i))\n    print('No. of images with',i,'label:',label_count[-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"726dda9e35cbb9476620cb0b4b0624435db2cf53"},"cell_type":"code","source":"x = np.arange(len(label_count))+1\nplt.bar(x,label_count)\nplt.title('Label Count Distribution in Train Set')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ab0b83a2c793ab2fba0860ad3fc1214ed1dc7643"},"cell_type":"markdown","source":"# It seems that most of the images contain only 1 class ! There are very images with 4 or more classes."},{"metadata":{"_uuid":"8723b4cc17855d0c878e946753bfe89c4f663271"},"cell_type":"markdown","source":"# Lets see the contents of train folder.\nAs expected it contains files in the format of** file_id + filter_name + .png**"},{"metadata":{"trusted":true,"_uuid":"05d5ccbf5ef47da8305ac68ebeb97b424b686854"},"cell_type":"code","source":"os.listdir('../input/train/')[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ebeb168805cb0f4406cfe54491f5ffb527f52f2e"},"cell_type":"markdown","source":"# Visualizing files with 5 targets"},{"metadata":{"_uuid":"187ec26ed396237e0c75002646320e9565df9459"},"cell_type":"markdown","source":"There are only 2 such files !"},{"metadata":{"trusted":true,"_uuid":"2a10e7f6bfcdd507814251b2e1730a3f73ba52e8"},"cell_type":"code","source":"img_filt = ['_green.png','_blue.png','_red.png','_yellow.png']\ntrain_img_path = '../input/train/'\nids = df[df.nclass==5].reset_index()\ni = 0; nimg = 2\nwhile True:\n    if i == nimg: break\n    img_id =  ids.Id[i]   \n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5251e751aef618776edc6dc73571afc65f9f953f"},"cell_type":"markdown","source":"# Visualizing files with 4 targets"},{"metadata":{"trusted":true,"_uuid":"9819e9a321ee68d9cd2ade6b1a6e4511aa1d8256"},"cell_type":"code","source":"ids = df[df.nclass==4].reset_index()\ni = 0; nimg = 5\nwhile True:\n    if i == nimg: break\n    img_id =  ids.Id[i]   \n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5097bce584d08f5a4498195743c348141a1d4280"},"cell_type":"markdown","source":"# Visualizing files with 3 targets"},{"metadata":{"trusted":true,"_uuid":"3ddb6ea307119f40bc40b200ba83134c80c56ff5"},"cell_type":"code","source":"ids = df[df.nclass==3].reset_index()\ni = 0; nimg = 5\nwhile True:\n    if i == nimg: break\n    img_id =  ids.Id[i]   \n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"277c6a9515a52c10aad178f0f1989cb442203fb5"},"cell_type":"markdown","source":"# Visualizing files with 2 targets"},{"metadata":{"trusted":true,"_uuid":"9e676c2a8717fb917d1b92ff3362ff211a27f649"},"cell_type":"code","source":"ids = df[df.nclass==2].reset_index()\ni = 0; nimg = 5\nwhile True:\n    if i == nimg: break\n    img_id =  ids.Id[i]   \n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f5526604eed95d339af87fa3baf095783c58804f"},"cell_type":"markdown","source":"# Visualizing files with 1 target"},{"metadata":{"trusted":true,"_uuid":"e3a544c10051273f5d1afb2c207727a85032eaec"},"cell_type":"code","source":"ids = df[df.nclass==1].reset_index()\ni = 0; nimg = 5\nwhile True:\n    if i == nimg: break\n    img_id =  ids.Id[i]   \n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"974c600513f58e1cf81a67cf8dbf0cff77d1b3bf"},"cell_type":"markdown","source":"# Mapping the labels\nThere are in total 28 different labels present in the dataset. \n\nThe labels are represented as integers that map to the following"},{"metadata":{"trusted":true,"_uuid":"7010a5817dc66c82081c5bd4358f8d81c9d062d6"},"cell_type":"code","source":"def mk_list(val):\n    return [int(label) for label in val.split(' ')]\ndf['target_list'] = df['Target'].map(mk_list)\nall_labels = list(chain.from_iterable(df['target_list'].values))\nlabel_count = Counter(all_labels)\narr = np.zeros((28,))\nfor key,value in label_count.items():\n    arr[key] = value","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f05ac5917048d0dd7147e6290f8304e697e8dbb0"},"cell_type":"code","source":"map_class_labels = {0:  'Nucleoplasm',\n1:  'Nuclear membrane',\n2:  'Nucleoli',\n3:  'Nucleoli fibrillar center',\n4:  'Nuclear speckles',\n5:  'Nuclear bodies',\n6: 'Endoplasmic reticulum',\n7:  'Golgi apparatus',\n8:  'Peroxisomes',\n9:  'Endosomes',\n10:  'Lysosomes',\n11:  'Intermediate filaments',\n12:  'Actin filaments',\n13:  'Focal adhesion sites',\n14: 'Microtubules',\n15: 'Microtubule ends',\n16: 'Cytokinetic bridge',\n17: 'Mitotic spindle',\n18: 'Microtubule organizing center',\n19: 'Centrosome',\n20: 'Lipid droplets',\n21: 'Plasma membrane',\n22: 'Cell junctions',\n23: 'Mitochondria',\n24: 'Aggresome',\n25: 'Cytosol',\n26: 'Cytoplasmic bodies',\n27: 'Rods and rings'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"317af55bb2c7046dc1fdcac44216aa2b89234f21"},"cell_type":"code","source":"plt.figure(figsize=(30,5))\nax = sns.barplot(x=np.arange(28),y=arr)\nax.set_xticklabels(list(map_class_labels.values()), fontsize=15, rotation=40, ha=\"right\")\nax.set(xlabel='Classes', ylabel='Class Counts')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5e3bec5dfba9e34b1de94c2f3bb4fb957335c81b"},"cell_type":"markdown","source":"From the figure it is clear that Nucleoplasm is the most common class followed by Cytosol.\n\nNucleoplasm  - the substance of a cell nucleus, especially that not forming part of a nucleolus.\n\nCytosol - the aqueous component of the cytoplasm of a cell, within which various organelles and particles are suspended."},{"metadata":{"_uuid":"0afaa6b0d15134c21d6fd8bcd2423a82015dc632"},"cell_type":"markdown","source":"# Visualizing the Nucleoplasm class images"},{"metadata":{"trusted":true,"_uuid":"2a37939411661a7616fd59c8596aa121d5f2f202"},"cell_type":"code","source":"ids = []\nnucleoplasm_class = [0]\nfor i,val in enumerate(df.target_list.values):\n    if nucleoplasm_class==val:\n        ids.append(i)\ni = 0; nimg = 3\nwhile True:\n    if i == nimg: break\n    img_id =  df.Id[ids[i]]\n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08a3e76a60e941f08c983f3def086bd626121421"},"cell_type":"markdown","source":"# Visualizing the Cytosol class images"},{"metadata":{"trusted":true,"_uuid":"07baa491ff7d278b6e4dc38d0e0e8f2ce8ee807d"},"cell_type":"code","source":"ids = []\ncytosol_class = [25]\nfor i,val in enumerate(df.target_list.values):\n    if cytosol_class==val:\n        ids.append(i)\ni = 0; nimg = 3\nwhile True:\n    if i == nimg: break\n    img_id =  df.Id[ids[i]]\n    plot_img(img_id)\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4592a198760375c13b06d2f5bd7ced42ca5b9982"},"cell_type":"markdown","source":"I will keep updating this notebook as i explore further.\n\nPS - Mitochondria is the power house of the cell :-)"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}