{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom collections import Counter\nimport os\nimport seaborn as sns\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b7d69ff5af8bedf4ad327bb44a3cc0c183a38804"},"cell_type":"markdown","source":"**Load dataset info\n**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntrain.head()\n# train_df.Target.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1b7a18fcb0def1c030c1f38d4d793c4fb46f26f0"},"cell_type":"markdown","source":"There are 28 different target proteins\ncreating label dict"},{"metadata":{"trusted":true,"_uuid":"cd8bda1caf6cf7ff658682e00dc6ae5d218507a6"},"cell_type":"code","source":"label_names={\n0:  \"Nucleoplasm\", \n1:  \"Nuclear membrane\",   \n2:  \"Nucleoli\",   \n3:  \"Nucleoli fibrillar center\" ,  \n4:  \"Nuclear speckles\"   ,\n5:  \"Nuclear bodies\"   ,\n6:  \"Endoplasmic reticulum\",   \n7:  \"Golgi apparatus\"   ,\n8:  \"Peroxisomes\"   ,\n9:  \"Endosomes\"   ,\n10:  \"Lysosomes\"   ,\n11:  \"Intermediate filaments\",   \n12:  \"Actin filaments\"   ,\n13:  \"Focal adhesion sites\",   \n14:  \"Microtubules\"   ,\n15:  \"Microtubule ends\",   \n16:  \"Cytokinetic bridge\",   \n17:  \"Mitotic spindle\"   ,\n18:  \"Microtubule organizing center\" ,  \n19:  \"Centrosome\"   ,\n20:  \"Lipid droplets\",   \n21:  \"Plasma membrane\",   \n22:  \"Cell junctions\"  , \n23:  \"Mitochondria\"   ,\n24:  \"Aggresome\"   ,\n25:  \"Cytosol\",\n26:  \"Cytoplasmic bodies\",   \n27:  \"Rods & rings\" \n}\n\nreverse_train_labels = dict((v,k) for k,v in label_names.items())\n\ndef fill_targets(row):\n    row.Target = np.array(row.Target.split(\" \")).astype(np.int)\n    for num in row.Target:\n        name = label_names[int(num)]\n        row.loc[name] = 1\n    return row","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9d4c4ba7e1caac1ea1b216a330109c7d1780bcca"},"cell_type":"markdown","source":"Count of proteins occur in each images "},{"metadata":{"trusted":true,"_uuid":"d80a6f1e790321a1f280c346d1fea593dcc608e3"},"cell_type":"code","source":"for key in label_names.keys():\n    train[label_names[key]] = 0\n\ntrain = train.apply(fill_targets, axis=1)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db9731e65a8648c80f5700870808fd43c2c7f909"},"cell_type":"code","source":"print(\"Total number of samples in the training data:\", train.shape[0])\nprint(\"Total number of unique IDs in the training data: \",len(train.Id.unique()))\ntrain[\"number_of_targets\"] = train.drop([\"Id\", \"Target\"],axis=1).sum(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"739154eda29d5ee876d5d2013052ac9cfa29d37c"},"cell_type":"markdown","source":"Count of Images for number of labes present in images\n\nAll counts:\n\n**        label count     and       Images count**\n        \n        1                 15126\n        \n        2                 12485\n        \n        3                  3160\n        \n        4                  299\n        \n        5                  2\n        "},{"metadata":{"trusted":true,"_uuid":"9ed9e40b16cda9b7d6b28c8f673e7f8bfd249aa5"},"cell_type":"code","source":"count_perc = np.round(100 * train[\"number_of_targets\"].value_counts() / train.shape[0], 2)\nplt.figure(figsize=(20,5))\nsns.barplot(x=count_perc.index.values, y=count_perc.values, palette=\"Oranges\")\nplt.xlabel(\"Number of targets per image\")\nplt.ylabel(\"% of data\")\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"87ecbd36cbabcf44479e594e1e07c49fb0c62264"},"cell_type":"markdown","source":"**Distribution of training labels**"},{"metadata":{"trusted":true,"_uuid":"09224cfc55e01e7b78b95cbee26dd23f72f1a0b3"},"cell_type":"code","source":"import gc\nimport matplotlib.pyplot as plt\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\ntarget_array = list(train.Target)\ntarget_array = [label_names[int(n)] for a in target_array for n in a]\nfig, ax = plt.subplots(figsize=(15, 5))\npd.Series(target_array).value_counts().plot('bar', fontsize=14 )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a95a6daa1c0f942ac6f52fe2547cba68bb412180"},"cell_type":"markdown","source":"1. **Single vs Multi label distribution of train data**"},{"metadata":{"trusted":true,"_uuid":"dbbd13c1e2b427adc3c7dca81828fb9d9172adea"},"cell_type":"code","source":"# train[\"nb_labels\"] = train[\"Target\"].apply(lambda x: len(x.split(\" \")))\nsingle_labels_count = train[train['number_of_targets']==1]['number_of_targets'].count()\nmulti_labels_count = train[train['number_of_targets']>1]['number_of_targets'].count()\n\nimport gc\nimport matplotlib.pyplot as plt\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\ndata=[go.Bar(x=['Single label', 'Multi-label'], y=[single_labels_count, multi_labels_count],marker=dict(color='rgb(58,200,225)'))]\nlayout=dict(height=10, width=10, title='Single vs Multi label distribution')\nfig=dict(data=data, layout=layout)\npy.iplot(data, filename='Label type vs Count')\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"41b7b220067cff9066068e2999e86daa6681b689"},"cell_type":"markdown","source":"**correlations between training labes**"},{"metadata":{"trusted":true,"_uuid":"5bf74bfee34c895023d47efbd85d99d53df2be9d"},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(15,15))\nsns.heatmap(train[train.number_of_targets>1].drop(\n    [\"Id\", \"Target\", \"number_of_targets\"],axis=1\n).corr(), cmap=\"YlGnBu\", vmin=-1, vmax=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ba5c70ceb9d4152ef3565b70738d6054204920f"},"cell_type":"code","source":"# def heatMap(self, df):\ndf = train[train.number_of_targets>1].drop( [\"Id\", \"Target\", \"number_of_targets\"],axis=1).corr()\n\nmirror = False\n# Create Correlation df\ncorr = df.corr()\n# Plot figsize\nfig, ax = plt.subplots(figsize=(20, 15))\n# Generate Color Map\ncolormap = sns.diverging_palette(220, 10, as_cmap=True)\n\nif mirror == True:\n   #Generate Heat Map, allow annotations and place floats in map\n   sns.heatmap(train[train.number_of_targets>1].drop( [\"Id\", \"Target\", \"number_of_targets\"],axis=1).corr(), cmap=colormap, annot=True, fmt=\".2f\")\n   #Apply xticks\n   plt.xticks(range(len(corr.columns)), corr.columns);\n   #Apply yticks\n   plt.yticks(range(len(corr.columns)), corr.columns)\n   #show plot\n\nelse:\n   # Drop self-correlations\n   dropSelf = np.zeros_like(corr)\n   dropSelf[np.triu_indices_from(dropSelf)] = True# Generate Color Map\n   colormap = sns.diverging_palette(220, 10, as_cmap=True)\n   # Generate Heat Map, allow annotations and place floats in map\n   sns.heatmap(train[train.number_of_targets>1].drop( [\"Id\", \"Target\", \"number_of_targets\"],axis=1).corr(), cmap='YlGnBu', annot=True, fmt=\".2f\", mask=dropSelf)\n   # Apply xticks\n   plt.xticks(range(len(corr.columns)), corr.columns);\n   # Apply yticks\n   plt.yticks(range(len(corr.columns)), corr.columns)\n# show plot\nplt.show()\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"816dbef41507e221e6b68d823675ffc242518bf1"},"cell_type":"markdown","source":"**Matrix for coreation values of labels**"},{"metadata":{"trusted":true,"_uuid":"fed03dbb3a7f3153351b4f32977f86c20a1c75e8","scrolled":true},"cell_type":"code","source":"# coreation values of protein\ncorr_matrix = train[train.number_of_targets>1].drop( [\"Id\", \"Target\", \"number_of_targets\"],axis=1).corr()\ncorr_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3d05f2e32777a54a03aebee9d7c4bc828b879366"},"cell_type":"markdown","source":"**High corelation between labels**"},{"metadata":{"trusted":true,"_uuid":"1ed14b8aa6e887541c18041e06d5195b24fd51bc"},"cell_type":"code","source":"# High corelation between proteins\n\nhigh_corr_var_=np.where(corr_matrix>0.02)\nhigh_corr_var=[(corr_matrix.columns[x],corr_matrix.columns[y], corr_matrix[corr_matrix.columns[x]][corr_matrix.columns[y]]) for x,y in zip(*high_corr_var_) if x!=y and x<y]\n\nhigh_corr_var","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f8ffd1dbf29a2b02848715f3498cc1fc15bb337a"},"cell_type":"markdown","source":"**Low corelation between labels**"},{"metadata":{"trusted":true,"_uuid":"29b270984a5c6a8c1c69fe6ddfa268eb357c676f"},"cell_type":"code","source":"\nlow_corr_var_=np.where(corr_matrix<=-.1)\nlow_corr_var=[(corr_matrix.columns[x],corr_matrix.columns[y], corr_matrix[corr_matrix.columns[x]][corr_matrix.columns[y]]) for x,y in zip(*low_corr_var_) if x!=y and x<y]\n\nlow_corr_var","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cda1a40cbcf2a66ce08bb11822e2b2113249fd0"},"cell_type":"code","source":" sns.pairplot(corr_matrix, diag_kind=\"kde\", palette=\"husl\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7713da385fb0e5f77f8b091981d217d9ce5363da"},"cell_type":"code","source":"import cv2\nfrom PIL import Image\nimport imageio\nfrom scipy.misc import imread\ntrain_path = \"../input/train/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acf322f178e21553f6d713cfec9669798df62ce4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b320f9efa5dbb8fc0a730942b4eed49e10945d5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}