{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"Credit: [@anokas](https://www.kaggle.com/anokas)"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\npal = sns.color_palette()\n\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\n\nprint('# File sizes')\nfor f in os.listdir('../input'):\n    if not os.path.isdir('../input/' + f):\n        print(f.ljust(30) + str(round(os.path.getsize('../input/' + f) / (1000000*1000), 7)) + 'GB')\n    else:\n        sizes = [os.path.getsize('../input/'+f+'/'+x)/(1000000*1000) for x in os.listdir('../input/' + f)]\n        print(f.ljust(30) + str(round(sum(sizes), 7)) + 'GB' + ' ({} files)'.format(len(sizes)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a85edf0b4866428c62e334d8af3f771494f6c7a"},"cell_type":"markdown","source":"# Training Data"},{"metadata":{"trusted":true,"_uuid":"8ffd86b5cbca4e485bf54fdf3c83c1a6e9bc79d3"},"cell_type":"code","source":"df_train = pd.read_csv('../input/train.csv')\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3cb48c28edb77cf297f062e5b777892528af2d48"},"cell_type":"markdown","source":"## A Useful Dictionary...."},{"metadata":{"trusted":true,"_uuid":"a24ef2609981b325813fb66faac28a7752f3319a"},"cell_type":"code","source":"dicts={\n0:  \"Nucleoplasm\", \n1:  \"Nuclear membrane\",   \n2:  \"Nucleoli\",   \n3:  \"Nucleoli fibrillar center\" ,  \n4:  \"Nuclear speckles\"   ,\n5:  \"Nuclear bodies\"   ,\n6:  \"Endoplasmic reticulum\",   \n7:  \"Golgi apparatus\"   ,\n8:  \"Peroxisomes\"   ,\n9:  \"Endosomes\"   ,\n10:  \"Lysosomes\"   ,\n11:  \"Intermediate filaments\",   \n12:  \"Actin filaments\"   ,\n13:  \"Focal adhesion sites\",   \n14:  \"Microtubules\"   ,\n15:  \"Microtubule ends\",   \n16:  \"Cytokinetic bridge\",   \n17:  \"Mitotic spindle\"   ,\n18:  \"Microtubule organizing center\" ,  \n19:  \"Centrosome\"   ,\n20:  \"Lipid droplets\",   \n21:  \"Plasma membrane\",   \n22:  \"Cell junctions\"  , \n23:  \"Mitochondria\"   ,\n24:  \"Aggresome\"   ,\n25:  \"Cytosol\",\n26:  \"Cytoplasmic bodies\",   \n27:  \"Rods & rings\" \n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bef269ec62b9493d3e679c4032158dc94bd717fe"},"cell_type":"code","source":"labels = df_train['Target'].apply(lambda x: x.split(' '))\nfrom collections import Counter, defaultdict\ncounts = defaultdict(int)\nfor l in labels:\n    for l2 in l:\n        counts[l2] += 1\nstrs=[]\nfor count in counts.keys(): strs.append(dicts[int(count)])\n\ndata=[go.Bar(x=list(strs), y=list(counts.values()))]\nlayout=dict(height=800, width=800, title='Distribution of training labels')\nfig=dict(data=data, layout=layout)\npy.iplot(data, filename='train-label-dist')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8c5e549625c7b9ccbbdc0002736e816b6cde1e2"},"cell_type":"code","source":"# Co-occurence Matrix\ncom = np.zeros([len(counts)]*2)\nfor i, l in enumerate(list(counts.keys())):\n    for i2, l2 in enumerate(list(counts.keys())):\n        c = 0\n        cy = 0\n        for row in labels.values:\n            if l in row:\n                c += 1\n                if l2 in row: cy += 1\n        com[i, i2] = cy / c\n\ndata=[go.Heatmap(z=com, x=list(strs), y=list(strs))]\nlayout=go.Layout(height=800, width=800, title='Co-occurence matrix of training labels')\nfig=dict(data=data, layout=layout)\npy.iplot(data, filename='train-com')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"42cf0255d2f5d5a39873743f6c62c53fb68a3ca7"},"cell_type":"markdown","source":"# Images"},{"metadata":{"trusted":true,"_uuid":"f29b47af27918c95fdbf15c76d9c4b8583d9e44d"},"cell_type":"code","source":"import cv2\nfrom PIL import Image\n\n\nnew_style = {'grid': False}\nplt.rc('axes', **new_style)\n_, ax = plt.subplots(2, 2, sharex='col', sharey='row', figsize=(20, 20))\ni = 0\nfor f, l in df_train[:4].values:\n    img = cv2.imread('../input/train/{}_green.png'.format(f)) + cv2.imread('../input/train/{}_red.png'.format(f)) + cv2.imread('../input/train/{}_blue.png'.format(f))\n    im = Image.fromarray(img)\n    \n    ax[i // 2, i % 2].imshow(im)\n    arr = l.split(\" \")\n    str = \"\"\n    for ind in arr:\n        str=str+dicts[int(ind)]+\", \"\n    ax[i // 2, i % 2].set_title('{} - {}'.format(f, str))\n    #ax[i // 4, i % 4].show()\n    i += 1\n    \nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}