{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"%load_ext autoreload\n%autoreload 2\n\n%matplotlib inline\n\nfrom fastai.imports import *\nfrom PIL import Image\nfrom imageio import imread\n\nfrom pandas_summary import DataFrameSummary\nfrom sklearn.ensemble import RandomForestRegressor, RandomForestClassifier\nfrom IPython.display import display\nimport cv2\nimport matplotlib.pyplot as plt\n\nfrom sklearn import metrics\n\nimport seaborn as sns\nsns.set()\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"PATH = '../input'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0f81125c96747bcbee610fb29c34a0e8e613167"},"cell_type":"code","source":"!ls {PATH}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6f82697c6fe62ba1a7381c3f39c19c404fe7302"},"cell_type":"code","source":"# upload training data\ndf_raw = pd.read_csv(f'{PATH}/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3f1c42c50295f3bba57363988cf93827698f21ed"},"cell_type":"markdown","source":"### Load images"},{"metadata":{"trusted":true,"_uuid":"1872370430deba33a77b5f1f277e47c0953accf3"},"cell_type":"code","source":"# load images into an array\ndef loadem(ids):\n    \"\"\"Load all channels of the given ids into a DataFrame\n    \"\"\"\n    images = np.empty((len(ids), 4, 512, 512))\n    for index, id in enumerate(ids):\n        images[index][0] = imread(f'{PATH}/train/'+id+'_red.png')    # red\n        images[index][1] = imread(f'{PATH}/train/'+id+'_green.png')  # green\n        images[index][2] = imread(f'{PATH}/train/'+id+'_blue.png')   # blue\n        images[index][3] = imread(f'{PATH}/train/'+id+'_yellow.png') # yellow\n    return images\n\n# load all channels of an ID to display it\ndef load_image(id):\n    \"\"\"Load all channels of the given \n    \"\"\"\n    red    = imread(f'{PATH}/train/'+id+'_red.png')\n    green  = imread(f'{PATH}/train/'+id+'_green.png')\n    blue   = imread(f'{PATH}/train/'+id+'_blue.png')\n    yellow = imread(f'{PATH}/train/'+id+'_yellow.png')\n    data = np.reshape(yellow, newshape=(512, 512, 1))\n    #rgb = Image.merge(\"RGB\",(red, green, blue ))\n    #array = np.array(rgb) + data/2\n    # add the yellow to the image\n    return [red, green, blue, yellow]\n\n# plot the different channels\ndef plotem(red, green, blue, yellow):\n    \"\"\"Plot the different channels in a row\n    \"\"\"\n    images = [red, green, blue, yellow]\n    titles = ['Reds', 'Greens', 'Blues', 'Oranges']\n    plt.figure(figsize=(15, 10))\n    for i in range(len(images)):\n        # plot \n        plt.subplot(1, 5, i+1)\n        plt.imshow(images[i], cmap=titles[i])\n        plt.title(titles[i])\n    \n    plt.tight_layout()\n    plt.show()\n\n#rgb, red, green, blue, yellow = combine('0032a07e-bba9-11e8-b2ba-ac1f6b6435d0')\n#plot(rgb, red, green, blue, yellow)\n#ids = df_raw.Id.head(6)\n#rgbs = ids.map(lambda id: combine(id))\n#display(rgbs[1])\n#for id in ids:\n#    rgb = combine(id)\n#    display(rgb)\n    #plt.figure()\n    #plt.imshow(rgb)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"159036397fd85969c72d82c417132d88592dcb62"},"cell_type":"code","source":"# for every label plot one image with all it's channels\nlabels = [\"Nucleoplasm\", \"Nuclear membrane\", \"Nucleoli\", \"Nucleoli fibrillar center\", \"Nuclear speckles\", \"Nuclear bodies\", \"Endoplasmic reticulum\", \"Golgi apparatus\", \"Peroxisomes\", \"Endosomes\", \"Lysosomes\", \"Intermediate filaments\", \"Actin filaments\", \"Focal adhesion sites\", \"Microtubules\", \"Microtubule ends\", \"Cytokinetic bridge\", \"Mitotic spindle\", \"Microtubule organizing center\", \"Centrosome\", \"Lipid droplets\", \"Plasma membrane\", \"Cell junctions\", \"Mitochondria\", \"Aggresome\", \"Cytosol\", \"Cytoplasmic bodies\", \"Rods & rings\"]\n\nfor index in range(len(labels)):\n    print(labels[index])\n    target = str(index)\n    df_raw[df_raw['Target']==target]['Id']\n    ids = df_raw[df_raw['Target']==target]['Id'].head(1)\n    for imgid in ids:\n        red, green, blue, yellow = load_image(imgid)\n        plotem(red, green, blue, yellow)\n        ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1352767d6609eda5e4e0f1dc6b2835106a364e27"},"cell_type":"markdown","source":"### Proteins count distribution"},{"metadata":{"trusted":true,"_uuid":"6fc39c0a523d190c493d0c0a4fa58f5c772e1ebb"},"cell_type":"code","source":"# split the string into list of indexes \ndf_raw['Target'] = df_raw['Target'].str.split()\n\ncounts = pd.DataFrame(df_raw['Target'].map(lambda x: len(x)), dtype=np.int8)\ncounts['count'] = 1\nax = counts.groupby(['Target']).agg(['count']).plot(marker='.', title='Distribution of number of proteins')\nax.set_xlabel(\"number of proteins\")\nax.set_ylabel(\"count\")\nax.set_label(None)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ced636d646f3fdbd2a05731dfa822b6ea389917a"},"cell_type":"markdown","source":"Most images has only one type or protein, may be two. The max number of proteins per image is 5."},{"metadata":{"_uuid":"bc2c74e59fa08fcdb93c7b80e6187321cabbd4c9"},"cell_type":"markdown","source":"### Featurise targets"},{"metadata":{"trusted":true,"_uuid":"a698c8b99ccd7638de4955091637b50f3b9275eb"},"cell_type":"code","source":"# generate one hot encoded columns for each protein\nfor index in range(len(labels)):\n    df_raw[labels[index]] = df_raw.apply(lambda _: int(str(index) in _.Target), axis=1)\n# drop the original Target column\ndf_raw = df_raw.drop('Target', axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d2b59207d91e587aec969eac68ea0c8266ecf2a0"},"cell_type":"markdown","source":"### Bar plot of the proteins"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"d37b2e7633d08cdda4027b47e29d07e6eafd6bc6"},"cell_type":"code","source":"# plot the distribution of proteins\nproteins = df_raw.sum(axis=0).drop('Id', axis=0)\nproteins = proteins.sort_values(ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ee85e6f054fd4dfab7fa38c9ad16f59cde7aefb"},"cell_type":"code","source":"# Initialize the matplotlib figure\nf, ax = plt.subplots(figsize=(15, 10))\n# Plot the total count of appearance for each protein\nsns.barplot(proteins.values, proteins.keys())\n    \n# Add a legend and informative axis label\nax.legend(ncol=2, loc=\"lower right\", frameon=True)\nax.set(xlim=(0, 13000), ylabel=\"Proteins\", xlabel=\"count\")\nsns.despine(left=True, bottom=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"687b99fef61b669a6f65253868ef729dbf0b62ed"},"cell_type":"markdown","source":"Bad news not that much of samples for some rare proteins, predicting'em will be a challenge!"},{"metadata":{"_uuid":"acb979ddc4c0b0d5c1468d47269508d976ed8bea"},"cell_type":"markdown","source":"### Correlation between proteins"},{"metadata":{"trusted":true,"_uuid":"b4c6e9116bd0b8aa964f5a9e1e0c30bc7d948328"},"cell_type":"code","source":"summary = DataFrameSummary(df_raw)\nsummary.corr","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49348378a2509ab20ab2b2d08cb8d841be420dc7"},"cell_type":"code","source":"# Compute the correlation matrix\ndf_noid = df_raw.drop(columns=['Id'])\ncorr = df_noid.corr()\n\n# Generate a mask for the upper triangle\nmask = np.zeros_like(corr, dtype=np.bool)\nmask[np.triu_indices_from(mask)] = True\n\n# Set up the matplotlib figure\nf, ax = plt.subplots(figsize=(11, 9))\n\n# Generate a custom diverging colormap\ncmap = sns.diverging_palette(160, 10, as_cmap=True)\n\n# Draw the heatmap with the mask and correct aspect ratio\nsns.heatmap(corr, mask=mask, cmap=cmap, vmax=.3, center=0, square=True, linewidths=.5, cbar_kws={\"shrink\": .5})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a783a1b288ec641fb472dbb249e64ae37f73cbb0"},"cell_type":"markdown","source":"Few red boxes, i.e. most of the proteins are not that much correlated with each other. There is two strong correlation:\n* **Lysosmes** and **Endosomes**\n* **Mitotic spindle** and **Cytokinetic bridge**"},{"metadata":{"trusted":true,"_uuid":"86b70e6085133e0ef21c5f274b7505f40d98b8c0"},"cell_type":"code","source":"# start small\nsize, size_validation = 1000, 20\ndf_keep = df_raw[: size]\ndf_train = df_keep[: size-size_validation]\ndf_valid = df_keep[size-size_validation: ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76d22dbf5c82fe9a349b953e5e22152789e1a121","scrolled":false},"cell_type":"code","source":"X_train = loadem(df_train['Id'])\nY_train = df_train[labels]\nX_train.shape, Y_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"0788e618dac8041d31f09005a30d7f6187540c3c"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Dense, Activation, Conv2D, MaxPooling2D, Flatten\n\nmodel = Sequential()\nmodel.add(Conv2D(64, (1, 1), padding='same', activation='relu'))\nmodel.add(Conv2D(16, (3, 3), padding='same', activation='relu'))\nmodel.add(MaxPooling2D((3, 3), strides=(1, 1), padding='same'))\n\nmodel.add(Flatten())\nmodel.add(Activation('relu'))\nmodel.add(Dense(28))\nmodel.add(Activation('softmax'))\n\nmodel.build(input_shape=((None, 4, 512, 512)))\n\nmodel.summary()\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy', 'categorical_accuracy'])\nhistory = model.fit(X_train, Y_train, batch_size=20, epochs=1, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e82a4ac0db87c9e4b1a4bba288071519d4517619"},"cell_type":"code","source":"def showImagesHorizontally(list_of_files):\n    fig = plt.figure()\n    number_of_files = len(list_of_files)\n    for i in range(number_of_files):\n        a=fig.add_subplot(1,number_of_files,i+1)\n        image = imread(list_of_files[i])\n        imshow(image,cmap='Greys_r')\n        axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f45e5ff4eeabf57414445212410710a71dcebf4"},"cell_type":"code","source":"!ls {PATH}/train/ | head -n 6","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0cf7a94fd0dc4843b4ee29a55590b213372e16f"},"cell_type":"code","source":"imgs = loadem_to_df(['00070df0-bbc3-11e8-b2bc-ac1f6b6435d0'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b76b92921920735ab405518805550e5437e12e46"},"cell_type":"code","source":"imgs.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a74ce90430fbd2e66866b1b2f31bb20f5730faf9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}