{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport os\nprint(os.listdir(\"../input\"))\nfrom matplotlib import pyplot as plt\nimport cv2\nfrom PIL import Image\nimport random\nfrom imgaug import augmenters as iaa\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential, load_model,Model\nfrom keras.layers import Activation,Dropout,Flatten,Dense,Input,BatchNormalization,Conv2D\nfrom keras.applications.inception_resnet_v2 import preprocess_input\nfrom keras.applications import InceptionResNetV2\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.callbacks import LambdaCallback\nfrom keras.callbacks import Callback\nfrom keras import metrics\nfrom keras.optimizers import Adam \nfrom keras import backend as K\nimport tensorflow as tf\nimport keras\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6ff0076a50a4e4659c56ade7df9d8f393b58abf2"},"cell_type":"markdown","source":" Some coding in this kernel is inspired(read blatantly stolen) from [this really great work](https://www.kaggle.com/allunia/protein-atlas-exploration-and-baseline) Also, suggestions are a welcome so far they are not about my grammer and spelling(which I admit is borderline horrible and there's no spell check in this environment)\n\n\n# LET's EDA\n\nLet's see the total available images in the dataset"},{"metadata":{"trusted":true,"_uuid":"ac5cedba5491fbe1b43241d61ec1bdd7a61c62fb","_kg_hide-input":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/train.csv\")\nprint(\"Total number of unique ids:\",df.Id.count())\nprint(\"Total number of images:\", df.Id.count()*4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40e9cf83ac3d6d3dfd48ee036fa5dfba97cae8af","_kg_hide-input":true,"_kg_hide-output":false},"cell_type":"code","source":"df.head(2)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"labels = {\n    0:  \"Nucleoplasm\",  \n    1:  \"Nuclear membrane\",   \n    2:  \"Nucleoli\",   \n    3:  \"Nucleoli fibrillar center\",   \n    4:  \"Nuclear speckles\",\n    5:  \"Nuclear bodies\",   \n    6:  \"Endoplasmic reticulum\",   \n    7:  \"Golgi apparatus\",   \n    8:  \"Peroxisomes\",   \n    9:  \"Endosomes\",   \n    10:  \"Lysosomes\",   \n    11:  \"Intermediate filaments\",   \n    12:  \"Actin filaments\",   \n    13:  \"Focal adhesion sites\",   \n    14:  \"Microtubules\",   \n    15:  \"Microtubule ends\",   \n    16:  \"Cytokinetic bridge\",   \n    17:  \"Mitotic spindle\",   \n    18:  \"Microtubule organizing center\",   \n    19:  \"Centrosome\",   \n    20:  \"Lipid droplets\",   \n    21:  \"Plasma membrane\",   \n    22:  \"Cell junctions\",   \n    23:  \"Mitochondria\",   \n    24:  \"Aggresome\",   \n    25:  \"Cytosol\",   \n    26:  \"Cytoplasmic bodies\",   \n    27:  \"Rods & rings\"\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bc4abac8368508fbca7ff613c04a0a89f6cb6c5","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"for key in labels.keys():\n    df[labels[key]] = 0\n    \ndef filltargets(row):\n    tar = row.Target.split(\" \")\n    for i in tar:\n        col = labels[int(i)]\n        row[str(col)]=1\n    return row\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df5da9ba57f54f974d0f0d6b8e59c6643629ba5f","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"df = df.apply(filltargets, axis=1)\ndf = df.drop([\"Target\"],axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1a055e75bbdf0c7a24503957214d804260f4759","_kg_hide-input":true,"_kg_hide-output":false},"cell_type":"code","source":"df.head(2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ab3b3c696c8ca1aaa74f4f98fc51ccac3c5acb05"},"cell_type":"markdown","source":"## Frequency of labels in the data"},{"metadata":{"trusted":true,"_uuid":"51371bb8d77d37b190d893440e168392bd542b81","_kg_hide-input":true},"cell_type":"code","source":"freq_df = df.drop([\"Id\"],axis=1).sum(axis=0).sort_values(ascending = True)\nplt.figure(figsize=(15,15))\nsns.barplot(y=freq_df.index.values, x=freq_df.values*100/len(df), order=freq_df.index,palette=\"Blues\")\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d9825642da8b56d5e564f0a72da5e0a898203049"},"cell_type":"markdown","source":"Insights: \n\n* Nucleoplasm is the most common occurance (>40%) followed by Cytosol (>27%)\n* Plasma Membrane, Nucleoli, Mitochondria(Power-house of the cell) and other relatively bigger objects are present in a quantity that may not surprise any one who knows the basic cell structure.\n* The most interesting things in the distribution  of less frequent objects such as Rods and Rings, Microtubules Ends, Lysosomes etc as compared to those of Nucleoplasm. Hence a clear classs imbalance that can lead to harmful bias in model. This also gives us the idea about the kind of metric that we might endup using. From my very limited knowledge I think it should be F1.\n"},{"metadata":{"_uuid":"07750a4fa499ca5639103645e654c5c495989abd"},"cell_type":"markdown","source":"## The grouping of labels\n\nLet's see the grouping of the least frequent variables first."},{"metadata":{"trusted":true,"_uuid":"2c592c2df1f8d241ca0cc4ebfb5d5e37e7fd12e9","_kg_hide-input":true},"cell_type":"code","source":"def correlated_distribution(label):\n    df_ = df[df[label]==1]\n    freq_df = df_.drop([\"Id\",label],axis=1).sum(axis=0).sort_values(ascending = True)\n    plt.figure(figsize=(5,5))\n    plt.xlabel(\"Distribution for \"+label)\n    sns.barplot(y=freq_df.index.values, x=freq_df.values*100/len(df_), order=freq_df.index,palette=\"Blues\")\n\ncorrelated_distribution(\"Endosomes\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f45a88b86c5a0082c7a4d7d02867ea85eedf68e5","_kg_hide-input":true},"cell_type":"code","source":"correlated_distribution(\"Rods & rings\")","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"b4b992d5be794b7e5673f0b609aa032323c0f89d"},"cell_type":"code","source":"correlated_distribution(\"Peroxisomes\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ce60dccbc3c68f2b4a1825dd881a710c3299e6a"},"cell_type":"markdown","source":"Let's see the grouping of most frequent labels"},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"de75e4efa9d871c660f489bc68f7eff43f9cd4ed"},"cell_type":"code","source":"correlated_distribution(\"Cytosol\")","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"7891b7beccf41e888cd19bb0f61b6ab40d4b25f4"},"cell_type":"code","source":"correlated_distribution(\"Mitochondria\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f207d9623775fa4fa7832620359a3c1d473daa7a"},"cell_type":"markdown","source":"Hmm... Interesting!\n\n\nThe key insight here is that the more frequent labels occure in more spread-out manner, i.e, they seems to be partnering up with way more variables while the least frequent variables are loyal to a small number of labels.\n\nWork in progress..."},{"metadata":{"trusted":true,"_uuid":"9a68e593c916455e8ff654c939d266e7aa6ef9d5"},"cell_type":"markdown","source":"# Multilabel Analysis"},{"metadata":{"trusted":true,"_uuid":"fdd708bf3fefbd819afa379059a5d78c6bc173ed","_kg_hide-input":true},"cell_type":"code","source":"freq_df = pd.DataFrame()\nfreq_df[\"number_of_targets\"] = df.drop([\"Id\"],axis=1).sum(axis=1).sort_values(ascending = True)\ncount_perc = np.round(100 * freq_df[\"number_of_targets\"].value_counts() / freq_df.shape[0], 2)\nplt.figure(figsize=(20,5))\nsns.barplot(x=count_perc.index.values, y=count_perc.values, palette=\"Blues\")\nplt.xlabel(\"Number of targets per image\")\nplt.ylabel(\"% of data\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ad14961f62dda9e54a193399f85fc7575604b89"},"cell_type":"markdown","source":"Okay so majority is either single or double labeled. "},{"metadata":{"trusted":true,"_uuid":"7e199d40b15fb1744bb2735fae74018cfd6be725","_kg_hide-input":true,"scrolled":false},"cell_type":"code","source":"shape = (299,299)\n\ndef load_image(id,path=\"../input/train/\"):\n    global shape\n    R = np.array(Image.open(path+id+'_red.png'))\n    G = np.array(Image.open(path+id+'_green.png'))\n    B = np.array(Image.open(path+id+'_blue.png'))\n    Y = np.array(Image.open(path+id+'_yellow.png'))\n\n    image = np.stack((\n        R/2+Y/2,\n        G/2+Y/2, \n        B),-1)\n\n    image = cv2.resize(image, (shape[0], shape[1]))\n    image = np.divide(image, 255)\n    return image  \n\nplt.figure(figsize=(15,15))\nfor n in range(4):\n    idx = np.random.randint(0,20000,1)\n    im_name = df.Id[idx[0]]\n    im = load_image(im_name)\n    plt.subplot(1,4,n+1)\n    plt.imshow(im)\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c100980a6cf331feb702d6c9b217cd7c254671aa"},"cell_type":"markdown","source":"Let's train a classifier for baseline. I'm choosing InceptionResnet50 but another interesting candidate is NasNet."},{"metadata":{"trusted":true,"_uuid":"f8a46f740b05574d55f5d48fd636d500ae14b7e6"},"cell_type":"code","source":"path_to_train = '/kaggle/input/train/'\ndata = pd.read_csv('/kaggle/input/train.csv')\n\ntrain_dataset_info = []\nfor name, labels in zip(data['Id'], data['Target'].str.split(' ')):\n    train_dataset_info.append({\n        'path':os.path.join(path_to_train, name),\n        'labels':np.array([int(label) for label in labels])})\ntrain_dataset_info = np.array(train_dataset_info)\n\nfrom sklearn.model_selection import train_test_split\ntrain_ids, test_ids, train_targets, test_target = train_test_split(\n    data['Id'], data['Target'], test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"997e0cfd24fde6ab5ba9ebb7ef881fbc9a0a07d6"},"cell_type":"code","source":"class data_generator:\n    \n    def create_train(dataset_info, batch_size, shape, augument=True):\n        assert shape[2] == 3\n        while True:\n            random_indexes = np.random.choice(len(dataset_info), batch_size)\n            batch_images = np.empty((batch_size, shape[0], shape[1], shape[2]))\n            batch_labels = np.zeros((batch_size, 28))\n            for i, idx in enumerate(random_indexes):\n                image = data_generator.load_image(\n                    dataset_info[idx]['path'], shape)   \n                if augument:\n                    image = data_generator.augment(image)\n                batch_images[i] = image\n                batch_labels[i][dataset_info[idx]['labels']] = 1\n            yield batch_images, batch_labels\n    \n    def load_image(path, shape):\n        R = np.array(Image.open(path+'_red.png'))\n        G = np.array(Image.open(path+'_green.png'))\n        B = np.array(Image.open(path+'_blue.png'))\n        Y = np.array(Image.open(path+'_yellow.png'))\n\n        image = np.stack((\n            R/2+Y/2,\n            G/2+Y/2, \n            B),-1)\n\n        image = cv2.resize(image, (shape[0], shape[1]))\n        image = np.divide(image, 255)\n        return image        \n    \n    def augment(image):\n        augment_img = iaa.Sequential([\n            iaa.OneOf([\n                iaa.Affine(rotate=0),\n                iaa.Affine(rotate=90),\n                iaa.Affine(rotate=180),\n                iaa.Affine(rotate=270),\n                iaa.Fliplr(0.5),\n                iaa.Flipud(0.5),\n            ])], random_order=True)\n        \n        image_aug = augment_img.augment_image(image)\n        return image_aug","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a4add9fa7daf95f5880d82fafdb43c053d96977"},"cell_type":"code","source":"train_datagen = data_generator.create_train(\n    train_dataset_info, 5, (299,299,3), augument=True)\n\nimages, labels = next(train_datagen)\n\nfig, ax = plt.subplots(1,5,figsize=(25,5))\nfor i in range(5):\n    ax[i].imshow(images[i])\nprint('min: {0}, max: {1}'.format(images.min(), images.max()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1598f679d67f3668f44b5493a0e0ac45b0ffbe80"},"cell_type":"code","source":"def create_model(input_shape, n_out):\n    model = Sequential()\n    model.add(InceptionResNetV2(include_top=False,input_shape= input_shape, pooling='avg', weights=\"imagenet\"))\n    model.add(Dense(512))\n    model.add(Activation('relu'))\n    model.add(Dropout(0.7))\n    model.add(Dense(n_out, activation='softmax'))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bfda06a1c00c0387e5fa16b628239e81571683e"},"cell_type":"code","source":"def f1(y_true, y_pred):\n    tp = K.sum(K.cast(y_true*y_pred, 'float'), axis=0)\n    fp = K.sum(K.cast((1-y_true)*y_pred, 'float'), axis=0)\n    fn = K.sum(K.cast(y_true*(1-y_pred), 'float'), axis=0)\n    p = tp / (tp + fp + K.epsilon())\n    r = tp / (tp + fn + K.epsilon())\n    f1 = 2*p*r / (p+r+K.epsilon())\n    f1 = tf.where(tf.is_nan(f1), tf.zeros_like(f1), f1)\n    return K.mean(f1)\ndef show_history(history):\n    fig, ax = plt.subplots(1, 3, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('f1')\n    ax[1].plot(history.epoch, history.history[\"f1\"], label=\"Train f1\")\n    ax[1].plot(history.epoch, history.history[\"val_f1\"], label=\"Validation f1\")\n    ax[2].set_title('acc')\n    ax[2].plot(history.epoch, history.history[\"acc\"], label=\"Train acc\")\n    ax[2].plot(history.epoch, history.history[\"val_acc\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()\n    ax[2].legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d71acd73dc9f5c4605f943485b7a3d8887832736","scrolled":true},"cell_type":"code","source":"from keras.optimizers import SGD\nmodel = create_model(\n    input_shape=(299,299,3), \n    n_out=28)\n\ncheckpointer = ModelCheckpoint(\n    '/kaggle/working/InceptionResNetV2.model',\n    verbose=2, save_best_only=True)\n\nBATCH_SIZE = 10\nINPUT_SHAPE = (299,299,3)\n\ntrain_generator = data_generator.create_train(\n    train_dataset_info[train_ids.index], BATCH_SIZE, INPUT_SHAPE, augument=False)\nvalidation_generator = data_generator.create_train(\n    train_dataset_info[test_ids.index], 256, INPUT_SHAPE, augument=False)\n\nmodel.layers[0].trainable = True\n\nmodel.compile(\n    loss='binary_crossentropy',  \n    optimizer=SGD(lr = 0.0001, momentum=0.9, nesterov=True),\n    metrics=['acc', f1])\n\nhistory = model.fit_generator(\n    train_generator,\n    steps_per_epoch=500,\n    validation_data=next(validation_generator),\n    epochs=15, \n    verbose=1,\n    callbacks=[checkpointer])\nshow_history(history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cb9ffd60a64a96596ca132728653f00efd0c64c"},"cell_type":"code","source":"from tqdm import tqdm\nsubmit = pd.read_csv('../input/sample_submission.csv')\npredicted = []\nfor name in tqdm(submit['Id']):\n    path = os.path.join('../input/test/', name)\n    image = data_generator.load_image(path, INPUT_SHAPE)\n    score_predict = model.predict(image[np.newaxis])[0]\n    label_predict = np.arange(28)[score_predict>=0.2]\n    str_predict_label = ' '.join(str(l) for l in label_predict)\n    predicted.append(str_predict_label)\n    \nsubmit['Predicted'] = predicted\nsubmit.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}