{"cells":[{"metadata":{},"cell_type":"markdown","source":"If you are forking, Please Upvote.\n\nAny suggeastions will be appreciated."},{"metadata":{},"cell_type":"markdown","source":"# Import Libraries"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport cv2\nimport seaborn as sns\nimport glob\nimport json\nfrom tqdm import tqdm, tqdm_notebook\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom keras.utils import to_categorical, Sequence\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom keras.optimizers import RMSprop,Adam\nfrom keras.applications import ResNet50","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Define Path"},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '/kaggle/input/hpa-single-cell-image-classification/'\nos.listdir(path)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Number train samples:', len(train_data.index))\nprint('Number train images:', len(os.listdir(path+'train')))\nprint('Number submission samples:', len(samp_subm.index))\nprint('Number submission images:', len(os.listdir(path+'test')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samp_subm.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load Images"},{"metadata":{},"cell_type":"markdown","source":"All image samples are represented by four filters (stored as individual files), the protein of interest (green) plus three cellular landmarks: nucleus (blue), microtubules (red), endoplasmic reticulum (yellow). "},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nimagepath=[]\n\nfor file in glob.glob('../input/hpa-single-cell-image-classification/train/*.png')[:50]:\n    imagepath.append(file)\n    \nimagedata=[]\n\nfor i in range(36):\n    plt.subplot(6, 6, i+1)\n    img = cv2.imread(imagepath[i])\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.tick_params(labelbottom=False,\n                    labelleft=False,\n                    labelright=False,\n                    labeltop=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nimagepath=[]\n\nfor file in glob.glob('../input/hpa-single-cell-image-classification/test/*.png')[:50]:\n    imagepath.append(file)\n    \nimagedata=[]\n\nfor i in range(36):\n    plt.subplot(6, 6, i+1)\n    img = cv2.imread(imagepath[i])\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.tick_params(labelbottom=False,\n                    labelleft=False,\n                    labelright=False,\n                    labeltop=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"colors_dict = {'blue': 'microtubule', 'green': 'nuclei', 'red': 'Endoplasmic Reticulum', 'yellow': 'protein'}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_id = train_data.loc[0, 'ID']\nimage_file = cv2.imread(path+'train/'+image_id+'_blue.png')\nimage_file.shape\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_id = train_data.loc[0, 'ID']\nfig, axs = plt.subplots(1, 4, figsize=(20, 5))\nfig.subplots_adjust(hspace = .2, wspace=.1)\naxs = axs.ravel()\ncolors = list(colors_dict.keys())\nfor i in range(len(colors)):\n    filename = ''.join([image_id, '_', colors[i], '.png'])\n    image_file = cv2.imread(path+'train/'+filename)\n    axs[i].imshow(image_file)\n    axs[i].set_title(colors_dict[colors[i]])\n    axs[i].set_xticklabels([])\n    axs[i].set_yticklabels([])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# The labels are represented as integers that map to the following:\n\n0. Nucleoplasm\n1. Nuclear membrane\n2. Nucleoli\n3. Nucleoli fibrillar center\n4. Nuclear speckles\n5. Nuclear bodies\n6. Endoplasmic reticulum\n7. Golgi apparatus\n8. Intermediate filaments\n9. Actin filaments 10. Microtubules\n11. Mitotic spindle\n12. Centrosome\n13. Plasma membrane\n14. Mitochondria\n15. Aggresome\n16. Cytosol\n17. Vesicles and punctate cytosolic patterns\n18. Negative"},{"metadata":{},"cell_type":"markdown","source":"# Encoding Labels\n\nThis is a multilabel classification.The labels are separeted by | in the train dataset."},{"metadata":{},"cell_type":"markdown","source":"We can do it by two types I am showing all two:"},{"metadata":{},"cell_type":"markdown","source":"# Type 1"},{"metadata":{"trusted":true},"cell_type":"code","source":"path_to_train = 'kaggle/input/hpa-single-cell-image-classification/train/'\ntrain_dataset_info = []\nfor name, labels in zip(train_data['ID'], train_data['Label'].str.split('|')):\n    train_dataset_info.append({\n        'path':os.path.join(path_to_train, name),\n        'labels':np.array([int(label) for label in labels])})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.DataFrame(columns=[\"path\", \"Label\"])\nn = 0\n\nfor i, D in enumerate(train_dataset_info[:7000]):\n    for x in D[\"labels\"]:\n        train_df.loc[n, \"path\"] = D[\"path\"]\n        train_df.loc[n, \"Label\"] = x\n        n = n + 1 \n\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA"},{"metadata":{"trusted":true},"cell_type":"code","source":"def ID_figure(LABEL_ID):\n    pathlist_ID = []\n\n    train_df_ID = train_df[train_df.Label == LABEL_ID][\"path\"].reset_index()\n\n    for i, D in enumerate(train_df_ID[\"path\"]):\n        pathlist_ID.append(D)\n        \n    for i in range(12):\n        plt.subplot(5, 3, i+1)\n        \n        red_image = cv2.imread(pathlist_ID[i]+\"_red.png\", cv2.IMREAD_UNCHANGED)\n        green_image = cv2.imread(pathlist_ID[i]+\"_green.png\", cv2.IMREAD_UNCHANGED)\n        blue_image = cv2.imread(pathlist_ID[i]+\"_blue.png\", cv2.IMREAD_UNCHANGED)\n        yellow_image = cv2.imread(pathlist_ID[i]+\"_yellow.png\", cv2.IMREAD_UNCHANGED)\n        \n        img_bgr = cv2.merge((blue_image, green_image, red_image ,yellow_image))  \n        \n        plt.title(pathlist_ID[i])\n        plt.imshow(img_bgr)\n        plt.tick_params(labelbottom=False,\n                        labelleft=False,\n                        labelright=False,\n                        labeltop=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pathlist_ID = []\nLABEL_ID = 0\n\ntrain_df_ID = train_df[train_df.Label == LABEL_ID][\"path\"].reset_index()\n\nfor i, D in enumerate(train_df_ID[\"path\"]):\n    pathlist_ID.append(D)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# label == 0.Nucleoplasm"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(0)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1.Nuclear membrane"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2.Nucleoli"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 3.Nucleoli fibrillar center"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(3)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4.Nuclear speckles"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 5.Nuclear bodies"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 6.Endoplasmic reticulum"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(6)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 7.Golgi apparatus"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(7)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 8.Intermediate filaments"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(8)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 9.Actin filaments"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(9)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 10.Microtubules"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 11.Mitotic spindle"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(11)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 12.Centrosome"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(12)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 13.Plasma membrane"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(13)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 14.Mitochondria"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(14)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 15.Aggresome"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(15)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 16.Cytosol"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(16)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 17.Vesicles and punctate cytosolic patterns"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(17)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 18.Negative"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\nID_figure(18)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Type 2"},{"metadata":{"trusted":true},"cell_type":"code","source":"# original label\nprint('input :', train_data.loc[0, 'Label'])\n# label as list\nprint('step 1:', train_data.loc[0, 'Label'].split('|'))\n# label as list on integers\nprint('step 2:', list(map(int, train_data.loc[0, 'Label'].split('|'))))\n# label to binary class matrix\nlabel = to_categorical(list(map(int, train_data.loc[0, 'Label'].split('|'))), num_classes=19)\nprint('step 3:', label)\n# sum the labels\nlabel = label.sum(axis=0)\nprint('step 4:', label)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"img_size = 64\nimg_channel = 3\nnum_classes = 19\nbatch_size = 256","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(Sequence):\n    def __init__(self, path, list_IDs, labels, batch_size, img_size, img_channel):\n        self.path = path\n        self.list_IDs = list_IDs\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_channel = img_channel\n        self.indexes = np.arange(len(self.list_IDs))\n        \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        return X, y\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, self.img_size, self.img_size, self.img_channel))\n        y = np.zeros((self.batch_size, num_classes), dtype=int)\n        for i, ID in enumerate(list_IDs_temp):\n            data_file = cv2.imread(self.path+ID+'_blue.png')\n            img = cv2.resize(data_file, (self.img_size, self.img_size))\n            X[i, ] = img/255.\n            # Prepare label\n            label = self.labels[i]\n            label = label.split('|')\n            label = list(map(int, label))\n            label = to_categorical(label, num_classes=num_classes)\n            label = label.sum(axis=0)\n            y[i, ] = label\n        return X, y","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modelling"},{"metadata":{"trusted":true},"cell_type":"code","source":"weights='/kaggle/input/models/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"metrics = [tf.keras.metrics.AUC(name='auc', multi_label=True)]\nlearning_rate = 1e-3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"conv_base = ResNet50(include_top=False,\n                     weights=weights,\n                     input_shape=(img_size, img_size, img_channel))\nconv_base.trainable = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\nmodel.add(conv_base)\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.1))\nmodel.add(Dense(num_classes, activation='sigmoid'))\n\nmodel.compile(optimizer=Adam(lr=learning_rate), loss=\"binary_crossentropy\", metrics=metrics)\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs = 5\n\ntrain_IDs, val_IDs, y_train, y_val = train_test_split(train_data['ID'], train_data['Label'], test_size=0.33, random_state=2021)\ntrain_IDs.index=range(len(train_IDs))\ny_train.index=range(len(train_IDs))\nval_IDs.index=range(len(val_IDs))\ny_val.index=range(len(val_IDs))\n\ntrain_generator = DataGenerator(path+'train/', train_IDs, y_train, batch_size, img_size, img_channel)\nval_generator = DataGenerator(path+'train/', val_IDs, y_val, batch_size, img_size, img_channel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit_generator(generator=train_generator,\n                              validation_data=val_generator,\n                              epochs = epochs)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Evaluation"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, axs = plt.subplots(1, 2, figsize=(20, 6))\nfig.subplots_adjust(hspace = .2, wspace=.2)\naxs = axs.ravel()\nloss = history.history['loss']\nloss_val = history.history['val_loss']\nepochs = range(1, len(loss)+1)\naxs[0].plot(epochs, loss, 'bo', label='loss_train')\naxs[0].plot(epochs, loss_val, 'ro', label='loss_val')\naxs[0].set_title('Value of the loss function')\naxs[0].set_xlabel('epochs')\naxs[0].set_ylabel('value of the loss function')\naxs[0].legend()\naxs[0].grid()\nacc = history.history['auc']\nacc_val = history.history['val_auc']\naxs[1].plot(epochs, acc, 'bo', label='accuracy_train')\naxs[1].plot(epochs, acc_val, 'ro', label='accuracy_val')\naxs[1].set_title('Accuracy')\naxs[1].set_xlabel('Epochs')\naxs[1].set_ylabel('Value of accuracy')\naxs[1].legend()\naxs[1].grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Submission = samp_subm.copy()\nSubmission.to_csv('submission.csv', index=False)\nSubmission.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Still Working"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}