{"cells":[{"metadata":{"_uuid":"a4028351a763ba54659a0152931e7d90cfd3b16d"},"cell_type":"markdown","source":"## Import necessary libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# tensorflow\nimport tensorflow as tf\n\n# Keras modules\nimport keras\nfrom keras.callbacks import ModelCheckpoint\nfrom keras import backend\n# from tensorflow.keras.applications import DenseNet --> Discarded due to too much memory usage\nfrom keras.applications import InceptionV3\nfrom keras.applications import InceptionResNetV2\nfrom keras.applications import MobileNet\nfrom keras.applications import ResNet50\nfrom keras.applications import VGG16\nfrom keras.applications import VGG19\nfrom keras.applications import Xception\nfrom keras.layers import BatchNormalization\nfrom keras.layers import Conv2D\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\nfrom keras.layers import Flatten\nfrom keras.layers import Input\nfrom keras.layers import MaxPooling2D\nfrom keras.models import Model\nfrom keras.models import Sequential\nfrom keras.optimizers import Adam\nfrom keras.utils import Sequence\n\n# data processing modules\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# plotting\nimport matplotlib.pyplot as plt\n\n# image processing\nfrom PIL import Image\nfrom imgaug import augmenters as iaa\n\n# python support libraries\nimport os\nimport datetime\nfrom zipfile import ZipFile\nfrom collections import Iterable\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"853c491b86fc2996eae89d3b8ac17036afc93652"},"cell_type":"markdown","source":"## Functions definitions "},{"metadata":{"trusted":true,"_uuid":"1d73541014c1e8cfcdcd02b420bb99494142ae7a"},"cell_type":"code","source":"def get_classes(x):\n    \"\"\"\n    Transform the Target column from the train dataset into an array of classes found in the image\n    to be later used as a target variable\n    :param x: row value for column Target from apply function\n    :return: array of length 27 with 0 and 1\n    \"\"\"\n    return np.array([0 if str(i) not in str(x).split(' ') else 1 for i in range(28)])\n\n\ndef load_image(file_path, image_name, shape=(512, 512, 3)):\n    \"\"\"\n    Load image contained within a zip file\n    :param file_path:  (string) Path to image on disk\n    :param image_name: (string) Image name contained in zip file\n    :param shape:      (tuple) Shape of image to be outputed\n    :return:           (array) 3D of RGBY divided by 255\n    \"\"\"\n    # load images by channel\n    channel_list = list()\n    for c in ['red', 'green', 'blue', 'yellow']:\n        img = Image.open(file_path + '/' + image_name + '_' + c + '.png')\n        img = img.resize((shape[0], shape[1]), Image.ANTIALIAS)\n        img = np.array(img)\n        channel_list.append(img)\n    \n    # stack pixels of image\n    if shape[2] == 3:\n        image = np.stack((\n            channel_list[0]/2 + channel_list[3]/2, \n            channel_list[1]/2 + channel_list[3]/2, \n            channel_list[2]\n        ),-1)\n    else:\n        image = np.stack(channel_list, -1)\n\n    # normalize pixels range\n    image = np.divide(image, 255)\n\n    # return array with normalized colors\n    return image\n\n\ndef f1(y_true, y_pred):\n    \"\"\"\n    Calculate the f1 score given the true values and predictions\n    :param y_true: (array) true value array\n    :param y_pred: (array) predictions array\n    :return: (float) f1 score\n    \"\"\"\n    tp = backend.sum(backend.cast(y_true * y_pred, 'float'), axis=0)\n    fp = backend.sum(backend.cast((1 - y_true) * y_pred, 'float'), axis=0)\n    fn = backend.sum(backend.cast(y_true * (1 - y_pred), 'float'), axis=0)\n\n    p = tp / (tp + fp + backend.epsilon())\n    r = tp / (tp + fn + backend.epsilon())\n\n    f1 = 2 * p * r / (p + r + backend.epsilon())\n    f1 = tf.where(tf.is_nan(f1), tf.zeros_like(f1), f1)\n    return backend.mean(f1)\n\n\ndef augment(image):\n    \"\"\"\n    Apply transformations to images\n    source: https://www.kaggle.com/rejpalcz/cnn-128x128x4-keras-from-scratch-lb-0-328\n    :param image:\n    :return:\n    \"\"\"\n    augment_img = iaa.Sequential([\n        iaa.OneOf([\n            # flip the image\n            iaa.Fliplr(0.5),\n            iaa.Flipud(0.5),\n\n            # random crops\n            iaa.Crop(percent=(0, 0.1)),\n\n            # Strengthen or weaken the contrast in each image\n            iaa.ContrastNormalization((0.75, 1.5)),\n\n            # Add gaussian noise\n            iaa.AdditiveGaussianNoise(loc=0, scale=(0.0, 0.05 * 255), per_channel=0.5),\n\n            # Make some images brighter and some darker\n            iaa.Multiply((0.8, 1.2), per_channel=0.2),\n\n            # Apply affine transformations to each image.\n            # Scale/zoom them, translate/move them, rotate them and shear them.\n            iaa.Affine(\n                scale={\"x\": (0.8, 1.2), \"y\": (0.8, 1.2)},\n                translate_percent={\"x\": (-0.2, 0.2), \"y\": (-0.2, 0.2)},\n                rotate=(-180, 180),\n                shear=(-8, 8)\n            )\n        ])], random_order=True)\n\n    image_aug = augment_img.augment_image(image)\n    return image_aug\n\n\ndef show_history(history):\n    fig, ax = plt.subplots(1, 3, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('f1')\n    ax[1].plot(history.epoch, history.history[\"f1\"], label=\"Train f1\")\n    ax[1].plot(history.epoch, history.history[\"val_f1\"], label=\"Validation f1\")\n    ax[2].set_title('acc')\n    ax[2].plot(history.epoch, history.history[\"acc\"], label=\"Train acc\")\n    ax[2].plot(history.epoch, history.history[\"val_acc\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()\n    ax[2].legend()\n\n\ndef sequencial_model(input_shape, n_out):\n    # Initialising the CNN\n    model = Sequential()\n\n    # ##### LAYER 1\n    # model.add(BatchNormalization())\n    model.add(Conv2D(32, (3, 3), input_shape=input_shape, data_format='channels_last', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.2))\n\n    # ##### LAYER 2\n    # model.add(BatchNormalization())\n    model.add(Conv2D(32, (3, 3), activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.2))\n\n    # ##### FLATTENING\n    model.add(Flatten())\n\n    # ##### ANN\n    model.add(Dense(activation='relu', units=1024))\n    model.add(Dense(activation='relu', units=128))\n    model.add(Dense(activation='softmax', units=n_out))\n\n    return model\n\n\ndef vgg_model(input_shape, n_out):\n    model = VGG19(\n        include_top=False, weights='imagenet', input_shape=input_shape\n    )\n\n    input_tensor = Input(shape=input_shape)\n    bn = BatchNormalization()(input_tensor)\n    x = model(bn)\n    x = Conv2D(128, kernel_size=(1, 1), activation='relu')(x)\n    x = Flatten()(x)\n    x = Dropout(0.5)(x)\n    x = Dense(512, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    output = Dense(n_out, activation='sigmoid')(x)\n    return Model(input_tensor, output)\n\n\ndef inception_res_net_model(input_shape, n_out):    \n    pretrain_model = InceptionResNetV2(\n        include_top=False, \n        weights='imagenet', \n        input_shape=input_shape\n    )    \n    \n    input_tensor = Input(shape=input_shape)\n    bn = BatchNormalization()(input_tensor)\n    x = pretrain_model(bn)\n    x = Conv2D(128, kernel_size=(1,1), activation='relu')(x)\n    x = Flatten()(x)\n    x = Dropout(0.5)(x)\n    x = Dense(512, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    output = Dense(n_out, activation='sigmoid')(x)\n    model = Model(input_tensor, output)\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f157fa82eab60de88212d934a21ced3f5bf2de5"},"cell_type":"markdown","source":"## Classes definition"},{"metadata":{"trusted":true,"_uuid":"d52134d793d911c6ffbb956bfb2e6a2cda3369b2"},"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    \"\"\"\n    Extend the Sequence class from the keras.utils module to create a class\n    capable of loading the images from the zipfile in batches, resizing it and\n    selecting specific channels\n\n    source: https://stanford.edu/~shervine/blog/keras-how-to-generate-data-on-the-fly\n    \"\"\"\n    def __init__(self, train_data, batch_size, file_path, shape, augment_flag=True):\n        assert isinstance(train_data, pd.DataFrame)\n        assert train_data.shape[0] > 0\n        assert batch_size > 0\n        assert (os.path.isfile(file_path) and '.zip' in file_path) or os.path.isdir(file_path)\n        assert isinstance(shape, Iterable)\n        assert len(shape) == 3\n        assert type(augment_flag) == bool\n\n        # saved arguments\n        self.train_data = train_data\n        self.batch_size = batch_size\n        self.file_path = file_path\n        self.shape = shape\n        self.augment_flag = augment_flag\n\n        # get the number of images in the train dataset\n        self.size = train_data.shape[0]\n\n        # get list of images\n        self.image_ids = train_data['Id'].values\n\n        # set a list of indexes to be extracted from the image\n        self.indexes = np.arange(len(self.image_ids))\n\n        # calculate the required number of batches\n        self.batches = int(np.floor(self.size / batch_size))\n\n    def __len__(self):\n        return self.batches\n\n    def __getitem__(self, index):\n        \"\"\"\n        Generate one batch of data\n        :param index: (int) index of the total batches size to be loaded\n        :return:\n        \"\"\"\n        # Generate indexes of the batch\n        indexes = self.indexes[index * self.batch_size: (index + 1) * self.batch_size]\n\n        # select images ids\n        batch_ids = [self.image_ids[k] for k in indexes]\n\n        # create array to hold images\n        batch_images = np.empty((self.batch_size, self.shape[0], self.shape[1], self.shape[2]))\n        batch_labels = np.zeros((self.batch_size, 28))\n\n        # load images into array\n        for i in range(self.batch_size):\n            # apply transformations to images based on augment flag\n            if self.augment_flag:\n                batch_images[i] = augment(self.__load_image(batch_ids[i]))\n            else:\n                batch_images[i] = self.__load_image(batch_ids[i])\n            batch_labels[i] = self.train_data.loc[self.train_data['Id'] == batch_ids[i], 'Classes'].values[0]\n\n        # return the images batch\n        return batch_images, batch_labels\n\n    def __iter__(self):\n        \"\"\"\n        Create a generator that iterate over the Sequence\n        \"\"\"\n        for i in range(self.batches):\n            yield self[i]\n\n    def on_epoch_end(self):\n        \"\"\"\n        Updates indexes after each epoch\n        \"\"\"\n        self.indexes = np.arange(len(self.image_ids))\n        np.random.shuffle(self.indexes)\n\n    def __load_image(self, image_name):\n        \"\"\"\n        Load image contained within a zip file\n        :param image_name: (string) image name contained in zip file\n        :return:           (array) 3D of RGBY divided by 255\n        \"\"\"\n        if '.zip' in self.file_path:\n            # open the zipfile\n            with ZipFile(self.file_path) as z:\n                # load images by channel\n                channel_list = list()\n                for c in ['red', 'green', 'blue', 'yellow']:\n                    with z.open(image_name + '_' + c + '.png') as file:\n                        img = Image.open(file)\n                        img = img.resize((self.shape[0], self.shape[1]), Image.ANTIALIAS)\n                        img = np.array(img)\n                        channel_list.append(img)\n\n        else:\n            # load images by channel\n            channel_list = list()\n            for c in ['red', 'green', 'blue', 'yellow']:\n                img = Image.open(self.file_path + '/' + image_name + '_' + c + '.png')\n                img = img.resize((self.shape[0], self.shape[1]), Image.ANTIALIAS)\n                img = np.array(img)\n                channel_list.append(img)\n        \n        # stack pixels of image\n        if self.shape[2] == 3:\n            image = np.stack((\n                channel_list[0]/2 + channel_list[3]/2, \n                channel_list[1]/2 + channel_list[3]/2, \n                channel_list[2]\n            ),-1)\n        else:\n            image = np.stack(channel_list, -1)\n        \n        # normalize pixels range\n        image = np.divide(image, 255)\n        \n        # return array with normalized colors\n        return image","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6886a00894250ba71f46fe6aab6b2097409e8849"},"cell_type":"markdown","source":"## Constants definition"},{"metadata":{"trusted":true,"_uuid":"fd738656b83fbd6d92f3c0e9fd83bafca8856d49"},"cell_type":"code","source":"INPUT_SHAPE = (299, 299, 3)\nTRAIN_BATCH = 10\nTEST_BATCH = 256\nTEST_SIZE = 0.2\nC_MIN = 5\nRANDOM_STATE = 42\nSTEPS_PER_EPOCH = 100\nEPOCHS = 15\nVALIDATION_STEPS = 50\nVERBOSE = 1\nVERBOSE_CK = 2\nMODEL_NAME = 'InceptionResNet'","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c0ab254bc28ad7a36d9179e8c0f2435ad9b6030c"},"cell_type":"markdown","source":"## Model setup"},{"metadata":{"trusted":true,"_uuid":"f64b0407ecf214d114c11789d272e1f08aad2f48"},"cell_type":"code","source":"keras.backend.clear_session()\n\nprint('DEFINING CNN MODEL')\nif MODEL_NAME == 'Sequential':\n    model = sequencial_model(INPUT_SHAPE, 28)\nelif MODEL_NAME == 'InceptionResNet':\n    model = inception_res_net_model(INPUT_SHAPE, 28)\nelif MODEL_NAME == 'VGG19':\n    model = vgg_model(INPUT_SHAPE, 28)\n\n# compile model\nmodel.compile(optimizer=Adam(1e-3), loss='binary_crossentropy', metrics=['accuracy', f1])\n\n# model.layers[2].trainable = False\n\n# export model summary\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"72c425f3aaa2645215a3adc808fc188c5b59106d"},"cell_type":"markdown","source":"## Data processing"},{"metadata":{"trusted":true,"_uuid":"90e52e0f3dae354bc3a3b0d6b255ca10385b88af"},"cell_type":"code","source":"print('LOADING TRAIN CSV FILE')\ntrain_csv = pd.read_csv('/kaggle/input/human-protein-atlas-image-classification/train.csv')\n\nprint('OBTAINING CLASSES ARRAY')\ntrain_csv['Classes'] = train_csv['Target'].apply(get_classes)\n\nprint('EVENLY DIVIDING CLASSES')\n# count classes\nvc = train_csv['Target'].value_counts()\n\n# create train and test dataframes\ntrain_df = pd.DataFrame()\ntest_df = pd.DataFrame()\n\nprint('    select well represented classes'.upper())\nfor target in vc[vc > C_MIN].index:\n    # filter images\n    images = train_csv[train_csv['Target'] == target]\n\n    # apply train test split\n    X_train, X_test, y_train, y_test = train_test_split(\n        images['Id'].values.flatten(), images['Classes'].values.flatten(),\n        test_size=TEST_SIZE#, random_state=RANDOM_STATE\n    )\n\n    # add to train and test dataframes\n    train_df = train_df.append(pd.DataFrame(data={'Id': X_train, 'Classes': y_train}))\n    test_df = test_df.append(pd.DataFrame(data={'Id': X_test, 'Classes': y_test}))\n\nprint('    select low represented classes'.upper())\npool = train_csv[train_csv['Target'].isin(vc[vc <= C_MIN].index)]\n\n# go through each class\nfor target in np.argsort(train_csv['Classes'].sum()):\n    # filter pool\n    images = pool[pool['Classes'].apply(lambda x: x[target] == 1)]\n    \n    if images.shape[0] == 0:\n        continue\n        \n    # apply train test split\n    X_train, X_test, y_train, y_test = train_test_split(\n        images['Id'].values.flatten(), images['Classes'].values.flatten(),\n        test_size=TEST_SIZE, random_state=RANDOM_STATE\n    )\n\n    # add to train and test dataframes\n    train_df = train_df.append(pd.DataFrame({'Id': X_train, 'Classes': y_train}))\n    test_df = test_df.append(pd.DataFrame({'Id': X_test, 'Classes': y_test}))\n\n    # remove from pool\n    pool = pool[~pool['Id'].isin(images['Id'])]\n\nprint('CREATING DATA GENERATOR')\ntrain_gen = DataGenerator(train_df, TRAIN_BATCH, '/kaggle/input/human-protein-atlas-image-classification/train', INPUT_SHAPE)\ntest_gen = DataGenerator(test_df, TEST_BATCH, '/kaggle/input/human-protein-atlas-image-classification/test', INPUT_SHAPE, augment_flag=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a437fd2556426333a41f70c5c02cdb0d5628d7df"},"cell_type":"raw","source":"test_df.shape[0] / (train_df.shape[0] + test_df.shape[0])"},{"metadata":{"_uuid":"ff3a3ce97521f24521686fb8a81b7fab060766ba"},"cell_type":"raw","source":"d = {\n    'Base': train_csv['Classes'].sum(),\n    'Train': train_df['Classes'].sum(),\n    'Test': test_df['Classes'].sum()\n}\ndf = pd.DataFrame(d)\ndf['Train'] = df['Train']/df['Base']\ndf['Test'] = df['Test']/df['Base']\ndf"},{"metadata":{"_uuid":"2fbffd71262d7d7e591ad5955feb6b0c5ee61737"},"cell_type":"markdown","source":"## Train model"},{"metadata":{"_kg_hide-output":false,"trusted":true,"_uuid":"9c946937a082843eea150dfdff77bda7d17a131d","scrolled":true},"cell_type":"code","source":"print('CREATING MODEL CHECK POINT')\ndt = datetime.datetime.today().strftime('%Y%m%d%H%M%S')\nmodel_path = '/kaggle/working/' + MODEL_NAME + dt + '.model'\ncheckpoint = ModelCheckpoint(model_path, verbose=VERBOSE_CK, save_best_only=True)\n\nprint('FITTING MODEL')\nhist = model.fit_generator(\n    generator=train_gen, \n    validation_data=test_gen,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS, \n    validation_steps=VALIDATION_STEPS, \n    verbose=VERBOSE,\n    callbacks=[checkpoint]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87c8e7f4219275aa7e2bcb99b195bcb4b5b276c2"},"cell_type":"code","source":"show_history(hist)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a52b306cd27fc030cfd3f7583009945de7828ac"},"cell_type":"markdown","source":"## Create submit"},{"metadata":{"trusted":true,"_uuid":"e3fc9b48ce40e9a392dcae98ce13790d8674d70a"},"cell_type":"code","source":"model_path = '/kaggle/working/' + MODEL_NAME + dt + '.model'\nmodel = load_model(\n    model_path, \n    custom_objects={'f1': f1}\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d9ee4a8c2b1fc75b9f8d58855e33b115e92134d"},"cell_type":"code","source":"submit = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"782cdb0d614add0f41c11ea180bd45e5600d60fb"},"cell_type":"code","source":"%%time\npredicted = []\nfor name in tqdm(submit['Id']):\n    path = os.path.join('../input/test/', name)\n    image = DataGenerator(train_data, TRAIN_BATCH, '/kaggle/input/train/', INPUT_SHAPE)\n    score_predict = model.predict(image[np.newaxis])[0]\n    label_predict = np.arange(28)[score_predict>=0.2]\n    str_predict_label = ' '.join(str(l) for l in label_predict)\n    predicted.append(str_predict_label)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07043ed35ad009ce6edc017e4aa88ac82571a750"},"cell_type":"code","source":"submit['Predicted'] = predicted\nsubmit.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}