{"cells":[{"metadata":{},"cell_type":"markdown","source":"# About this kernel\n\nThis is a rather quick and dirty kernel I created, with two ideas in mind: Training a \"2-headed\" network that will learn to predict siRNA using images from both sites at the same time, and split the learning process into two stages, namely first training on all data, then training the CNN on data from a single experiment at a time. The second idea comes from [this thread by Phalanx](https://www.kaggle.com/c/recursion-cellular-image-classification/discussion/100414#latest-586901). The data comes from my previous kernel on preprocessing.\n\nHere are the relevant sections:\n* **Data Generator**: The `__generate_X` method is pretty different, since it loads two images at the same time. Everything else is standard\n* **Model**: The CNN architecture used here is `EfficientNetB0`. With the right learning rates and enough time, you can probably try B1-B5; they have unfortunately not succeeded in my case. The inputs are two images, i.e. from site 1 and site 2. The two images are passed through the same CNN, then global-average-pooled, and added to form a single 1280-dimensional vector, which is ultimately used to perform predictions. This means that the networks will be updated simultaneously from the gradients of both sites.\n* **Phase 1**: Train the model on all data from 10 epochs, and save results to `model.h5`.\n* **Phase 2**: Load `model.h5` and train the model for 15 epochs on data from a single cell line, i.e. *HEPG2, HUVEC, RPE, U2OS*.\n\n\n\n"},{"metadata":{"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!pip install efficientnet\nimport efficientnet","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs_0 = 10\nepochs_1 = 28","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import json\nimport math\nimport os\n\nimport cv2\nimport tensorflow as tf\nfrom PIL import Image\nimport numpy as np\nimport keras\nfrom keras import layers\nfrom keras.applications import MobileNetV2\nfrom keras.callbacks import Callback, ModelCheckpoint\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Model, load_model\nfrom keras.layers import GlobalAveragePooling2D, Dense, Dropout, BatchNormalization, concatenate, Input, add\nfrom keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nimport scipy\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Preprocessing"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/recursion-cellular-image-classification/train.csv')\ntest_df = pd.read_csv('../input/recursion-cellular-image-classification/test.csv')\n\ntrain_df['category'] = train_df['experiment'].apply(lambda x: x.split('-')[0])\ntest_df['category'] = test_df['experiment'].apply(lambda x: x.split('-')[0])\n\ntrain_target_df = pd.get_dummies(train_df['sirna'])\n\nprint(train_df.shape)\nprint(test_df.shape)\nprint(train_target_df.shape)\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_idx, val_idx = train_test_split(\n    train_df.index, test_size=0.15, random_state=1337\n)\n\nprint(train_idx.shape)\nprint(val_idx.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Data Generator"},{"metadata":{"trusted":true},"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, list_IDs, df, target_df=None, mode='fit',\n                 base_path = '../input/recursion-cellular-image-classification-224-jpg/train/train',\n                 batch_size=32, dim=(224, 224), n_channels=3,\n                 n_classes=5, random_state=1337, shuffle=True):\n        self.dim = dim\n        self.batch_size = batch_size\n        self.df = df\n        self.mode = mode\n        self.base_path = base_path\n        self.target_df = target_df\n        self.list_IDs = list_IDs\n        self.n_channels = n_channels\n        self.n_classes = n_classes\n        self.shuffle = shuffle\n        self.random_state = random_state\n        \n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Find list of IDs\n        list_IDs_batch = [self.list_IDs[k] for k in indexes]\n        \n        X = self.__generate_X(list_IDs_batch)\n        \n        if self.mode == 'fit':\n            y = self.__generate_y(list_IDs_batch)\n            return X, y\n        \n        elif self.mode == 'predict':\n            return X\n        else:\n            raise AttributeError('The parameter mode should be set to \"fit\" or \"predict\".')\n        \n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.seed(self.random_state)\n            np.random.shuffle(self.indexes)\n    \n    def __generate_X(self, list_IDs_batch):\n        'Generates data containing batch_size samples'\n        # Initialization\n        X_1 = np.empty((self.batch_size, *self.dim, self.n_channels))\n        X_2 = np.empty((self.batch_size, *self.dim, self.n_channels))\n        \n        # Generate data\n        for i, ID in enumerate(list_IDs_batch):\n            ext = 'jpeg'\n            \n            code = self.df['id_code'].iloc[ID]\n            \n            img_path_1 = f\"{self.base_path}/{code}_s1.{ext}\"\n            img_path_2 = f\"{self.base_path}/{code}_s2.{ext}\"\n            \n            img1 = self.__load_image(img_path_1)\n            img2 = self.__load_image(img_path_2)\n            \n            # Store samples\n            X_1[i,] = img1\n            X_2[i,] = img2\n\n        return [X_1, X_2]\n    \n    def __generate_y(self, list_IDs_batch):\n        y = np.empty((self.batch_size, self.n_classes), dtype=int)\n        \n        for i, ID in enumerate(list_IDs_batch):\n            sirna = self.target_df.iloc[ID]\n            y[i, ] = sirna\n        \n        return y\n    \n    def __load_image(self, img_path):\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n        img = img.astype(np.float32) / 255.\n\n        return img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE = 32\ntrain_generator = DataGenerator(\n    train_idx, \n    df=train_df,\n    target_df=train_target_df,\n    batch_size=BATCH_SIZE, \n    n_classes=train_target_df.shape[1]\n)\n\nval_generator = DataGenerator(\n    val_idx, \n    df=train_df,\n    target_df=train_target_df,\n    batch_size=BATCH_SIZE, \n    n_classes=train_target_df.shape[1]\n)\n\ntest_generator = DataGenerator(\n    test_df.index, \n    df=test_df,\n    batch_size=1, \n    shuffle=False,\n    mode='predict',\n    n_classes=train_target_df.shape[1],\n    base_path='../input/recursion-cellular-image-classification-224-jpg/test/test/'\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model(n_classes):\n    # First load mobilenet\n    backbone = efficientnet.EfficientNetB1(\n        weights='imagenet', \n        include_top=False,\n        input_shape=(224, 224, 3)\n    )\n    \n    im_inp_1 = Input(shape=(224, 224, 3))\n    im_inp_2 = Input(shape=(224, 224, 3))\n\n    x1 = backbone(im_inp_1)\n    x2 = backbone(im_inp_2)\n\n    x1 = GlobalAveragePooling2D()(x1)\n    x2 = GlobalAveragePooling2D()(x2)\n\n    out = add([x1, x2])\n    out = Dropout(0.5)(out)\n\n    out = Dense(n_classes, activation='softmax')(out)\n\n    model = Model(inputs=[im_inp_1, im_inp_2], outputs=out)\n    \n    model.compile(Adam(0.0001), loss='categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Phase 1: Train on all data"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = build_model(n_classes=train_target_df.shape[1])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    'model.h5', \n    monitor='val_loss', \n    verbose=0, \n    save_best_only=True, \n    save_weights_only=False,\n    mode='auto'\n)\n\nhistory = model.fit_generator(\n    train_generator,\n    validation_data=val_generator,\n    callbacks=[checkpoint],\n    use_multiprocessing=False,\n    workers=1,\n    epochs=epochs_0\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"with open('history.json', 'w') as f:\n    json.dump(history.history, f)\n\nhistory_df = pd.DataFrame(history.history)\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['acc', 'val_acc']].plot()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Phase 2: train on each cell line"},{"metadata":{"trusted":true},"cell_type":"code","source":"categories = train_df['category'].unique()\noutput_df = []\n\nfor category in categories:\n    # Retrieve desired category\n    category_df = train_df[train_df['category'] == category]\n    cat_test_df = test_df[test_df['category'] == category].copy()\n    \n    print('\\n' + '=' * 40)\n    print(\"CURRENT CATEGORY:\", category)\n    print('-' * 40)\n    \n    train_idx, val_idx = train_test_split(\n        category_df.index, \n        random_state=1337,\n        test_size=0.15\n    )\n    \n    # Create new generators\n    train_generator = DataGenerator(\n        train_idx, \n        df=train_df,\n        target_df=train_target_df,\n        batch_size=BATCH_SIZE, \n        n_classes=train_target_df.shape[1]\n    )\n\n    val_generator = DataGenerator(\n        val_idx, \n        df=train_df,\n        target_df=train_target_df,\n        batch_size=BATCH_SIZE, \n        n_classes=train_target_df.shape[1]\n    )\n\n    test_generator = DataGenerator(\n        cat_test_df.index, \n        df=test_df,\n        batch_size=1, \n        shuffle=False,\n        mode='predict',\n        n_classes=train_target_df.shape[1],\n        base_path='../input/recursion-cellular-image-classification-224-jpg/test/test/'\n    )\n\n    # Restore previously trained model\n    model.load_weights('model.h5')\n    model.compile(\n        Adam(0.0001), \n        loss='categorical_crossentropy', \n        metrics=['accuracy']\n    )\n\n    # Train model only on data for specific category\n    checkpoint = ModelCheckpoint(\n        f'model_{category}.h5', \n        monitor='val_loss', \n        verbose=0, \n        save_best_only=True, \n        save_weights_only=False,\n        mode='auto'\n    )\n\n    history_category = model.fit_generator(\n        train_generator,\n        validation_data=val_generator,\n        callbacks=[checkpoint],\n        use_multiprocessing=False,\n        workers=1,\n        verbose=2,\n        epochs=epochs_1\n    )\n\n    # Make prediction and add to output dataframe\n    y_pred = model.predict_generator(\n        test_generator,\n        workers=2,\n        use_multiprocessing=True,\n        verbose=1\n    )\n\n    cat_test_df['sirna'] = y_pred.argmax(axis=1)\n    output_df.append(cat_test_df[['id_code', 'sirna']])\n\n    # Save history\n    with open(f'history_{category}.json', 'w') as f:\n        json.dump(history_category.history, f)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"output_df = pd.concat(output_df)\noutput_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"}},"nbformat":4,"nbformat_minor":1}