{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import os, sys, math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2a26c3d31d91582668924e0a58d22041bc0368d"},"cell_type":"code","source":"INPUT_SHAPE = (232,232,4)\nBATCH_SIZE = 32","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b92f76f60d58ea0d17da8305cbb50aa7de4b5aca"},"cell_type":"markdown","source":"### Load dataset info"},{"metadata":{"trusted":true,"_uuid":"ae261988b47b05319d7ebb0342a82852e0184d12"},"cell_type":"code","source":"path_to_train = '/kaggle/input/human-protein-atlas-image-classification/train/'\ndata = pd.read_csv('/kaggle/input/human-protein-atlas-image-classification/train.csv')\nweight_path = '/kaggle/input/dna-first-kaggle/'\n\ntrain_dataset_info = []\nfor name, labels in zip(data['Id'], data['Target'].str.split(' ')):\n    train_dataset_info.append({\n        'path':os.path.join(path_to_train, name),\n        'labels':np.array([int(label) for label in labels])})\ntrain_dataset_info = np.array(train_dataset_info)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82a27accf7b689b60f7b88b17cb2dcdaf7ef346e"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_ids, test_ids, train_targets, test_target = train_test_split(\n    data['Id'], data['Target'], test_size=0.2, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1181e4c83d22b7fa7161e070db40c3e8c44ddc62"},"cell_type":"markdown","source":"### Create datagenerator"},{"metadata":{"trusted":true,"_uuid":"a56616d0ae1b8cc1e8c5ccd5638b69cd59f57322"},"cell_type":"code","source":"class data_generator:\n    \n    def create_train(dataset_info, batch_size, shape, augument=True):\n        assert shape[2] == 4\n        while True:\n            random_indexes = np.random.choice(len(dataset_info), batch_size)\n            batch_images = np.empty((batch_size, shape[0], shape[1], shape[2]))\n            batch_labels = np.zeros((batch_size, 28))\n            for i, idx in enumerate(random_indexes):\n                image = data_generator.load_image(\n                    dataset_info[idx]['path'], shape)   \n                if augument:\n                    image = data_generator.augment(image)\n                batch_images[i] = image\n                batch_labels[i][dataset_info[idx]['labels']] = 1\n            yield batch_images, batch_labels\n            \n    \n    def load_image(path, shape):\n        R = np.array(Image.open(path+'_red.png'))\n        G = np.array(Image.open(path+'_green.png'))\n        B = np.array(Image.open(path+'_blue.png'))\n        Y = np.array(Image.open(path+'_yellow.png'))\n\n        image = np.stack((R,G,B,Y),-1)\n        image = cv2.resize(image, (shape[0], shape[1]))\n        image = np.divide(image, 255)\n        return image  \n                \n            \n    def augment(image):\n        augment_img = iaa.Sequential([\n            iaa.OneOf([\n                iaa.Affine(rotate=0),\n                iaa.Affine(rotate=90),\n                iaa.Affine(rotate=180),\n                iaa.Affine(rotate=270),\n                iaa.Fliplr(0.5),\n                iaa.Flipud(0.5),\n            ])], random_order=True)\n        \n        image_aug = augment_img.augment_image(image)\n        return image_aug","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfcca6e92addeda86ac7635cccfeec94534c3de4"},"cell_type":"code","source":"# create train datagen\ntrain_datagen = data_generator.create_train(train_dataset_info, BATCH_SIZE, INPUT_SHAPE, augument=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5750c93722a938540b43cdee72b48fe2c7634a87"},"cell_type":"code","source":"images, labels = next(train_datagen)\n\nfig, ax = plt.subplots(1,4,figsize=(25,5))\nfor i in range(4):\n    ax[i].imshow(images[i])\nprint('min: {0}, max: {1}'.format(images.min(), images.max()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd9d81e4df3b0013559ed601a1ebd83af21e5019"},"cell_type":"code","source":"images.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3c83cd8a919c6e546fb3cb2b573c5eca72ee91eb"},"cell_type":"markdown","source":"### Create Model"},{"metadata":{"trusted":true,"_uuid":"f225000b743a7053d8be74f2e0778ce25f317302"},"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Sequential, load_model\nfrom keras.layers import Activation\nfrom keras.layers import Dropout\nfrom keras.layers import Flatten\nfrom keras.layers import Dense\nfrom keras.layers import Input\nfrom keras.layers import BatchNormalization\nfrom keras.layers import Conv2D, GlobalAveragePooling2D\nfrom keras.models import Model\nfrom keras.applications.inception_resnet_v2 import InceptionResNetV2\nfrom keras.applications.resnet50 import ResNet50\nfrom keras.applications.densenet import DenseNet121\nfrom keras.applications.densenet import DenseNet169\nfrom keras.applications.densenet import DenseNet201\nfrom keras.callbacks import ModelCheckpoint\nfrom keras.callbacks import LambdaCallback\nfrom keras.callbacks import Callback\nfrom keras import metrics\nfrom keras.optimizers import Adam \nfrom keras import backend as K\nimport tensorflow as tf\nimport keras","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4662d128d38352d6963730127b07d14744e869fd"},"cell_type":"code","source":"def DnAInput(img_shape = INPUT_SHAPE) :\n    model = Sequential()\n    # 230 * 230\n    model.add(Conv2D(32, kernel_size=(3,3), activation=\"relu\", kernel_initializer='he_normal', input_shape = img_shape))\n    model.add(BatchNormalization())\n    # 228 * 228\n    model.add(Conv2D(32, kernel_size=(3,3), activation=\"relu\", kernel_initializer='he_normal'))\n    model.add(BatchNormalization())\n    # 226 * 226\n    model.add(Conv2D(32, kernel_size=(3,3), activation=\"relu\", kernel_initializer='he_normal'))\n    model.add(BatchNormalization())\n    # 224 * 224\n    model.add(Conv2D(3, kernel_size=(3,3), activation=\"relu\", kernel_initializer='he_normal'))\n    model.add(BatchNormalization())\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25485a2f550323ecce3d4796e4cf75eedaa89ffd"},"cell_type":"code","source":"def f1(y_true, y_pred):\n    tp = K.sum(K.cast(y_true*y_pred, 'float'), axis=0)\n    fp = K.sum(K.cast((1-y_true)*y_pred, 'float'), axis=0)\n    fn = K.sum(K.cast(y_true*(1-y_pred), 'float'), axis=0)\n\n    p = tp / (tp + fp + K.epsilon())\n    r = tp / (tp + fn + K.epsilon())\n\n    f1 = 2*p*r / (p+r+K.epsilon())\n    f1 = tf.where(tf.is_nan(f1), tf.zeros_like(f1), f1)\n    return K.mean(f1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1523b9ec1899737be2f38e8f7c61f8fe7ef1d3c0"},"cell_type":"code","source":"def show_history(history):\n    fig, ax = plt.subplots(1, 3, figsize=(15,5))\n    ax[0].set_title('loss')\n    ax[0].plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\n    ax[0].plot(history.epoch, history.history[\"val_loss\"], label=\"Validation loss\")\n    ax[1].set_title('f1')\n    ax[1].plot(history.epoch, history.history[\"f1\"], label=\"Train f1\")\n    ax[1].plot(history.epoch, history.history[\"val_f1\"], label=\"Validation f1\")\n    ax[2].set_title('acc')\n    ax[2].plot(history.epoch, history.history[\"acc\"], label=\"Train acc\")\n    ax[2].plot(history.epoch, history.history[\"val_acc\"], label=\"Validation acc\")\n    ax[0].legend()\n    ax[1].legend()\n    ax[2].legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"564b2c15a9a617fe12d338205fbd4998638469a5"},"cell_type":"code","source":"input_model = DnAInput()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9271503b34f807615feac192eeeb4fddc942790d"},"cell_type":"code","source":"densenet169 = DenseNet169(include_top = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d55e089c8221aa6641e8ee0657e7005ebd76202e"},"cell_type":"code","source":"densenet169.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c24d87bd7d9c5dc39809559bf5a90ef90213c57e"},"cell_type":"code","source":"for layer in densenet169.layers:\n    layer.trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf596356689ff0d1489ca7a18bba31ca85fe63bc"},"cell_type":"code","source":"x = Conv2D(28,kernel_size=(3,3), activation='relu', kernel_initializer=\"he_normal\")(densenet169.output)\nx = BatchNormalization()(x)\nflat = GlobalAveragePooling2D()(x)\nfc = Dense(28, activation='sigmoid')(flat)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc64ed3d97293e3fe9737bac9d96e5925deaa05b"},"cell_type":"code","source":"inputToDensnet169 = Model(inputs=densenet169.input, outputs=fc)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06db57ba7866df961a877c0abb964825eb1a4a61"},"cell_type":"code","source":"inputToDensnet169.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c3004a68a48cea65247e7b5b42a470d584fcb6e"},"cell_type":"code","source":"tmp_output = input_model.output\nfinal_output = inputToDensnet169(tmp_output)\nDnaNet = Model(inputs = input_model.input, output = final_output)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a26cd480f02f4396f7e0f3fba38e1f95383dea4a"},"cell_type":"code","source":"DnaNet.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92a8c6dcb94576b159d4537645778cb4d747eb0e"},"cell_type":"code","source":"DnaNet.compile(optimizer=Adam(), loss='binary_crossentropy', metrics=['accuracy',f1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee64d4715b0ea85e8069102a3998c60d92480773"},"cell_type":"code","source":"import glob\nglob.glob(weight_path + '*')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f9ce8c387fa1d5c733c144b863b9e1bf78c20d42"},"cell_type":"code","source":"DnaNet.load_weights(weight_path + 'DnAnet64-10.hdf5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ddb1cf1cca68ef93da0c8e440574c1793cc27fe7"},"cell_type":"code","source":"checkpointer = ModelCheckpoint('DnAnet64-{epoch:02d}_ver2.hdf5',\n    verbose=2, save_best_only=False)\n\ntrain_generator = data_generator.create_train(\n    train_dataset_info[train_ids.index], BATCH_SIZE, INPUT_SHAPE, augument=False)\nvalidation_generator = data_generator.create_train(\n    train_dataset_info[test_ids.index], 256, INPUT_SHAPE, augument=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96907ce994b40765771fb600a960cd8f8e0ba434","_kg_hide-output":false},"cell_type":"code","source":"history = DnaNet.fit_generator(\n    train_generator,\n    steps_per_epoch=int(data.shape[0]/BATCH_SIZE),\n    validation_data=next(validation_generator),\n    epochs=20, \n    verbose=1,\n    callbacks=[checkpointer])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"457aced0d9fbd5678cafc06c01a33b8bc4e09364"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}