{"cells":[{"metadata":{"trusted":true,"_uuid":"5bf61e64c3561a577afb09faa8764f1bc5560bc0","_kg_hide-output":true},"cell_type":"code","source":"import tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from keras import backend as K\nfrom keras.optimizers import Adam, SGD, Adagrad, Adadelta\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau, LearningRateScheduler, CSVLogger\nfrom keras.models import Model, Sequential, load_model, model_from_json\nfrom keras.layers import Flatten, Dense, Activation, Input, Dropout, Activation, BatchNormalization, Reshape\nfrom keras.layers import Conv2D, MaxPooling2D, GlobalAveragePooling2D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11b6fbca7cb70d597caa095e8fa441594b4ba30d"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e6889a7ba4147d5c69d25c16a5c3982f0597e6f","_kg_hide-output":false},"cell_type":"code","source":"import random\nimport os\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e32001c038eecad03422a5d21966915be12ebce8"},"cell_type":"code","source":"data_path = '../input'\ntrain_path = os.path.join(data_path, 'train')\ntest_path = os.path.join(data_path,'test')\nlabels_path = os.path.join(data_path, 'train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c4bfa8199e979a195da6c6031d63ee49d198ef89","_kg_hide-output":true},"cell_type":"code","source":"os.listdir(data_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"abdf31b18e28b054e41fcc6c2de155bc8b406eff"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ab3d1160aadb6272f175d1476011b2b939ef87f"},"cell_type":"code","source":"def pick_label(x):\n    # x list of strings\n    t  = x\n    if '27' in t:\n        return '27'\n    elif '15' in t:\n        return '15'\n    elif '10' in t:\n        return '10'\n    elif '9' in t:\n        return '9'\n    elif '8' in t:\n        return '8'\n    elif '0' in t:\n        return '0'\n    elif '25' in t:\n        return '25'\n    else:\n        return t[0]\n    \n    \n\n\ntrain_labels = pd.read_csv(labels_path,index_col=False)\nlabels_dict = dict(zip(train_labels.values[ :  ,0], train_labels.values[ : , 1]))\nt = {k:pick_label(v.split()) for (k, v) in labels_dict.items()}\ntrain_labels['t'] = t.values()\nt_arr = np.array(list(t.items()))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c708a1cf712048a95ad98f321e09504d5177780"},"cell_type":"code","source":"\ntrain_ids, val_ids = train_test_split(train_labels, stratify=t_arr[ :, 1],\n                                        test_size=0.1, random_state=48)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f375c5d28052aa4301a0652a566e86ad8e32cfb"},"cell_type":"code","source":"labels = [item.split() for item in train_labels['Target']]\n\nmlb = MultiLabelBinarizer()\nmlb.fit(labels)\nclasses = mlb.classes_\ny_val = mlb.transform([item.split() for item in val_ids['Target']])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"189b6aa2a09147f112a3cb1b79bbca2a3317d77b"},"cell_type":"code","source":"y_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58258d6f30869761152f67ddcaec99fcae69e850"},"cell_type":"code","source":"def model(sample_shape):\n    \n\n    model = Sequential()\n\n    model.add(Conv2D(32, kernel_size=(3, 3), input_shape=sample_shape, name='conv1'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Conv2D(32, kernel_size=(3, 3), name='conv2'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Conv2D(32, kernel_size=(3, 3), name='conv2b'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2), name='pool1'))\n\n    model.add(Conv2D(64, kernel_size=(3, 3), name='conv3'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Conv2D(64, kernel_size=(3, 3), name='conv4'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2), name='pool2'))\n    \n    \n    model.add(Conv2D(128, kernel_size=(3, 3), name='conv5'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Conv2D(128, kernel_size=(3, 3), name='conv6'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2), name='pool3'))\n    \n    model.add(Conv2D(256, kernel_size=(3, 3), name='conv7'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Conv2D(256, kernel_size=(3, 3), name='conv8'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(GlobalAveragePooling2D())\n    model.add(Dense(4096, name='fc1'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    \n    model.add(Dense(28))\n    model.add(Activation('sigmoid'))\n    return model\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b36135c974413149d38bbedb45816d1460be0b77"},"cell_type":"code","source":"def get_rgb_img(image_folder,img_id):\n    img = []\n    img.append(plt.imread(os.path.join(image_folder,img_id+'_red.png')))\n    img.append(plt.imread(os.path.join(image_folder, img_id+'_blue.png')))\n    img.append(plt.imread(os.path.join(image_folder, img_id+'_green.png')))\n    return np.stack(img, axis=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a79047af8096b48a7cec3df6318ceb78144010ab"},"cell_type":"code","source":"img = get_rgb_img(train_path,val_ids.values[0][0])\nplt.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d49a9878c89f3b454c13cc5ec5d76db17852dc0e"},"cell_type":"code","source":"def val_generator(BATCH_SIZE):\n   \n\n    image_folder = train_path\n    while True:\n        \n        \n        val_imgs = []\n        val_labels = []\n        \n        \n        for f in val_ids.values:\n            img = get_rgb_img(image_folder,f[0])\n            val_imgs.append(cv2.resize(img, (IMG_SIZE, IMG_SIZE)))\n            val_labels.append(f[1])\n            if len(val_imgs) == BATCH_SIZE:\n                imgs = np.stack(val_imgs, axis=0)\n                labels = mlb.transform([item.split() for item in val_labels])\n                if len(imgs.shape[ 1: ]) == 2:\n                    imgs = np.expand_dims(imgs, axis=3)\n                yield (imgs, labels)\n                val_imgs =[]\n                val_labels =[]\n        if len(val_imgs) > 0:\n            imgs = np.stack(val_imgs, axis=0)\n            labels = mlb.transform([item.split() for item in val_labels])\n            if len(imgs.shape[ 1: ]) == 2:\n                imgs = np.expand_dims(imgs, axis=3)\n            yield (imgs, labels)\n  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94d451f5f837757acb5244c3832544b52c7448ea"},"cell_type":"code","source":"def train_generator(BATCH_SIZE):\n    import random\n    image_folder = train_path\n    while True:\n        #sample = train_ids.values\n        #np.random.shuffle(sample)\n        \n        train_imgs = []\n        train_labels = []\n        \n        for f in train_ids.values:\n            img = get_rgb_img(image_folder,f[0])\n            train_imgs.append(cv2.resize(img, (IMG_SIZE, IMG_SIZE)))\n            train_labels.append(f[1])\n            if len(train_imgs) == BATCH_SIZE:\n                imgs = np.stack(train_imgs, axis=0)\n                labels = mlb.transform([item.split() for item in train_labels])\n                if len(imgs.shape[ 1: ]) == 2:\n                    imgs = np.expand_dims(imgs, axis=3)\n                yield (imgs, labels)\n                train_imgs = []\n                train_labels = []\n        if len(train_imgs) > 0:\n            imgs = np.stack(train_imgs, axis=0)\n            labels = mlb.transform([item.split() for item in train_labels])\n            if len(imgs.shape[ 1: ]) == 2:\n                imgs = np.expand_dims(imgs, axis=3)\n            \n            yield (imgs, labels)\n  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ac14a130632da7344535d930bce231fc896dffc"},"cell_type":"code","source":"## model parameters\nDEPTH = 3\nBATCH_SIZE = 32\nIMG_SIZE = 256\nSAMPLE_SHAPE = (IMG_SIZE, IMG_SIZE, DEPTH)\n\nSEED = 1234\nrandom.seed(SEED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c2988897e1eb37540fab6f6f166e44c4e86b345"},"cell_type":"code","source":"K.clear_session()\nmodel = model(SAMPLE_SHAPE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"144ef77af33f8a968d5e13fc2b93b0d2eabe4093"},"cell_type":"code","source":"EPOCHS = 50\nval_steps = int(np.ceil(len(val_ids)/BATCH_SIZE))\nnum_steps = int(np.ceil(len(train_ids)/BATCH_SIZE))\nprint('train_size: ', len(train_ids), 'batch: ', BATCH_SIZE, 'num steps: ', num_steps)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12c527a949a80ce5906b3f225b88a4aae175d1b7"},"cell_type":"code","source":"def f1(y_true, y_pred, thresholds= np.array(28*[0.5])):\n    m, n = y_true.shape\n    p = tf.cast(tf.greater(y_pred, thresholds), tf.float32)\n    tp = tf.reduce_sum(y_true * p, 0)\n    num_pos = tf.reduce_sum(tf.cast(y_true, tf.float32), 0)\n    pred_pos = tf.reduce_sum(p, 0)\n    precision = tp/(pred_pos + K.epsilon())\n    recall = tp /(num_pos + K.epsilon())\n    f1 = tf.reduce_mean(tf.divide(2*precision*recall, (precision + recall + K.epsilon())))\n    K.get_session().run(tf.local_variables_initializer())\n    return f1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"22b7ef61d057aa4d5351172c0ef24f674175e2c2"},"cell_type":"code","source":"lr = 1e-3\nadam = Adam(lr=lr)\nmodel.compile(optimizer=adam, \n                  loss='binary_crossentropy',\n                metrics=[f1])\n\n\nearlyStopping = EarlyStopping(monitor='val_loss', min_delta=0, mode='min', patience=6, verbose=0,restore_best_weights=True)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', mode='min',factor=0.2, patience=3, min_lr = 1e-6, cooldown=1,verbose=1)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"8360e2fe7dfc004189817dcff7a1cd391fd29739"},"cell_type":"code","source":"history = model.fit_generator(train_generator(BATCH_SIZE),\n                                  steps_per_epoch = num_steps,\n                                  validation_data=val_generator(BATCH_SIZE),\n                                  validation_steps=val_steps,\n                                  epochs=EPOCHS,\n                                  callbacks=[earlyStopping, reduce_lr], verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f78b693f431caec5ca4c7164e89c2973cb410aa0"},"cell_type":"code","source":"predictions = model.predict_generator(val_generator(BATCH_SIZE),steps=val_steps)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d57f9161dec1f78e3b0833a831788091468c801","_kg_hide-output":false},"cell_type":"code","source":"with tf.Session() as sess:\n    print(sess.run(f1(y_val, predictions)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6ed89c2098f430afb35ff0483a6521d3e16711e"},"cell_type":"code","source":"submission = pd.read_csv(os.path.join(data_path, 'sample_submission.csv'), index_col=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b14670bb13a88dd11fead8240bfcb90cc7dff38"},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7927f8869ec97e0c083b474c5090f4423bfb9d1"},"cell_type":"code","source":"def test_generator(BATCH_SIZE):\n    \n    \n    image_folder = test_path\n    \n    while True:\n        \n        \n        test_imgs = []\n        \n        for f in submission['Id']:\n            \n            img = get_rgb_img(image_folder, f)\n            test_imgs.append(cv2.resize(img, (IMG_SIZE, IMG_SIZE)))\n            \n            if len(test_imgs) == BATCH_SIZE:\n                imgs = np.stack(test_imgs, axis=0)\n                if len(imgs.shape[ 1: ]) == 2:\n                    imgs = np.expand_dims(imgs, axis=3)\n                yield imgs\n                test_imgs =[]\n        if len(test_imgs) > 0:\n            imgs = np.stack(test_imgs, axis=0)\n            if len(imgs.shape[ 1: ]) == 2:\n                imgs = np.expand_dims(imgs, axis=3)\n            yield imgs\n  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ecccc753a98a3ea0f3e12f1dd57d2f3509be598"},"cell_type":"code","source":"num_steps = int(np.ceil(len(submission)/BATCH_SIZE))\ntest_pred = model.predict_generator(test_generator(BATCH_SIZE), steps=num_steps )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1eb04f814ba62a9d54b64a3b24e543e9e4bd44e6"},"cell_type":"code","source":"submission.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff172199ebbfc0a57d461561fb9cba7363ba222a"},"cell_type":"code","source":"def binary_prf1(y_true, y_pred):\n    \n    # y_true (ground truth)  - 1d array of 1, 0 \n    # y_pred (predictions) - id arry of 1, 0\n    # 1 - positive , 0 - negative\n    num_pos = np.sum(y_true)\n    pred_pos = np.sum(y_pred)\n    tp = np.sum(y_true * y_pred)\n    if pred_pos > 0:\n        precision = tp/pred_pos\n    else:\n        precision = 0\n    if num_pos > 0:\n        recall = tp/num_pos\n    else:\n        recall = 0\n        print('no pos cases for this class')\n    if precision >0 or recall > 0:\n        f1 = 2*precision*recall/(precision + recall)\n    else:\n        f1 = 0\n    return precision, recall, f1\n\n\ndef max_thresh(y_val, predictions, n=100):\n    x = np.linspace(0,1,n+1)[1 : -1]\n    f1_matrix = np.zeros((len(x), 28))\n\n    for i in range(28):\n        class_f1 = []\n        for thresh in x:\n            pred_class = (predictions > thresh).astype(int)\n            class_f1.append((binary_prf1(y_val[ :, i], pred_class[ : , i])[2]))\n        f1_matrix[ :, i] = np.array(class_f1)\n    #np.round(np.max(f1_matrix, axis=0),3)\n    #f1_max = np.max(f1_matrix, axis=0)\n    max_loc = np.argmax(f1_matrix, axis=0)\n    max_thresh = [x[i] for i in max_loc]\n    #print(max_thresh)\n    #pc = (predictions > max_thresh).astype(int)\n    return max_thresh","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75b7cb33ace6e433218c2ac776d9bfeb94edfd37"},"cell_type":"code","source":"max_t = max_thresh(y_val, predictions)\nwith tf.Session() as sess:\n    print(sess.run(f1(y_val, predictions, thresholds=max_t)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7dace5a219ef9cb0ef171841f54737edf84bae51"},"cell_type":"code","source":"\npred_classes = (test_pred > max_t).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64bcc9bfddce973a763c87431d64253e2cdbf1d4"},"cell_type":"code","source":"pred_labels = mlb.inverse_transform(pred_classes)\npred_labels = [' '.join(item) for item in pred_labels]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18635f70d47b43de05d073638f2afb43cac85ab3"},"cell_type":"code","source":"submission['Predicted'] = pd.Series(pred_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"965843b60b5c4d1b2039da1d1db76d6807f7e445"},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"847a25e8cfbf0733104a8229820ed9d267525f8d"},"cell_type":"code","source":"np.sum(submission[\"Predicted\"] == '')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"885a8334c8ee4d2f4b943fb30f6e8d2e95f09067"},"cell_type":"code","source":"submission[submission[\"Predicted\"] == ''] = '0 25'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb4e20367f7b1cff4dab11cb65545044b59ccd1f"},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}