{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import math\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go\nimport os, gc, cv2, random, warnings, math, sys, json, pprint\nfrom glob import glob\nfrom pylab import rcParams\nfrom sklearn.model_selection import  GroupKFold\nfrom sklearn.metrics import roc_auc_score\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\nfrom tensorflow.keras import models, layers\nfrom tensorflow.keras.optimizers import Adam\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print(f'Running on TPU {tpu.master()}')\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy()\n\nAUTO = tf.data.experimental.AUTOTUNE\nREPLICAS = strategy.num_replicas_in_sync","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/ranzcr-clip-catheter-line-classification'\nos.listdir(data_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train images: %d' %len(os.listdir(os.path.join(data_dir, \"train\"))))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\n\ndata_dir ='../input/ranzcr-clip-catheter-line-classification'\npath_dir = KaggleDatasets().get_gcs_path('ranzcr-clip-catheter-line-classification')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = os.path.join(data_dir, 'train.csv')\ntrain_df = pd.read_csv(train)  \nprint(\"data_train_csv : \" + str(train_df.shape[0]))\n\n\nsub = os.path.join(data_dir,'sample_submission.csv')\nsub_df = pd.read_csv(sub)\nprint(\"data_submission_csv : \" + str(sub_df.shape[0]) )\n\n\nlabel_cols = sub_df.columns[1:]\nlabels = train_df[label_cols].values\n\ntrain_images = path_dir + \"/train/\" + train_df['StudyInstanceUID'] + '.jpg'   \ntest_images = path_dir + \"/test/\" + sub_df['StudyInstanceUID'] + '.jpg'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_cols.shape[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE=8*REPLICAS \nSTEPS_PER_EPOCH = len(train_df) * 0.8 / BATCH_SIZE\nVALIDATION_STEPS = len(train_df) * 0.2 / BATCH_SIZE\nEPOCHS = 30\nTARGET_SIZE = 750\nN_LABELS=label_cols.shape[0]\n\ndef build_decoder(with_labels = True,\n                  target_size = (TARGET_SIZE, TARGET_SIZE), \n                  ext = 'jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels = 3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels = 3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels = True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        img = tf.image.random_brightness(img,0.2)\n        img = tf.image.random_contrast(img,0.2,0.5)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels = None, bsize = BATCH_SIZE, cache = True,\n                  decode_fn = None, augment_fn = None,\n                  augment = True, repeat = True, shuffle = 1024, \n                  cache_dir = \"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls = AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls = AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img, valid_img, train_labels, valid_labels = train_test_split(train_images, labels, train_size = 0.8, random_state = 42)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_labels.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = build_dataset(\n     train_img, train_labels, bsize = BATCH_SIZE, \n     cache = True)\n\nvalid_df = build_dataset(\n     valid_img, valid_labels, bsize = BATCH_SIZE, \n     repeat = False, shuffle = False, augment = False, \n     cache = True)\n\ntest_df = build_dataset(\n    test_images, bsize = BATCH_SIZE, repeat = False, \n    shuffle = False, augment = False, cache = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionResNetV2,Xception,ResNet152,EfficientNetB7,NASNetLarge\n\ndef create_model_Xception():\n    conv_base = Xception(include_top = False, weights = 'imagenet',\n                          input_shape = (TARGET_SIZE, TARGET_SIZE, 3))\n    model = conv_base.output\n    print(conv_base.output.shape)\n    model = layers.GlobalAveragePooling2D()(model)\n    model = layers.Dropout(0.25)(model)\n    model = layers.Dense(11, activation = \"sigmoid\")(model)\n    model = models.Model(conv_base.input, model)\n\n    model.compile(optimizer = Adam(lr = 0.0001),\n                   loss = \"binary_crossentropy\",\n                   metrics = [tf.keras.metrics.AUC(multi_label = True)])\n    return model\n\ndef create_model_InceptionResNetV2():\n    conv_base = InceptionResNetV2(include_top = False, weights = 'imagenet',\n                          input_shape = (TARGET_SIZE, TARGET_SIZE, 3))\n    model = conv_base.output\n    print(conv_base.output.shape)\n\n    model = layers.GlobalAveragePooling2D()(model)\n    model = layers.Dropout(0.25)(model)\n    model = layers.Dense(11, activation = \"sigmoid\")(model)\n    model = models.Model(conv_base.input, model)\n\n    model.compile(optimizer = Adam(lr = 0.0001),\n                   loss = \"binary_crossentropy\",\n                   metrics = [tf.keras.metrics.AUC(multi_label = True)])\n    return model\n\n# def create_model_ResNet152():\n#     conv_base = ResNet152(include_top = False, weights = 'imagenet',\n#                           input_shape = (TARGET_SIZE, TARGET_SIZE, 3))\n#     model = conv_base.output\n#     model = layers.GlobalAveragePooling2D()(model)\n#     model = layers.Dropout(0.25)(model)\n#     model = layers.Dense(11, activation = \"sigmoid\")(model)\n#     model = models.Model(conv_base.input, model)\n\n#     model.compile(optimizer = Adam(lr = 0.001),\n#                    loss = \"binary_crossentropy\",\n#                    metrics = [tf.keras.metrics.AUC(multi_label = True)])\n#     return model\n\n# def create_model_EfficientNetB7():\n#     conv_base = EfficientNetB7(include_top = False, weights = 'imagenet',\n#                           input_shape = (TARGET_SIZE, TARGET_SIZE, 3))\n#     model = conv_base.output\n#     model = layers.GlobalAveragePooling2D()(model)\n#     model = layers.Dropout(0.25)(model)\n#     model = layers.Dense(11, activation = \"sigmoid\")(model)\n#     model = models.Model(conv_base.input, model)\n\n#     model.compile(optimizer = Adam(lr = 0.001),\n#                    loss = \"binary_crossentropy\",\n#                    metrics = [tf.keras.metrics.AUC(multi_label = True)])\n#     return model\n\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model_Xception = create_model_Xception()\n    model_InceptionResNetV2 = create_model_InceptionResNetV2()\n    model_Xception.load_weights('../input/ranzcr-models/model_Xception_best.h5')\n    model_InceptionResNetV2.load_weights('../input/ranzcr-models/model_InceptionResNetV2_best.h5')\n\n#     model_ResNet152 = create_model_ResNet152()\n#     model_EfficientNetB7 = create_model_EfficientNetB7()\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping,ReduceLROnPlateau\n\nearly_stop = EarlyStopping(monitor = 'val_loss', min_delta = 0.0001, \n                           patience = 5, mode = 'min', verbose = 1,\n                           restore_best_weights = True)\n\nreduce_lr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.4, \n                              patience = 2, min_delta = 0.0001, \n                              mode = 'min', verbose = 1)\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_Xception.save('./model_Xception.h5')\nmodel_Xception_save = ModelCheckpoint('./model_Xception_best.h5', \n                             save_best_only = True, \n                             save_weights_only = True,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\n\n\nmodel_InceptionResNetV2.save('./model_InceptionResNetV2.h5')\nmodel_InceptionResNetV2_save = ModelCheckpoint('./model_InceptionResNetV2_best.h5', \n                             save_best_only = True, \n                             save_weights_only = True,\n                             monitor = 'val_loss', \n                             mode = 'min', verbose = 1)\n\n\n# model_ResNet152.save('./model_ResNet152.h5')\n# model_ResNet152_save = ModelCheckpoint('./model_ResNet152_best.h5', \n#                              save_best_only = True, \n#                              save_weights_only = True,\n#                              monitor = 'val_loss', \n#                              mode = 'min', verbose = 1)\n\n# model_EfficientNetB7.save('./model_EfficientNetB7.h5')\n# model_EfficientNetB7_save = ModelCheckpoint('./model_EfficientNetB7_best.h5', \n#                              save_best_only = True, \n#                              save_weights_only = True,\n#                              monitor = 'val_loss', \n#                              mode = 'min', verbose = 1)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_model_Xception = model_Xception.fit(train_df,\n                    epochs=EPOCHS,\n                    steps_per_epoch = STEPS_PER_EPOCH,\n                    validation_data=valid_df,\n                    validation_steps = VALIDATION_STEPS,\n                    callbacks = [model_Xception_save,early_stop,reduce_lr]\n                   )\n\nhistory_model_InceptionResNetV2 = model_InceptionResNetV2.fit(train_df,\n                    epochs=EPOCHS,\n                    steps_per_epoch = STEPS_PER_EPOCH,\n                    validation_data=valid_df,\n                    validation_steps = VALIDATION_STEPS,\n                    callbacks = [model_InceptionResNetV2_save,early_stop,reduce_lr]\n                   )\n# history_model_ResNet152 = model_ResNet152.fit(train_df,\n#                     epochs=EPOCHS,\n#                     steps_per_epoch = STEPS_PER_EPOCH,\n#                     validation_data=valid_df,\n#                     validation_steps = VALIDATION_STEPS,\n#                     callbacks = [model_ResNet152_save,early_stop,reduce_lr]\n#                    )\n# history_model_EfficientNetB7 = model_EfficientNetB7.fit(train_df,\n#                     epochs=EPOCHS,\n#                     steps_per_epoch = STEPS_PER_EPOCH,\n#                     validation_data=valid_df,\n#                     validation_steps = VALIDATION_STEPS,\n#                     callbacks = [model_EfficientNetB7_save,early_stop,reduce_lr]\n#                    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(history_model_Xception.history.keys())\nprint(history_model_InceptionResNetV2.history.keys())\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(history_model_Xception.history['loss'])\nplt.plot(history_model_Xception.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(history_model_Xception.history['auc'])\nplt.plot(history_model_Xception.history['val_auc'])\n\nplt.title('Xception AUC')\nplt.ylabel('AUC score')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(history_model_InceptionResNetV2.history['loss'])\nplt.plot(history_model_InceptionResNetV2.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nplt.plot(history_model_InceptionResNetV2.history['auc_1'])\nplt.plot(history_model_InceptionResNetV2.history['val_auc_1'])\n\nplt.title('InceptionResNetV2 AUC')\nplt.ylabel('AUC score')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v1 = valid_labels.reshape(66187,)\ndef getaccuracy(r1):\n    count=0\n    for i in range(len(r1)):\n        if (r1[i]<0.5 and v1[i]<0.5) or (r1[i]>=0.5 and v1[i]>=0.5):\n            count+=1\n    return count/len(r1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nresult = model_Xception.predict(valid_df)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(getaccuracy(result.reshape(66187,)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result2 = model_InceptionResNetV2.predict(valid_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(getaccuracy(result2.reshape(66187,)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = [model_Xception,model_InceptionResNetV2]\nyhats = [model.predict(valid_df) for model in models]\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yhats=np.array(yhats)\nsummed = np.sum(yhats, axis=0)\nresult=summed/2\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nv1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(getaccuracy(result.reshape(66187,)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ensemble accuracy: 0.9455","metadata":{}},{"cell_type":"code","source":"v1=valid_labels.reshape(66187,)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.metrics import roc_curve\nfpr_ensemble, tpr_ensemble, thresholds_ensemble = roc_curve(valid_labels.reshape(66187,), result.reshape(66187,))\nfrom sklearn.metrics import auc\nauc_ensemble = auc(fpr_ensemble, tpr_ensemble)\nplt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_ensemble, tpr_ensemble, label='Ensemble (area = {:.3f})'.format(auc_ensemble))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.savefig('Ensemble')\nplt.show()\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"whats=yhats\nweights=[0.1,0.3,0.5,0.7]\nweightedresults=np.zeros((4,6017,11))\nfor w in range(len(weights)):\n    for i in range(whats.shape[0]):\n        for j in range(whats.shape[1]):\n            for k in range(whats.shape[2]):\n                if i == 0:\n                    whats[i][j][k] *= weights[w]\n                else:\n                    whats[i][j][k] *= weights[(1-w)]\n    summed = np.sum(whats, axis=0)\n    weightedresults[w] = summed\n    whats=yhats\n    \n\n\n                    \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve\ny_pred_Xception = model_Xception.predict(valid_df).ravel()\n# fpr_keras, tpr_keras, thresholds_keras = roc_curve(y_test, y_pred_keras)\ny_pred_Incep = model_InceptionResNetV2.predict(valid_df).ravel()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_Xception.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_labels.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nvl.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fpr_Xception, tpr_Xception, thresholds_Xception = roc_curve(valid_labels.reshape(66187,), y_pred_Xception)\nfpr_Incep, tpr_Incep, thresholds_Incep = roc_curve(valid_labels.reshape(66187,), y_pred_Incep)\n\nfrom sklearn.metrics import auc\nauc_Xception = auc(fpr_Xception, tpr_Xception)\nauc_incep = auc(fpr_Incep,tpr_Incep)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_Xception, tpr_Xception, label='Xception (area = 0.976)')\nplt.plot(fpr_Incep, tpr_Incep, label='InceptionResNetV2 (area = 0.973)')\nplt.plot(fpr_ensemble, tpr_ensemble, label='Ensemble (area = {:.3f})'.format(auc_ensemble))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.savefig('AUCfinal')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result_Xception =  model_Xception.predict(test_df, verbose = 1)\n# result_InceptionResNetV2 =  model_InceptionResNetV2.predict(test_df, verbose = 1)\n# # result_ResNet152 =  model_ResNet152.predict(test_df, verbose = 1)\n# # result_EfficientNetB7 =  model_EfficientNetB7.predict(test_df, verbose = 1)\n# result = (result_Xception+result_InceptionResNetV2)/2\n\n# sub_df[label_cols] = result\n# sub_df.to_csv('submission.csv', index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"done!\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}