{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom matplotlib import image\nfrom matplotlib import pyplot\nimport os\nimport cv2\nimport random\nimport concurrent.futures\nimport time\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom itertools import cycle\nfrom sklearn import svm, datasets\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.multiclass import OneVsRestClassifier\nfrom scipy import interp\nfrom sklearn.metrics import roc_auc_score\nimport datetime\n\nprint(tf.__version__)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/ranzcr-clip-catheter-line-classification/train.csv\", dtype=str)\ntrain['image_names'] = [i+\".jpg\" for i in train['StudyInstanceUID'].values]\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = [[int(train['ETT - Abnormal'][i]),int(train['ETT - Borderline'][i]), \\\n           int(train['ETT - Normal'][i]),int(train['NGT - Abnormal'][i]), \\\n           int(train['NGT - Borderline'][i]),int(train['NGT - Incompletely Imaged'][i]),\\\n           int(train['NGT - Normal'][i]),int(train['CVC - Abnormal'][i]),\\\n           int(train['CVC - Borderline'][i]),int(train['CVC - Normal'][i]),\\\n           int(train['Swan Ganz Catheter Present'][i])] for i in range(len(train))]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_names = train[\"image_names\"]\n\ndef myfunc():\n    return 0.5\nc = list(zip(img_names, labels))\nrandom.shuffle(c, myfunc)\nimg_names, labels = zip(*c)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMSIZE = 256\ndef read_img(image):\n    img = tf.keras.preprocessing.image.load_img(image, color_mode='rgb', target_size=(IMSIZE, IMSIZE))\n    return img\ndef prepare_dataset(namelist, labels, path):\n    start = time.time()\n    labels = np.array(labels)\n    labels = tf.convert_to_tensor(labels)\n    labels = tf.cast(labels, tf.int8)\n    namelist = [os.path.join(path, ele) for ele in namelist]\n    imgs = []\n    with concurrent.futures.ThreadPoolExecutor(max_workers = 8) as executor:\n        i = 0\n        for value in executor.map(read_img, namelist):\n            i+=1\n            print(\"\\rFetching: [{}/{}]\".format(i, len(namelist)), end=\"\", flush=True)\n            imgs.append(value)\n        imgs = np.stack(imgs)\n        imgs = tf.convert_to_tensor(imgs)\n    print(\"\\nExecution time: \",time.time() - start, \"s\")\n    return imgs, labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with tf.device('/cpu:0'):\n    path = '../input/ranzcr-clip-catheter-line-classification/train'\n    OFFSET = 10240\n    TRAIN_SIZE = 6400\n    VAL_SIZE = 2560\n    train_images, train_labels = prepare_dataset(img_names[OFFSET:OFFSET+TRAIN_SIZE], \\\n                                                 labels[OFFSET:OFFSET+TRAIN_SIZE], path)\n    val_images, val_labels = prepare_dataset(img_names[OFFSET+TRAIN_SIZE:OFFSET+VAL_SIZE+TRAIN_SIZE], \\\n                                             labels[OFFSET+TRAIN_SIZE:OFFSET+VAL_SIZE+TRAIN_SIZE], path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Training Image tensor shape\", train_images.shape)\nprint(\"Training Labels tensor shape\", train_labels.shape)\nprint(\"Testing Image tensor shape\", val_images.shape)\nprint(\"Tesing Labels tensor shape\", val_labels.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"resolver = tf.distribute.cluster_resolver.TPUClusterResolver(tpu='')\ntf.config.experimental_connect_to_cluster(resolver)\ntf.tpu.experimental.initialize_tpu_system(resolver)\nprint(\"All devices: \", tf.config.list_logical_devices('TPU'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEED = 1000\nrandom_rotation = tf.keras.layers.experimental.preprocessing.RandomRotation(3.142/2, seed=SEED)\nrandom_flip = tf.keras.layers.experimental.preprocessing.RandomFlip(mode=\"horizontal_and_vertical\", seed=SEED)\nrandom_zoom = tf.keras.layers.experimental.preprocessing.RandomZoom((0, 0.25), seed=SEED)\nrandom_translate = tf.keras.layers.experimental.preprocessing.RandomTranslation((-0, 0.25), (-0, 0.25), seed=SEED)\n\ndef preprocess(imgs, label):\n    imgs = random_rotation.call(imgs)\n    imgs = random_flip.call(imgs)\n    imgs = random_zoom.call(imgs)\n    imgs = random_translate.call(imgs)\n    return imgs, label\n\ndef normalize(imgs, label):\n    return tf.cast(imgs, tf.float16)/255, label\n\nstrategy = tf.distribute.experimental.TPUStrategy(resolver)\nwith tf.device('/cpu:0'):\n    TRAIN_BATCH_SIZE = 16 * strategy.num_replicas_in_sync\n    VAL_BATCH_SIZE = 8 * strategy.num_replicas_in_sync\n    train_dataset = tf.data.Dataset.from_tensor_slices((tf.cast(train_images, tf.uint8), \\\n                                                        tf.cast(train_labels, tf.uint8)))\\\n                    .shuffle(TRAIN_SIZE).repeat().batch(TRAIN_BATCH_SIZE).map(preprocess, num_parallel_calls=tf.data.experimental.AUTOTUNE)\\\n                    .cache()\n    del train_images, train_labels\n    val_dataset = tf.data.Dataset.from_tensor_slices((tf.cast(val_images, tf.uint8), tf.cast(val_labels, tf.uint8))).repeat().batch(VAL_BATCH_SIZE)\n    del val_images, val_labels\n    train_dataset = train_dataset.map(normalize, num_parallel_calls=tf.data.experimental.AUTOTUNE).prefetch(tf.data.AUTOTUNE)\n    val_dataset = val_dataset.map(normalize, num_parallel_calls=tf.data.experimental.AUTOTUNE).prefetch(tf.data.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_test = []\ni = 0\nfor x, y in val_dataset:\n    i+=1\n    if(i > VAL_SIZE/VAL_BATCH_SIZE):\n        break\n    print('\\r',i, end='')\n    for ele in y:\n        y_test.append(list(ele.numpy()))\ny_test = np.array(y_test)\n\ndef roc_auc_plot(y_test, y_score):\n    lw = 1\n    # Compute ROC curve and ROC area for each class\n    fpr = dict()\n    tpr = dict()\n    roc_auc = dict()\n    n_classes = y_score.shape[1]\n    for i in range(n_classes):\n        fpr[i], tpr[i], _ = roc_curve(y_test[:, i], y_score[:, i])\n        roc_auc[i] = auc(fpr[i], tpr[i])\n\n    # Compute micro-average ROC curve and ROC area\n    fpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(y_test.ravel(), y_score.ravel())\n    roc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\n\n    # First aggregate all false positive rates\n    all_fpr = np.unique(np.concatenate([fpr[i] for i in range(n_classes)]))\n\n    # Then interpolate all ROC curves at this points\n    mean_tpr = np.zeros_like(all_fpr)\n    for i in range(n_classes):\n        mean_tpr += np.interp(all_fpr, fpr[i], tpr[i])\n\n    # Finally average it and compute AUC\n    mean_tpr /= n_classes\n\n    fpr[\"macro\"] = all_fpr\n    tpr[\"macro\"] = mean_tpr\n    roc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\n\n    # Plot all ROC curves\n    plt.figure()\n    plt.plot(fpr[\"micro\"], tpr[\"micro\"],\n             label='micro-average ROC curve (area = {0:0.2f})'\n                   ''.format(roc_auc[\"micro\"]),\n             color='deeppink', linestyle=':', linewidth=4)\n\n    plt.plot(fpr[\"macro\"], tpr[\"macro\"],\n             label='macro-average ROC curve (area = {0:0.2f})'\n                   ''.format(roc_auc[\"macro\"]),\n             color='navy', linestyle=':', linewidth=4)\n\n    colors = cycle(['aqua', 'darkorange', 'cornflowerblue'])\n    for i, color in zip(range(n_classes), colors):\n        plt.plot(fpr[i], tpr[i], color=color, lw=lw,\n                 label='ROC curve of class {0} (area = {1:0.2f})'\n                 ''.format(i, roc_auc[i]))\n\n    plt.plot([0, 1], [0, 1], 'k--', lw=lw)\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Some extension of Receiver operating characteristic to multi-class')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n    return roc_auc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%load_ext tensorboard\n\nIMSIZE = 256\nwith strategy.scope():\n    base_model = tf.keras.applications.InceptionResNetV2(include_top=False,\\\n                                                   weights='imagenet', pooling = 'max')\n    base_model.trainable = True\n    for layer in base_model.layers[:150]:\n        layer.trainable = False\n    model = tf.keras.Sequential([\n        tf.keras.layers.Input(shape=(IMSIZE, IMSIZE, 3)),\n        #tf.keras.layers.experimental.preprocessing.RandomRotation(3.142/2, seed=SEED),\n        #tf.keras.layers.experimental.preprocessing.RandomFlip(mode=\"horizontal_and_vertical\", seed=SEED),\n        #tf.keras.layers.experimental.preprocessing.RandomZoom((0, 0.25), seed=SEED),\n        #tf.keras.layers.experimental.preprocessing.RandomTranslation((-0, 0.25), (-0, 0.25), seed=SEED),\n        base_model,\n        tf.keras.layers.Dense(11, activation='sigmoid')\n    ])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class myCallback_EarlyStopping(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs = {}): \n        print(\"Predict: \", end='')\n        y_score = model.predict(val_dataset, batch_size=VAL_BATCH_SIZE, steps = int(VAL_SIZE/VAL_BATCH_SIZE), verbose=1)\n        THRESH=0.5\n        y_pred = []\n        for li in y_score:\n            vec = []\n            for ele in li:\n                if(ele >= THRESH):\n                    vec.append(1)\n                else:\n                    vec.append(0)\n            y_pred.append(vec)\n        y_score = np.array(y_pred)\n        roc = roc_auc_plot(y_test, y_score)\n        if(roc['macro']>0.75):\n            model.save('./macro.h5')\n        if(roc['micro']>0.75):\n            model.save('./micro.h5')\n        elif(roc['macro']>0.85):\n            print(\"\\n Validation Macro-Avg AUC of 90% has reached!\")\n            model.save('./best_macro.h5')\n            self.model.stop_training = True\n        \ncallback_EarlyStopping = myCallback_EarlyStopping()\nlog_dir = \"./logs/fit/\" + datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\")\ntensorboard_callback = tf.keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1)    \nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath='./mobilenetv2-1.h5',\n    save_weights_only=False,\n    monitor='val_auc',\n    mode='max',\n    save_best_only=True)\nlr_decay_plateau = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.8,\n    patience=5,\n    verbose=1,\n    mode=\"min\",\n)\ndist_train_dataset = strategy.experimental_distribute_dataset(train_dataset)\ndist_val_dataset = strategy.experimental_distribute_dataset(val_dataset)\nmetric_auc = tf.keras.metrics.AUC(num_thresholds=200, multi_label=True, name='auc')\nmodel.compile(optimizer=tf.keras.optimizers.Adam(0.0001),\\\n              loss=tf.keras.losses.BinaryCrossentropy(), \\\n              metrics=['acc',metric_auc])\n#%tensorboard --logdir ./logs/fit\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(train_dataset, steps_per_epoch = int(TRAIN_SIZE/TRAIN_BATCH_SIZE), \\\n                    validation_data=val_dataset, validation_steps=int(VAL_SIZE/VAL_BATCH_SIZE),\\\n                    epochs=100, callbacks=[model_checkpoint_callback, callback_EarlyStopping, lr_decay_plateau])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./mobilenetv2.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.models.load_model('./mobilenetv2-1.h5', compile=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dist_val_dataset = strategy.experimental_distribute_dataset(val_dataset)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_score=model.predict(val_dataset, batch_size=VAL_BATCH_SIZE, steps = int(VAL_SIZE/VAL_BATCH_SIZE), verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_test = []\ni = 0\nfor x, y in val_dataset:\n    i+=1\n    if(i > VAL_SIZE/VAL_BATCH_SIZE):\n        break\n    print('\\r',i, end='')\n    for ele in y:\n        y_test.append(list(ele.numpy()))\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"THRESH=0.5\ny_pred = []\nfor li in y_score:\n    vec = []\n    for ele in li:\n        if(ele >= THRESH):\n            vec.append(1)\n        else:\n            vec.append(0)\n    y_pred.append(vec)\ny_score = np.array(y_pred)\ny_test = np.array(y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(y_score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc = roc_auc_plot(y_test, y_score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(roc['macro'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./mobilenetv2.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def roc_auc_plot(y_test, y_score):\n    lw = 1\n    # Compute ROC curve and ROC area for each class\n    fpr = dict()\n    tpr = dict()\n    roc_auc = dict()\n    n_classes = y_score.shape[1]\n    for i in range(n_classes):\n        fpr[i], tpr[i], _ = roc_curve(y_test[:, i], y_score[:, i])\n        roc_auc[i] = auc(fpr[i], tpr[i])\n    # Compute micro-average ROC curve and ROC area\n    fpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(y_test.ravel(), y_score.ravel())\n    roc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\n\n    # First aggregate all false positive rates\n    all_fpr = np.unique(np.concatenate([fpr[i] for i in range(n_classes)]))\n\n    # Then interpolate all ROC curves at this points\n    mean_tpr = np.zeros(len(all_fpr))\n    for i in range(n_classes):\n        mean_tpr += np.interp(all_fpr, fpr[i], tpr[i])\n\n    # Finally average it and compute AUC\n    mean_tpr /= n_classes\n\n    fpr[\"macro\"] = all_fpr\n    tpr[\"macro\"] = mean_tpr\n    roc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\n\n    # Plot all ROC curves\n    plt.figure()\n    plt.plot(fpr[\"micro\"], tpr[\"micro\"],\n             label='micro-average ROC curve (area = {0:0.2f})'\n                   ''.format(roc_auc[\"micro\"]),\n             color='deeppink', linestyle=':', linewidth=4)\n\n    plt.plot(fpr[\"macro\"], tpr[\"macro\"],\n             label='macro-average ROC curve (area = {0:0.2f})'\n                   ''.format(roc_auc[\"macro\"]),\n             color='navy', linestyle=':', linewidth=4)\n\n    colors = cycle(['aqua', 'darkorange', 'cornflowerblue'])\n    for i, color in zip(range(n_classes), colors):\n        plt.plot(fpr[i], tpr[i], color=color, lw=lw,\n                 label='ROC curve of class {0} (area = {1:0.2f})'\n                 ''.format(i, roc_auc[i]))\n\n    plt.plot([0, 1], [0, 1], 'k--', lw=lw)\n    plt.xlim([0.0, 1.0])\n    plt.ylim([0.0, 1.05])\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Some extension of Receiver operating characteristic to multi-class')\n    plt.legend(loc=\"lower right\")\n    plt.show()\n    return roc_auc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./mobilenetv2.h5')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}