{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n'''\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow import keras\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tensorflow_addons as tfa\nfrom sklearn import model_selection\nfrom sklearn import metrics\nimport tensorflow_hub as hub\nimport pandas as pd\nfrom tensorflow.keras.preprocessing import image\nimport glob\nimport random\n# because jupyter doesn't make auto completions for me I don't know why\n%config Completer.use_jedi = False\n\n#Please Upvote :) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df.set_index(train_df.StudyInstanceUID) #for easier search for images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_test, y_train, y_test = model_selection.train_test_split(train_df.iloc[:,0].values, train_df.iloc[:, 1:-1].values,train_size=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#generator to get batches of data\nclass RanzcrDataGenerator(keras.utils.Sequence):\n    def __init__(self, data_path, x, y, target_shape, batch_size):\n        self.data_path = data_path\n        self.batch_size = batch_size\n        self.target_shape = target_shape\n        self.x = x\n        self.y = y\n        self.images = [os.path.join(self.data_path, curr_img+'.jpg') for curr_img in x]\n        self.dataset_length = len(x)\n    \n    def __len__(self):\n        return self.dataset_length // self.batch_size\n    \n    def __getitem__(self, index):\n        idx = index * self.batch_size\n        imgs_batch = self.images[idx: idx+self.batch_size]\n        labels = self.y[idx: idx+self.batch_size]\n        decoded_batch = self.decode(imgs_batch).astype('float32')\n        return decoded_batch, labels.astype('int32')\n    \n    def decode(self, batch):\n        decoded_batch = np.zeros((self.batch_size,) + self.target_shape + (3,))\n        for i, current_img in enumerate(batch):\n            decoded_batch[i] = image.img_to_array(image.load_img(current_img, target_size=self.target_shape))\n        return decoded_batch","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = RanzcrDataGenerator('../input/ranzcr-clip-catheter-line-classification/train/', x=x_train, y=y_train, \n                                    batch_size=64, target_shape=(200, 200))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_dataset = RanzcrDataGenerator('../input/ranzcr-clip-catheter-line-classification/train/', x=x_test, y=y_test, \n                                  batch_size=64, target_shape=(200, 200))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Big Transfer Model\n#check out https://blog.tensorflow.org/2020/05/bigtransfer-bit-state-of-art-transfer-learning-computer-vision.html\nmodule_handle='https://tfhub.dev/google/bit/m-r152x4/1'\nmodule=hub.KerasLayer(module_handle)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_augmentation = keras.Sequential([layers.experimental.preprocessing.RandomFlip(\"horizontal\"),\n        layers.experimental.preprocessing.RandomRotation(0.1),])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class BiTModel(keras.Model):\n    def __init__(self, module, num_classes, activation, augmentation=None):\n        super(BiTModel, self).__init__()\n        self.num_classes = num_classes\n        self.module = module\n        self.head = layers.Dense(num_classes, kernel_initializer='zeros')\n        self.augmentation=augmentation\n        self.activation=keras.activations.get(activation)\n    def call(self, inputs):\n        if self.augmentation:\n            inputs = self.augmentation(inputs)\n        inputs = self.module(inputs)\n        return self.activation(self.head(inputs))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.backend.clear_session() #to free up ram\nmodel = BiTModel(module, 11, 'sigmoid', data_augmentation)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# Define optimiser and loss\n# Decay learning rate by factor of 10 at SCHEDULE_BOUNDARIES.\nlr = 0.003\nSCHEDULE_BOUNDARIES = [200, 300, 400]\nlr_schedule = tf.keras.optimizers.schedules.PiecewiseConstantDecay(boundaries=SCHEDULE_BOUNDARIES,\n                                                                  values=[lr, lr*0.1, lr*0.001, lr*0.0001])\noptimizer = tf.keras.optimizers.SGD(learning_rate=lr_schedule, momentum=0.9)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"loss_fn = tf.keras.losses.BinaryCrossentropy()\nmodel.compile(optimizer=optimizer,\n             loss=loss_fn,\n             metrics=[keras.metrics.AUC()])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"callbacks = [keras.callbacks.EarlyStopping(patience=4), keras.callbacks.ModelCheckpoint(\"chest_xray_classification.h5\", save_best_only=True)]\nmodel.fit(train_dataset,\n   epochs=15, validation_data=val_dataset, callbacks=callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_test_images(data_path, model):\n    predictions = dict()\n    images = glob.glob(data_path)\n    for i, img in enumerate(images):\n        img_name = img.split('/')[-1][:-4]\n        decoded_img = np.expand_dims(image.load_img(img, target_size=(200, 200)), 0)\n        preds = model(decoded_img)\n        preds = (preds > 0.5).numpy().astype('int32')\n        predictions[img_name] = preds\n        if i % 50 == 0:\n            print('Finished 50 imgs')\n    return predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = get_test_images('../input/ranzcr-clip-catheter-line-classification/test/*', model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ids = list(preds.keys())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_data = np.array(list(preds.values()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_data = np.squeeze(predicted_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame({'StudyInstanceUID': x, \n                           'ETT - Abnormal': predicted_data[:, 0], 'ETT - Borderline': predicted_data[:, 1], \n                           'ETT - Normal': predicted_data[:, 2], 'NGT - Abnormal': predicted_data[:, 3], \n                           'NGT - Borderline': predicted_data[:, 4],\n                          'NGT - Incompletely Imaged': predicted_data[:,5], 'NGT - Normal': predicted_data[:, 6], \n                           'CVC - Abnormal': predicted_data[:, 7], \n                          'CVC - Borderline': predicted_data[:, 8], 'CVC - Normal': predicted_data[:, 9], \n                           'Swan Ganz Catheter Present': predicted_data[:, 10]})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}