{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import BatchNormalization, Conv2D, MaxPool2D, Conv3D, MaxPool3D, UpSampling2D, GlobalMaxPool2D, GlobalMaxPool3D, GlobalAveragePooling2D, GlobalAveragePooling3D, Conv2DTranspose, concatenate\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Reshape, Flatten, Input\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.utils import to_categorical, plot_model\n\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications import NASNetMobile, Xception, DenseNet121, MobileNetV2, InceptionV3, InceptionResNetV2, vgg16, resnet50, inception_v3, xception, DenseNet201\n\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.callbacks import CSVLogger\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras import metrics\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport sklearn\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import jaccard_score\nfrom sklearn.cluster import KMeans\n\nfrom scipy import stats\n\nimport seaborn as sns\n\nimport skimage\nfrom skimage.transform import rotate\n\nfrom tqdm.notebook import tqdm\nfrom datetime import datetime\n\nfrom sklearn.metrics import f1_score, recall_score, precision_score, accuracy_score, roc_auc_score, roc_curve\nimport numpy as np\nimport os\nimport cv2\nimport pandas as pd\n# import imutils\nimport random\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nimport pickle\nimport urllib\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom PIL import Image\n\nimport tensorflow_addons as tfa\n\nfrom IPython.display import HTML\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_color_lut\nimport re\n\nimport itertools\n\nfrom sklearn.utils import shuffle\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-14T16:56:21.175096Z","iopub.execute_input":"2021-10-14T16:56:21.175491Z","iopub.status.idle":"2021-10-14T16:56:21.190178Z","shell.execute_reply.started":"2021-10-14T16:56:21.175459Z","shell.execute_reply":"2021-10-14T16:56:21.187634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define","metadata":{}},{"cell_type":"code","source":"im_size = (256,256)\nnum_im = 64\nbatch_size = 48\nsplit_size = (0.7,0.3,0.0)\nrandom_state = 42\nimage_threshold = 0.5\ntest_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/test'","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.191786Z","iopub.execute_input":"2021-10-14T16:56:21.192527Z","iopub.status.idle":"2021-10-14T16:56:21.220167Z","shell.execute_reply.started":"2021-10-14T16:56:21.192487Z","shell.execute_reply":"2021-10-14T16:56:21.21938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Images","metadata":{}},{"cell_type":"code","source":"def index_train_valid_test_split(data, is_shuffle=True, split_size=split_size):\n    \n    length = len(data)\n    train_end = int(split_size[0] * length)\n    \n    valid_start = int(train_end)\n    valid_end = int((split_size[0] + split_size[1]) * length)\n    \n    test_start = int(valid_end)\n    \n    if is_shuffle:\n        data = shuffle(data)\n        \n    return data[:train_end], data[valid_start:valid_end], data[test_start:]","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.223182Z","iopub.execute_input":"2021-10-14T16:56:21.224475Z","iopub.status.idle":"2021-10-14T16:56:21.232912Z","shell.execute_reply.started":"2021-10-14T16:56:21.224415Z","shell.execute_reply":"2021-10-14T16:56:21.232187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\ndf_test['BraTS21ID'] = df_test['BraTS21ID'].apply(lambda x: f'{\"0\"*(5-len(str(x)))}{x}')\ndf_test","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.235Z","iopub.execute_input":"2021-10-14T16:56:21.235628Z","iopub.status.idle":"2021-10-14T16:56:21.257273Z","shell.execute_reply.started":"2021-10-14T16:56:21.23559Z","shell.execute_reply":"2021-10-14T16:56:21.25664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(im, im_size=im_size):\n    im = cv2.resize(im, dsize=im_size)\n    return im\n\ndef scale_image(im):\n    max_im = np.max(im)\n    min_im = np.min(im)\n    \n    if max_im == min_im:\n        return im - min_im\n    \n    return (im - min_im) / (max_im - min_im)\n\ndef get_image(path, im_size=im_size):\n    ds = pydicom.dcmread(path)\n\n    im = ds.pixel_array\n\n    im = scale_image(im)\n    \n    im = preprocess(im, im_size)\n    \n    return np.array(im)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.259289Z","iopub.execute_input":"2021-10-14T16:56:21.259475Z","iopub.status.idle":"2021-10-14T16:56:21.266557Z","shell.execute_reply.started":"2021-10-14T16:56:21.259453Z","shell.execute_reply":"2021-10-14T16:56:21.265586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_id_image(scan_id, types, path_dir=test_dir, im_size=im_size, num_im=num_im, black_threshold=0.8, is_tqdm=True):\n    files = sorted(os.listdir(f'{path_dir}/{scan_id}/{types}'),\n               key=lambda x: int(''.join(re.findall(r'[0-9]+', x))))\n    \n    im_list = []\n    \n    step = len(files) / num_im\n    \n    files = [files[int(i*step)] for i in range(num_im)]\n    \n    if is_tqdm:\n        files = tqdm(files)\n        \n    for file in files:\n        file_path = f'{path_dir}/{scan_id}/{types}/{file}'\n        \n        im = get_image(file_path, im_size=im_size)\n            \n        im_list += [im]\n        \n    im_list = np.moveaxis(np.array(im_list), 0, -1)\n         \n    im_list = im_list.reshape(*im_list.shape, 1)\n    \n    return tf.image.convert_image_dtype(im_list, tf.float32)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.268332Z","iopub.execute_input":"2021-10-14T16:56:21.268776Z","iopub.status.idle":"2021-10-14T16:56:21.279982Z","shell.execute_reply.started":"2021-10-14T16:56:21.268733Z","shell.execute_reply":"2021-10-14T16:56:21.279022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"with tf.device('/device:GPU:0'):\n\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\n# with tpu_strategy.scope():\n    def get_model():\n        inputs = Input(shape=(*im_size, num_im, 1))\n    \n        x = Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n        x = MaxPool3D(pool_size=2)(x)\n        x = BatchNormalization()(x)\n\n        x = Conv3D(filters=64, kernel_size=3, activation=\"relu\")(x)\n        x = MaxPool3D(pool_size=2)(x)\n        x = BatchNormalization()(x)\n\n        x = Conv3D(filters=128, kernel_size=3, activation=\"relu\")(x)\n        x = MaxPool3D(pool_size=2)(x)\n        x = BatchNormalization()(x)\n\n        x = Conv3D(filters=256, kernel_size=3, activation=\"relu\")(x)\n        x = MaxPool3D(pool_size=2)(x)\n        x = BatchNormalization()(x)\n\n        x = GlobalAveragePooling3D()(x)\n        x = Dense(units=512, activation=\"relu\")(x)\n        x = Dropout(0.3)(x)\n\n        outputs = Dense(units=1, activation=\"sigmoid\")(x)\n\n        model = Model(inputs=inputs, outputs=outputs, name='3dcnn')\n        model.compile(loss='binary_crossentropy',\n                      optimizer='adam', \n                      metrics=['accuracy', tf.keras.metrics.AUC(name='auc')])\n\n        return model\n\n    get_model().summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.281714Z","iopub.execute_input":"2021-10-14T16:56:21.2821Z","iopub.status.idle":"2021-10-14T16:56:21.775644Z","shell.execute_reply.started":"2021-10-14T16:56:21.282062Z","shell.execute_reply":"2021-10-14T16:56:21.774928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\nimport math\nclass Dataset(Sequence):\n    def __init__(self, df, types, is_train=True, batch_size=3, shuffle=True):\n        self.paths = df['BraTS21ID'].values\n        self.y =  df['MGMT_value'].values\n        self.is_train = is_train\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.types = types\n\n    def __len__(self):\n        return math.ceil(len(self.paths)/self.batch_size)\n\n    def __getitem__(self, ids):\n\n        batch_paths = self.paths[ids * self.batch_size:(ids + 1) * self.batch_size]\n\n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n\n        list_x =  [get_id_image(id_path, self.types, is_tqdm=False) for id_path in batch_paths]\n\n        batch_X = np.stack(list_x)\n\n        if self.is_train:\n            return batch_X, batch_y\n        else:\n            return batch_X\n\n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.paths, self.y))\n            shuffle(ids_y)\n            self.paths, self.y = list(zip(*ids_y))","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.778378Z","iopub.execute_input":"2021-10-14T16:56:21.778595Z","iopub.status.idle":"2021-10-14T16:56:21.789595Z","shell.execute_reply.started":"2021-10-14T16:56:21.778562Z","shell.execute_reply":"2021-10-14T16:56:21.788897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict Model","metadata":{}},{"cell_type":"code","source":"# with tpu_strategy.scope():\nwith tf.device('/device:GPU:0'):\n    model_FLAIR = get_model()\n    model_FLAIR.load_weights('../input/3d-brian-tumor-weight/best_3d_FLAIR_model_numim_64_AUC.h5')\n    \n    test_dataset_FLAIR = Dataset(df_test, 'FLAIR', is_train=False)\n\n    FLAIR_pred = model_FLAIR.predict(test_dataset_FLAIR)[:, 0]","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:56:21.791164Z","iopub.execute_input":"2021-10-14T16:56:21.791705Z","iopub.status.idle":"2021-10-14T16:57:14.801732Z","shell.execute_reply.started":"2021-10-14T16:56:21.791665Z","shell.execute_reply":"2021-10-14T16:57:14.800831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nplt.hist(FLAIR_pred);","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:59:06.500164Z","iopub.execute_input":"2021-10-14T16:59:06.500899Z","iopub.status.idle":"2021-10-14T16:59:06.730319Z","shell.execute_reply.started":"2021-10-14T16:59:06.50083Z","shell.execute_reply":"2021-10-14T16:59:06.729653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with tpu_strategy.scope():\nwith tf.device('/device:GPU:0'):\n    model_T1wCE = get_model()\n    model_T1wCE.load_weights('../input/3d-brian-tumor-weight/best_3d_T1wCE_model_numim_64_AUC.h5')\n    \n    test_dataset_T1wCE = Dataset(df_test, 'T1wCE', is_train=False)\n\n    T1wCE_pred = model_T1wCE.predict(test_dataset_T1wCE)[:, 0]","metadata":{"execution":{"iopub.status.busy":"2021-10-14T16:57:14.803155Z","iopub.execute_input":"2021-10-14T16:57:14.803415Z","iopub.status.idle":"2021-10-14T16:58:00.983892Z","shell.execute_reply.started":"2021-10-14T16:57:14.803382Z","shell.execute_reply":"2021-10-14T16:58:00.983051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nplt.hist(T1wCE_pred);","metadata":{"execution":{"iopub.status.busy":"2021-10-14T17:00:48.507809Z","iopub.execute_input":"2021-10-14T17:00:48.508371Z","iopub.status.idle":"2021-10-14T17:00:48.739524Z","shell.execute_reply.started":"2021-10-14T17:00:48.508332Z","shell.execute_reply":"2021-10-14T17:00:48.738777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_prediction = np.mean([FLAIR_pred, T1wCE_pred], axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T17:03:16.974674Z","iopub.execute_input":"2021-10-14T17:03:16.975402Z","iopub.status.idle":"2021-10-14T17:03:16.979285Z","shell.execute_reply.started":"2021-10-14T17:03:16.975362Z","shell.execute_reply":"2021-10-14T17:03:16.978404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nplt.hist(mean_prediction);","metadata":{"execution":{"iopub.status.busy":"2021-10-14T17:03:34.927143Z","iopub.execute_input":"2021-10-14T17:03:34.927417Z","iopub.status.idle":"2021-10-14T17:03:35.144784Z","shell.execute_reply.started":"2021-10-14T17:03:34.927389Z","shell.execute_reply":"2021-10-14T17:03:35.144094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\nsubmission['MGMT_value'] = T1wCE_pred\nsubmission","metadata":{"execution":{"iopub.status.busy":"2021-10-14T17:03:17.954771Z","iopub.execute_input":"2021-10-14T17:03:17.955362Z","iopub.status.idle":"2021-10-14T17:03:17.971259Z","shell.execute_reply.started":"2021-10-14T17:03:17.955325Z","shell.execute_reply":"2021-10-14T17:03:17.97062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}