{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I'm reading through several existing notebooks and trying to distill down the information into a new notebook to help me understand the project.  All help appreciated!\n\n# References\n\n- [Advanced EDA - Brain Tumor Data](https://www.kaggle.com/smoschou55/advanced-eda-brain-tumor-data)\n- [Team 9 Second Week](https://www.kaggle.com/evanyao27/team-9-second-week)\n  - The only model that is working. get_model02()\n- [Dataset to Model with Tensorflow](https://www.kaggle.com/ohbewise/dataset-to-model-with-tensorflow)\n- [Brain Tumer Train Class Flair](https://www.kaggle.com/lucamtb/brain-tumer-train-class-flair)\n  - Uses TPU\n  - Generates a Tensorflow model: Brain_flair_model_effect_3e-05_0.0001.h5\n- [Brain Tumor very basic inference](https://www.kaggle.com/lucamtb/brain-tumor-very-basice-inference)\n  - Uses the above mentioned model: Brain_flair_model_effect_3e-05_0.0001.h5\n  - Add this Kaggle Dataset: https://www.kaggle.com/lucamtb/effect0-brain","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\n\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\nimport random\nfrom tqdm import tqdm\nimport pydicom # Handle MRI images\n\nimport cv2  # OpenCV - https://docs.opencv.org/master/d6/d00/tutorial_py_root.html\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\n\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:29.732359Z","iopub.execute_input":"2021-10-15T18:04:29.732613Z","iopub.status.idle":"2021-10-15T18:04:34.712009Z","shell.execute_reply.started":"2021-10-15T18:04:29.732544Z","shell.execute_reply":"2021-10-15T18:04:34.711272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration, Constants, Setup","metadata":{}},{"cell_type":"code","source":"data_dir = Path('../input/rsna-miccai-brain-tumor-radiogenomic-classification/')\n\nmri_types = [\"FLAIR\", \"T1w\", \"T2w\", \"T1wCE\"]\nexcluded_images = [109, 123, 709] # Bad images","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:36.901091Z","iopub.execute_input":"2021-10-15T18:04:36.901466Z","iopub.status.idle":"2021-10-15T18:04:36.908153Z","shell.execute_reply.started":"2021-10-15T18:04:36.901421Z","shell.execute_reply":"2021-10-15T18:04:36.907484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Datasets","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(data_dir / \"train_labels.csv\",\n#                        index='id',\n#                       nrows=100000\n                      )\ntest_df = pd.read_csv(data_dir / \"sample_submission.csv\")\nsample_submission = pd.read_csv(data_dir / \"sample_submission.csv\")\n\ntrain_df = train_df[~train_df.BraTS21ID.isin(excluded_images)]\nval_df = train_df[-50:]\ntrain_df = train_df[:-50]\n\nprint(f\"train data: Rows={train_df.shape[0]}, Columns={train_df.shape[1]}\")\nprint(f\"val data: Rows={val_df.shape[0]}, Columns={val_df.shape[1]}\")\n\n# print(f\"test data : Rows={test_df.shape[0]}, Columns={test_df.shape[1]}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:37.471479Z","iopub.execute_input":"2021-10-15T18:04:37.472128Z","iopub.status.idle":"2021-10-15T18:04:37.510398Z","shell.execute_reply.started":"2021-10-15T18:04:37.472094Z","shell.execute_reply":"2021-10-15T18:04:37.509624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"markdown","source":"### There's a version that converts into grayscale: \n\n- https://www.kaggle.com/smoschou55/advanced-eda-brain-tumor-data\n","metadata":{}},{"cell_type":"code","source":"def load_dicom(path, size = 224):\n    ''' \n    Reads a DICOM image, standardizes so that the pixel values are between 0 and 1, then rescales to 0 and 255\n    \n    Not super sure if this kind of scaling is appropriate, but everyone seems to do it. \n    '''\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    # transform data into black and white scale / grayscale\n#     data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return cv2.resize(data, (size, size))","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:37.757486Z","iopub.execute_input":"2021-10-15T18:04:37.758056Z","iopub.status.idle":"2021-10-15T18:04:37.763828Z","shell.execute_reply.started":"2021-10-15T18:04:37.758025Z","shell.execute_reply":"2021-10-15T18:04:37.762929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_niifile(niifile,size=64,filtering=0.15):  # Read niifile file\n\n    img = nib.load(niifile)  # Download niifile file (actually extract the file)\n    img_fdata = img.get_fdata()  # Get niifile data\n\n#     if np.max(img_fdata) != 0:\n#         img_fdata = img_fdata / np.max(img_fdata)\n#     img_fdata = (img_fdata * 255).astype(np.uint8)\n#     print(img_fdata.shape,'1111')\n    img_fdata = cv2.resize(img_fdata,(size,size))\n\n    num_images = len(img_fdata[0,0,:])\n        \n    start = int(num_images * filtering)\n    end = int(num_images * (1-filtering))\n\n    interval = 3\n    \n    if num_images < 10: \n        interval = 1\n        \n    return np.array(img_fdata[:,:,start:end:interval])\n\ndef load_niifile(niifile,size=64,filtering=0.35):  # Read niifile file\n\n#     if np.max(img_fdata) != 0:\n#         img_fdata = img_fdata / np.max(img_fdata)\n#     img_fdata = (img_fdata * 255).astype(np.uint8)\n    print(img_fdata.shape,'1111')\n    img_fdata = cv2.resize(img_fdata,(size,size))\n\n    num_images = len(img_fdata[0,0,:])\n        \n    start = int(num_images * filtering)\n    end = int(num_images * (1-filtering))\n\n    interval = 3\n    \n    if num_images < 10: \n        interval = 1\n        \n    return np.array(img_fdata[:,:,start:end:interval])\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:37.799047Z","iopub.execute_input":"2021-10-15T18:04:37.799600Z","iopub.status.idle":"2021-10-15T18:04:37.809609Z","shell.execute_reply.started":"2021-10-15T18:04:37.799569Z","shell.execute_reply":"2021-10-15T18:04:37.808925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel,delayed\nimport multiprocessing as mp\nimport nibabel as nib\nfrom glob import glob\ndef get_all_image_paths(brats21id, image_type, folder='train',filtering=0.35): \n    '''\n    Returns an arry of all the images of a particular type for a particular patient ID\n    '''\n    assert(image_type in mri_types)\n    \n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/%s/\" % folder, \n        str(brats21id).zfill(5),\n    )\n\n    paths = sorted(\n        glob.glob(os.path.join(patient_path, image_type, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    \n    num_images = len(paths)\n\n    start = int(num_images * filtering)\n    end = int(num_images * (1-filtering))\n\n    interval = 3\n    \n    if num_images < 10: \n        interval = 1\n    \n    return np.array(paths[start:end:interval])\n\ndef get_all_nifiti_paths(brats21id, folder='train',folername=None): \n    fullidname = str(brats21id).zfill(5)\n    if folername == 'mask_img': \n        patient_path = f'../input/segresult/{folername}/{fullidname}.nii'\n    elif folername == 'sample': \n        patient_path = f'../input/segresult/{folername}/{fullidname}_0001.nii'\n    elif folder == 'test': \n        patient_path = f'./rsna-preprocessed/test/{fullidname}_T1wCE.nii.gz'\n\n    return str(patient_path)\n\n# def get_all_images(brats21id, image_type, folder='train', size=225):\n#     return [load_dicom(path, size) for path in get_all_image_paths(brats21id, image_type, folder)]\n\ndef get_all_images(brats21id, image_type, folder='train', size=225):\n    return Parallel(n_jobs=mp.cpu_count(),prefer='threads')(delayed(load_dicom)(path, size) for path in get_all_image_paths(brats21id, image_type, folder))\n\n# def get_all_nifit(img_list, folder='train', size=225):\n#     return Parallel(n_jobs=mp.cpu_count(),prefer='threads')(delayed(read_niifile)(path, size) for path in get_all_nifiti_paths(img_list,folder))\ndef get_all_nifit(img_list, folder='train',folername=None, size=225):\n    path = get_all_nifiti_paths(img_list,folder,folername)\n    return read_niifile(path, size)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:37.892513Z","iopub.execute_input":"2021-10-15T18:04:37.892707Z","iopub.status.idle":"2021-10-15T18:04:37.974737Z","shell.execute_reply.started":"2021-10-15T18:04:37.892685Z","shell.execute_reply":"2021-10-15T18:04:37.974070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Images We Will Need","metadata":{}},{"cell_type":"code","source":"def crop_dim(voxel):\n    keep = (voxel.mean(axis=(0, 1)) > 0)\n    return keep\n\ndef get_all_data_for_train(df, image_type, image_size=32,use_mask=True):\n    global train_df\n    \n    X = []\n    y = []\n    tumory = []\n    M = []\n    train_ids = []\n    gt = []\n\n    for i in tqdm(df.index):\n        x = df.loc[i]\n#         images = get_all_images(int(x['BraTS21ID']), image_type, 'train', image_size)\n        images = get_all_nifit(int(x['BraTS21ID']), 'train', 'sample',image_size)\n        if use_mask == True: \n            maskes = get_all_nifit(int(x['BraTS21ID']),'train','mask_img',image_size)\n            keepdim = crop_dim(maskes)\n            M.append(maskes)\n            \n        label = x['MGMT_value']\n        X.append(images)\n        if label == 1: \n            y.append(keepdim*1)\n\n        elif label == 0: \n            y.append([label] * len(maskes[0,0,:]))\n\n        tumory.append(keepdim)\n        gt += [label]  * len(maskes[0,0,:])\n        train_ids += [int(x['BraTS21ID'])] * len(images[0,0,:])\n#         assert(len(X) == len(y))\n    if use_mask == True: \n        return X, y,tumory, np.array(train_ids), M , np.array(gt)\n    else: \n        return np.array(X), np.array(y), np.array(train_ids)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:38.071992Z","iopub.execute_input":"2021-10-15T18:04:38.072190Z","iopub.status.idle":"2021-10-15T18:04:38.081911Z","shell.execute_reply.started":"2021-10-15T18:04:38.072167Z","shell.execute_reply":"2021-10-15T18:04:38.081288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import chain","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:38.128487Z","iopub.execute_input":"2021-10-15T18:04:38.128951Z","iopub.status.idle":"2021-10-15T18:04:38.132727Z","shell.execute_reply.started":"2021-10-15T18:04:38.128925Z","shell.execute_reply":"2021-10-15T18:04:38.131852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y,tumory, trainidt,M,train_GT = get_all_data_for_train(train_df, 'T1wCE', image_size=64,use_mask=True)\n# print(np.concatenate(X,axis=-1).shape,len(list(chain(*y)))),len(list(chain(*tumory)),np.concatenate(M,axis=-1).shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T18:04:38.225464Z","iopub.execute_input":"2021-10-15T18:04:38.225651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_val, y_val,y_val_tumor, validt,M_val,val_GT = get_all_data_for_train(val_df, 'T1wCE', image_size=64,use_mask=True)\nprint(np.concatenate(X_val,axis=-1).shape,len(y_val),len(list(chain(*y_val_tumor))))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"M_valT = np.concatenate(M_val,axis=-1).transpose((2,0,1))\nX_valT = np.concatenate(X_val,axis=-1).transpose((2,0,1))\ny_val_tumor = list(chain(*y_val_tumor))\nmgmt_y_val = list(chain(*y_val))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt \nfig,axs=plt.subplots(1,2,figsize=(10,10))\nn = 0\nw = 100\nMT = np.concatenate(M,axis=-1).transpose((2,0,1))\nXT = np.concatenate(X,axis=-1).transpose((2,0,1))\ntumor_y = list(chain(*tumory))\nmgmt_y = list(chain(*y))\naxs[0].imshow(MT[w])\naxs[1].imshow(XT[w])\nprint(f'MGMT:{mgmt_y[w]},Tumor:{tumor_y[w]*1}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train/Validation Split","metadata":{}},{"cell_type":"code","source":"X_train, _, y_train, _, trainidt_train, _ = train_test_split(XT, np.stack([mgmt_y,tumor_y],axis=1), trainidt, test_size=0.1, random_state=3452)\nX_valid, _, y_valid, _, trainidt_valid, _ = train_test_split(X_valT, np.stack([mgmt_y_val,y_val_tumor],axis=1), validt, test_size=0.01, random_state=3452)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train.shape, X_valid.shape, y_train.shape, y_valid.shape, trainidt_train.shape, trainidt_valid.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Remove dimension","metadata":{}},{"cell_type":"code","source":"X_train = tf.expand_dims(X_train, axis=-1)\nX_valid = tf.expand_dims(X_valid, axis=-1)\nprint(X_train.shape, X_valid.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One-hot encode labels","metadata":{}},{"cell_type":"code","source":"y_train = to_categorical(y_train)\ny_valid = to_categorical(y_valid)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Models","metadata":{}},{"cell_type":"markdown","source":"## Model from:  https://www.kaggle.com/ohbewise/dataset-to-model-with-tensorflow","metadata":{}},{"cell_type":"code","source":"# # Define, train, and evaluate model\n# # source: https://keras.io/examples/vision/3D_image_classification/\n# def get_model01(width=128, height=128, depth=64, name='3dcnn'):\n#     \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n#     inputs = tf.keras.Input((width, height, depth, 1))\n\n#     x = tf.keras.layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n#     x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n#     x = tf.keras.layers.BatchNormalization()(x)\n\n#     x = tf.keras.layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(x)\n#     x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n#     x = tf.keras.layers.BatchNormalization()(x)\n\n#     x = tf.keras.layers.Conv3D(filters=128, kernel_size=3, activation=\"relu\")(x)\n#     x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n#     x = tf.keras.layers.BatchNormalization()(x)\n\n#     x = tf.keras.layers.Conv3D(filters=256, kernel_size=3, activation=\"relu\")(x)\n#     x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n#     x = tf.keras.layers.BatchNormalization()(x)\n\n#     x = tf.keras.layers.GlobalAveragePooling3D()(x)\n#     x = tf.keras.layers.Dense(units=512, activation=\"relu\")(x)\n#     x = tf.keras.layers.Dropout(0.3)(x)\n\n#     outputs = tf.keras.layers.Dense(units=1, activation=\"sigmoid\")(x)\n\n#     # Define the model.\n#     model = tf.keras.Model(inputs, outputs, name=name)\n    \n#     # Compile model.\n#     initial_learning_rate = 0.0001\n#     lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n#         initial_learning_rate, decay_steps=100000, decay_rate=0.96, staircase=True\n#     )\n#     model.compile(\n#         loss=\"binary_crossentropy\",\n#         optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule),\n#         metrics=[\"acc\"],\n#     )\n    \n#     return model\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model from: https://www.kaggle.com/evanyao27/team-9-second-week/notebook\n\n- Validation AUC=0.9148664856146349","metadata":{}},{"cell_type":"code","source":"def get_model2D():\n    np.random.seed(42)\n    random.seed(42)\n    tf.random.set_seed(42)\n\n    inpt = keras.Input(shape=X_train.shape[1:])\n    print(inpt.shape)\n    h = keras.layers.experimental.preprocessing.Rescaling(1.0 / 255)(inpt)\n    \n    h = keras.layers.Conv2D(6, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_1\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(2, 2))(h)\n    \n    h = keras.layers.Conv2D(16, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_2\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(2, 2))(h)\n\n    h = keras.layers.Conv2D(24, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_3\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(2, 2))(h)\n    h = keras.layers.Dropout(0.1)(h)\n\n    h = keras.layers.Flatten()(h)\n    h = keras.layers.Dense(24, activation=\"relu\")(h)\n    \n    mgmt_output = keras.layers.Dense(2, activation=\"sigmoid\", name='mgmt_output')(h)\n    tumor_output = keras.layers.Dense(2, activation=\"sigmoid\", name='tumor_output')(h)\n    \n    model = keras.Model(inpt, [mgmt_output,tumor_output])\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up Model Checkpoint","metadata":{}},{"cell_type":"code","source":"class PlotProgress(keras.callbacks.Callback):\n    \n    def __init__(self, entity='loss'):\n        self.entity = entity\n        \n    def on_train_begin(self, logs={}):\n        self.i = 0\n        self.x = []\n        self.losses = []\n        self.val_losses = []\n        \n        self.fig = plt.figure()\n        \n        self.logs = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        \n        self.logs.append(logs)\n        self.x.append(self.i)\n        self.losses.append(logs.get('{}'.format(self.entity)))\n        self.val_losses.append(logs.get('val_{}'.format(self.entity)))\n        self.i += 1\n        \n        clear_output(wait=True)\n        plt.plot(self.x, self.losses, label=\"{}\".format(self.entity))\n        plt.plot(self.x, self.val_losses, label=\"val_{}\".format(self.entity))\n        plt.legend()\n        plt.show();","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_filepath = \"chpt_{epoch}.h5\"\nfirst_decay_steps = 1000\nscheduler = tf.keras.experimental.CosineDecayRestarts(1e-3,first_decay_steps,t_mul=2.0,m_mul=1.0)\nplot_progress = PlotProgress(entity='loss')\n\n\nmodel_checkpoint_callback = [\n    tf.keras.callbacks.LearningRateScheduler(scheduler),\n    tf.keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=False,\n    monitor=\"val_auc\",\n    mode=\"max\",\n    save_best_only=False,\n    save_freq=\"epoch\",\n    verbose=10,\n)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Note that rerunning the cell below will change val_acc to val_acc_N and the model will not be saved.","metadata":{}},{"cell_type":"code","source":"model = get_model2D()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T15:56:32.974967Z","iopub.execute_input":"2021-10-15T15:56:32.975497Z","iopub.status.idle":"2021-10-15T15:56:33.102318Z","shell.execute_reply.started":"2021-10-15T15:56:32.975459Z","shell.execute_reply":"2021-10-15T15:56:33.101611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T15:56:36.438352Z","iopub.execute_input":"2021-10-15T15:56:36.4389Z","iopub.status.idle":"2021-10-15T15:56:36.452749Z","shell.execute_reply.started":"2021-10-15T15:56:36.438863Z","shell.execute_reply":"2021-10-15T15:56:36.451508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.compile(\n    loss={'mgmt_output':\"categorical_crossentropy\",'tumor_output':\"categorical_crossentropy\"},optimizer=tf.keras.optimizers.Adam(learning_rate=2e-3), metrics=[tf.keras.metrics.AUC()]\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T15:56:38.465964Z","iopub.execute_input":"2021-10-15T15:56:38.466236Z","iopub.status.idle":"2021-10-15T15:56:38.493797Z","shell.execute_reply.started":"2021-10-15T15:56:38.466209Z","shell.execute_reply":"2021-10-15T15:56:38.493062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(x = X_train,\n                    y = {'mgmt_output':y_train[:,0],'tumor_output':y_train[:,1]},\n#                     y = {'dense_7':np.array(y_train)[:,0,:],'dense_8':np.array(y_train)[:,1,:]},\n                    epochs=50, \n                    callbacks=[model_checkpoint_callback], \n                    validation_data= (X_valid, {'mgmt_output':y_valid[:,0],'tumor_output':y_valid[:,1]}))","metadata":{"execution":{"iopub.status.busy":"2021-10-15T15:56:43.584071Z","iopub.execute_input":"2021-10-15T15:56:43.584361Z","iopub.status.idle":"2021-10-15T16:01:03.153669Z","shell.execute_reply.started":"2021-10-15T15:56:43.584332Z","shell.execute_reply":"2021-10-15T16:01:03.152809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_val_test= tf.expand_dims(X_valT, axis=-1)\ny_val_test = to_categorical(list(chain(*y_val)))\nbest = 0\nepoch = -1\nfor i in range(30):\n    model_best = tf.keras.models.load_model(filepath=f'chpt_{i + 1}.h5')\n    y_pred = model_best.predict(X_val_test)\n    \n\n    tumor_pred = y_pred[1][:,1] > 0.3 # filtering tumor\n\n    pred = np.argmax(y_pred[0],axis=1)\n    result = pd.DataFrame(validt)\n    result[1] = pred\n    result.columns = [\"BraTS21ID\", \"MGMT_value\"]\n    result = result.iloc[tumor_pred]\n    \n    result2 = result.groupby(\"BraTS21ID\", as_index=False).mean()\n    result2 = val_df.merge(result2, on=\"BraTS21ID\")\n    auc = roc_auc_score(\n        result2.MGMT_value_x,\n        result2.MGMT_value_y,\n    )\n    print(f\"Validation AUC={auc}\")\n    if best < auc:\n        best = auc\n        epoch = i + 1\n        \nprint(f'Best AUC: {best}, epoch: {epoch}')","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:13:53.778928Z","iopub.execute_input":"2021-10-15T17:13:53.779221Z","iopub.status.idle":"2021-10-15T17:14:03.745387Z","shell.execute_reply.started":"2021-10-15T17:13:53.77919Z","shell.execute_reply":"2021-10-15T17:14:03.74445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Our Best Model","metadata":{}},{"cell_type":"markdown","source":"# Predictions on Validation Set","metadata":{}},{"cell_type":"code","source":"!pip install --force-reinstall /kaggle/input/torchio/SimpleITK-2.1.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install /kaggle/input/torchio/Deprecated-1.2.13-py2.py3-none-any.whl\n!pip install /kaggle/input/torchio/torchio-0.18.56-py2.py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2021-10-15T16:01:03.155513Z","iopub.execute_input":"2021-10-15T16:01:03.155967Z","iopub.status.idle":"2021-10-15T16:02:30.726856Z","shell.execute_reply.started":"2021-10-15T16:01:03.155927Z","shell.execute_reply":"2021-10-15T16:02:30.726004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def preprocess_dataset(dataset, out_dir, parallel=True, demo=False):\n#     change_num = {'T1w':'0','T1wCE':'1','T2w':'2','FLAIR':'3'}\n#     import shutil\n#     import multiprocessing as mp\n#     from pathlib import Path\n#     from tqdm import tqdm\n#     if demo:  # just to showcase TorchIO\n#         dataset._subjects = dataset._subjects[:5]\n#     out_dir = Path(out_dir)\n#     labels_name = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv'\n#     out_dir.mkdir(exist_ok=True, parents=True)\n#     subjects_dir = out_dir / ('train' if dataset.train else 'test')\n#     if parallel:\n#         loader = torch.utils.data.DataLoader(\n#             dataset,\n#             num_workers=mp.cpu_count(),\n#             collate_fn=lambda x: x[0],\n#         )\n#         iterable = loader\n#     else:\n#         iterable = dataset\n#     for subject in tqdm(iterable):\n#         subject_dir = subjects_dir \n# #         print(subject_dir)\n#         for name, image in tqdm(subject.get_images_dict().items(), leave=False):\n#             image_dir = subject_dir \n# #             image_dir.mkdir(exist_ok=True, parents=True)\n#             image_path = out_dir / f'{subject.BraTS21ID}_{change_num[name].zfill(4)}.nii.gz'\n#             image.save(image_path)\n\n# out_dir = './'\n# # if not Path(out_dir).is_dir():\n# preprocess_dataset(test_set, out_dir, parallel=True, demo=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T16:02:30.7304Z","iopub.execute_input":"2021-10-15T16:02:30.730851Z","iopub.status.idle":"2021-10-15T16:02:30.735232Z","shell.execute_reply.started":"2021-10-15T16:02:30.730816Z","shell.execute_reply":"2021-10-15T16:02:30.734537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nimport torch\nimport torchio as tio\nimport matplotlib.pyplot as plt\n\nroot_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\npreprocessing_transforms = (    \n#     tio.ZNormalization(),\n    tio.ToCanonical(),\n    tio.Resample(1, image_interpolation='bspline'),\n    tio.Resample('T1wCE', image_interpolation='nearest'),\n)\npreprocess = tio.Compose(preprocessing_transforms)\ntest_set = tio.datasets.RSNAMICCAI(root_dir, train=False, transform=preprocess)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:40:11.344383Z","iopub.execute_input":"2021-10-15T17:40:11.34504Z","iopub.status.idle":"2021-10-15T17:40:16.654124Z","shell.execute_reply.started":"2021-10-15T17:40:11.345004Z","shell.execute_reply":"2021-10-15T17:40:16.653223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef preprocess_dataset(dataset, out_dir, parallel=True, demo=False):\n    import shutil\n    import multiprocessing as mp\n    from pathlib import Path\n    from tqdm import tqdm\n#     if demo:  # just to showcase TorchIO\n#         dataset._subjects = dataset._subjects[:5]\n    out_dir = Path(out_dir)\n#     labels_name = 'train_labels.csv'\n#     out_dir.mkdir(exist_ok=True, parents=True)\n#     shutil.copy(dataset.root_dir / labels_name, out_dir / labels_name)\n    subjects_dir = out_dir / ('train' if dataset.train else 'test')\n    if parallel:\n        loader = torch.utils.data.DataLoader(\n            dataset,\n            num_workers=mp.cpu_count(),\n            collate_fn=lambda x: x[0],\n        )\n        iterable = loader\n    else:\n        iterable = dataset\n    for subject in tqdm(iterable):\n        subject_dir = subjects_dir \n#         print(subject_dir)\n        for name, image in tqdm(subject.get_images_dict().items(), leave=False):\n            if name == 'T1wCE': \n                image_dir = subject_dir \n                image_dir.mkdir(exist_ok=True, parents=True)\n                image_path = image_dir / f'{subject.BraTS21ID.zfill(4)}_{name}.nii.gz'\n                image.save(image_path)\n        \n    return dataset\n\nout_dir = 'rsna-preprocessed'\n# if not Path(out_dir).is_dir():\ndicom_images = preprocess_dataset(test_set,out_dir, parallel=True, demo=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:40:24.758267Z","iopub.execute_input":"2021-10-15T17:40:24.758564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_data_for_test(image_type, image_size=32):\n    global test_df\n    \n    X = []\n    test_ids = []\n\n    for i in tqdm(test_df.index):\n        x = test_df.loc[i]\n        images = get_all_nifit(int(x['BraTS21ID']), 'test', None,image_size)\n        X.append(images)\n#         images = get_all_images(int(x['BraTS21ID']), image_type, 'test', image_size)\n#         X += images\n        test_ids += [int(x['BraTS21ID'])] * len(images[0,0,:])\n\n    return X, np.array(test_ids)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T16:17:27.706956Z","iopub.execute_input":"2021-10-15T16:17:27.707229Z","iopub.status.idle":"2021-10-15T16:17:27.713324Z","shell.execute_reply.started":"2021-10-15T16:17:27.7072Z","shell.execute_reply":"2021-10-15T16:17:27.712464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get_all_images = Parallel(n_jobs=mp.cpu_count(),prefer='threads')(delayed(load_niifile)(path.get_images_dict()['T1wCE']) for path in tqdm(dicom_images))","metadata":{"execution":{"iopub.status.busy":"2021-10-15T16:13:03.836659Z","iopub.execute_input":"2021-10-15T16:13:03.837434Z","iopub.status.idle":"2021-10-15T16:13:03.85299Z","shell.execute_reply.started":"2021-10-15T16:13:03.837396Z","shell.execute_reply":"2021-10-15T16:13:03.852056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test, testidt = get_all_data_for_test('T1wCE', image_size=64)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T16:17:29.8811Z","iopub.execute_input":"2021-10-15T16:17:29.881828Z","iopub.status.idle":"2021-10-15T16:18:20.504973Z","shell.execute_reply.started":"2021-10-15T16:17:29.881791Z","shell.execute_reply":"2021-10-15T16:18:20.50399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_testT =  np.concatenate(X_test,axis=-1).transpose((2,0,1))\nX_testT= tf.expand_dims(X_testT, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:14:58.334946Z","iopub.execute_input":"2021-10-15T17:14:58.335226Z","iopub.status.idle":"2021-10-15T17:14:58.677508Z","shell.execute_reply.started":"2021-10-15T17:14:58.335197Z","shell.execute_reply":"2021-10-15T17:14:58.676695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions on the Test Set","metadata":{}},{"cell_type":"code","source":"# X_val_test= tf.expand_dims(X_valT, axis=-1)\n# y_val_test = to_categorical(list(chain(*y_val)))\n# best = 0\n# epoch = -1\n# for i in range(30):\n#     model_best = tf.keras.models.load_model(filepath=f'chpt_{i + 1}.h5')\n#     y_pred = model_best.predict(X_val_test)\n    \n\n#     tumor_pred = y_pred[1][:,1] > 0.3 # filtering tumor\n\n#     pred = np.argmax(y_pred[0],axis=1)\n#     result = pd.DataFrame(validt)\n#     result[1] = pred\n#     result.columns = [\"BraTS21ID\", \"MGMT_value\"]\n#     result = result.iloc[tumor_pred]\n    \n#     result2 = result.groupby(\"BraTS21ID\", as_index=False).mean()\n#     result2 = val_df.merge(result2, on=\"BraTS21ID\")\n#     auc = roc_auc_score(\n#         result2.MGMT_value_x,\n#         result2.MGMT_value_y,\n#     )\n#     print(f\"Validation AUC={auc}\")\n#     if best < auc:\n#         best = auc\n#         epoch = i + 1\n        \n# print(f'Best AUC: {best}, epoch: {epoch}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport torchio as tio\nfrom pathlib import Path\n\n# Parameters to limit the processing power needed.\ndemo  = False # if True limits to 10 patients\nscan_types    = ['FLAIR','T1w','T1wCE','T2w'] # uses all scan types\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:31:10.233929Z","iopub.execute_input":"2021-10-15T17:31:10.234963Z","iopub.status.idle":"2021-10-15T17:31:10.241123Z","shell.execute_reply.started":"2021-10-15T17:31:10.234917Z","shell.execute_reply":"2021-10-15T17:31:10.240039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir   = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/'\nout_dir    = './processed'\n\nfor dataset in ['train']:\n    dataset_dir = f'{data_dir}{dataset}'\n    patients = os.listdir(dataset_dir)\n    if demo:\n        patients = patients[:10]\n    \n    # Remove cases the competion host said to exclude \n    # https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/discussion/262046\n    if '00109' in patients: patients.remove('00109')\n    if '00123' in patients: patients.remove('00123')\n    if '00709' in patients: patients.remove('00709')\n    \n    print(f'Total patients in {dataset} dataset: {len(patients)}')\n\n    count = 0\n    for patient in patients:\n        count = count + 1\n        print(f'{dataset}: {count}/{len(patients)}')\n\n        for scan_type in scan_types:\n            scan_src  = f'{dataset_dir}/{patient}/{scan_type}/'\n            scan_dest = f'{out_dir}/{dataset}/{patient}/{scan_type}/'\n            Path(scan_dest).mkdir(parents=True, exist_ok=True)\n            image = tio.ScalarImage(scan_src)\n            transforms = [\n                tio.ToCanonical(),\n                tio.Resample(1),\n                tio.ZNormalization(masking_method=tio.ZNormalization.mean),\n                tio.CropOrPad((128,128,64)),\n                tio.RescaleIntensity((-1, 1)),\n            ]\n            transform = tio.Compose(transforms)\n            preprocessed = transform(image)\n            preprocessed.save(f'{scan_dest}/{scan_type}.nii.gz')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_best = tf.keras.models.load_model(filepath=f'chpt_{epoch}.h5')\n\ny_pred = model_best.predict(X_testT)\n# tumor_filter = y_pred[1][:,1] > 0.3 # filtering tumor\n\npred = np.argmax(y_pred[0], axis=1) #\n\nresult = pd.DataFrame(testidt)\nresult[1] = pred\npred","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:18:39.227375Z","iopub.execute_input":"2021-10-15T17:18:39.227885Z","iopub.status.idle":"2021-10-15T17:18:39.728046Z","shell.execute_reply.started":"2021-10-15T17:18:39.227848Z","shell.execute_reply":"2021-10-15T17:18:39.727319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission File","metadata":{}},{"cell_type":"code","source":"# result = result.iloc[tumor_filter]\nresult.columns=['BraTS21ID','MGMT_value']\n\nresult2 = result.groupby('BraTS21ID',as_index=False).mean()\nresult2['BraTS21ID'] = sample_submission['BraTS21ID']\n\n# Rounding...\nresult2['MGMT_value'] = result2['MGMT_value'].apply(lambda x:round(x*10)/10)\nresult2.to_csv('submission.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:18:42.054346Z","iopub.execute_input":"2021-10-15T17:18:42.055065Z","iopub.status.idle":"2021-10-15T17:18:42.079421Z","shell.execute_reply.started":"2021-10-15T17:18:42.055025Z","shell.execute_reply":"2021-10-15T17:18:42.078373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.countplot(data=result2, x=\"MGMT_value\")","metadata":{"execution":{"iopub.status.busy":"2021-10-15T17:18:11.91632Z","iopub.execute_input":"2021-10-15T17:18:11.916912Z","iopub.status.idle":"2021-10-15T17:18:12.149359Z","shell.execute_reply.started":"2021-10-15T17:18:11.916877Z","shell.execute_reply":"2021-10-15T17:18:12.14861Z"},"trusted":true},"execution_count":null,"outputs":[]}]}