{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing libraries ","metadata":{}},{"cell_type":"code","source":"import os\nimport sys \nimport json \nimport random \nimport collections \nimport time\nimport re \nimport math \nimport numpy as np \nimport pandas as pd \nimport cv2\nimport glob\nimport matplotlib.pyplot as plt \nimport seaborn as sns \n\nimport pydicom \nfrom  pydicom.pixel_data_handlers.util import apply_voi_lut\n\nfrom random import shuffle \nfrom sklearn import model_selection as sk_model_selection \n\nfrom tensorflow import keras \nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.metrics import AUC \nfrom tensorflow.keras.utils import Sequence\nimport tensorflow as tf ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-12T13:57:54.347165Z","iopub.execute_input":"2021-10-12T13:57:54.347482Z","iopub.status.idle":"2021-10-12T13:57:59.418226Z","shell.execute_reply.started":"2021-10-12T13:57:54.347448Z","shell.execute_reply":"2021-10-12T13:57:59.41745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading data","metadata":{}},{"cell_type":"code","source":"data_direc ='../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\nmri_types = ['FLAIR', 'T1w', 'T1wCE', 'T2w']\nmri_types_id = [0,1,2,3]\n\nimage_size = 128 \nnum_images  = 64 \nbatch_size = 4 \n\nnum_folds = 4 \n","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:58:01.61138Z","iopub.execute_input":"2021-10-12T13:58:01.612127Z","iopub.status.idle":"2021-10-12T13:58:01.61617Z","shell.execute_reply.started":"2021-10-12T13:58:01.612094Z","shell.execute_reply":"2021-10-12T13:58:01.615543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directories of image files \ntrain_folder = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'\ntest_folder = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/test'\ntrain_label_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv'\nsample_sub_path  = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv'\n\nsample_sheet = pd.read_csv(sample_sub_path)\n# get list of all files \ntrain_files = glob.glob(train_folder +  '/**/*.dcm', recursive  = True)\ntest_files = glob.glob(test_folder + '/**/*.dcm', recursive = True)\n\n#  get the labels labels \ntrain_labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')          \n#train_df = pd.read_csv(train_label_path) \n                          \n# start adding data to the dataframe\ntrain_df = pd.DataFrame(train_files, columns = {'FilePath'})\ntest_df = pd.DataFrame(test_files, columns = {'FilePath'})\n\n#train_df['BraTS21ID_5'] = [format(x , '05d') for x in train_df.BraTS21ID]\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:58:02.343742Z","iopub.execute_input":"2021-10-12T13:58:02.3443Z","iopub.status.idle":"2021-10-12T13:58:53.233259Z","shell.execute_reply.started":"2021-10-12T13:58:02.344264Z","shell.execute_reply":"2021-10-12T13:58:53.232534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parseString(filepath, indx_val):\n    parts  = filepath.split('/')\n    return parts[indx_val] ","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:58:53.2349Z","iopub.execute_input":"2021-10-12T13:58:53.23518Z","iopub.status.idle":"2021-10-12T13:58:53.239614Z","shell.execute_reply.started":"2021-10-12T13:58:53.235146Z","shell.execute_reply":"2021-10-12T13:58:53.238924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add data columns to the test and train dataframes \ntrain_df['Type'] = train_df.apply(lambda row: parseString(row['FilePath'], -2) , axis =1 )\ntrain_df['BraTS21ID'] = train_df.apply(lambda row: parseString(row['FilePath'], -3) , axis =1 )\n\n\ntest_df['Type'] = test_df.apply(lambda row: parseString(row['FilePath'], -2) , axis =1 )\ntest_df['BraTS21ID'] = test_df.apply(lambda row: parseString(row['FilePath'], -3) , axis =1 )\n\n# Get a dictionary of  BraID and labels and map them to the dataframes \ntrain_labels = train_labels.astype({'BraTS21ID': 'int'})\ntrain_labels['BraTS21ID_5'] = [format(x , '05d') for x in train_labels.BraTS21ID]\nlabel_dict = train_labels.set_index('BraTS21ID_5').to_dict()['MGMT_value']\n\n# map all values from the dicitonary to the dataframe \ntrain_df['MGMT_value'] = train_df['BraTS21ID'].map(label_dict)\ntest_df['MGMT_value'] = test_df['BraTS21ID'].map(label_dict)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:58:53.240872Z","iopub.execute_input":"2021-10-12T13:58:53.241375Z","iopub.status.idle":"2021-10-12T13:59:02.128831Z","shell.execute_reply.started":"2021-10-12T13:58:53.241326Z","shell.execute_reply":"2021-10-12T13:59:02.128125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:59:02.130702Z","iopub.execute_input":"2021-10-12T13:59:02.130955Z","iopub.status.idle":"2021-10-12T13:59:02.143123Z","shell.execute_reply.started":"2021-10-12T13:59:02.130924Z","shell.execute_reply":"2021-10-12T13:59:02.142247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom data generators ","metadata":{}},{"cell_type":"code","source":"# reference: 🧠Brain Tumor 3D [Training]\ndef load_dicom(path , img_size= image_size, voi_lut = True, rotate = 0):\n    dicom  = pydicom.read_file(path)\n    data = dicom.pixel_array \n    if voi_lut: \n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else: \n        data = dicom.pixel_array\n        \n    if rotate > 0: \n        rot_choices  = [0, cv2.ROTATE_90_CLOCKWISE,\n                       cv2.ROTATE_90_COUNTERCLOCKWISE, \n                       cv2.ROTATE_180]\n        data = cv2.rotate(data , rot_choices[rotate])\n        \n    data = cv2.resize(data ,  (img_size , img_size))\n    return data ","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:59:49.755823Z","iopub.execute_input":"2021-10-12T13:59:49.756529Z","iopub.status.idle":"2021-10-12T13:59:49.762956Z","shell.execute_reply.started":"2021-10-12T13:59:49.756491Z","shell.execute_reply":"2021-10-12T13:59:49.762088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reference: 🧠Brain Tumor 3D [Training]\nclass Dataset(Sequence): \n    def __init__(self, df_test, is_train = True, batch_size = batch_size, shuffle = True):\n        self.y = df_test['MGMT_value'].values \n        self.paths = df_test['FilePath'].values\n        self.is_train = is_train\n        self.batch_size = batch_size\n        self.shuffle = shuffle \n    def __len__(self):\n        return math.ceil(len(self.y)/self.batch_size)\n    \n    def __getitem__(self, ids): \n        id_path =self.paths[ids]\n        batch_paths = self.paths[ids*self.batch_size:(ids + 1)*self.batch_size]\n        \n        if self.y is not None: \n            batch_y = self.y[ids*self.batch_size:(ids + 1)*self.batch_size]\n            \n        if self.is_train:\n            list_x = [load_dicom(x) for x in batch_paths]\n            #print(list_x[0].shape)\n            batch_x = np.stack(list_x, axis = 0)\n            return batch_x, batch_y\n        else: \n            list_x = [load_dicom(x) for x in batch_paths]\n            batch_x  = np.stack(list_x, axis =0)\n            return batch_x\n        \n    def on_epoch_end(self): \n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.paths, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))\n            ","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:59:51.528074Z","iopub.execute_input":"2021-10-12T13:59:51.528929Z","iopub.status.idle":"2021-10-12T13:59:51.539636Z","shell.execute_reply.started":"2021-10-12T13:59:51.52888Z","shell.execute_reply":"2021-10-12T13:59:51.538914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = Dataset(train_df, batch_size=batch_size)\ntest_dataset = Dataset(test_df, batch_size=batch_size, is_train= False)","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:59:52.166619Z","iopub.execute_input":"2021-10-12T13:59:52.167358Z","iopub.status.idle":"2021-10-12T13:59:52.172275Z","shell.execute_reply.started":"2021-10-12T13:59:52.167319Z","shell.execute_reply":"2021-10-12T13:59:52.171338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1):\n    images, label = train_dataset[i]\n    print(\"Dimension of the CT scan is:\", images.shape)\n    print(\"label=\",label)\n    print(images.shape)\n    plt.imshow(images[1,:,:], cmap=\"gray\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T13:59:54.214706Z","iopub.execute_input":"2021-10-12T13:59:54.214977Z","iopub.status.idle":"2021-10-12T13:59:54.51316Z","shell.execute_reply.started":"2021-10-12T13:59:54.214949Z","shell.execute_reply":"2021-10-12T13:59:54.512524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\ndef get_model(width=image_size, height=image_size):\n    \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n    inputs = tf.keras.Input(shape = (width, height, 1))\n     \n    x = layers.Conv2D(filters=32, kernel_size=3, activation=\"relu\",padding = 'same')(inputs)\n    x = layers.MaxPool2D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n    \n    x = layers.Conv2D(filters=64, kernel_size=3, activation=\"relu\",padding = 'same')(x)\n    x = layers.MaxPool2D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.6)(x)\n\n    x = layers.Conv2D(filters=128, kernel_size=3, activation=\"relu\",padding = 'same')(x)\n    x = layers.MaxPool2D(pool_size=3)(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.6)(x)\n\n    x = layers.Conv2D(filters=256, kernel_size=3, activation=\"relu\",padding = 'same')(x)\n    x = layers.MaxPool2D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.4)(x)\n\n    x = layers.GlobalAveragePooling2D()(x)\n    #x = layers.Dense(units=1024, activation=\"relu\")(x)\n    #x = layers.Dropout(0.2)(x)\n\n    outputs = layers.Dense(units=1, activation=\"sigmoid\")(x)\n\n    # Define the model.\n    model = keras.Model(inputs, outputs, name=\"2DCNN\")\n\n    return model\n\n# Build model.\nmodel = get_model(width=image_size, height=image_size)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-12T14:13:37.214355Z","iopub.execute_input":"2021-10-12T14:13:37.215059Z","iopub.status.idle":"2021-10-12T14:13:37.320087Z","shell.execute_reply.started":"2021-10-12T14:13:37.215026Z","shell.execute_reply":"2021-10-12T14:13:37.319457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile model.\ninitial_learning_rate = 0.0001\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate, decay_steps=100000, decay_rate=0.96, staircase=True\n)\nmodel.compile(\n    loss=\"binary_crossentropy\",\n    optimizer=keras.optimizers.Adam(learning_rate=lr_schedule),\n    metrics=[AUC(name='auc'),\"acc\"],\n)\n# Define callbacks.\nfilepath = '../output/model.epoch{epoch:02d}.hdf5'\n\nmodel_save = ModelCheckpoint(filepath, \n                             save_best_only = True, \n                             monitor = 'val_auc', \n                             mode = 'max', verbose = 1)\n\n# Train the model, doing validation at the end of each epoch\nepochs = 5\nmodel.fit(\n    train_dataset,\n    epochs=epochs,\n    shuffle=True,\n    verbose=1,\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T05:11:29.546018Z","iopub.execute_input":"2021-10-11T05:11:29.546295Z","iopub.status.idle":"2021-10-11T06:20:33.478339Z","shell.execute_reply.started":"2021-10-11T05:11:29.546268Z","shell.execute_reply":"2021-10-11T06:20:33.477627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_dataset, verbose = True)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T06:22:21.409601Z","iopub.execute_input":"2021-10-11T06:22:21.409871Z","iopub.status.idle":"2021-10-11T06:30:19.513911Z","shell.execute_reply.started":"2021-10-11T06:22:21.409842Z","shell.execute_reply":"2021-10-11T06:30:19.513228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['MGMT_value'] = predictions\naverage_values = test_df.groupby(['BraTS21ID']).mean().reset_index()\nlabel_dict_submission = average_values.set_index('BraTS21ID').to_dict()['MGMT_value']","metadata":{"execution":{"iopub.status.busy":"2021-10-11T06:37:54.814001Z","iopub.execute_input":"2021-10-11T06:37:54.81426Z","iopub.status.idle":"2021-10-11T06:37:54.834949Z","shell.execute_reply.started":"2021-10-11T06:37:54.814226Z","shell.execute_reply":"2021-10-11T06:37:54.834165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_df['BraTS21ID'] = test_df.apply(lambda row: parseString(row['FilePath'], -3) , axis =1 average_valuesission = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\naverage_values.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T06:37:56.791174Z","iopub.execute_input":"2021-10-11T06:37:56.791746Z","iopub.status.idle":"2021-10-11T06:37:56.804534Z","shell.execute_reply.started":"2021-10-11T06:37:56.79171Z","shell.execute_reply":"2021-10-11T06:37:56.803636Z"},"trusted":true},"execution_count":null,"outputs":[]}]}