{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys \nimport json\nimport glob\nimport random\nimport collections\nimport time\nimport re\nimport math\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nfrom random import shuffle\nfrom sklearn import model_selection as sk_model_selection\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.metrics import AUC","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:34.803445Z","iopub.execute_input":"2021-10-15T20:50:34.804057Z","iopub.status.idle":"2021-10-15T20:50:34.810271Z","shell.execute_reply.started":"2021-10-15T20:50:34.804018Z","shell.execute_reply":"2021-10-15T20:50:34.809343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#global variables,initialisations\n\ndata_directory = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\n \nmri_types = ['FLAIR','T1w','T1wCE','T2w']\nmri_types_id=0 # 0,1,2,3\n\n#3-D image parametrs\nIMAGE_SIZE = 128\nNUM_IMAGES = 64\nBATCH_SIZE= 4","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:34.826421Z","iopub.execute_input":"2021-10-15T20:50:34.826707Z","iopub.status.idle":"2021-10-15T20:50:34.830892Z","shell.execute_reply.started":"2021-10-15T20:50:34.826677Z","shell.execute_reply":"2021-10-15T20:50:34.830145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ../","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:34.899307Z","iopub.execute_input":"2021-10-15T20:50:34.900614Z","iopub.status.idle":"2021-10-15T20:50:34.907372Z","shell.execute_reply.started":"2021-10-15T20:50:34.900554Z","shell.execute_reply":"2021-10-15T20:50:34.906521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\ntest=sample_submission\ntest['BraTS21ID5'] = [format(x, '05d') for x in test.BraTS21ID]\ntest.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:35.010435Z","iopub.execute_input":"2021-10-15T20:50:35.011215Z","iopub.status.idle":"2021-10-15T20:50:35.031505Z","shell.execute_reply.started":"2021-10-15T20:50:35.011174Z","shell.execute_reply":"2021-10-15T20:50:35.030487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom_image(path, img_size=IMAGE_SIZE, voi_lut=True, rotate=0):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n        \n    if rotate > 0:\n        rot_choices = [0, cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE, cv2.ROTATE_180]\n        data = cv2.rotate(data, rot_choices[rotate])\n        \n    data = cv2.resize(data, (img_size, img_size))\n    return data\n\n\ndef load_dicom_images_3d(scan_id, num_imgs=NUM_IMAGES, img_size=IMAGE_SIZE, mri_type=\"FLAIR\", split=\"test\", rotate=0):\n\n    files = sorted(glob.glob(f\"{data_directory}/{split}/{scan_id}/{mri_type}/*.dcm\"), \n               key=lambda var:[int(x) if x.isdigit() else x for x in re.findall(r'[^0-9]|[0-9]+', var)])\n\n    middle = len(files)//2\n    num_imgs2 = num_imgs//2\n    p1 = max(0, middle - num_imgs2)\n    p2 = min(len(files), middle + num_imgs2)\n    img3d = np.stack([load_dicom_image(f, rotate=rotate) for f in files[p1:p2]]).T \n    if img3d.shape[-1] < num_imgs:\n        n_zero = np.zeros((img_size, img_size, num_imgs - img3d.shape[-1]))\n        img3d = np.concatenate((img3d,  n_zero), axis = -1)\n        \n    if np.min(img3d) < np.max(img3d):\n        img3d = img3d - np.min(img3d)\n        img3d = img3d / np.max(img3d)\n            \n    return np.expand_dims(img3d,0)\n\na = load_dicom_images_3d(\"00001\")\nprint(a.shape)\nprint(np.min(a), np.max(a), np.mean(a), np.median(a))\nimage = a[0]\nprint(\"Dimension of the CT scan is:\", image.shape)\nplt.imshow(np.squeeze(image[:, :, 30]), cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:35.085493Z","iopub.execute_input":"2021-10-15T20:50:35.085779Z","iopub.status.idle":"2021-10-15T20:50:36.336489Z","shell.execute_reply.started":"2021-10-15T20:50:35.085749Z","shell.execute_reply":"2021-10-15T20:50:36.33575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''''#splitting into train and validation datasets, stratifying.\n\ndf_train, df_valid = sk_model_selection.train_test_split(\n    train_df, \n    test_size=0.2, #valid dataset is 1/5 of total, i.e 1/5*585=117\n    random_state=12, \n    stratify=train_df[\"MGMT_value\"], #stratifying.\n                            #the train and valid sets will have the same proportion of \"1s\" and \"0s\" as \n                            #the original. For eg, if there were 75% 1s and 25% 0s, then the splits (train and valid)\n                            #will each have 75% 1s and 25% 0s.\n)\n#checking the data frames\n#len(df_train) #468\n#len(df_valid) #117\n#df_valid\n#df_train '''","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.406125Z","iopub.status.idle":"2021-10-15T20:50:36.406821Z","shell.execute_reply.started":"2021-10-15T20:50:36.406536Z","shell.execute_reply":"2021-10-15T20:50:36.406566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#use of sequences is helpful as it ensures that, per epoch, each input will be trained only once.\n\n''''from tensorflow.keras.utils import Sequence\n\nclass Dataset(Sequence):\n    def __init__(self,df,is_train=True,batch_size=4 #BATCH_SIZE\n                 ,shuffle=True):\n        \n        self.idx = df[\"BraTS21ID\"].values #values 0,2,3..\n        self.paths = df[\"BraTS21ID5\"].values #values 000000,00002,00003..\n        self.y =  df[\"MGMT_value\"].values #MGMT values 0/1\n        \n        #is_train and shuffle are set to true\n        self.is_train = is_train\n        self.shuffle = shuffle\n        \n        self.batch_size = batch_size\n        \n    def __len__(self):\n        \n        #eg if self.idx is the list [0,3,7.....](10 values)\n        #then len attribute will be 10/5=2.\n        \n        return math.ceil(len(self.idx)/self.batch_size)\n   \n    def __getitem__(self,ids):\n        \n        id_path= self.paths[ids] \n        \n      \n        start= ids*self.batch_size\n        end= (ids+1)*self.batch_size\n        \n        batch_paths = self.paths[start:end]   #eg if ids=0, then batch_paths will be self.paths[0:4]\n        \n        if self.y is not None: \n            batch_y = self.y[start: end] #eg if ids=0, then batch_y will be self.y[0:4], i.e MGMT values for \n                                         #first 4 patients\n        \n        if self.is_train: #set to true\n            list_x =  [load_dicom_images_3d(x,split=\"train\") for x in batch_paths] #list of all the arrays corresponding to\n                                                                                   #the images having IDs in batch_paths\n            batch_X = np.stack(list_x, axis=4) #stacking the arrays along a 4th axis\n            return batch_X,batch_y #returning the arrays for the batch, alog with MGMT values\n        \n        else: \n            list_x =  load_dicom_images_3d(id_path,split=\"test\")#str(scan_id).zfill(5)\n            batch_X = np.stack(list_x)\n            return batch_X\n    \n    def on_epoch_end(self): #after one epoch completed, we will shuffle around the values for the next epoch\n        \n        if self.shuffle and self.is_train: #both set to true\n            ids_y = list(zip(self.idx, self.y))  #will \"zip\" the two lists into one list , containg entries like\n                                        #[(0,1),(2,1),(3,0),(5,1)...]\n                                        #[(index,MGMT value),(index,MGMT value)..]\n            shuffle(ids_y) #shuffle the contents\n            self.idx, self.y = list(zip(*ids_y)) #unzips. so now the order of values in idx ,y lists will be changed'''","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.408386Z","iopub.status.idle":"2021-10-15T20:50:36.40907Z","shell.execute_reply.started":"2021-10-15T20:50:36.408775Z","shell.execute_reply":"2021-10-15T20:50:36.408802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import Sequence\nclass Dataset(Sequence):\n    def __init__(self,df,is_train=True,batch_size=1,shuffle=True):\n        self.idx = df[\"BraTS21ID\"].values\n        self.paths = df[\"BraTS21ID5\"].values\n        self.y =  df[\"MGMT_value\"].values\n        self.is_train = is_train\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n    def __len__(self):\n        return math.ceil(len(self.idx)/self.batch_size)\n   \n    def __getitem__(self,ids):\n        id_path= self.paths[ids]\n        batch_paths = self.paths[ids * self.batch_size:(ids + 1) * self.batch_size]\n        \n        if self.y is not None:\n            batch_y = self.y[ids * self.batch_size: (ids + 1) * self.batch_size]\n            \n        list_x =  load_dicom_images_3d(id_path)#str(scan_id).zfill(5)\n        #list_x =  [load_dicom_images_3d(x) for x in batch_paths]\n        batch_X = np.stack(list_x)\n        if self.is_train:\n            return batch_X,batch_y\n        else:\n            return batch_X\n    \n    def on_epoch_end(self):\n        if self.shuffle and self.is_train:\n            ids_y = list(zip(self.idx, self.y))\n            shuffle(ids_y)\n            self.idx, self.y = list(zip(*ids_y))","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.410421Z","iopub.status.idle":"2021-10-15T20:50:36.411106Z","shell.execute_reply.started":"2021-10-15T20:50:36.410825Z","shell.execute_reply":"2021-10-15T20:50:36.410851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test datasets\n\n\n#train_dataset = Dataset(df_train)\n#valid_dataset = Dataset(df_valid)\ntest_dataset = Dataset(test,is_train=False)\n ","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.412438Z","iopub.status.idle":"2021-10-15T20:50:36.413064Z","shell.execute_reply.started":"2021-10-15T20:50:36.412784Z","shell.execute_reply":"2021-10-15T20:50:36.41281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#build model based on 3D CNN architecture\n\n\ndef get_model(width=128, height=128, depth=64):\n    \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n    inputs = keras.Input((width, height, depth, 1))\n\n    x = layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=128, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.Conv3D(filters=256, kernel_size=3, activation=\"relu\")(x)\n    x = layers.MaxPool3D(pool_size=2)(x)\n    x = layers.BatchNormalization()(x)\n\n    x = layers.GlobalAveragePooling3D()(x)\n    x = layers.Dense(units=512, activation=\"relu\")(x)\n    \n    \n    outputs = layers.Dense(units=1, activation=\"sigmoid\")(x)\n\n    # Define the model.\n    model = keras.Model(inputs, outputs, name=\"3dcnn\")\n    return model\n\n\n# Build model.\nmodel = get_model(width=128, height=128, depth=64)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.41553Z","iopub.status.idle":"2021-10-15T20:50:36.416037Z","shell.execute_reply.started":"2021-10-15T20:50:36.415785Z","shell.execute_reply":"2021-10-15T20:50:36.415809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd modelarch","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.416867Z","iopub.status.idle":"2021-10-15T20:50:36.417793Z","shell.execute_reply.started":"2021-10-15T20:50:36.417558Z","shell.execute_reply":"2021-10-15T20:50:36.417584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=keras.models.load_model(\"Model (1).h5\")\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.419009Z","iopub.status.idle":"2021-10-15T20:50:36.419343Z","shell.execute_reply.started":"2021-10-15T20:50:36.419155Z","shell.execute_reply":"2021-10-15T20:50:36.419176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ../","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.42045Z","iopub.status.idle":"2021-10-15T20:50:36.420792Z","shell.execute_reply.started":"2021-10-15T20:50:36.420595Z","shell.execute_reply":"2021-10-15T20:50:36.420624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ../input/weights\n","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.421785Z","iopub.status.idle":"2021-10-15T20:50:36.422104Z","shell.execute_reply.started":"2021-10-15T20:50:36.421946Z","shell.execute_reply":"2021-10-15T20:50:36.421966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.423262Z","iopub.status.idle":"2021-10-15T20:50:36.423809Z","shell.execute_reply.started":"2021-10-15T20:50:36.423623Z","shell.execute_reply":"2021-10-15T20:50:36.423642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(\"weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:50:36.425058Z","iopub.status.idle":"2021-10-15T20:50:36.425377Z","shell.execute_reply.started":"2021-10-15T20:50:36.425215Z","shell.execute_reply":"2021-10-15T20:50:36.425237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_dataset)\n#preds = preds.reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:56:29.238139Z","iopub.execute_input":"2021-10-15T20:56:29.238486Z","iopub.status.idle":"2021-10-15T20:57:33.765665Z","shell.execute_reply.started":"2021-10-15T20:56:29.238453Z","shell.execute_reply":"2021-10-15T20:57:33.764611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=preds.reshape(-1)\npreds","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:58:18.085401Z","iopub.execute_input":"2021-10-15T20:58:18.085762Z","iopub.status.idle":"2021-10-15T20:58:18.09424Z","shell.execute_reply.started":"2021-10-15T20:58:18.085724Z","shell.execute_reply":"2021-10-15T20:58:18.093234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'BraTS21ID':sample_submission['BraTS21ID'],'MGMT_value':preds})","metadata":{"execution":{"iopub.status.busy":"2021-10-15T20:59:13.946287Z","iopub.execute_input":"2021-10-15T20:59:13.94663Z","iopub.status.idle":"2021-10-15T20:59:13.9536Z","shell.execute_reply.started":"2021-10-15T20:59:13.946593Z","shell.execute_reply":"2021-10-15T20:59:13.952478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2021-10-15T21:06:33.519407Z","iopub.execute_input":"2021-10-15T21:06:33.519721Z","iopub.status.idle":"2021-10-15T21:06:33.535255Z","shell.execute_reply.started":"2021-10-15T21:06:33.519689Z","shell.execute_reply":"2021-10-15T21:06:33.534235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd ../","metadata":{"execution":{"iopub.status.busy":"2021-10-15T21:12:43.657283Z","iopub.execute_input":"2021-10-15T21:12:43.657594Z","iopub.status.idle":"2021-10-15T21:12:43.663442Z","shell.execute_reply.started":"2021-10-15T21:12:43.657562Z","shell.execute_reply":"2021-10-15T21:12:43.66271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd working","metadata":{"execution":{"iopub.status.busy":"2021-10-15T21:13:00.474135Z","iopub.execute_input":"2021-10-15T21:13:00.474496Z","iopub.status.idle":"2021-10-15T21:13:00.481761Z","shell.execute_reply.started":"2021-10-15T21:13:00.474461Z","shell.execute_reply":"2021-10-15T21:13:00.480437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submit.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-15T21:13:12.915488Z","iopub.execute_input":"2021-10-15T21:13:12.91597Z","iopub.status.idle":"2021-10-15T21:13:12.926236Z","shell.execute_reply.started":"2021-10-15T21:13:12.915893Z","shell.execute_reply":"2021-10-15T21:13:12.924769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}