{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras.models import Model, Sequential\nfrom keras import metrics,layers\nimport pydicom\nimport cv2\nimport glob\nimport os\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-16T11:10:28.560433Z","iopub.execute_input":"2021-07-16T11:10:28.560850Z","iopub.status.idle":"2021-07-16T11:10:33.567676Z","shell.execute_reply.started":"2021-07-16T11:10:28.560727Z","shell.execute_reply":"2021-07-16T11:10:33.566896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nprint(train.shape)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:10:33.569045Z","iopub.execute_input":"2021-07-16T11:10:33.569363Z","iopub.status.idle":"2021-07-16T11:10:33.601268Z","shell.execute_reply.started":"2021-07-16T11:10:33.569329Z","shell.execute_reply":"2021-07-16T11:10:33.600536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['study'] = train['BraTS21ID'].apply(lambda x: '0'*(5-len(str(x))) + str(x))\ntrain.tail()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:10:33.603142Z","iopub.execute_input":"2021-07-16T11:10:33.603471Z","iopub.status.idle":"2021-07-16T11:10:33.617788Z","shell.execute_reply.started":"2021-07-16T11:10:33.603437Z","shell.execute_reply":"2021-07-16T11:10:33.616725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.study.values\ny = train.MGMT_value.values","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:10:33.619719Z","iopub.execute_input":"2021-07-16T11:10:33.620146Z","iopub.status.idle":"2021-07-16T11:10:33.624938Z","shell.execute_reply.started":"2021-07-16T11:10:33.620107Z","shell.execute_reply":"2021-07-16T11:10:33.624087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_array(batch_x,SERIES='FLAIR',kind='train'):\n    ARRAY = np.empty((0,5,64,64,1))\n    SERIES= SERIES\n    bs = len(batch_x)\n    for i in range(bs):\n        study = batch_x[i]\n        array = np.empty((0,64,64,1))\n        file_names = glob.glob(f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/{kind}/\" + str(study) + '/' + str(SERIES) + '/*')\n        file_names.sort()\n        file_names = file_names[-5:]\n        \n        for fn in file_names:\n            img = pydicom.dcmread(fn).pixel_array\n            img = ((img - img.min())/(img.max()+1e-4)).astype(float)\n            img = cv2.resize(img,(64,64))\n            array = np.append(array,img.reshape(1,64,64,1),axis=0)\n        ARRAY = np.append(ARRAY,array.reshape(1,5,64,64,1),axis=0)\n    \n    return ARRAY\n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:36:56.389174Z","iopub.execute_input":"2021-07-16T12:36:56.389501Z","iopub.status.idle":"2021-07-16T12:36:56.397272Z","shell.execute_reply.started":"2021-07-16T12:36:56.389470Z","shell.execute_reply":"2021-07-16T12:36:56.396412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset(tf.keras.utils.Sequence):\n    def __init__(self, study, target, batch_size=8,test=0):\n        self.study, self.target = study, target\n        self.batch_size = batch_size\n        self.test = test\n\n    def __len__(self):\n        return int(np.ceil(len(self.study) / float(self.batch_size)))\n\n    def __getitem__(self, idx):\n        \n        if self.test==0:\n            batch_x = self.study[idx * self.batch_size:(idx + 1) * self.batch_size]\n            FLAIR_ARRAY = get_array(batch_x,'FLAIR','train')\n            T1W_ARRAY = get_array(batch_x,'T1w','train')\n            T1WCE_ARRAY = get_array(batch_x,'T1wCE','train')\n            T2W_ARRAY = get_array(batch_x,'T2w','train')\n            batch_y = self.target[idx * self.batch_size:(idx + 1) * self.batch_size]\n            \n            return [FLAIR_ARRAY,T1W_ARRAY,T1WCE_ARRAY,T2W_ARRAY],[np.array(batch_y)]\n        \n        \n        else:\n            batch_x = self.study[idx * self.batch_size:(idx + 1) * self.batch_size]\n            FLAIR_ARRAY = get_array(batch_x,'FLAIR','test')\n            T1W_ARRAY = get_array(batch_x,'T1w','test')\n            T1WCE_ARRAY = get_array(batch_x,'T1wCE','test')\n            T2W_ARRAY = get_array(batch_x,'T2w','test')\n            \n            return [FLAIR_ARRAY,T1W_ARRAY,T1WCE_ARRAY,T2W_ARRAY]","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:37:03.847494Z","iopub.execute_input":"2021-07-16T12:37:03.847810Z","iopub.status.idle":"2021-07-16T12:37:03.856701Z","shell.execute_reply.started":"2021-07-16T12:37:03.847779Z","shell.execute_reply":"2021-07-16T12:37:03.855928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train,X_test,y_train,y_test = train_test_split(X,y,test_size=0.2,stratify=y,random_state=2021)\nX_train.shape,X_test.shape,y_train.shape,y_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:13:49.475328Z","iopub.execute_input":"2021-07-16T11:13:49.475658Z","iopub.status.idle":"2021-07-16T11:13:49.488351Z","shell.execute_reply.started":"2021-07-16T11:13:49.475626Z","shell.execute_reply":"2021-07-16T11:13:49.487016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = RSNADataset(X_train,y_train)\nval_ds = RSNADataset(X_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:25:22.675572Z","iopub.execute_input":"2021-07-16T12:25:22.675925Z","iopub.status.idle":"2021-07-16T12:25:22.681073Z","shell.execute_reply.started":"2021-07-16T12:25:22.675887Z","shell.execute_reply":"2021-07-16T12:25:22.680201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    base_model = tf.keras.applications.ResNet50(include_top=False,weights=\"imagenet\",input_shape=(64,64,3))\n    \n    input1 = layers.Input(shape=(5,64,64,1))\n    inp1 = layers.Concatenate(axis=-1)([input1,input1,input1])\n\n    input2 = layers.Input(shape=(5,64,64,1))\n    inp2 = layers.Concatenate(axis=-1)([input2,input2,input2])\n\n    input3 = layers.Input(shape=(5,64,64,1))\n    inp3 = layers.Concatenate(axis=-1)([input3,input3,input3])\n\n    input4 = layers.Input(shape=(5,64,64,1))\n    inp4 = layers.Concatenate(axis=-1)([input4,input4,input4])\n    \n    time_dist_layer = Sequential(\n                        [base_model,\n                        layers.GlobalAveragePooling2D(),\n                        ]\n                        ) \n    \n    \n    x1 = tf.keras.layers.TimeDistributed(time_dist_layer)(inp1)\n    x2 = tf.keras.layers.TimeDistributed(time_dist_layer)(inp2)\n    x3 = tf.keras.layers.TimeDistributed(time_dist_layer)(inp3)\n    x4 = tf.keras.layers.TimeDistributed(time_dist_layer)(inp4)\n\n    #x = x1\n    x = layers.Concatenate(axis=-1)([x1,x2,x3,x4])\n    \n    x = layers.LSTM(256)(x)\n    out = layers.Dense(1,activation='sigmoid')(x)\n    model = tf.keras.Model(inputs=[input1,input2,input3,input4], outputs=out)\n    return model\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:33:22.158294Z","iopub.execute_input":"2021-07-16T11:33:22.158654Z","iopub.status.idle":"2021-07-16T11:33:22.171437Z","shell.execute_reply.started":"2021-07-16T11:33:22.158615Z","shell.execute_reply":"2021-07-16T11:33:22.170058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:33:26.385620Z","iopub.execute_input":"2021-07-16T11:33:26.385990Z","iopub.status.idle":"2021-07-16T11:33:31.148021Z","shell.execute_reply.started":"2021-07-16T11:33:26.385954Z","shell.execute_reply":"2021-07-16T11:33:31.147104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=\"adam\",loss=tf.keras.losses.BinaryCrossentropy(),metrics=tf.keras.metrics.AUC())","metadata":{"execution":{"iopub.status.busy":"2021-07-16T11:33:47.838669Z","iopub.execute_input":"2021-07-16T11:33:47.839000Z","iopub.status.idle":"2021-07-16T11:33:47.868011Z","shell.execute_reply.started":"2021-07-16T11:33:47.838967Z","shell.execute_reply":"2021-07-16T11:33:47.867227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(filepath='/kaggle/working/model.h5',monitor='val_loss',mode='min',save_best_only=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:06:36.900216Z","iopub.execute_input":"2021-07-16T12:06:36.900575Z","iopub.status.idle":"2021-07-16T12:06:36.905566Z","shell.execute_reply.started":"2021-07-16T12:06:36.900540Z","shell.execute_reply":"2021-07-16T12:06:36.904509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_ds,validation_data=val_ds,epochs=10,callbacks=model_checkpoint_callback)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:06:41.469735Z","iopub.execute_input":"2021-07-16T12:06:41.470150Z","iopub.status.idle":"2021-07-16T12:14:06.776889Z","shell.execute_reply.started":"2021-07-16T12:06:41.470111Z","shell.execute_reply":"2021-07-16T12:14:06.776123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:14:12.796332Z","iopub.execute_input":"2021-07-16T12:14:12.796688Z","iopub.status.idle":"2021-07-16T12:14:12.822713Z","shell.execute_reply.started":"2021-07-16T12:14:12.796625Z","shell.execute_reply":"2021-07-16T12:14:12.821641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['study'] = test['BraTS21ID'].apply(lambda x: '0'*(5-len(str(x))) + str(x))\ntest.tail()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:14:19.639708Z","iopub.execute_input":"2021-07-16T12:14:19.640088Z","iopub.status.idle":"2021-07-16T12:14:19.653090Z","shell.execute_reply.started":"2021-07-16T12:14:19.640055Z","shell.execute_reply":"2021-07-16T12:14:19.651497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = RSNADataset(test.study.values,None,test=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:36:24.909363Z","iopub.execute_input":"2021-07-16T12:36:24.909696Z","iopub.status.idle":"2021-07-16T12:36:24.915325Z","shell.execute_reply.started":"2021-07-16T12:36:24.909664Z","shell.execute_reply":"2021-07-16T12:36:24.914333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model(\"/kaggle/working/model.h5\")\n\npreds = model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:37:44.579465Z","iopub.execute_input":"2021-07-16T12:37:44.579792Z","iopub.status.idle":"2021-07-16T12:38:23.100308Z","shell.execute_reply.started":"2021-07-16T12:37:44.579761Z","shell.execute_reply":"2021-07-16T12:38:23.099416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['MGMT_value'] = preds\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:39:08.287663Z","iopub.execute_input":"2021-07-16T12:39:08.288010Z","iopub.status.idle":"2021-07-16T12:39:08.299657Z","shell.execute_reply.started":"2021-07-16T12:39:08.287977Z","shell.execute_reply":"2021-07-16T12:39:08.298863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[['BraTS21ID','MGMT_value']].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T12:40:01.005326Z","iopub.execute_input":"2021-07-16T12:40:01.005654Z","iopub.status.idle":"2021-07-16T12:40:01.023832Z","shell.execute_reply.started":"2021-07-16T12:40:01.005616Z","shell.execute_reply":"2021-07-16T12:40:01.023032Z"},"trusted":true},"execution_count":null,"outputs":[]}]}