{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport matplotlib.pyplot as plt\nimport os\nimport random\nfrom sklearn import model_selection as sk_model_selection\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import activations\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom keras_preprocessing.image.dataframe_iterator import DataFrameIterator\nfrom tensorflow.keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\n\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-14T22:16:16.41267Z","iopub.execute_input":"2021-12-14T22:16:16.413031Z","iopub.status.idle":"2021-12-14T22:16:16.420751Z","shell.execute_reply.started":"2021-12-14T22:16:16.412994Z","shell.execute_reply":"2021-12-14T22:16:16.419751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    INPUT_PATH_DCM = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/'\n    INPUT_PATH_PNG = '../input/rsna-miccai-png/'\n    TENSORBOARD_LOG_DIR = '../working/log_tensorboard/'\n    SEED = 42\n    #These should be removed from the dataset\n    EXCLUDED_STR = ['00109', '00123', '00709']\n    EXCLUDED_INT = [109, 123, 709]\n\n    #Defining target size of image\n    IMG_SIZE = 224\n    NUM_SLICES_3D = 64\n    MIN_SLICES = 12\n    \n    BATCH_SIZE = 64\n    \n    CLASS_MODE = 'binary'\n    COLOR_MODE = 'rgb'\n    TARGET_SIZE = (224, 224)\n    def __self__():\n        pass\n    @staticmethod\n    def set_seed(seed_val):\n        tf.random.set_seed(seed_val)\n        random.seed(seed_val)\n        os.environ['PYTHONHASHSEED'] = str(seed_val)\n        np.random.seed(seed_val)\n        ","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:16.422963Z","iopub.execute_input":"2021-12-14T22:16:16.423369Z","iopub.status.idle":"2021-12-14T22:16:16.438561Z","shell.execute_reply.started":"2021-12-14T22:16:16.423322Z","shell.execute_reply":"2021-12-14T22:16:16.43767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config.set_seed(Config.SEED)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:16.439853Z","iopub.execute_input":"2021-12-14T22:16:16.44011Z","iopub.status.idle":"2021-12-14T22:16:16.461987Z","shell.execute_reply.started":"2021-12-14T22:16:16.440078Z","shell.execute_reply":"2021-12-14T22:16:16.461222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Getting to know the data","metadata":{}},{"cell_type":"code","source":"!ls ../input/rsna-miccai-brain-tumor-radiogenomic-classification/\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:16.465393Z","iopub.execute_input":"2021-12-14T22:16:16.466423Z","iopub.status.idle":"2021-12-14T22:16:17.265906Z","shell.execute_reply.started":"2021-12-14T22:16:16.466366Z","shell.execute_reply":"2021-12-14T22:16:17.264558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(Config.INPUT_PATH_DCM+'train_labels.csv', dtype={\n    'BraTS21ID': str, 'MGMT_value':int\n})\n\ndf_test = pd.read_csv(Config.INPUT_PATH_DCM+'sample_submission.csv', dtype={\n    'BraTS21ID': str, 'MGMT_value':float\n})\n\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.269066Z","iopub.execute_input":"2021-12-14T22:16:17.269349Z","iopub.status.idle":"2021-12-14T22:16:17.293095Z","shell.execute_reply.started":"2021-12-14T22:16:17.269315Z","shell.execute_reply":"2021-12-14T22:16:17.292467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lendo dataset \ndf = df[~df['BraTS21ID'].isin(Config.EXCLUDED_STR)]\n\n#Para testes, usando 1/3 do dataset\n# df = df[:int(len(df)/3)]\n\n\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.294227Z","iopub.execute_input":"2021-12-14T22:16:17.294566Z","iopub.status.idle":"2021-12-14T22:16:17.301362Z","shell.execute_reply.started":"2021-12-14T22:16:17.294537Z","shell.execute_reply":"2021-12-14T22:16:17.300724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adapted from: https://www.kaggle.com/leandronidas/tf-efficientnet-transfer-learning-strat-split/edit\n\ndf['flair'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/FLAIR/')\ndf['t1w'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T1w/')\ndf['t1wce'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T1wCE/')\ndf['t2w'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T2w/')\n\ndf['brats21idInt'] = df['BraTS21ID'].astype(int)\n\ndf_test['flair'] = df_test['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'test/'+file_id+'/FLAIR/')\ndf_test['t1w'] = df_test['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'test/'+file_id+'/T1w/')\ndf_test['t1wce'] = df_test['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'test/'+file_id+'/T1wCE/')\ndf_test['t2w'] = df_test['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'test/'+file_id+'/T2w/')\n\ndf_test['brats21idInt'] = df_test['BraTS21ID'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.302612Z","iopub.execute_input":"2021-12-14T22:16:17.302847Z","iopub.status.idle":"2021-12-14T22:16:17.327198Z","shell.execute_reply.started":"2021-12-14T22:16:17.302818Z","shell.execute_reply":"2021-12-14T22:16:17.326263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\npred = sample_y\npred['BraTS21ID5'] = [format(x, '05d') for x in pred.BraTS21ID]\npred.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.328895Z","iopub.execute_input":"2021-12-14T22:16:17.329225Z","iopub.status.idle":"2021-12-14T22:16:17.345814Z","shell.execute_reply.started":"2021-12-14T22:16:17.32918Z","shell.execute_reply":"2021-12-14T22:16:17.345087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Divisao estratificada em treino, teste e validação\ndf_trainval, df_test = sk_model_selection.train_test_split(\n    df, \n    test_size=0.15, \n    random_state=Config.SEED, \n    stratify=df[\"MGMT_value\"],\n)\ndf_train, df_val = sk_model_selection.train_test_split(\n    df_trainval, \n    test_size=0.2, \n    random_state=Config.SEED, \n    stratify=df_trainval[\"MGMT_value\"],\n)\n\nprint(df_train.shape)\nprint(df_val.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.347155Z","iopub.execute_input":"2021-12-14T22:16:17.347532Z","iopub.status.idle":"2021-12-14T22:16:17.362647Z","shell.execute_reply.started":"2021-12-14T22:16:17.347493Z","shell.execute_reply":"2021-12-14T22:16:17.36168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.363856Z","iopub.execute_input":"2021-12-14T22:16:17.364736Z","iopub.status.idle":"2021-12-14T22:16:17.384544Z","shell.execute_reply.started":"2021-12-14T22:16:17.364698Z","shell.execute_reply":"2021-12-14T22:16:17.383615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_val[df_val['brats21idInt'] == 109].any())\n\nprint(df_train[df_train['brats21idInt'] == 109].any())","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.385811Z","iopub.execute_input":"2021-12-14T22:16:17.386247Z","iopub.status.idle":"2021-12-14T22:16:17.398559Z","shell.execute_reply.started":"2021-12-14T22:16:17.386212Z","shell.execute_reply":"2021-12-14T22:16:17.397615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    \ndef get_iterating_dataframe(df, mri_type):\n    \n    all_img_files = []\n    all_img_labels = []\n    all_img_patient_ids = []\n    for row in df.iterrows():\n\n        img_dir = row[1][mri_type]\n        img_files = os.listdir(img_dir)\n        img_nums = sorted([int(ele.replace('Image-', '').replace('.png', '')) for ele in img_files])\n        totalnum_images = len(img_nums)\n        mid_point = int(totalnum_images//2)\n        start_point = mid_point - max(int(mid_point*0.1), Config.MIN_SLICES//2)\n        end_point = mid_point + max(int(mid_point*0.1), Config.MIN_SLICES//2)\n\n        img_names = [f'Image-{img_nums[i]}.png' for i in range(start_point, end_point+1)]\n\n        img_paths = [img_dir+ele for ele in img_names]\n        img_labels = [row[1]['MGMT_value']]*len(img_paths)\n        img_patient_ids = [row[1]['brats21idInt']]*len(img_paths)\n        all_img_files.extend(img_paths)\n        all_img_labels.extend(img_labels)\n        all_img_patient_ids.extend(img_patient_ids)\n\n    new_df = pd.DataFrame({'patient_ids': all_img_patient_ids,\n                  'labels': all_img_labels,\n                  'file_paths': all_img_files})\n            \n    return new_df","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.400153Z","iopub.execute_input":"2021-12-14T22:16:17.400652Z","iopub.status.idle":"2021-12-14T22:16:17.411785Z","shell.execute_reply.started":"2021-12-14T22:16:17.400611Z","shell.execute_reply":"2021-12-14T22:16:17.410852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adapted from gist: https://gist.github.com/abr-98/4e12ef6d7eca7cbcc3d9f37a526140b7\n\nclass PNGDataFrameIterator(DataFrameIterator):\n    def __init__(self, *arg, **kwargs):\n        self.white_list_formats = ('png')\n        super(PNGDataFrameIterator, self).__init__(*arg, **kwargs)\n        self.dataframe = kwargs['dataframe']\n        self.x = self.dataframe[kwargs['x_col']]\n        self.y = self.dataframe[kwargs['y_col']]\n        self.color_mode = kwargs['color_mode']\n        self.target_size = kwargs['target_size']\n\n    def _get_batches_of_transformed_samples(self, indices_array):\n        # get batch of images\n        batch_x = np.array([self.read_png_as_array(path, self.target_size, \n                                                   color_mode=self.color_mode)\n                            for path in self.x.iloc[indices_array]])\n\n        batch_y = np.array(self.y.iloc[indices_array].astype(np.uint8))  # astype because y was passed as str\n\n        # transform images\n        if self.image_data_generator is not None:\n            for i, (x, y) in enumerate(zip(batch_x, batch_y)):\n                transform_params = self.image_data_generator.get_random_transform(x.shape)\n                batch_x[i] = self.image_data_generator.apply_transform(x, transform_params)\n      \n\n        return batch_x, batch_y\n\n    \n    @staticmethod\n    def read_png_as_array(path, target_size=(Config.IMG_SIZE, Config.IMG_SIZE),\n                          color_mode='rgb'):\n        im_gray = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        pixels = im_gray - np.min(im_gray)\n        pixels = pixels / np.max(pixels)\n        image_manual_norm = (pixels * 255).astype(np.uint8)\n        image_array = cv2.resize(image_manual_norm, target_size, interpolation=cv2.INTER_CUBIC) \n        \n        if color_mode == 'rgb':\n            image_array = np.dstack((image_array, np.zeros_like(image_array), np.zeros_like(image_array)))\n        return image_array\n\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.413006Z","iopub.execute_input":"2021-12-14T22:16:17.413644Z","iopub.status.idle":"2021-12-14T22:16:17.430045Z","shell.execute_reply.started":"2021-12-14T22:16:17.413604Z","shell.execute_reply":"2021-12-14T22:16:17.428955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exam_list = ['flair','t1w','t1wce','t2w']\nexam_list = ['flair']\n\ndftrainIter_dict = {}\ndfvalIter_dict = {}\ndftestIter_dict = {}\ntrain_gen_dict = {}\nval_gen_dict = {}\ntest_gen_dict = {}\nmodel_check_dict = {}\nmodel_dict = {}\n\nfor exam in exam_list:\n    dftrainIter = get_iterating_dataframe(df_train, exam)\n    dftrainIter['labels_str'] = dftrainIter['labels'].astype(str)\n\n    dfvalIter = get_iterating_dataframe(df_val, exam)\n    dfvalIter['labels_str'] = dfvalIter['labels'].astype(str)\n\n    dftestIter = get_iterating_dataframe(df_test, exam)\n    dftestIter['labels_str'] = dftestIter['labels'].astype(str)\n\n    datagen = ImageDataGenerator(\n            preprocessing_function=tf.keras.applications.resnet50.preprocess_input,\n            zoom_range=0.2,\n            rotation_range=45,\n            fill_mode='nearest',\n            height_shift_range= 0.15,\n            width_shift_range=0.15,\n            horizontal_flip=True,\n            vertical_flip=True,\n            brightness_range = [0.8, 1.2],\n            rescale=1.0/255,\n    )\n    test_datagen = ImageDataGenerator(\n        preprocessing_function=tf.keras.applications.resnet50.preprocess_input,\n        rescale=1.0/255,\n    )\n\n    train_generator = PNGDataFrameIterator(dataframe=dftrainIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n\n    val_generator = PNGDataFrameIterator(dataframe=dfvalIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=test_datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n\n    test_generator = PNGDataFrameIterator(dataframe=dftestIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=test_datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n    model_resnet = tf.keras.applications.ResNet50(weights='imagenet',\n                                              input_shape=(Config.IMG_SIZE, Config.IMG_SIZE, 3),\n                                              include_top=False)\n    model_effnetb0 = tf.keras.applications.EfficientNetB0(\n                                                    weights='imagenet',\n                                                  input_shape=(Config.IMG_SIZE, Config.IMG_SIZE, 3),\n                                                  include_top=False)\n\n    model_deep = model_resnet\n    \n        \n    # Congela todas as camadas exceto as ultimas\n    for layer in model_deep.layers[:143]:\n        layer.trainable = False\n\n#     # Debug - checa qual layer está congelado    \n#     for i, layer in enumerate(model_deep.layers):\n#         print( i, layer.name, \"-\" ,layer.trainable)\n        \n        \n    x = layers.GlobalMaxPooling2D()(model_deep.output)\n#     x = layers.Flatten()(model_deep.output)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.Dense(32, activation='relu')(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dense(32, activation='relu')(x)\n    x = layers.Dropout(0.5)(x)\n    x = layers.BatchNormalization()(x)\n#     x = layers.Dense(128, activation='relu')(x)\n#     x = layers.Dropout(0.5)(x)\n#     x = layers.BatchNormalization()(x)\n    out = layers.Dense(1, activation='sigmoid')(x)\n\n    model = Model(inputs=model_deep.input, outputs=out)\n\n    opt = tf.keras.optimizers.Adam(learning_rate=1e-4)\n\n#     model.summary()\n    \n    model_checkpoint = ModelCheckpoint('../working/resnet2d_congelada_' + exam + '.hdf5', \n                                       monitor='val_loss',verbose=1, \n                                       save_best_only=True)\n    early_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss',\n                                                      patience=10,\n                                                      restore_best_weights=True)\n    tensorboard = tf.keras.callbacks.TensorBoard(log_dir=Config.TENSORBOARD_LOG_DIR,\n                                                histogram_freq=1,\n                                                )\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5,\n                                       patience=2, min_lr=1e-7,\n                                      verbose=1)\n    #model_ResNet50_2d.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy'])\n    model.compile(loss='binary_crossentropy', optimizer=opt, \n                              metrics=[tf.keras.metrics.BinaryAccuracy(),\n                                       tf.keras.metrics.AUC(from_logits=False)] )\n    \n\n    dftrainIter_dict[exam] = dftrainIter\n    dfvalIter_dict[exam] = dfvalIter\n    dftestIter_dict[exam] = dftestIter \n    train_gen_dict[exam] = train_generator\n    val_gen_dict[exam] = val_generator\n    test_gen_dict[exam] = test_generator\n    model_check_dict[exam] = model_checkpoint\n    model_dict[exam] = model","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:17.433472Z","iopub.execute_input":"2021-12-14T22:16:17.433732Z","iopub.status.idle":"2021-12-14T22:16:24.425839Z","shell.execute_reply.started":"2021-12-14T22:16:17.433704Z","shell.execute_reply":"2021-12-14T22:16:24.422651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_dict","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:24.427106Z","iopub.execute_input":"2021-12-14T22:16:24.42734Z","iopub.status.idle":"2021-12-14T22:16:24.433671Z","shell.execute_reply.started":"2021-12-14T22:16:24.427311Z","shell.execute_reply":"2021-12-14T22:16:24.432912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NUM_EPOCHS = 6 * 5 #Should take ~1hr to run for each 5 epochs\n# history_dict = {}\n# for exam in exam_list:\n#     history_dict[exam] = model_dict[exam].fit(train_gen_dict[exam],\n#                                     validation_data=val_gen_dict[exam],\n#                                     epochs = NUM_EPOCHS,\n#                                     verbose = 1,\n#                                     callbacks=[\n#                                         early_stopping, \n#                                         model_check_dict[exam],\n#                                         reduce_lr,\n#                                               ]\n#                                     )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:24.43506Z","iopub.execute_input":"2021-12-14T22:16:24.435295Z","iopub.status.idle":"2021-12-14T22:16:24.444963Z","shell.execute_reply.started":"2021-12-14T22:16:24.435265Z","shell.execute_reply":"2021-12-14T22:16:24.443915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for exam in exam_list:\n#     history = history_dict[exam]\n#     plt.plot(history.history['loss'])\n#     plt.plot(history.history['val_loss'])\n#     plt.title('model loss '+ exam)\n#     plt.ylabel('loss')\n#     plt.xlabel('epoch')\n#     plt.legend(['train', 'val'], loc='upper left')\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:24.446503Z","iopub.execute_input":"2021-12-14T22:16:24.447047Z","iopub.status.idle":"2021-12-14T22:16:24.458535Z","shell.execute_reply.started":"2021-12-14T22:16:24.447013Z","shell.execute_reply":"2021-12-14T22:16:24.45764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Loading previously trained resnets (5 epochs for now)\n#!ls ../input/rsna-miccai-resnet50-4exams-5epochs/resnet2d_congelada_' + exam + '.hdf5'\nfor exam in exam_list:\n    model_dict[exam].load_weights('../input/pesosrenset2/resnet2d_partially_congelada_' + exam + '.hdf5')\n#     model_dict[exam].load_weights('../input/rsna-miccai-resnet50-4exams-5epochs/resnet2d_congelada_' + exam + '.hdf5')","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:24.459966Z","iopub.execute_input":"2021-12-14T22:16:24.460282Z","iopub.status.idle":"2021-12-14T22:16:26.963949Z","shell.execute_reply.started":"2021-12-14T22:16:24.460239Z","shell.execute_reply":"2021-12-14T22:16:26.963045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def aggregate_predictions(df):\n    # aggregate the predictions on all image for each person (take the most confident prediction out of all image predictions)\n    mean_pred = df['pred_y'].mean()\n    test_pred_agg = df.groupby('patient_ids').apply(\n        lambda x: x['pred_y'].max()\n        if (x['pred_y'].max() - mean_pred) > (mean_pred - x['pred_y'].min()) \n        else x['pred_y'].min()\n    )\n    return test_pred_agg\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:16:26.965094Z","iopub.execute_input":"2021-12-14T22:16:26.965327Z","iopub.status.idle":"2021-12-14T22:16:26.971521Z","shell.execute_reply.started":"2021-12-14T22:16:26.965297Z","shell.execute_reply":"2021-12-14T22:16:26.97064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## val set analise","metadata":{}},{"cell_type":"code","source":"def generate_predictions(gen_dict, dfIter_dict):\n    val_pred_dict = {}\n    for exam in exam_list:\n        print('Computing prediction for ' + exam + '...')\n        val_pred = model_dict[exam].predict(gen_dict[exam], steps=len(gen_dict[exam]))\n        dfIter_dict[exam]['pred_y'] = val_pred\n        mean_pred = val_pred.mean()\n        val_pred_agg = dfIter_dict[exam].groupby('patient_ids').apply(\n            lambda x: x['pred_y'].mean()\n#             if (x['pred_y'].max() - mean_pred) > (mean_pred - x['pred_y'].min()) \n#             else x['pred_y'].min()\n        )\n        val_pred_dict[exam] = val_pred_agg\n    print('done!')\n    return val_pred_dict\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:30:24.424066Z","iopub.execute_input":"2021-12-14T22:30:24.424374Z","iopub.status.idle":"2021-12-14T22:30:24.430667Z","shell.execute_reply.started":"2021-12-14T22:30:24.424338Z","shell.execute_reply":"2021-12-14T22:30:24.429947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred_dict = generate_predictions(val_gen_dict, dfvalIter_dict)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:30:24.600087Z","iopub.execute_input":"2021-12-14T22:30:24.600742Z","iopub.status.idle":"2021-12-14T22:32:06.305915Z","shell.execute_reply.started":"2021-12-14T22:30:24.600699Z","shell.execute_reply":"2021-12-14T22:32:06.305045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ROC_AUC_graph(df, pred_dict, colname):\n    ns_probs = [0 for _ in range(len(df.MGMT_value))]\n    print(exam_list)\n    for exam in exam_list:\n\n    #     dfvalIter['pred_y'] = predictions\n    #     pred_agg = aggregate_predictions(dfvalIter)\n\n        predagg_sorted = pred_dict[exam].sort_index()\n\n\n        dfsorted = df.sort_values(by='brats21idInt')\n        dfsorted[colname] = predagg_sorted.values\n\n        labels = df.sort_values(by='brats21idInt').MGMT_value.tolist()\n\n        ns_probs = [0 for _ in range(len(labels))]\n        probs = dfsorted[colname]\n\n        ns_auc = roc_auc_score(labels, ns_probs)\n        lr_auc = roc_auc_score(labels, probs)\n        print('No skill: ROC AUC=%.3f' % (ns_auc))\n        print('CNN: ROC AUC=%.3f' % (lr_auc))\n        lr_fpr, lr_tpr, _ = roc_curve(labels, probs)\n        plt.plot(lr_fpr, lr_tpr, marker='.', label='CNN')\n\n        ns_fpr, ns_tpr, _ = roc_curve(labels, ns_probs)\n        plt.plot(ns_fpr, ns_tpr, marker='.', label='No Skill')\n\n        plt.xlabel('False Positive Rate')\n        plt.ylabel('True Positive Rate')\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:32:06.307566Z","iopub.execute_input":"2021-12-14T22:32:06.307836Z","iopub.status.idle":"2021-12-14T22:32:06.317858Z","shell.execute_reply.started":"2021-12-14T22:32:06.307804Z","shell.execute_reply":"2021-12-14T22:32:06.317017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_AUC_graph(df_val, val_pred_dict, 'y_pred')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:32:06.319042Z","iopub.execute_input":"2021-12-14T22:32:06.319266Z","iopub.status.idle":"2021-12-14T22:32:06.477863Z","shell.execute_reply.started":"2021-12-14T22:32:06.31923Z","shell.execute_reply":"2021-12-14T22:32:06.476927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test set ROC AUC","metadata":{}},{"cell_type":"code","source":"test_pred_dict = genereate_predictions(test_gen_dict, dftestIter_dict)\nROC_AUC_graph(df_test, test_pred_dict, 'y_pred')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:25:50.813045Z","iopub.execute_input":"2021-12-14T22:25:50.813286Z","iopub.status.idle":"2021-12-14T22:27:22.289825Z","shell.execute_reply.started":"2021-12-14T22:25:50.813257Z","shell.execute_reply":"2021-12-14T22:27:22.288895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"predagg_sorted = val_pred_dict[exam].sort_index()\ndfsorted = df_val.sort_values(by='brats21idInt')\ndfsorted['y_pred'] = predagg_sorted.values\ndfsorted[(dfsorted['y_pred'] > 0.5) & (dfsorted['MGMT_value'] == 0)]\n# dfsorted[(dfsorted['y_pred'] > 0.5)]","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:43:02.210068Z","iopub.execute_input":"2021-12-14T22:43:02.210389Z","iopub.status.idle":"2021-12-14T22:43:02.240301Z","shell.execute_reply.started":"2021-12-14T22:43:02.210358Z","shell.execute_reply":"2021-12-14T22:43:02.239283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Grad CAM","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\n\n# Display\nfrom IPython.display import Image, display\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\n\ndef get_img_array(img_path, size):\n    # `img` is a PIL image of size 299x299\n    img = keras.preprocessing.image.load_img(img_path, target_size=size)\n    # `array` is a float32 Numpy array of shape (299, 299, 3)\n    array = keras.preprocessing.image.img_to_array(img)\n    # We add a dimension to transform our array into a \"batch\"\n    # of size (1, 299, 299, 3)\n    array = np.expand_dims(array, axis=0)\n    return array\n\n\ndef make_gradcam_heatmap(img_array, model, last_conv_layer_name, pred_index=None):\n    # First, we create a model that maps the input image to the activations\n    # of the last conv layer as well as the output predictions\n    grad_model = tf.keras.models.Model(\n        [model.inputs], [model.get_layer(last_conv_layer_name).output, model.output]\n    )\n\n    # Then, we compute the gradient of the top predicted class for our input image\n    # with respect to the activations of the last conv layer\n    with tf.GradientTape() as tape:\n        last_conv_layer_output, preds = grad_model(img_array)\n        if pred_index is None:\n            pred_index = tf.argmax(preds[0])\n        class_channel = preds[:, pred_index]\n\n    # This is the gradient of the output neuron (top predicted or chosen)\n    # with regard to the output feature map of the last conv layer\n    grads = tape.gradient(class_channel, last_conv_layer_output)\n\n    # This is a vector where each entry is the mean intensity of the gradient\n    # over a specific feature map channel\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n\n    # We multiply each channel in the feature map array\n    # by \"how important this channel is\" with regard to the top predicted class\n    # then sum all the channels to obtain the heatmap class activation\n    last_conv_layer_output = last_conv_layer_output[0]\n    heatmap = last_conv_layer_output @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n\n    # For visualization purpose, we will also normalize the heatmap between 0 & 1\n    heatmap = tf.maximum(heatmap, 0) / tf.math.reduce_max(heatmap)\n    return heatmap.numpy()\n\n\ndef save_and_display_gradcam(img, heatmap, cam_path=\"cam.jpg\", alpha=0.5):\n    # Load the original image\n#     img = keras.preprocessing.image.load_img(img_path)\n#     img = keras.preprocessing.image.img_to_array(img)\n\n    # Rescale heatmap to a range 0-255\n    heatmap = np.uint8(255 * heatmap)\n\n    # Use jet colormap to colorize heatmap\n    jet = cm.get_cmap(\"jet\")\n\n    # Use RGB values of the colormap\n    jet_colors = jet(np.arange(256))[:, :3]\n    jet_heatmap = jet_colors[heatmap]\n\n    # Create an image with RGB colorized heatmap\n    jet_heatmap = keras.preprocessing.image.array_to_img(jet_heatmap)\n    jet_heatmap = jet_heatmap.resize((img.shape[1], img.shape[0]))\n    jet_heatmap = keras.preprocessing.image.img_to_array(jet_heatmap)\n\n    # Superimpose the heatmap on original image\n    superimposed_img = jet_heatmap * alpha + img\n    superimposed_img = keras.preprocessing.image.array_to_img(superimposed_img)\n\n    # Save the superimposed image\n    superimposed_img.save(cam_path)\n\n    # Display Grad CAM\n    display(Image(cam_path))\n\n\ndef gradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=None,\n           last_conv_layer_name='conv5_block3_3_conv',\n           output_prefix = None):\n\n    # Prepare image\n    if model is None:\n        model_builder = keras.applications.ResNet50\n        model = model_builder(weights=\"imagenet\")\n    \n    \n    img_size = (Config.IMG_SIZE, Config.IMG_SIZE)\n\n    for i, img_path in enumerate(files_paths):\n        # The local path to our target image\n        img_array = preprocess_input(get_img_array(img_path, size=img_size))\n\n        model.layers[-1].activation = None\n\n        # Generate class activation heatmap\n        heatmap = make_gradcam_heatmap(img_array, model, last_conv_layer_name)\n        \n\n        output_path = '../working/gradCam_out_' + output_prefix + '_' + str(i) +'.png'\n        \n        save_and_display_gradcam(img_array[0], heatmap, output_path)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:19:42.846484Z","iopub.execute_input":"2021-12-14T22:19:42.84675Z","iopub.status.idle":"2021-12-14T22:19:42.870868Z","shell.execute_reply.started":"2021-12-14T22:19:42.846715Z","shell.execute_reply":"2021-12-14T22:19:42.870101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_paths = [\n    # mgmt = 1\n    '../input/rsna-miccai-png/train/00188/FLAIR/Image-42.png',\n    '../input/rsna-miccai-png/train/00360/FLAIR/Image-32.png',\n     '../input/rsna-miccai-png/train/00366/FLAIR/Image-34.png',\n    #mgmt = 0\n    '../input/rsna-miccai-png/train/00024/FLAIR/Image-302.png',\n    '../input/rsna-miccai-png/train/00003/FLAIR/Image-448.png',\n    '../input/rsna-miccai-png/train/00343/FLAIR/Image-28.png',\n    \n]\ngradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=model_dict['flair'],\n            output_prefix = 'flair'\n           )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:45:32.83383Z","iopub.execute_input":"2021-12-14T22:45:32.834646Z","iopub.status.idle":"2021-12-14T22:45:36.553416Z","shell.execute_reply.started":"2021-12-14T22:45:32.834602Z","shell.execute_reply":"2021-12-14T22:45:36.552533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_paths = [\n    '../input/rsna-miccai-png/train/00000/FLAIR/Image-242.png',\n    '../input/rsna-miccai-png/train/00045/FLAIR/Image-116.png',\n     '../input/rsna-miccai-png/train/00133/FLAIR/Image-35.png',\n    \n]\ngradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=model_dict['flair'],\n            output_prefix = 'flair'\n           )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:19:44.979641Z","iopub.execute_input":"2021-12-14T22:19:44.979875Z","iopub.status.idle":"2021-12-14T22:19:46.853915Z","shell.execute_reply.started":"2021-12-14T22:19:44.979846Z","shell.execute_reply":"2021-12-14T22:19:46.8533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_paths = [\n    '../input/rsna-miccai-png/train/00000/FLAIR/Image-242.png',\n    '../input/rsna-miccai-png/train/00000/FLAIR/Image-147.png',\n    '../input/rsna-miccai-png/train/00000/T1w/Image-15.png',\n    '../input/rsna-miccai-png/train/00000/T1w/Image-24.png',\n    '../input/rsna-miccai-png/train/00000/T1wCE/Image-80.png',\n    '../input/rsna-miccai-png/train/00000/T1wCE/Image-59.png',\n    '../input/rsna-miccai-png/train/00000/T2w/Image-150.png',\n    '../input/rsna-miccai-png/train/00000/T2w/Image-210.png',\n    '../input/rsna-miccai-png/train/00733/FLAIR/Image-75.png',\n    '../input/rsna-miccai-png/train/00733/FLAIR/Image-106.png',\n    '../input/rsna-miccai-png/train/00733/T1w/Image-74.png',\n    '../input/rsna-miccai-png/train/00733/T1w/Image-53.png',\n    '../input/rsna-miccai-png/train/00733/T1wCE/Image-74.png',\n    '../input/rsna-miccai-png/train/00733/T1wCE/Image-53.png',\n    '../input/rsna-miccai-png/train/00733/T2w/Image-135.png',\n    '../input/rsna-miccai-png/train/00733/T2w/Image-101.png',\n]\ngradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=model_dict['flair'],\n            output_prefix = 'flair'\n           )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:19:46.85519Z","iopub.execute_input":"2021-12-14T22:19:46.85562Z","iopub.status.idle":"2021-12-14T22:19:56.701543Z","shell.execute_reply.started":"2021-12-14T22:19:46.855588Z","shell.execute_reply":"2021-12-14T22:19:56.700748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_paths = [\n    '../input/rsna-miccai-png/train/00000/T2w/Image-126.png',\n    '../input/rsna-miccai-png/train/00045/T2w/Image-323.png',\n     '../input/rsna-miccai-png/train/00133/T2w//Image-19.png',\n    \n]\ngradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=model_dict['t2w'],\n            output_prefix = 't2w'\n           )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:19:56.702715Z","iopub.execute_input":"2021-12-14T22:19:56.702939Z","iopub.status.idle":"2021-12-14T22:19:56.740865Z","shell.execute_reply.started":"2021-12-14T22:19:56.70291Z","shell.execute_reply":"2021-12-14T22:19:56.73983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_paths = [\n    '../input/rsna-miccai-png/train/00000/T1wCE/Image-78.png',\n    '../input/rsna-miccai-png/train/00045/T1wCE/Image-77.png',\n     '../input/rsna-miccai-png/train/00133/T1wCE//Image-35.png',\n    \n]\ngradCAM(files_paths, \n            preprocess_input=keras.applications.resnet50.preprocess_input,\n            model=model_dict['t1wce'],\n            output_prefix = 't1wce'\n           )","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:19:56.742011Z","iopub.status.idle":"2021-12-14T22:19:56.742408Z","shell.execute_reply.started":"2021-12-14T22:19:56.742218Z","shell.execute_reply":"2021-12-14T22:19:56.742237Z"},"trusted":true},"execution_count":null,"outputs":[]}]}