{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2\nimport matplotlib.pyplot as plt\nimport os\nimport random\nfrom sklearn import model_selection as sk_model_selection\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import activations\nfrom tensorflow.keras import Model, Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom keras_preprocessing.image.dataframe_iterator import DataFrameIterator\nfrom tensorflow.keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\n\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import roc_auc_score\n\nfrom operator import itemgetter\nfrom scipy import ndimage\nimport matplotlib.image as mpimg\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-17T13:41:57.283558Z","iopub.execute_input":"2021-12-17T13:41:57.283965Z","iopub.status.idle":"2021-12-17T13:42:02.918943Z","shell.execute_reply.started":"2021-12-17T13:41:57.283852Z","shell.execute_reply":"2021-12-17T13:42:02.918163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    INPUT_PATH_DCM = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/'\n    INPUT_PATH_PNG = '../input/rsna-miccai-png/'\n    TENSORBOARD_LOG_DIR = '../working/log_tensorboard/'\n    SEED = 42\n    #These should be removed from the dataset\n    EXCLUDED_STR = ['00109', '00123', '00709']\n    EXCLUDED_INT = [109, 123, 709]\n\n    #Defining target size of image\n    IMG_SIZE = 224\n    NUM_SLICES_3D = 64\n    MIN_SLICES = 12\n    \n    BATCH_SIZE = 64\n    \n    CLASS_MODE = 'binary'\n    COLOR_MODE = 'rgb'\n    TARGET_SIZE = (256, 256)\n    def __self__():\n        pass\n    @staticmethod\n    def set_seed(seed_val):\n        tf.random.set_seed(seed_val)\n        random.seed(seed_val)\n        os.environ['PYTHONHASHSEED'] = str(seed_val)\n        np.random.seed(seed_val)\n        ","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:02.920439Z","iopub.execute_input":"2021-12-17T13:42:02.920681Z","iopub.status.idle":"2021-12-17T13:42:02.926977Z","shell.execute_reply.started":"2021-12-17T13:42:02.920646Z","shell.execute_reply":"2021-12-17T13:42:02.926408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Config.set_seed(Config.SEED)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:02.928031Z","iopub.execute_input":"2021-12-17T13:42:02.928431Z","iopub.status.idle":"2021-12-17T13:42:02.943073Z","shell.execute_reply.started":"2021-12-17T13:42:02.928395Z","shell.execute_reply":"2021-12-17T13:42:02.942437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Getting to know the data","metadata":{}},{"cell_type":"code","source":"!ls ../input/rsna-miccai-brain-tumor-radiogenomic-classification/","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:02.945134Z","iopub.execute_input":"2021-12-17T13:42:02.945388Z","iopub.status.idle":"2021-12-17T13:42:03.614610Z","shell.execute_reply.started":"2021-12-17T13:42:02.945352Z","shell.execute_reply":"2021-12-17T13:42:03.612013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(Config.INPUT_PATH_DCM+'train_labels.csv', dtype={\n    'BraTS21ID': str, 'MGMT_value':int\n})\n\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.616539Z","iopub.execute_input":"2021-12-17T13:42:03.616901Z","iopub.status.idle":"2021-12-17T13:42:03.667065Z","shell.execute_reply.started":"2021-12-17T13:42:03.616864Z","shell.execute_reply":"2021-12-17T13:42:03.665659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lendo dataset \ndf = df[~df['BraTS21ID'].isin(Config.EXCLUDED_STR)]\n\n#Para testes, usando 1/3 do dataset\n# df = df[:int(len(df)/3)]\n\n\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.671860Z","iopub.execute_input":"2021-12-17T13:42:03.674812Z","iopub.status.idle":"2021-12-17T13:42:03.695088Z","shell.execute_reply.started":"2021-12-17T13:42:03.674772Z","shell.execute_reply":"2021-12-17T13:42:03.694177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adapted from: https://www.kaggle.com/leandronidas/tf-efficientnet-transfer-learning-strat-split/edit\n\ndf['flair'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/FLAIR/')\ndf['t1w'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T1w/')\ndf['t1wce'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T1wCE/')\ndf['t2w'] = df['BraTS21ID'].apply(lambda file_id : Config.INPUT_PATH_PNG+'train/'+file_id+'/T2w/')\n\ndf['brats21idInt'] = df['BraTS21ID'].astype(int)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.699269Z","iopub.execute_input":"2021-12-17T13:42:03.701580Z","iopub.status.idle":"2021-12-17T13:42:03.720831Z","shell.execute_reply.started":"2021-12-17T13:42:03.701544Z","shell.execute_reply":"2021-12-17T13:42:03.720215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_y = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\npred = sample_y\npred['BraTS21ID5'] = [format(x, '05d') for x in pred.BraTS21ID]\npred.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.725087Z","iopub.execute_input":"2021-12-17T13:42:03.727247Z","iopub.status.idle":"2021-12-17T13:42:03.752544Z","shell.execute_reply.started":"2021-12-17T13:42:03.727212Z","shell.execute_reply":"2021-12-17T13:42:03.751769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Divisao estratificada em treino, teste e validação\ndf_trainval, df_test = sk_model_selection.train_test_split(\n    df, \n    test_size=0.15, \n    random_state=Config.SEED, \n    stratify=df[\"MGMT_value\"],\n)\ndf_train, df_val = sk_model_selection.train_test_split(\n    df_trainval, \n    test_size=0.2, \n    random_state=Config.SEED, \n    stratify=df_trainval[\"MGMT_value\"],\n)\n\nprint(df_train.shape)\nprint(df_val.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.756879Z","iopub.execute_input":"2021-12-17T13:42:03.758927Z","iopub.status.idle":"2021-12-17T13:42:03.779348Z","shell.execute_reply.started":"2021-12-17T13:42:03.758887Z","shell.execute_reply":"2021-12-17T13:42:03.778673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking if sample 109 is neither present on train and validation sets","metadata":{}},{"cell_type":"code","source":"    \ndef get_iterating_dataframe(df, mri_type):\n    \n    all_img_files = []\n    all_img_labels = []\n    all_img_patient_ids = []\n    for row in df.iterrows():\n\n        img_dir = row[1][mri_type]\n        img_files = os.listdir(img_dir)\n        img_nums = sorted([int(ele.replace('Image-', '').replace('.png', '')) for ele in img_files])\n        totalnum_images = len(img_nums)\n        mid_point = int(totalnum_images//2)\n        start_point = mid_point - max(int(mid_point*0.1), Config.MIN_SLICES//2)\n        end_point = mid_point + max(int(mid_point*0.1), Config.MIN_SLICES//2)\n\n        img_names = [f'Image-{img_nums[i]}.png' for i in range(start_point, end_point+1)]\n\n        img_paths = [img_dir+ele for ele in img_names]\n        img_labels = [row[1]['MGMT_value']]*len(img_paths)\n        img_patient_ids = [row[1]['brats21idInt']]*len(img_paths)\n        all_img_files.extend(img_paths)\n        all_img_labels.extend(img_labels)\n        all_img_patient_ids.extend(img_patient_ids)\n\n    new_df = pd.DataFrame({'patient_ids': all_img_patient_ids,\n                  'labels': all_img_labels,\n                  'file_paths': all_img_files})\n            \n    return new_df","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.784521Z","iopub.execute_input":"2021-12-17T13:42:03.786561Z","iopub.status.idle":"2021-12-17T13:42:03.800707Z","shell.execute_reply.started":"2021-12-17T13:42:03.786523Z","shell.execute_reply":"2021-12-17T13:42:03.799918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PNGDataFrameIterator(DataFrameIterator):\n    def __init__(self, *arg, **kwargs):\n        self.white_list_formats = ('png')\n        super(PNGDataFrameIterator, self).__init__(*arg, **kwargs)\n        self.dataframe = kwargs['dataframe']\n        self.x = self.dataframe[kwargs['x_col']]\n        self.y = self.dataframe[kwargs['y_col']]\n        self.color_mode = kwargs['color_mode']\n        self.target_size = kwargs['target_size']\n\n    def _get_batches_of_transformed_samples(self, indices_array):\n        # get batch of images\n        batch_x = np.array([self.read_png_as_array(path, self.target_size, \n                                                   color_mode=self.color_mode)\n                            for path in self.x.iloc[indices_array]])\n\n        batch_y = np.array(self.y.iloc[indices_array].astype(np.uint8))  # astype because y was passed as str\n\n        # transform images\n        if self.image_data_generator is not None:\n            for i, (x, y) in enumerate(zip(batch_x, batch_y)):\n                transform_params = self.image_data_generator.get_random_transform(x.shape)\n                batch_x[i] = self.image_data_generator.apply_transform(x, transform_params)\n      \n\n        return batch_x, batch_y\n\n    \n    @staticmethod\n    def read_png_as_array(path, target_size=(Config.IMG_SIZE, Config.IMG_SIZE),\n                          color_mode='rgb'):\n        im_gray = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        pixels = im_gray - np.min(im_gray)\n        pixels = pixels / np.max(pixels)\n        image_manual_norm = (pixels * 255).astype(np.uint8)\n        image_array = cv2.resize(image_manual_norm, target_size, interpolation=cv2.INTER_CUBIC) \n        \n        if color_mode == 'rgb':\n            image_array = np.dstack((image_array,\n                                     image_array,\n                                     image_array,\n#                                      np.zeros_like(image_array),\n#                                      np.zeros_like(image_array)\n                                    ))\n        return image_array\n\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.806555Z","iopub.execute_input":"2021-12-17T13:42:03.808442Z","iopub.status.idle":"2021-12-17T13:42:03.833670Z","shell.execute_reply.started":"2021-12-17T13:42:03.808398Z","shell.execute_reply":"2021-12-17T13:42:03.833007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exam_list = ['flair','t1w','t1wce','t2w']\nexam_list = ['flair', 't1wce']\n\ndftrainIter_dict = {}\ndfvalIter_dict = {}\ndftestIter_dict = {}\ntrain_gen_dict = {}\nval_gen_dict = {}\ntest_gen_dict = {}\nmodel_check_dict = {}\nmodel_dict = {}\n\nfor exam in exam_list:\n    dftrainIter = get_iterating_dataframe(df_train, exam)\n    dftrainIter['labels_str'] = dftrainIter['labels'].astype(str)\n\n    dfvalIter = get_iterating_dataframe(df_val, exam)\n    dfvalIter['labels_str'] = dfvalIter['labels'].astype(str)\n\n    dftestIter = get_iterating_dataframe(df_test, exam)\n    dftestIter['labels_str'] = dftestIter['labels'].astype(str)\n\n    datagen = ImageDataGenerator(\n            preprocessing_function=tf.keras.applications.resnet50.preprocess_input,\n            zoom_range=0.2,\n            rotation_range=45,\n            fill_mode='nearest',\n            height_shift_range= 0.15,\n            width_shift_range=0.15,\n            horizontal_flip=True,\n            vertical_flip=True,\n            brightness_range = [0.8, 1.2],\n            rescale=1.0/255,\n    )\n    test_datagen = ImageDataGenerator(\n        preprocessing_function=tf.keras.applications.resnet50.preprocess_input,\n        rescale=1.0/255,\n    )\n\n    train_generator = PNGDataFrameIterator(dataframe=dftrainIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n\n    val_generator = PNGDataFrameIterator(dataframe=dfvalIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=test_datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n\n    test_generator = PNGDataFrameIterator(dataframe=dftestIter,\n                                     x_col='file_paths',\n                                     y_col='labels_str',\n                                     image_data_generator=test_datagen,\n                                     seed=Config.SEED,\n                                     batch_size=Config.BATCH_SIZE,\n                                     class_mode=Config.CLASS_MODE,\n                                     color_mode=Config.COLOR_MODE,\n                                     target_size=Config.TARGET_SIZE,  \n                                    )\n\n    dftrainIter_dict[exam] = dftrainIter\n    dfvalIter_dict[exam] = dfvalIter\n    dftestIter_dict[exam] = dftestIter \n    train_gen_dict[exam] = train_generator\n    val_gen_dict[exam] = val_generator\n    test_gen_dict[exam] = test_generator\n","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:03.834908Z","iopub.execute_input":"2021-12-17T13:42:03.835878Z","iopub.status.idle":"2021-12-17T13:42:33.816797Z","shell.execute_reply.started":"2021-12-17T13:42:03.835838Z","shell.execute_reply":"2021-12-17T13:42:33.815984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_deep = tf.keras.applications.ResNet50(weights='imagenet',\n                                              input_shape=(Config.IMG_SIZE, Config.IMG_SIZE, 3),\n                                              include_top=False)\n\n# Congela todas as camadas exceto as ultimas\nfor layer in model_deep.layers:\n    layer.trainable = False\n\n#     # Debug - checa qual layer está congelado    \n#     for i, layer in enumerate(model_deep.layers):\n#         print( i, layer.name, \"-\" ,layer.trainable)\n\n\nx = layers.GlobalAveragePooling2D()(model_deep.output)\nmodel = Model(inputs=model_deep.input, outputs=x)\n\nmodel.summary()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-12-17T13:42:33.818054Z","iopub.execute_input":"2021-12-17T13:42:33.818286Z","iopub.status.idle":"2021-12-17T13:42:38.251651Z","shell.execute_reply.started":"2021-12-17T13:42:33.818253Z","shell.execute_reply":"2021-12-17T13:42:38.250977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Leitura dos Feature Vectors","metadata":{}},{"cell_type":"code","source":"feats_train_dict = {}\nfeats_val_dict = {}\nfeats_test_dict = {}\n\nfor exam in exam_list:\n    feats_train = model.predict(train_gen_dict[exam],)\n    feats_val = model.predict(val_gen_dict[exam])\n    feats_test = model.predict(test_gen_dict[exam])\n    feats_train_dict[exam] = feats_train\n    feats_val_dict[exam] = feats_val\n    feats_test_dict[exam] = feats_test\n    \n    print('done!')","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:42:38.253026Z","iopub.execute_input":"2021-12-17T13:42:38.253256Z","iopub.status.idle":"2021-12-17T13:48:25.907641Z","shell.execute_reply.started":"2021-12-17T13:42:38.253224Z","shell.execute_reply":"2021-12-17T13:48:25.906822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_mean_train = {}\nfeats_mean_val = {}\nfeats_mean_test = {}\n\nfor exam in exam_list:\n    list_ = [feats_train_dict[exam][i] for i in range(feats_train_dict[exam].shape[0])]\n    dftrainIter_dict[exam]['featvecs'] =  list_\n    mean_ = dftrainIter_dict[exam][['patient_ids','featvecs']].groupby('patient_ids').mean()\n    feats_mean_train[exam] = mean_\n    \n    list_ = [feats_val_dict[exam][i] for i in range(feats_val_dict[exam].shape[0])]\n    dfvalIter_dict[exam]['featvecs'] =  list_\n    mean_ = dfvalIter_dict[exam][['patient_ids','featvecs']].groupby('patient_ids').mean()\n    feats_mean_val[exam] = mean_\n    \n    list_ = [feats_test_dict[exam][i] for i in range(feats_test_dict[exam].shape[0])]\n    dftestIter_dict[exam]['featvecs'] =  list_\n    mean_ = dftestIter_dict[exam][['patient_ids','featvecs']].groupby('patient_ids').mean()\n    feats_mean_test[exam] = mean_","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:49:55.711382Z","iopub.execute_input":"2021-12-17T13:49:55.711644Z","iopub.status.idle":"2021-12-17T13:49:55.918198Z","shell.execute_reply.started":"2021-12-17T13:49:55.711613Z","shell.execute_reply":"2021-12-17T13:49:55.917495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featvecs_dict_train = {}\nfeatvecs_dict_val = {}\nfeatvecs_dict_test = {}\n\nfor exam in exam_list:\n    featvecs_unstacked = feats_mean_train[exam]['featvecs'].values\n    featvecs_train = np.stack(featvecs_unstacked[:])\n    featvecs_dict_train[exam] = featvecs_train\n    \n    featvecs_unstacked = feats_mean_val[exam]['featvecs'].values\n    featvecs_val = np.stack(featvecs_unstacked[:])\n    featvecs_dict_val[exam] = featvecs_val\n    \n    featvecs_unstacked = feats_mean_test[exam]['featvecs'].values\n    featvecs_test = np.stack(featvecs_unstacked[:])\n    featvecs_dict_test[exam] = featvecs_test","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:49:57.049877Z","iopub.execute_input":"2021-12-17T13:49:57.050139Z","iopub.status.idle":"2021-12-17T13:49:57.062777Z","shell.execute_reply.started":"2021-12-17T13:49:57.050110Z","shell.execute_reply":"2021-12-17T13:49:57.062009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for exam, array in featvecs_dict_train.items():\nX_train = np.hstack(list(featvecs_dict_train.values()))\nX_val = np.hstack(list(featvecs_dict_val.values()))\nX_test = np.hstack(list(featvecs_dict_test.values()))","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:58:36.057962Z","iopub.execute_input":"2021-12-17T13:58:36.058241Z","iopub.status.idle":"2021-12-17T13:58:36.065043Z","shell.execute_reply.started":"2021-12-17T13:58:36.058210Z","shell.execute_reply":"2021-12-17T13:58:36.064269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = df_train.sort_values(by='brats21idInt')['MGMT_value']\ny_train = y_train.values\n\ny_val = df_val.sort_values(by='brats21idInt')['MGMT_value']\ny_val = y_val.values\n\ny_test = df_test.sort_values(by='brats21idInt')['MGMT_value']\ny_test = y_test.values","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:58:58.865686Z","iopub.execute_input":"2021-12-17T13:58:58.865945Z","iopub.status.idle":"2021-12-17T13:58:58.874192Z","shell.execute_reply.started":"2021-12-17T13:58:58.865917Z","shell.execute_reply":"2021-12-17T13:58:58.873472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\n\nxshuffled, yshuffled = shuffle(X_train, y_train, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-17T13:59:00.190941Z","iopub.execute_input":"2021-12-17T13:59:00.191789Z","iopub.status.idle":"2021-12-17T13:59:00.199446Z","shell.execute_reply.started":"2021-12-17T13:59:00.191737Z","shell.execute_reply":"2021-12-17T13:59:00.198652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\n\nclf = SVC(C=5,\n          kernel='rbf',\n          gamma='scale', class_weight='balanced',\n          verbose=True,\n          probability=True,\n)\nclf.fit(xshuffled, yshuffled)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:05:16.330249Z","iopub.execute_input":"2021-12-17T14:05:16.330730Z","iopub.status.idle":"2021-12-17T14:05:20.520600Z","shell.execute_reply.started":"2021-12-17T14:05:16.330693Z","shell.execute_reply":"2021-12-17T14:05:20.519887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred_train = clf.predict(X_train)\n# y_pred_val = clf.predict(X_val)\n# y_pred_test = clf.predict(X_test)\n\ny_pred_train = clf.predict_proba(X_train)\ny_pred_val = clf.predict_proba(X_val)\ny_pred_test = clf.predict_proba(X_test)\ny_pred_train = y_pred_train[:,1]\ny_pred_val = y_pred_val[:,1]\ny_pred_test = y_pred_test[:,1]","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:05:20.522073Z","iopub.execute_input":"2021-12-17T14:05:20.522759Z","iopub.status.idle":"2021-12-17T14:05:21.730292Z","shell.execute_reply.started":"2021-12-17T14:05:20.522722Z","shell.execute_reply":"2021-12-17T14:05:21.729484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ROC_AUC(probs, labels):\n    ns_probs = [0 for _ in range(len(labels))]\n\n    ns_auc = roc_auc_score(labels, ns_probs)\n    lr_auc = roc_auc_score(labels, probs)\n    print('No skill: ROC AUC=%.3f' % (ns_auc))\n    print('CNN: ROC AUC=%.3f' % (lr_auc))\n    lr_fpr, lr_tpr, _ = roc_curve(labels, probs)\n    plt.plot(lr_fpr, lr_tpr, marker='.', label='CNN')\n\n    ns_fpr, ns_tpr, _ = roc_curve(labels, ns_probs)\n    plt.plot(ns_fpr, ns_tpr, marker='.', label='No Skill')\n\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:05:21.731788Z","iopub.execute_input":"2021-12-17T14:05:21.732086Z","iopub.status.idle":"2021-12-17T14:05:21.739758Z","shell.execute_reply.started":"2021-12-17T14:05:21.732042Z","shell.execute_reply":"2021-12-17T14:05:21.739120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_AUC(y_pred_train, y_train)\nROC_AUC(y_pred_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:05:21.741977Z","iopub.execute_input":"2021-12-17T14:05:21.742449Z","iopub.status.idle":"2021-12-17T14:05:22.097857Z","shell.execute_reply.started":"2021-12-17T14:05:21.742395Z","shell.execute_reply":"2021-12-17T14:05:22.097223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_AUC(y_pred_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T14:05:22.099128Z","iopub.execute_input":"2021-12-17T14:05:22.099409Z","iopub.status.idle":"2021-12-17T14:05:22.278894Z","shell.execute_reply.started":"2021-12-17T14:05:22.099374Z","shell.execute_reply":"2021-12-17T14:05:22.278151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}