{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\"\"\"\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\"\"\"\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-13T20:56:00.395618Z","iopub.execute_input":"2021-08-13T20:56:00.396095Z","iopub.status.idle":"2021-08-13T20:56:00.404541Z","shell.execute_reply.started":"2021-08-13T20:56:00.396019Z","shell.execute_reply":"2021-08-13T20:56:00.403055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nimport pydicom\nimport glob\nfrom datetime import datetime\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import regularizers\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LogisticRegression as LogReg\nfrom sklearn.metrics import log_loss\n\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:01.145773Z","iopub.execute_input":"2021-08-13T20:56:01.146126Z","iopub.status.idle":"2021-08-13T20:56:01.153339Z","shell.execute_reply.started":"2021-08-13T20:56:01.146094Z","shell.execute_reply":"2021-08-13T20:56:01.151867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ndata = pd.read_csv(root_dir + 'train_labels.csv')\n\nto_exclude = [109, 123, 709]\ndata = data[~data['BraTS21ID'].isin(to_exclude)]","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:02.085295Z","iopub.execute_input":"2021-08-13T20:56:02.085722Z","iopub.status.idle":"2021-08-13T20:56:02.099434Z","shell.execute_reply.started":"2021-08-13T20:56:02.08569Z","shell.execute_reply":"2021-08-13T20:56:02.098476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_samples = data.shape[0]\nnum_positives = np.sum(data['MGMT_value'] == 1)\nnum_negatives = np.sum(data['MGMT_value'] == 0)\ndata.hist(column=\"MGMT_value\")","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:02.755638Z","iopub.execute_input":"2021-08-13T20:56:02.758679Z","iopub.status.idle":"2021-08-13T20:56:03.081609Z","shell.execute_reply.started":"2021-08-13T20:56:02.758628Z","shell.execute_reply":"2021-08-13T20:56:03.080578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of samples: \" + str(num_samples))\nprint(\"Number of positive labels: \" + str(num_positives))\nprint(\"Number of negative labels: \" + str(num_negatives))","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:05.422962Z","iopub.execute_input":"2021-08-13T20:56:05.42331Z","iopub.status.idle":"2021-08-13T20:56:05.431215Z","shell.execute_reply.started":"2021-08-13T20:56:05.423279Z","shell.execute_reply":"2021-08-13T20:56:05.429929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Baseline AUC (always predict the most common class)","metadata":{}},{"cell_type":"code","source":"y_true = data['MGMT_value'].values\ny_pred = np.array([1]*num_samples)\nbaseline_auc = roc_auc_score(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:06.283097Z","iopub.execute_input":"2021-08-13T20:56:06.283555Z","iopub.status.idle":"2021-08-13T20:56:06.293013Z","shell.execute_reply.started":"2021-08-13T20:56:06.283512Z","shell.execute_reply":"2021-08-13T20:56:06.291467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_auc","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:06.737188Z","iopub.execute_input":"2021-08-13T20:56:06.737608Z","iopub.status.idle":"2021-08-13T20:56:06.745011Z","shell.execute_reply.started":"2021-08-13T20:56:06.737576Z","shell.execute_reply":"2021-08-13T20:56:06.743471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This makes sense because the two classes are roughly balanced.","metadata":{}},{"cell_type":"code","source":"def full_ids(data):\n    zeros = 5 - len(str(data))\n    if zeros > 0:\n        prefix = ''.join(['0' for i in range(zeros)])\n    \n    return prefix+str(data)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:07.410498Z","iopub.execute_input":"2021-08-13T20:56:07.410822Z","iopub.status.idle":"2021-08-13T20:56:07.418049Z","shell.execute_reply.started":"2021-08-13T20:56:07.410792Z","shell.execute_reply":"2021-08-13T20:56:07.415477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['BraTS21ID_full'] = data['BraTS21ID'].apply(full_ids)\n\n# Add all the paths to the df for easy access\ndata['flair'] = data['BraTS21ID_full'].apply(lambda file_id : root_dir+'train/'+file_id+'/FLAIR/')\ndata['t1w'] = data['BraTS21ID_full'].apply(lambda file_id : root_dir+'train/'+file_id+'/T1w/')\ndata['t1wce'] = data['BraTS21ID_full'].apply(lambda file_id : root_dir+'train/'+file_id+'/T1wCE/')\ndata['t2w'] = data['BraTS21ID_full'].apply(lambda file_id : root_dir+'train/'+file_id+'/T2w/')\ndata","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:08.795257Z","iopub.execute_input":"2021-08-13T20:56:08.795687Z","iopub.status.idle":"2021-08-13T20:56:08.829806Z","shell.execute_reply.started":"2021-08-13T20:56:08.795655Z","shell.execute_reply":"2021-08-13T20:56:08.828357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(root_dir + 'sample_submission.csv')\ntest_data['BraTS21ID_full'] = test_data['BraTS21ID'].apply(full_ids)\ntest_data['flair'] = test_data['BraTS21ID_full'].apply(lambda file_id : root_dir+'test/'+file_id+'/FLAIR/')\ntest_data['t1w'] = test_data['BraTS21ID_full'].apply(lambda file_id : root_dir+'test/'+file_id+'/T1w/')\ntest_data['t1wce'] = test_data['BraTS21ID_full'].apply(lambda file_id : root_dir+'test/'+file_id+'/T1wCE/')\ntest_data['t2w'] = test_data['BraTS21ID_full'].apply(lambda file_id : root_dir+'test/'+file_id+'/T2w/')\ntest_data","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:09.767194Z","iopub.execute_input":"2021-08-13T20:56:09.767609Z","iopub.status.idle":"2021-08-13T20:56:09.806718Z","shell.execute_reply.started":"2021-08-13T20:56:09.767575Z","shell.execute_reply":"2021-08-13T20:56:09.805491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_image(data):\n    '''\n    Returns the image data as a numpy array.\n    '''  \n    if np.max(data.pixel_array)==0:\n        img = data.pixel_array\n    else:\n        img = data.pixel_array/np.max(data.pixel_array)\n        img = (img * 255).astype(np.uint8)\n        \n    return img","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:10.753326Z","iopub.execute_input":"2021-08-13T20:56:10.75372Z","iopub.status.idle":"2021-08-13T20:56:10.760833Z","shell.execute_reply.started":"2021-08-13T20:56:10.753687Z","shell.execute_reply":"2021-08-13T20:56:10.759207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_mri_image_data = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/T1w/Image-25.dcm'","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:16.071801Z","iopub.execute_input":"2021-08-13T20:56:16.072187Z","iopub.status.idle":"2021-08-13T20:56:16.077788Z","shell.execute_reply.started":"2021-08-13T20:56:16.072156Z","shell.execute_reply":"2021-08-13T20:56:16.075768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_data = pydicom.dcmread(test_mri_image_data)\nimg = get_image(img_data)\nplt.imshow(img, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:16.632682Z","iopub.execute_input":"2021-08-13T20:56:16.633089Z","iopub.status.idle":"2021-08-13T20:56:16.856247Z","shell.execute_reply.started":"2021-08-13T20:56:16.633055Z","shell.execute_reply":"2021-08-13T20:56:16.854955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sorted_image_dirs(path: str, sort=True):\n    '''\n    Sorts the list of image directories by image number in a path\n    '''\n    dirs = glob.glob(path+'*')\n    if sort:\n        dirs.sort(key=lambda x: int(x.split('/')[-1].split('-')[-1].split('.')[0]))\n    \n    return dirs\n\n\ndef get_all_images(path: str, sort=True):\n    '''\n    Returns a list of (non blank) images from a given path (of shape [non_blank_image_count, 512, 512])\n    '''\n    image_dirs = sorted_image_dirs(path, sort)\n    images = []\n    \n    for directory in image_dirs:\n        data = pydicom.dcmread(directory)\n        img = get_image(data)\n        \n        # Exclude the blank images\n        if np.max(img)!=0:\n            images.append(img)\n        else:\n            pass\n    \n    return images\n\ndef show_animation(images: list):\n    '''\n    Displays an animation from the list of images.\n    \n    set: matplotlib.rcParams['animation.html'] = 'jshtml'\n    \n    '''\n    fig = plt.figure(figsize=(6, 6))\n    plt.axis('off')\n    im = plt.imshow(images[0], cmap='gray')\n    \n    def animate_func(i):\n        im.set_array(images[i])\n        return [im]\n    \n    return matplotlib.animation.FuncAnimation(fig, animate_func, frames = len(images), interval = 20)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:20.183526Z","iopub.execute_input":"2021-08-13T20:56:20.183976Z","iopub.status.idle":"2021-08-13T20:56:20.205095Z","shell.execute_reply.started":"2021-08-13T20:56:20.183943Z","shell.execute_reply":"2021-08-13T20:56:20.202044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\npatient = 10\n\nflair_images = get_all_images(data['flair'][patient])\nprint('No of images:', len(flair_images))\nprint('MGMT: ', data['MGMT_value'][patient])\n\nfig = plt.figure(figsize=(30,30))\n\nc = 1\nfor image in flair_images:\n    ax = fig.add_subplot(len(flair_images)//10+1, 10, c)\n    ax.imshow(image, cmap='gray')\n    c+=1\n    \n    plt.axis('off')\n    \nfig.tight_layout()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:25.974014Z","iopub.execute_input":"2021-08-13T20:56:25.974366Z","iopub.status.idle":"2021-08-13T20:56:25.981193Z","shell.execute_reply.started":"2021-08-13T20:56:25.974333Z","shell.execute_reply":"2021-08-13T20:56:25.979936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"excluded_indexes = np.setdiff1d(list(range(585)), list(data.index))","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:27.897258Z","iopub.execute_input":"2021-08-13T20:56:27.897655Z","iopub.status.idle":"2021-08-13T20:56:27.90633Z","shell.execute_reply.started":"2021-08-13T20:56:27.897622Z","shell.execute_reply":"2021-08-13T20:56:27.905204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\npd.set_option('mode.chained_assignment', None)\nflattened_image_df = data.copy(deep=True)\nshape_df = data.copy(deep=True)\nMRI_types = ['flair', 't1w', 't1wce', 't2w']\nstart = datetime.now() \nfor img_type in MRI_types:\n    for patient_id in np.setdiff1d(list(range(585)), excluded_indexes):\n        images = get_all_images(data[img_type][patient_id], sort=False)\n        shape_df[img_type][patient_id] = images[0].shape\n        \n        images = np.array(images)\n        flattened_image = np.mean(images, axis=0)\n        flattened_image_df[img_type][patient_id] = flattened_image\n        \n        if patient_id % 40 == 0:\n            print(str(img_type) + \" \" + str(patient_id))\n             \nend = datetime.now()\nduration = end - start\nseconds_elapsed = duration.total_seconds()\nprint(\"Time elapsed: \" + str(seconds_elapsed))\n\nshape_df.to_pickle(\"/kaggle/working/input_image_shapes.p\")\nflattened_image_df.to_pickle(\"/kaggle/working/input_flattened_images.p\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:30.835515Z","iopub.execute_input":"2021-08-13T20:56:30.835923Z","iopub.status.idle":"2021-08-13T20:56:30.846721Z","shell.execute_reply.started":"2021-08-13T20:56:30.83589Z","shell.execute_reply":"2021-08-13T20:56:30.845194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\npd.set_option('mode.chained_assignment', None)\nflattened_image_df = test_data.copy(deep=True)\nMRI_types = ['flair', 't1w', 't1wce', 't2w']\nstart = datetime.now() \nfor img_type in MRI_types:\n    for patient_id in list(test_data.index):\n        images = get_all_images(test_data[img_type][patient_id], sort=False)\n        images = np.array(images)\n        flattened_image = np.mean(images, axis=0)\n        flattened_image_df[img_type][patient_id] = flattened_image\n        \n        if patient_id % 10 == 0:\n            print(str(img_type) + \" \" + str(patient_id))\n             \nend = datetime.now()\nduration = end - start\nseconds_elapsed = duration.total_seconds()\nprint(\"Time elapsed: \" + str(seconds_elapsed))\n\n#flattened_image_df.to_pickle(\"/kaggle/working/input_flattened_images.p\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:56:40.163905Z","iopub.execute_input":"2021-08-13T20:56:40.164296Z","iopub.status.idle":"2021-08-13T21:04:41.654929Z","shell.execute_reply.started":"2021-08-13T20:56:40.164263Z","shell.execute_reply":"2021-08-13T21:04:41.653878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shape_df = pd.read_pickle('/kaggle/input/brain-tumour-image-data-zip/input_image_shapes.p')\n#flattened_image_df = pd.read_pickle('/kaggle/input/brain-tumour-image-data-zip/input_flattened_images.p')","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:50:29.227488Z","iopub.execute_input":"2021-08-13T19:50:29.228259Z","iopub.status.idle":"2021-08-13T19:50:29.235036Z","shell.execute_reply.started":"2021-08-13T19:50:29.228212Z","shell.execute_reply":"2021-08-13T19:50:29.233528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# First Model","metadata":{}},{"cell_type":"code","source":"#from tensorflow.keras.applications.vgg16 import VGG16\n#from tensorflow.keras.applications.inception_v3 import InceptionV3\n#from tensorflow.keras.applications import ResNet50","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:28.823125Z","iopub.execute_input":"2021-08-13T19:51:28.823665Z","iopub.status.idle":"2021-08-13T19:51:28.829763Z","shell.execute_reply.started":"2021-08-13T19:51:28.823605Z","shell.execute_reply":"2021-08-13T19:51:28.82815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device_name = tf.test.gpu_device_name()\nif \"GPU\" not in device_name:\n    print(\"GPU device not found\")\nprint('Found GPU at: {}'.format(device_name))","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:29.610228Z","iopub.execute_input":"2021-08-13T19:51:29.610659Z","iopub.status.idle":"2021-08-13T19:51:31.762771Z","shell.execute_reply.started":"2021-08-13T19:51:29.610625Z","shell.execute_reply":"2021-08-13T19:51:31.761085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"width = 256\nheight = 256","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:31.765807Z","iopub.execute_input":"2021-08-13T19:51:31.766774Z","iopub.status.idle":"2021-08-13T19:51:31.772232Z","shell.execute_reply.started":"2021-08-13T19:51:31.766723Z","shell.execute_reply":"2021-08-13T19:51:31.77062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('mode.chained_assignment', None)\ndef compute_embeddings(mri_type, batch_size):\n\n    assert mri_type in ['flair', 't1w', 't1wce', 't2w']\n    \n    model_input = np.zeros((batch_size, width, height, 3))\n    raw_data = flattened_image_df[mri_type].values\n    for index in range(batch_size):\n        model_input[index, :, :, 0] = raw_data[index]\n        model_input[index, :, :, 1] = raw_data[index]\n        model_input[index, :, :, 2] = raw_data[index]\n\n    image_embedding = ResNet_model.predict(model_input)\n    image_embedding = image_embedding.reshape([batch_size, -1])\n    \n    entry_index = 0\n    #for patient_id in np.setdiff1d(list(range(585)), excluded_indexes):\n    for patient_id in list(flattened_image_df.index):\n        flattened_image_df[mri_type][patient_id] = image_embedding[entry_index, :]\n        entry_index += 1\n    \n    return","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:34.204007Z","iopub.execute_input":"2021-08-13T19:51:34.204369Z","iopub.status.idle":"2021-08-13T19:51:34.213519Z","shell.execute_reply.started":"2021-08-13T19:51:34.204338Z","shell.execute_reply":"2021-08-13T19:51:34.211847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n#ResNet_model = ResNet50(input_shape=(width, height, 3), include_top=False, weights=\"imagenet\")\n#ResNet_model = keras.models.load_model(\"/kaggle/input/resnet50-weights/ResNet50.h5\")\nflattened_image_df['flair'] = flattened_image_df['flair'].apply(lambda x: cv2.resize(x, (width, height)))\nflattened_image_df['t1w'] = flattened_image_df['t1w'].apply(lambda x: cv2.resize(x, (width, height)))\nflattened_image_df['t1wce'] = flattened_image_df['t1wce'].apply(lambda x: cv2.resize(x, (width, height)))\nflattened_image_df['t2w'] = flattened_image_df['t2w'].apply(lambda x: cv2.resize(x, (width, height)))\n\nfor mri_type in ['flair', 't1w', 't1wce', 't2w']:\n    compute_embeddings(mri_type, flattened_image_df.shape[0])\ny = flattened_image_df['MGMT_value'].values.reshape((-1,1))\nX = np.zeros((y.shape[0], len(flattened_image_df['flair'][0]) * 4))\n\nfor index in range(y.shape[0]):\n    data = flattened_image_df.iloc[index]\n    features = np.concatenate((data['flair'], data['t1w'], data['t1wce'], data['t2w']))\n    X[index, :] = features\n    \nX_test = X\n#np.save(\"/kaggle/working/y.npy\", y)\n#np.save(\"/kaggle/working/X_test.npy\", X)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:34.569887Z","iopub.execute_input":"2021-08-13T19:51:34.570185Z","iopub.status.idle":"2021-08-13T19:51:34.580056Z","shell.execute_reply.started":"2021-08-13T19:51:34.570157Z","shell.execute_reply":"2021-08-13T19:51:34.578761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = np.load(\"/kaggle/input/d/nicholasjohnson2020/brain-tumour-features-resnet50-256/256_ResNet50_features/y.npy\")\nX = np.load(\"/kaggle/input/d/nicholasjohnson2020/brain-tumour-features-resnet50-256/256_ResNet50_features/X.npy\")\nX_test = np.load(\"/kaggle/input/d/nicholasjohnson2020/brain-tumour-features-resnet50-256/256_ResNet50_features/X_test.npy\")","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:51:35.328081Z","iopub.execute_input":"2021-08-13T19:51:35.328459Z","iopub.status.idle":"2021-08-13T19:51:51.730842Z","shell.execute_reply.started":"2021-08-13T19:51:35.328428Z","shell.execute_reply":"2021-08-13T19:51:51.729713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca = PCA(n_components = 100)\nX_trim = pca.fit_transform(X)\nX_test_trim = pca.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:52:23.783309Z","iopub.execute_input":"2021-08-13T19:52:23.783773Z","iopub.status.idle":"2021-08-13T19:53:19.254417Z","shell.execute_reply.started":"2021-08-13T19:52:23.783738Z","shell.execute_reply":"2021-08-13T19:53:19.252803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape)\nprint(X_trim.shape)\nprint(X_test.shape)\nprint(X_test_trim.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:53:19.256857Z","iopub.execute_input":"2021-08-13T19:53:19.257793Z","iopub.status.idle":"2021-08-13T19:53:19.268844Z","shell.execute_reply.started":"2021-08-13T19:53:19.257736Z","shell.execute_reply":"2021-08-13T19:53:19.267366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(hist, last = None):\n    if last == None:\n        last = len(hist.history[\"loss\"])\n    plt.plot(hist.history[\"loss\"][-last:])\n    plt.plot(hist.history[\"val_loss\"][-last:])\n    plt.title(\"model accuracy\")\n    plt.ylabel(\"accuracy\")\n    plt.xlabel(\"epoch\")\n    plt.legend([\"train\", \"validation\"], loc=\"upper left\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:38.93889Z","iopub.execute_input":"2021-08-13T19:57:38.939391Z","iopub.status.idle":"2021-08-13T19:57:38.945897Z","shell.execute_reply.started":"2021-08-13T19:57:38.939339Z","shell.execute_reply":"2021-08-13T19:57:38.944747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"saved_X = X","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:39.605635Z","iopub.execute_input":"2021-08-13T19:57:39.606025Z","iopub.status.idle":"2021-08-13T19:57:39.6122Z","shell.execute_reply.started":"2021-08-13T19:57:39.605995Z","shell.execute_reply":"2021-08-13T19:57:39.610674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def l3_res_model(input_shape, no_classes, lr):\n    inputs = tf.keras.Input(shape=input_shape)\n    x = layers.Dense(128, activation='sigmoid')(inputs)\n    x = layers.BatchNormalization()(x)\n    b_1 = layers.Dropout(0.2)(x)\n    x = layers.Dense(128, activation='sigmoid')(b_1)\n    x = layers.BatchNormalization()(x)\n    b_2 = layers.Dropout(0.2)(x)\n    x = layers.Dense(128, activation='sigmoid')(b_2)\n    x = layers.BatchNormalization()(x)\n    b_3 = layers.Dropout(0.2)(x)\n    tot_op = tf.keras.layers.add([b_1, b_2, b_3])\n    outputs = layers.Dense(no_classes, activation='sigmoid')(tot_op)\n    model = tf.keras.Model(inputs, outputs)\n    model.compile(loss='binary_crossentropy', optimizer=tf.keras.optimizers.Adam(learning_rate = lr), metrics=['binary_crossentropy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:40.707119Z","iopub.execute_input":"2021-08-13T19:57:40.707498Z","iopub.status.idle":"2021-08-13T19:57:40.719545Z","shell.execute_reply.started":"2021-08-13T19:57:40.707462Z","shell.execute_reply":"2021-08-13T19:57:40.716275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X_trim\ninput_dim = X.shape[1]\n\nlosses_NN=[]\nauc_NN=[]\nkf = KFold(n_splits=10)\ntf.random.set_seed(1010)\nnp.random.seed(1010)\n\nfor train_index, test_index in kf.split(X):\n    X_train, X_test = X[train_index], X[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n\n    nnclf = l3_res_model((input_dim,),1,0.00001)\n    hist = nnclf.fit(X_train, y_train, batch_size=64, epochs=50, validation_data=(X_test, y_test), verbose=0)\n    plot_hist(hist, last=20)\n\n    preds = nnclf.predict(X_test) # list of preds per class\n\n    loss = log_loss(np.ravel(y_test), np.ravel(preds))\n    auc = roc_auc_score(y_test[:, 0], preds[:, 0])\n    print('Loss: '+str(loss))\n    print('AUC: '+str(auc))\n    losses_NN.append(loss)\n    auc_NN.append(auc)\n\nprint('Average Loss: '+str(np.average(losses_NN)))\nprint('Average AUC: '+str(np.average(auc_NN)))","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:42.426528Z","iopub.execute_input":"2021-08-13T19:57:42.42707Z","iopub.status.idle":"2021-08-13T19:57:42.435335Z","shell.execute_reply.started":"2021-08-13T19:57:42.427024Z","shell.execute_reply":"2021-08-13T19:57:42.434233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X_trim\ninput_dim = X.shape[1]\nweek1_model = l3_res_model((input_dim,),1,0.00001)\nweek1_model.fit(X, y, batch_size=16, epochs=50, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:43.709562Z","iopub.execute_input":"2021-08-13T19:57:43.709959Z","iopub.status.idle":"2021-08-13T19:57:47.846986Z","shell.execute_reply.started":"2021-08-13T19:57:43.709926Z","shell.execute_reply":"2021-08-13T19:57:47.845908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = week1_model.predict(X_test_trim)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T19:57:57.058874Z","iopub.execute_input":"2021-08-13T19:57:57.059306Z","iopub.status.idle":"2021-08-13T19:57:57.522538Z","shell.execute_reply.started":"2021-08-13T19:57:57.059272Z","shell.execute_reply":"2021-08-13T19:57:57.521257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_file = pd.read_csv(\"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\")\nsubmission_file[\"BraTS21ID\"] = submission_file['BraTS21ID'].apply(full_ids)\nsubmission_file[\"MGMT_value\"] = preds\nsubmission_file.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T20:12:26.327105Z","iopub.execute_input":"2021-08-13T20:12:26.327501Z","iopub.status.idle":"2021-08-13T20:12:26.34398Z","shell.execute_reply.started":"2021-08-13T20:12:26.327467Z","shell.execute_reply":"2021-08-13T20:12:26.342725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Miscellaneous Prototyping","metadata":{}},{"cell_type":"code","source":"\"\"\"\nMRI_types = ['flair', 't1w', 't1wce', 't2w']\nfor mri_type in MRI_types:\n    model_input = np.zeros((num_patients, img_width, img_height, 3))\n        for patient_id in range(num_patients):\n            image_array = get_all_images(data[image_type][patient_id], sort=False)\n            image_array = np.array(image_array)\n            print(\"Patient \" + str(patient_id) + \" data shape: \" + str(image_array.shape))\n            flattened_image = np.mean(image_array, axis=0)\n            \n            model_input[patient_id, :, :, 0] = flattened_image\n            model_input[patient_id, :, :, 1] = flattened_image\n            model_input[patient_id, :, :, 2] = flattened_image\n            \n        image_embedding = embedding_model.predict(model_input)\n        image_embedding = image_embedding.reshape([num_patients, -1])\n        \n        for patient_id in range(num_patients):\n            output_df[image_type][patient_id] = image_embedding[patient_id, :]\n\"\"\"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nflair_count = shape_df['flair'].value_counts()\nt1w_count = shape_df['t1w'].value_counts()\nt1wce_count = shape_df['t1wce'].value_counts()\nt2w_count = shape_df['t2w'].value_counts()\n\nflair_shapes = list(flair_count.index)\nt1w_shapes = list(t1w_count.index)\nt1wce_shapes = list(t1wce_count.index)\nt2w_shapes = list(t2w_count.index)\nunique_shapes = list(set(flair_shapes + t1w_shapes + t1wce_shapes + t2w_shapes))\nshape_counts = {}\nfor shape in unique_shapes:\n    shape_counts[shape] = 0\nfor shape in flair_shapes:\n    shape_counts[shape] += flair_count[shape]\nfor shape in t1w_shapes:\n    shape_counts[shape] += t1w_count[shape]\nfor shape in t1wce_shapes:\n    shape_counts[shape] += t1wce_count[shape]\nfor shape in t2w_shapes:\n    shape_counts[shape] += t2w_count[shape]\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-12T22:21:33.616152Z","iopub.execute_input":"2021-08-12T22:21:33.616527Z","iopub.status.idle":"2021-08-12T22:21:33.632442Z","shell.execute_reply.started":"2021-08-12T22:21:33.616497Z","shell.execute_reply":"2021-08-12T22:21:33.631588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndef compute_embeddings(mri_types, batch_size, output_df, img_width, img_height):\n\n    model_input = np.zeros((batch_size, img_width, img_height, 3))\n    temp_legend = {}\n    entry_index = 0\n    for mri_type in mri_types:\n        for patient_id in np.setdiff1d(list(range(585)), excluded_indexes):\n            if shape_df[mri_type][patient_id] == (img_width, img_height):\n                temp_legend[entry_index] = (mri_type, patient_id)\n                model_input[entry_index, :, :, 0] = flattened_image_df[mri_type][patient_id]\n                model_input[entry_index, :, :, 1] = flattened_image_df[mri_type][patient_id]\n                model_input[entry_index, :, :, 2] = flattened_image_df[mri_type][patient_id]\n                entry_index += 1\n    assert entry_index == batch_size            \n\n    ResNet_model = ResNet50(input_shape=(img_width, img_height,3), include_top=False, weights=\"imagenet\")\n    image_embedding = ResNet_model.predict(model_input)\n    image_embedding = image_embedding.reshape([batch_size, -1])\n    for entry_index in range(batch_size):\n        (mri_type, patient_id) = temp_legend[entry_index]\n        output_df[mri_type][patient_id] = image_embedding[entry_index, :]\n\n    return\n\nstart = datetime.now()      \nMRI_types = ['flair', 't1w', 't1wce', 't2w']\npd.set_option('mode.chained_assignment', None)\n#embedding_df = flattened_image_df.copy(deep=True)\nembedding_df = flattened_image_df\n\nfor (img_width, img_height) in unique_shapes:\n    if (img_width, img_height) == (512, 512):\n        continue\n    compute_embeddings(MRI_types, shape_counts[(img_width, img_height)], embedding_df,\n                       img_width, img_height)\n    print(\"Finished \" + str((img_width, img_height)))\n\n\n    \ncompute_embeddings(['flair', 't1w'], flair_count[(512, 512)] + t1w_count[(512, 512)],\n                   embedding_df, 512, 512)\ncompute_embeddings([\"t1wce\", 't2w'], t1wce_count[(512, 512)] + t2w_count[(512, 512)],\n                   embedding_df, 512, 512)\n\nend = datetime.now()\nduration = end - start\nseconds_elapsed = duration.total_seconds()\nprint(\"Time elapsed: \" + str(seconds_elapsed))\n\nembedding_df.to_pickle(\"/kaggle/working/embeddeding_512.p\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-12T22:28:22.155957Z","iopub.execute_input":"2021-08-12T22:28:22.156368Z","iopub.status.idle":"2021-08-12T22:28:53.868482Z","shell.execute_reply.started":"2021-08-12T22:28:22.156333Z","shell.execute_reply":"2021-08-12T22:28:53.867204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ndef transform_patient_data_v1(embedding_model, num_patients=num_samples,\n                              img_width=512, img_height=512):\n    \n    pd.set_option('mode.chained_assignment', None)\n    output_df = data.copy(deep=True)\n    \n    MRI_types = ['flair', 't1w', 't1wce', 't2w']\n    for image_type in MRI_types:\n        model_input = np.zeros((num_patients, img_width, img_height, 3))\n        for patient_id in range(num_patients):\n            image_array = get_all_images(data[image_type][patient_id], sort=False)\n            image_array = np.array(image_array)\n            print(\"Patient \" + str(patient_id) + \" data shape: \" + str(image_array.shape))\n            flattened_image = np.mean(image_array, axis=0)\n            \n            model_input[patient_id, :, :, 0] = flattened_image\n            model_input[patient_id, :, :, 1] = flattened_image\n            model_input[patient_id, :, :, 2] = flattened_image\n            \n        image_embedding = embedding_model.predict(model_input)\n        image_embedding = image_embedding.reshape([num_patients, -1])\n        \n        for patient_id in range(num_patients):\n            output_df[image_type][patient_id] = image_embedding[patient_id, :]\n            \n    return output_df\n    \ndef transform_patient_data_v2(patient_id, embedding_model):\n    \n    MRI_types = ['flair', 't1w', 't1wce', 't2w']\n    output_dict = {}\n    \n    for image_type in MRI_types:\n        image_array = get_all_images(data[image_type][patient_id])\n        model_input = np.zeros((len(image_array), image_array[0].shape[0],\n                                image_array[0].shape[1], 3))\n        for i in range(model_input.shape[0]):\n            model_input[i, :, :, 0] = image_array[i]\n            model_input[i, :, :, 1] = image_array[i]\n            model_input[i, :, :, 2] = image_array[i]\n        \n        image_embedding = embedding_model.predict(model_input)\n        image_embedding = image_embedding.reshape([image_embedding.shape[0], -1])\n        image_embedding = np.mean(image_embedding, axis=0)\n        \n        output_dict[image_type] = image_embedding\n\n    return output_dict\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-08-11T17:23:28.673776Z","iopub.execute_input":"2021-08-11T17:23:28.674103Z","iopub.status.idle":"2021-08-11T17:23:28.686739Z","shell.execute_reply.started":"2021-08-11T17:23:28.674072Z","shell.execute_reply":"2021-08-11T17:23:28.685578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scan_width = 512\n#scan_height = 512","metadata":{"execution":{"iopub.status.busy":"2021-08-11T17:14:38.67971Z","iopub.execute_input":"2021-08-11T17:14:38.680038Z","iopub.status.idle":"2021-08-11T17:14:38.684282Z","shell.execute_reply.started":"2021-08-11T17:14:38.680006Z","shell.execute_reply":"2021-08-11T17:14:38.683147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#VGG_model = VGG16(input_shape = (scan_width, scan_height, 3), include_top = False, weights = 'imagenet')\n#Inception_model = InceptionV3(input_shape = (scan_width, scan_height, 3), include_top = False, weights = 'imagenet')\n#ResNet_model = ResNet50(input_shape=(scan_width, scan_height,3), include_top=False, weights=\"imagenet\")","metadata":{"execution":{"iopub.status.busy":"2021-08-11T17:14:45.651989Z","iopub.execute_input":"2021-08-11T17:14:45.652364Z","iopub.status.idle":"2021-08-11T17:14:54.710208Z","shell.execute_reply.started":"2021-08-11T17:14:45.652329Z","shell.execute_reply":"2021-08-11T17:14:54.709287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#start = datetime.now()      \n#VGG_embeddings = transform_patient_data_v2(0, VGG_model)\n#end = datetime.now()\n#duration = end - start\n#seconds_elapsed = duration.total_seconds()\n#print(\"Time elapsed: \" + str(seconds_elapsed))","metadata":{"execution":{"iopub.status.busy":"2021-08-11T17:14:57.386506Z","iopub.execute_input":"2021-08-11T17:14:57.386831Z","iopub.status.idle":"2021-08-11T17:15:53.65299Z","shell.execute_reply.started":"2021-08-11T17:14:57.386802Z","shell.execute_reply":"2021-08-11T17:15:53.6517Z"},"trusted":true},"execution_count":null,"outputs":[]}]}