{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import packages","metadata":{}},{"cell_type":"code","source":"# !pip install imutils\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom as dicom\nimport matplotlib.pylab as plt\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom skimage import measure\n# ipywidgets for some interactive plots\nfrom ipywidgets.widgets import * \nimport ipywidgets as widgets\nfrom PIL import Image\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.vgg16 import VGG16, preprocess_input\nfrom keras import layers\nfrom keras.models import Model, Sequential\nfrom keras.optimizers import Adam, RMSprop\nfrom keras.callbacks import EarlyStopping\nimport plotly.express as px\nimport keras\nfrom keras.datasets import mnist\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Conv2D, MaxPool2D, Flatten\nfrom keras.utils import np_utils\nfrom sklearn.metrics import accuracy_score\nimport gc","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:44:48.986507Z","iopub.execute_input":"2021-08-28T23:44:48.986900Z","iopub.status.idle":"2021-08-28T23:44:48.995006Z","shell.execute_reply.started":"2021-08-28T23:44:48.986862Z","shell.execute_reply":"2021-08-28T23:44:48.993736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"sample_submission_file = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\ntrain_data_labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.594518Z","iopub.execute_input":"2021-08-28T23:39:15.594918Z","iopub.status.idle":"2021-08-28T23:39:15.612009Z","shell.execute_reply.started":"2021-08-28T23:39:15.594869Z","shell.execute_reply":"2021-08-28T23:39:15.611092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"print('Shape of train data:', train_data_labels.shape)\ntrain_data_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.615340Z","iopub.execute_input":"2021-08-28T23:39:15.615599Z","iopub.status.idle":"2021-08-28T23:39:15.638410Z","shell.execute_reply.started":"2021-08-28T23:39:15.615572Z","shell.execute_reply":"2021-08-28T23:39:15.637644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of image folders:', len(os.listdir('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train')))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.640073Z","iopub.execute_input":"2021-08-28T23:39:15.640441Z","iopub.status.idle":"2021-08-28T23:39:15.677485Z","shell.execute_reply.started":"2021-08-28T23:39:15.640395Z","shell.execute_reply":"2021-08-28T23:39:15.676692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set1 = set(os.listdir('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train'))\nset2 = set(train_data_labels['BraTS21ID'].unique())\nprint('Intersection between IDs in train data labels file and train data folders:', len(set1.intersection(set2)))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.678765Z","iopub.execute_input":"2021-08-28T23:39:15.679151Z","iopub.status.idle":"2021-08-28T23:39:15.687510Z","shell.execute_reply.started":"2021-08-28T23:39:15.679113Z","shell.execute_reply":"2021-08-28T23:39:15.686648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert folder IDs to numbers\nset1 = set([int(x) for x in list(set1)])\nprint('Intersection between IDs in train data labels file and train data folders:', len(set1.intersection(set2)))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.688791Z","iopub.execute_input":"2021-08-28T23:39:15.689345Z","iopub.status.idle":"2021-08-28T23:39:15.698393Z","shell.execute_reply.started":"2021-08-28T23:39:15.689302Z","shell.execute_reply":"2021-08-28T23:39:15.697339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Shape of test data:', sample_submission_file.shape)\nsample_submission_file.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.699913Z","iopub.execute_input":"2021-08-28T23:39:15.700295Z","iopub.status.idle":"2021-08-28T23:39:15.714198Z","shell.execute_reply.started":"2021-08-28T23:39:15.700258Z","shell.execute_reply":"2021-08-28T23:39:15.713104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FLAIR\n# specify your image path\n# using images of a patient with brain tumor\nimage_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/FLAIR/Image-101.dcm'\nds = dicom.dcmread(image_path)\n\nplt.imshow(ds.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.718116Z","iopub.execute_input":"2021-08-28T23:39:15.718748Z","iopub.status.idle":"2021-08-28T23:39:15.917409Z","shell.execute_reply.started":"2021-08-28T23:39:15.718707Z","shell.execute_reply":"2021-08-28T23:39:15.916549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# T1w\nimage_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/T1w/Image-11.dcm'\nds = dicom.dcmread(image_path)\n\nplt.imshow(ds.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:15.919358Z","iopub.execute_input":"2021-08-28T23:39:15.919921Z","iopub.status.idle":"2021-08-28T23:39:16.118591Z","shell.execute_reply.started":"2021-08-28T23:39:15.919873Z","shell.execute_reply":"2021-08-28T23:39:16.117792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# T1wCE\nimage_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/T1wCE/Image-101.dcm'\nds = dicom.dcmread(image_path)\n\nplt.imshow(ds.pixel_array)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.120018Z","iopub.execute_input":"2021-08-28T23:39:16.120385Z","iopub.status.idle":"2021-08-28T23:39:16.307231Z","shell.execute_reply.started":"2021-08-28T23:39:16.120340Z","shell.execute_reply":"2021-08-28T23:39:16.306338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# T2w\nimage_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/T2w/Image-101.dcm'\nds = dicom.dcmread(image_path)\n\nimg = ds.pixel_array\n\nplt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.308636Z","iopub.execute_input":"2021-08-28T23:39:16.309030Z","iopub.status.idle":"2021-08-28T23:39:16.490479Z","shell.execute_reply.started":"2021-08-28T23:39:16.308989Z","shell.execute_reply":"2021-08-28T23:39:16.489401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds.pixel_array","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.492034Z","iopub.execute_input":"2021-08-28T23:39:16.492466Z","iopub.status.idle":"2021-08-28T23:39:16.499237Z","shell.execute_reply.started":"2021-08-28T23:39:16.492411Z","shell.execute_reply":"2021-08-28T23:39:16.498158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.500843Z","iopub.execute_input":"2021-08-28T23:39:16.501281Z","iopub.status.idle":"2021-08-28T23:39:16.515061Z","shell.execute_reply.started":"2021-08-28T23:39:16.501242Z","shell.execute_reply":"2021-08-28T23:39:16.514168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The unit of measurement in CT scans is the Hounsfield Unit (HU), which is a measure of radiodensity. CT scanners are carefully calibrated to accurately measure this\n# Load the scans in given folder path\ndef load_scan(path):\n    slices = [dicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: float(x.ImagePositionPatient[2])) # Sort by patient's position while scan was taken\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n    return slices\n        \n    return slices","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.518098Z","iopub.execute_input":"2021-08-28T23:39:16.518365Z","iopub.status.idle":"2021-08-28T23:39:16.526531Z","shell.execute_reply.started":"2021-08-28T23:39:16.518339Z","shell.execute_reply":"2021-08-28T23:39:16.525732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pixels_hu(slices):\n    image = np.stack([s.pixel_array for s in slices])\n    # Convert to int16 (from sometimes int16), \n    # should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n\n    # Set outside-of-scan pixels to 0\n    # The intercept is usually -1024, so air is approximately 0\n    image[image == -2000] = 0\n    \n    # Convert to Hounsfield units (HU)\n    for slice_number in range(len(slices)):\n        \n        intercept = slices[slice_number].RescaleIntercept\n        slope = slices[slice_number].RescaleSlope\n        \n        if slope != 1:\n            image[slice_number] = slope * image[slice_number].astype(np.float64)\n            image[slice_number] = image[slice_number].astype(np.int16)\n            \n        image[slice_number] += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.527983Z","iopub.execute_input":"2021-08-28T23:39:16.528565Z","iopub.status.idle":"2021-08-28T23:39:16.537383Z","shell.execute_reply.started":"2021-08-28T23:39:16.528524Z","shell.execute_reply":"2021-08-28T23:39:16.536477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/FLAIR/'\n\n# slide through dicom images using a slide bar \nplt.figure(1)\ndef dicom_animation(x):\n    first_patient = load_scan(path)\n    first_patient_pixels = get_pixels_hu(first_patient)\n    plt.hist(first_patient_pixels.flatten(), bins=80, color='c')\n    plt.xlabel(\"Hounsfield Units (HU)\")\n    plt.ylabel(\"Frequency\")\n    plt.show()\n\n    # Show some slice in the middle\n    # cmap=plt.cm.gray\n    plt.imshow(first_patient_pixels[x], cmap=plt.cm.bone)\n    plt.show()\n    return x\ninteract(dicom_animation, x=(0, len(os.listdir(path))-1))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:16.539805Z","iopub.execute_input":"2021-08-28T23:39:16.540078Z","iopub.status.idle":"2021-08-28T23:39:22.708497Z","shell.execute_reply.started":"2021-08-28T23:39:16.540044Z","shell.execute_reply":"2021-08-28T23:39:22.707722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_patient = load_scan(path)\nfirst_patient_pixels = get_pixels_hu(first_patient)\n\ndef sample_stack(stack, rows=6, cols=6, start_with=10, show_every=3):\n    fig,ax = plt.subplots(rows,cols,figsize=[12,12])\n    for i in range(rows*cols):\n        ind = start_with + i*show_every\n        ax[int(i/rows),int(i % rows)].set_title('slice %d' % ind)\n        ax[int(i/rows),int(i % rows)].imshow(stack[ind],cmap='gray')\n        ax[int(i/rows),int(i % rows)].axis('off')\n    plt.show()\n\nsample_stack(first_patient_pixels)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:22.709800Z","iopub.execute_input":"2021-08-28T23:39:22.710184Z","iopub.status.idle":"2021-08-28T23:39:26.154809Z","shell.execute_reply.started":"2021-08-28T23:39:22.710145Z","shell.execute_reply":"2021-08-28T23:39:26.153933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_scan(scan, new_shape):\n    # read slice as 32 bit signed integers\n    img = Image.fromarray(scan)\n    # do the resizing\n    img = img.resize(new_shape, resample=Image.LANCZOS)\n    # convert back to 16 bit integers\n    resized_scan = np.array(img, dtype=np.int16)\n    return resized_scan","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:26.155931Z","iopub.execute_input":"2021-08-28T23:39:26.156268Z","iopub.status.idle":"2021-08-28T23:39:26.161488Z","shell.execute_reply.started":"2021-08-28T23:39:26.156234Z","shell.execute_reply":"2021-08-28T23:39:26.160529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_scan(scan):\n    img = Image.fromarray(scan)\n    \n    left = (scan.shape[0]-512)/2\n    right = (scan.shape[0]+512)/2\n    top = (scan.shape[1]-512)/2\n    bottom = (scan.shape[1]+512)/2\n\n    img = img.crop((left, top, right, bottom))\n    # convert back to 16 bit integers\n    cropped_scan = np.array(img, dtype=np.int16)\n    return cropped_scan","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:26.163062Z","iopub.execute_input":"2021-08-28T23:39:26.163627Z","iopub.status.idle":"2021-08-28T23:39:26.172631Z","shell.execute_reply.started":"2021-08-28T23:39:26.163568Z","shell.execute_reply":"2021-08-28T23:39:26.171886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_and_resize(scan, new_shape):\n    img = Image.fromarray(scan)\n    \n    left = (scan.shape[0]-512)/2\n    right = (scan.shape[0]+512)/2\n    top = (scan.shape[1]-512)/2\n    bottom = (scan.shape[1]+512)/2\n    \n    img = img.crop((left, top, right, bottom))\n    img = img.resize(new_shape, resample=Image.LANCZOS)\n    \n    cropped_resized_scan = np.array(img, dtype=np.int16)\n    return cropped_resized_scan","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:26.173895Z","iopub.execute_input":"2021-08-28T23:39:26.174457Z","iopub.status.idle":"2021-08-28T23:39:26.182094Z","shell.execute_reply.started":"2021-08-28T23:39:26.174417Z","shell.execute_reply":"2021-08-28T23:39:26.181100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_patient = load_scan(path)\nfirst_patient_pixels = get_pixels_hu(first_patient)\nprocessed_img = crop_and_resize(first_patient_pixels[242,:,:], new_shape = [512,512])\nplt.imshow(processed_img, cmap=plt.cm.bone)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:26.184736Z","iopub.execute_input":"2021-08-28T23:39:26.185211Z","iopub.status.idle":"2021-08-28T23:39:27.477266Z","shell.execute_reply.started":"2021-08-28T23:39:26.185177Z","shell.execute_reply":"2021-08-28T23:39:27.476348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create data for modelling","metadata":{}},{"cell_type":"code","source":"train_data = train_data_labels.copy()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:27.478475Z","iopub.execute_input":"2021-08-28T23:39:27.478833Z","iopub.status.idle":"2021-08-28T23:39:27.483172Z","shell.execute_reply.started":"2021-08-28T23:39:27.478804Z","shell.execute_reply":"2021-08-28T23:39:27.482302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['# FLAIR images'] = np.NaN\ntrain_data['# T1w images'] = np.NaN\ntrain_data['# T1wCE images'] = np.NaN\ntrain_data['# T2w images'] = np.NaN","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:27.488393Z","iopub.execute_input":"2021-08-28T23:39:27.488991Z","iopub.status.idle":"2021-08-28T23:39:27.496329Z","shell.execute_reply.started":"2021-08-28T23:39:27.488926Z","shell.execute_reply":"2021-08-28T23:39:27.495427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using a subset of the complete train data\nnum_patients = 30\ntrain_data = train_data.iloc[0:num_patients]","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:27.498754Z","iopub.execute_input":"2021-08-28T23:39:27.499310Z","iopub.status.idle":"2021-08-28T23:39:27.504585Z","shell.execute_reply.started":"2021-08-28T23:39:27.499268Z","shell.execute_reply":"2021-08-28T23:39:27.503661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing images with issues\ninvalid_ids = [109, 123, 709]\nprint('Train data shape before:', train_data.shape)\ntrain_data = train_data[~train_data['BraTS21ID'].isin(invalid_ids)].reset_index(drop = True)\nprint('Train data shape after:', train_data.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:27.506216Z","iopub.execute_input":"2021-08-28T23:39:27.506785Z","iopub.status.idle":"2021-08-28T23:39:27.517232Z","shell.execute_reply.started":"2021-08-28T23:39:27.506747Z","shell.execute_reply":"2021-08-28T23:39:27.516044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create master data containing FLAIR, T1w, T1wCE, T2w for every patient\n\nflair_dict = {}\nT1w_dict = {}\nT1wCE_dict = {}\nT2w_dict = {}\n\nfor i in tqdm(range(0, len(train_data))):\n    pat_id_for_folder = str(train_data['BraTS21ID'].iloc[i]).rjust(5, \"0\")\n    \n    for image_type in ['FLAIR', 'T1w', 'T1wCE', 'T2w']:\n        image_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/' + pat_id_for_folder + f'/{image_type}/'    \n        image_list = os.listdir(image_path)\n        train_data[f'# {image_type} images'].iloc[i] = len(image_list)\n\n        all_scans_of_patient = load_scan(image_path)\n        all_scans_of_patient = get_pixels_hu(all_scans_of_patient)\n        temp_dict = {}\n        for j in range(0, len(image_list)):\n            temp_dict[image_list[j]] = np.resize(all_scans_of_patient[j], (512,512)) # Resize into same shape\n            \n        if image_type == 'FLAIR':\n            flair_dict[train_data['BraTS21ID'].iloc[i]] = temp_dict\n        elif image_type == 'T1w':\n            T1w_dict[train_data['BraTS21ID'].iloc[i]] = temp_dict\n        elif image_type == 'T1wCE':\n            T1wCE_dict[train_data['BraTS21ID'].iloc[i]] = temp_dict\n        elif image_type == 'T2w':\n            T2w_dict[train_data['BraTS21ID'].iloc[i]] = temp_dict","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:39:27.518425Z","iopub.execute_input":"2021-08-28T23:39:27.518940Z","iopub.status.idle":"2021-08-28T23:42:49.337349Z","shell.execute_reply.started":"2021-08-28T23:39:27.518901Z","shell.execute_reply":"2021-08-28T23:42:49.336359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:42:49.338782Z","iopub.execute_input":"2021-08-28T23:42:49.339151Z","iopub.status.idle":"2021-08-28T23:42:49.353177Z","shell.execute_reply.started":"2021-08-28T23:42:49.339112Z","shell.execute_reply":"2021-08-28T23:42:49.352007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grp_data = train_data.groupby(['MGMT_value']).count().reset_index().iloc[:,0:2].rename(columns = {'BraTS21ID':'# Patients'})\nfig = px.bar(grp_data, x='MGMT_value', y='# Patients', color = 'MGMT_value')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:42:49.354716Z","iopub.execute_input":"2021-08-28T23:42:49.355256Z","iopub.status.idle":"2021-08-28T23:42:50.342276Z","shell.execute_reply.started":"2021-08-28T23:42:49.355212Z","shell.execute_reply":"2021-08-28T23:42:50.341492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To convert all slices to a single entry, I am averaging all the scans under the 4 types for each patient. Let's see the created image after averaging.","metadata":{}},{"cell_type":"code","source":"flair_avg_dict = {}\nT1w_avg_dict = {}\nT1wCE_avg_dict = {}\nT2w_avg_dict = {}\n\nfor i in tqdm(range(0, len(train_data))):\n    pat_id_for_folder = str(train_data['BraTS21ID'].iloc[i]).rjust(5, \"0\")\n    flair_avg_dict[train_data['BraTS21ID'].iloc[i]] = np.array([v for k, v in flair_dict[train_data['BraTS21ID'].iloc[i]].items()]).mean(axis = 0)\n    T1w_avg_dict[train_data['BraTS21ID'].iloc[i]] = np.array([v for k, v in T1w_dict[train_data['BraTS21ID'].iloc[i]].items()]).mean(axis = 0)\n    T1wCE_avg_dict[train_data['BraTS21ID'].iloc[i]] = np.array([v for k, v in T1wCE_dict[train_data['BraTS21ID'].iloc[i]].items()]).mean(axis = 0)\n    T2w_avg_dict[train_data['BraTS21ID'].iloc[i]] = np.array([v for k, v in T2w_dict[train_data['BraTS21ID'].iloc[i]].items()]).mean(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:42:50.346281Z","iopub.execute_input":"2021-08-28T23:42:50.348494Z","iopub.status.idle":"2021-08-28T23:42:58.891069Z","shell.execute_reply.started":"2021-08-28T23:42:50.348449Z","shell.execute_reply":"2021-08-28T23:42:58.889731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del flair_dict\ndel T1w_dict\ndel T1wCE_dict\ndel T2w_dict\ndel all_scans_of_patient","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:42:58.892660Z","iopub.execute_input":"2021-08-28T23:42:58.893173Z","iopub.status.idle":"2021-08-28T23:42:58.917876Z","shell.execute_reply.started":"2021-08-28T23:42:58.893131Z","shell.execute_reply":"2021-08-28T23:42:58.917068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = 4\ncols = 5\nstack = [v for k,v in flair_avg_dict.items()]\nstack_ids = [k for k,v in flair_avg_dict.items()]\npat_with_tumor = train_data[train_data['MGMT_value']==1]['BraTS21ID'].tolist()\nfig,ax = plt.subplots(cols,rows,figsize=[12,12])\nfor i in range(rows*cols):\n    ind = stack_ids[i]\n    if ind in pat_with_tumor:\n        ax[max(int(i/rows),0),int(i % rows)].set_title('With tumor', color = 'red')\n    else:\n        ax[max(int(i/rows),0),int(i % rows)].set_title('No tumor', color = 'green')\n    ax[max(int(i/rows),0),int(i % rows)].imshow(stack[i],cmap='gray')\n    ax[max(int(i/rows),0),int(i % rows)].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:42:58.919106Z","iopub.execute_input":"2021-08-28T23:42:58.919471Z","iopub.status.idle":"2021-08-28T23:43:00.253673Z","shell.execute_reply.started":"2021-08-28T23:42:58.919432Z","shell.execute_reply":"2021-08-28T23:43:00.252791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Treating all slices as a single entry through averaging makes it hard to distinguish the tumor very well","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(train_data['BraTS21ID'], train_data['MGMT_value'], test_size=0.3, random_state=42, stratify = train_data['MGMT_value'])","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.257445Z","iopub.execute_input":"2021-08-28T23:43:00.259494Z","iopub.status.idle":"2021-08-28T23:43:00.272422Z","shell.execute_reply.started":"2021-08-28T23:43:00.259450Z","shell.execute_reply":"2021-08-28T23:43:00.271350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['is_train_flag'] = np.where(train_data['BraTS21ID'].isin(X_train), 1, 0)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.276147Z","iopub.execute_input":"2021-08-28T23:43:00.278261Z","iopub.status.idle":"2021-08-28T23:43:00.286105Z","shell.execute_reply.started":"2021-08-28T23:43:00.278217Z","shell.execute_reply":"2021-08-28T23:43:00.285128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grp_data = train_data.groupby(['is_train_flag']).agg({'BraTS21ID':'count', 'MGMT_value':'sum'}).reset_index().rename(columns = {'BraTS21ID':'# Patients', 'MGMT_value':'# Patients with brain tumor'})\ngrp_data['% Patients with tumor'] = grp_data['# Patients with brain tumor']/grp_data['# Patients'] * 100\ngrp_data","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.291000Z","iopub.execute_input":"2021-08-28T23:43:00.293088Z","iopub.status.idle":"2021-08-28T23:43:00.321683Z","shell.execute_reply.started":"2021-08-28T23:43:00.293048Z","shell.execute_reply":"2021-08-28T23:43:00.320826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_for_modelling = np.concatenate([np.array([v for k,v in flair_avg_dict.items() if k in list(X_train)]),\n         np.array([v for k,v in T1w_avg_dict.items() if k in list(X_train)]),\n         np.array([v for k,v in T1wCE_avg_dict.items() if k in list(X_train)]),\n         np.array([v for k,v in T2w_avg_dict.items() if k in list(X_train)])])\n\ny_train_for_modelling = np.array(list(train_data[train_data['BraTS21ID'].isin([k for k,v in flair_avg_dict.items() if k in list(X_train)])]['MGMT_value']) * 4)\n\nX_valid_for_modelling = np.concatenate([np.array([v for k,v in flair_avg_dict.items() if k in list(X_valid)]),\n         np.array([v for k,v in T1w_avg_dict.items() if k in list(X_valid)]),\n         np.array([v for k,v in T1wCE_avg_dict.items() if k in list(X_valid)]),\n         np.array([v for k,v in T2w_avg_dict.items() if k in list(X_valid)])])\n\ny_valid_for_modelling = np.array(list(train_data[train_data['BraTS21ID'].isin([k for k,v in flair_avg_dict.items() if k in list(X_valid)])]['MGMT_value']) * 4)\n\nprint('X train shape:', X_train_for_modelling.shape)\nprint('y train shape:', y_train_for_modelling.shape)\nprint('X validation shape:', X_valid_for_modelling.shape)\nprint('y validation shape:', y_valid_for_modelling.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.325398Z","iopub.execute_input":"2021-08-28T23:43:00.327564Z","iopub.status.idle":"2021-08-28T23:43:00.489492Z","shell.execute_reply.started":"2021-08-28T23:43:00.327524Z","shell.execute_reply":"2021-08-28T23:43:00.488642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# building the input vector from the 512x512 pixels\nX_train_for_modelling = X_train_for_modelling.reshape(X_train_for_modelling.shape[0], 512, 512, 1)\nX_valid_for_modelling = X_valid_for_modelling.reshape(X_valid_for_modelling.shape[0], 512, 512, 1)\nX_train_for_modelling = X_train_for_modelling.astype('float32')\nX_valid_for_modelling = X_valid_for_modelling.astype('float32')\n\n# one-hot encoding using keras' numpy-related utilities\nn_classes = 1\n# print(\"Shape before one-hot encoding: \", y_train_for_modelling.shape)\n# y_train_for_modelling = np_utils.to_categorical(y_train_for_modelling, n_classes)\n# y_test_for_modelling = np_utils.to_categorical(y_test_for_modelling, n_classes)\n# print(\"Shape after one-hot encoding: \", y_train_for_modelling.shape)\n\nprint('X train shape:', X_train_for_modelling.shape)\nprint('y train shape:', y_train_for_modelling.shape)\nprint('X validation shape:', X_valid_for_modelling.shape)\nprint('y validation shape:', y_valid_for_modelling.shape)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.493731Z","iopub.execute_input":"2021-08-28T23:43:00.495871Z","iopub.status.idle":"2021-08-28T23:43:00.549935Z","shell.execute_reply.started":"2021-08-28T23:43:00.495826Z","shell.execute_reply":"2021-08-28T23:43:00.549044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:47:00.551230Z","iopub.execute_input":"2021-08-28T23:47:00.551656Z","iopub.status.idle":"2021-08-28T23:47:00.809101Z","shell.execute_reply.started":"2021-08-28T23:47:00.551597Z","shell.execute_reply":"2021-08-28T23:47:00.807831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modelling","metadata":{}},{"cell_type":"code","source":"# Transfer learning code to be updated\n# import keras\n# base_model = keras.applications.Xception(\n#     weights='imagenet',  # Load weights pre-trained on ImageNet.\n#     input_shape=(512, 512, 1),\n#     include_top=False)  # Do not include the ImageNet classifier at the top.","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.553747Z","iopub.execute_input":"2021-08-28T23:43:00.555976Z","iopub.status.idle":"2021-08-28T23:43:00.561342Z","shell.execute_reply.started":"2021-08-28T23:43:00.555934Z","shell.execute_reply":"2021-08-28T23:43:00.560490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# base_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.566170Z","iopub.execute_input":"2021-08-28T23:43:00.568544Z","iopub.status.idle":"2021-08-28T23:43:00.573869Z","shell.execute_reply.started":"2021-08-28T23:43:00.568505Z","shell.execute_reply":"2021-08-28T23:43:00.572920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inputs = keras.Input(shape=(512, 512, 1))\n# # We make sure that the base_model is running in inference mode here,\n# # by passing `training=False`. This is important for fine-tuning, as you will\n# # learn in a few paragraphs.\n# x = base_model(inputs, training=False)\n# # Convert features of shape `base_model.output_shape[1:]` to vectors\n# x = keras.layers.GlobalAveragePooling2D()(x)\n# # A Dense classifier with a single unit (binary classification)\n# outputs = keras.layers.Dense(1)(x)\n# model = keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:43:00.578126Z","iopub.execute_input":"2021-08-28T23:43:00.579832Z","iopub.status.idle":"2021-08-28T23:43:00.586202Z","shell.execute_reply.started":"2021-08-28T23:43:00.579793Z","shell.execute_reply":"2021-08-28T23:43:00.585249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# building a linear stack of layers with the sequential model\nmodel = Sequential()\n# convolutional layer\nmodel.add(Conv2D(25, kernel_size=(3,3), strides=(1,1), activation='relu', input_shape=(512, 512, 1)))\nmodel.add(MaxPool2D(pool_size=(50,50)))\n# flatten output of conv\nmodel.add(Flatten())\n# hidden layer\nmodel.add(Dense(100, activation='relu'))\n# output layer\nmodel.add(Dense(n_classes, activation='sigmoid'))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:47:14.785510Z","iopub.execute_input":"2021-08-28T23:47:14.785916Z","iopub.status.idle":"2021-08-28T23:47:14.831048Z","shell.execute_reply.started":"2021-08-28T23:47:14.785875Z","shell.execute_reply":"2021-08-28T23:47:14.830010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=keras.optimizers.Adam(),\n              loss=keras.losses.BinaryCrossentropy(from_logits=True),\n              metrics=[keras.metrics.BinaryAccuracy()])\nmodel.fit(X_train_for_modelling, y_train_for_modelling, epochs=20, validation_data=(X_valid_for_modelling, y_valid_for_modelling))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T23:47:34.893497Z","iopub.execute_input":"2021-08-28T23:47:34.893890Z","iopub.status.idle":"2021-08-28T23:47:48.718329Z","shell.execute_reply.started":"2021-08-28T23:47:34.893853Z","shell.execute_reply":"2021-08-28T23:47:48.716396Z"},"trusted":true},"execution_count":null,"outputs":[]}]}