{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Fork of v14 RSNA-BTRC-xception2D-noGPU v4\n\nv1 load imagenet weight\n\nv2 less unit on FCN: 2048 per layers\n\nv3 add dropout layer between FCN layers\n\nv4 1 FCN layer only, 70 epochs, patient = 25, drop_prob = 0.3\n\nv5 2 FCN layer with dropout (v3 with longer epoch)\n\nv6 dropout = 0.5\n\nv7 3rd layer FCN\n\nv8 4th layer FCN and batchnorm\n\nv9 del batchnorm\n\nv10 fcn 4096\n\nv12 dropout = 0.3\n\nv13 fcn layers = 3, rotation -90 - 90, initializer = he\n\nv14 drop_prob = 0 on fcn 1, back to v1 with more rotation\n\nv15 add L2 regularization on layer 1\n\nv16 300 epochs\n\nv17 no LR schedule\n\nv18 add 1 fcn layer\n\nv19 drop prob 0.5 for 1st fcn layer\n\nv20 del 1st fcn layer, DO for fcn 2 = 0.2, add LR schedule decay = 1,036\n\nv21 no LR schedule\n\nv22 LR schedule, DO 1st layer = 0, 1000 epochs, no GPU\n\nv23 DO 1st = 0.2, DO 2nd = 0.3\n\nv24 DO1st = 0.2, no regularization\n\nv25 3 fcn layers with regularization\n\nv28 drop prob fcn 1 = 0\n\nv29 drop prob fnc 1 = 0.3, use T2W only\n\nv30 add FLAIR as second channel\n\nv31 return FLAIR to first channel, layer 1 & 2 DO = 0.2\n\nv32 LR decay rate = 0.92, staircase = False, layer 1 DO = 0.1\n\nv33 LR decay rate = 0.9, decay steps = 280, data from notebook output\n\nv34 layer 1 DO = 0.2, try RMSprop\n\nv35 Adam\n\nv36 overfit train set, see v14\n\nv37 more augmentation on random resize crop\n\nv38 undo augmentaion on random resize crop, L2 reg\n\nv39 no L2 reg, del layer 1, dense units = 2048\n\nv40 L2 reg\n\nv41 dropout = 0.5, layer 1 active, units = 4096\n\nv42 no L2 reg\n\nv43 L2 reg, units = 2048\n\nv44 del layer 1","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport os\nimport pydicom\nfrom pydicom import dcmread\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom scipy import ndimage\nimport random\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.lines as lines\nfrom glob import glob\nfrom PIL import Image\nimport gc\n\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-28T16:35:57.985235Z","iopub.execute_input":"2021-09-28T16:35:57.9856Z","iopub.status.idle":"2021-09-28T16:36:02.594809Z","shell.execute_reply.started":"2021-09-28T16:35:57.985516Z","shell.execute_reply":"2021-09-28T16:36:02.594053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)\ntf.random.set_seed(42)\nrandom.seed(42)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.596258Z","iopub.execute_input":"2021-09-28T16:36:02.596529Z","iopub.status.idle":"2021-09-28T16:36:02.60028Z","shell.execute_reply.started":"2021-09-28T16:36:02.596496Z","shell.execute_reply":"2021-09-28T16:36:02.59965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Global Variables and Data","metadata":{}},{"cell_type":"code","source":"# Read train labels data and drop data with issues\nPATH = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\ntrain_data_path = os.path.join(PATH,'train_labels.csv')\ntrain_data = pd.read_csv(train_data_path, dtype={'BraTS21ID': object})\ntrain_data = train_data.set_index('BraTS21ID')\ntrain_data = train_data.drop(['00109', '00123', '00709'], axis = 0)\ntrain_data = train_data.reset_index()\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.601465Z","iopub.execute_input":"2021-09-28T16:36:02.601851Z","iopub.status.idle":"2021-09-28T16:36:02.63876Z","shell.execute_reply.started":"2021-09-28T16:36:02.601812Z","shell.execute_reply":"2021-09-28T16:36:02.638147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Global variables\nVOL = 1 # Cap image depth/volume\nSIZE = 256 # Image size width and heigth\nBATCH_SIZE = 32\nDATA_LEN = len(train_data)\nVAL_RATIO = 0.2\nTRAIN_RATIO = 1 - VAL_RATIO\nTRAIN_SIZE = int(((DATA_LEN * TRAIN_RATIO) // BATCH_SIZE) * BATCH_SIZE) # To ensure train data batched evenly (no empty numpy)\nmpMRIs = ['FLAIR', 'T2w']\nFILE_TYPES = len(mpMRIs)\nCHANNEL = 3","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.64126Z","iopub.execute_input":"2021-09-28T16:36:02.64146Z","iopub.status.idle":"2021-09-28T16:36:02.648591Z","shell.execute_reply.started":"2021-09-28T16:36:02.641439Z","shell.execute_reply":"2021-09-28T16:36:02.64782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Define function to Load Dicom Files and Normalize Data\n\nGet 3D data code source: \nhttps://www.kaggle.com/rluethy/efficientnet3d-with-one-mri-type\n\nCrop 3D images:\nhttps://www.kaggle.com/josecarmona/btrc-a-simple-tool-to-crop-3d-images\n\nAdjustting contrast on MR images:\nhttps://www.kaggle.com/davidbroberts/adjusting-contrast-on-mr-images","metadata":{}},{"cell_type":"markdown","source":"def get_image_plane(data):\n    x1, y1, _, x2, y2, _ = [round(j) for j in data.ImageOrientationPatient]\n    cords = [x1, y1, x2, y2]\n\n    if cords == [1, 0, 0, 0]:\n        return 'Coronal'\n    elif cords == [1, 0, 0, 1]:\n        return 'Axial'\n    elif cords == [0, 1, 0, 0]:\n        return 'Sagittal'\n    else:\n        return 'Unknown'","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.050397Z","iopub.execute_input":"2021-09-18T16:12:14.050664Z","iopub.status.idle":"2021-09-18T16:12:14.062741Z","shell.execute_reply.started":"2021-09-18T16:12:14.050634Z","shell.execute_reply":"2021-09-18T16:12:14.0616Z"}}},{"cell_type":"markdown","source":"def get_voxel(study_id, cohort, mpMRI):\n    imgs = []\n    dcm_dir = os.path.join(PATH, cohort, study_id, mpMRI)\n    dcm_paths = sorted(glob(f\"{dcm_dir}/*.dcm\"),key=lambda f: int(f.split('Image-')[1].split('.')[0]))\n    positions = []\n    \n    for dcm_path in dcm_paths:\n        img = pydicom.dcmread(str(dcm_path))\n        imgs.append(img.pixel_array)\n        positions.append(img.ImagePositionPatient)\n    \n    img = pydicom.dcmread(str(dcm_paths[0]))    \n    plane = get_image_plane(img)\n    voxel = np.stack(imgs)\n    p_i = img.PhotometricInterpretation\n        \n    # reorder planes if needed and rotate voxel\n    if plane == \"Coronal\":\n        if positions[0][1] < positions[-1][1]:\n            voxel = voxel[::-1]\n        voxel = voxel.transpose((1, 0, 2))\n    elif plane == \"Sagittal\":\n        if positions[0][0] < positions[-1][0]:\n            voxel = voxel[::-1]\n        voxel = voxel.transpose((1, 2, 0))\n        voxel = np.rot90(voxel, 2, axes=(1, 2))\n    elif plane == \"Axial\":\n        if positions[0][2] > positions[-1][2]:\n            voxel = voxel[::-1]\n        voxel = np.rot90(voxel, 2)\n    else:\n        raise ValueError(f\"Unknown plane {plane}\")\n    return voxel, p_i","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.064339Z","iopub.execute_input":"2021-09-18T16:12:14.064603Z","iopub.status.idle":"2021-09-18T16:12:14.077735Z","shell.execute_reply.started":"2021-09-18T16:12:14.064574Z","shell.execute_reply":"2021-09-18T16:12:14.076939Z"}}},{"cell_type":"markdown","source":"# Find top brilliant image:\ndef stats_image(image):\n    nonzero_pixels = image[np.nonzero(image)]\n    if nonzero_pixels.shape == (0,):\n        mean = 0\n        std = 0\n    else:\n        mean = np.mean(nonzero_pixels)\n        std = np.std(nonzero_pixels)\n    return (mean,std)\n\ndef calc_idx(image):\n    (mean,std) = stats_image(image)\n    nonzero_pixels = np.count_nonzero(image > (mean + 1.7 * std))\n    return nonzero_pixels\n    \ndef top_brilliant_image(images):\n    idx = [calc_idx(image) for image in images]\n    top_image = np.argsort(idx)[::-1][0]\n    return top_image","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.078701Z","iopub.execute_input":"2021-09-18T16:12:14.078942Z","iopub.status.idle":"2021-09-18T16:12:14.093618Z","shell.execute_reply.started":"2021-09-18T16:12:14.078917Z","shell.execute_reply":"2021-09-18T16:12:14.092443Z"}}},{"cell_type":"markdown","source":"def cropped_images(image):\n    ''' Crop images heigth and width'''\n    min_pix=np.array(np.nonzero(image)).min(axis=1)\n    max_pix=np.array(np.nonzero(image)).max(axis=1)\n    return image[min_pix[0]:max_pix[0],min_pix[1]:max_pix[1]]","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.094822Z","iopub.execute_input":"2021-09-18T16:12:14.095585Z","iopub.status.idle":"2021-09-18T16:12:14.104838Z","shell.execute_reply.started":"2021-09-18T16:12:14.095448Z","shell.execute_reply":"2021-09-18T16:12:14.104165Z"}}},{"cell_type":"markdown","source":"#https://scikit-image.org/docs/dev/auto_examples/color_exposure/plot_adapt_hist_eq_3d.html#sphx-glr-auto-examples-color-exposure-plot-adapt-hist-eq-3d-py\n\nfrom skimage import exposure, util\nimport imageio as io\n\ndef hist_normalize(im_orig):\n    # Perform clahe\n    im_orig = exposure.equalize_adapthist(im_orig)\n    \n    return im_orig","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.105791Z","iopub.execute_input":"2021-09-18T16:12:14.106492Z","iopub.status.idle":"2021-09-18T16:12:14.309945Z","shell.execute_reply.started":"2021-09-18T16:12:14.106453Z","shell.execute_reply":"2021-09-18T16:12:14.308912Z"}}},{"cell_type":"markdown","source":"def normalize(im_orig):\n    \n    # Rescale image data to range [0, 1]\n    #im_orig = np.clip(im_orig,\n    #                  np.percentile(im_orig, 5),\n    #                  np.percentile(im_orig, 95))\n    if (im_orig.max() - im_orig.min()) !=0:\n        im_orig = (im_orig - im_orig.min()) / (im_orig.max() - im_orig.min())\n    \n    return im_orig","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.315136Z","iopub.execute_input":"2021-09-18T16:12:14.315414Z","iopub.status.idle":"2021-09-18T16:12:14.323342Z","shell.execute_reply.started":"2021-09-18T16:12:14.315385Z","shell.execute_reply":"2021-09-18T16:12:14.32256Z"}}},{"cell_type":"markdown","source":"#https://www.kaggle.com/davidbroberts/standardizing-mr-images\n\ndef clamped_value(value):\n    return min(max(value, 0), 255) #(use 255 or 1)\n\n# Make a simple linear VOI LUT from the raw (stored) pixel data and Apply the LUT to a pixel array\ndef apply_lut(pixels, p_i, center=500, width=350, invert = False):\n    # Invert pixels and cent for MONOCHROME1. We invert the specified center so that \n    # increasing the center value makes the images brighter regardless of photometric intrepretation\n    max_pix = pixels.max()\n    min_pix = pixels.min()\n    if p_i == \"MONOCHROME1\":\n        invert = True\n    else:\n        center = (max_pix - min_pix) - center\n    \n    # Slope and Intercept set to 1 and 0 for MR. Get these from DICOM tags instead if using \n    # on a modality that requires them (CT, PT etc)\n    slope = 1.0\n    intercept = 0.0\n    \n    pixels = pixels * slope + intercept  \n    pixels = ((pixels - center) / width + 0.5) * 255.0 #(use 255 or 1)\n    #pixels = np.vectorize(clamped_value)(pixels)  \n    \n    if invert:\n        pixels = 255-pixels # (use 255 or 1 or max pix)\n    \n    return pixels","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.324878Z","iopub.execute_input":"2021-09-18T16:12:14.325111Z","iopub.status.idle":"2021-09-18T16:12:14.335012Z","shell.execute_reply.started":"2021-09-18T16:12:14.325083Z","shell.execute_reply":"2021-09-18T16:12:14.334185Z"}}},{"cell_type":"markdown","source":"def get_dcm_data(study_id, cohort):\n    '''Combine all 2D images in mpMRI folder into a 3D images \n    with dimension: (batch_size, height, width, volume, channel)'''\n    image_3d = np.empty((SIZE, SIZE, CHANNEL), dtype=np.float32)\n    for i, mpMRI in enumerate(mpMRIs):\n        images, p_i = get_voxel(study_id, cohort, mpMRI)\n        \n        image = images[top_brilliant_image(images)]\n        \n        # Crop volume of 3D image:\n        image = cropped_images(image)\n        \n        #clahe = cv2.createCLAHE(clipLimit=0.8, tileGridSize=(8,8))\n        #for i in range(VOL):\n        #    image_3d[i,:,:] = clahe.apply(image_3d[i,:,:]) # choose either one\n        #    image_3d[i,:,:] = cv2.equalizeHist(image_3d[i,:,:]) # choose either one\n        \n        # Standardize image\n        #image = hist_normalize(image)\n        image = apply_lut(image, p_i, center=500, width=350, invert = False).astype('float32')\n            \n        # Resize to SIZE, SIZE, VOL  \n        image = cv2.resize(image, (SIZE, SIZE))\n        image_3d[:,:,i] = image\n    \n    if FILE_TYPES < 3:\n        image_3d[:,:,2] = image_3d[:,:,0]\n    if FILE_TYPES < 2:\n        image_3d[:,:,1] = image_3d[:,:,0]\n        \n    return image_3d","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.336652Z","iopub.execute_input":"2021-09-18T16:12:14.336931Z","iopub.status.idle":"2021-09-18T16:12:14.347825Z","shell.execute_reply.started":"2021-09-18T16:12:14.336905Z","shell.execute_reply":"2021-09-18T16:12:14.347139Z"}}},{"cell_type":"markdown","source":"## Data Augmentation and Generation\n\nDataGenerator code source: \nhttps://stanford.edu/~shervine/blog/keras-how-to-generate-data-on-the-fly","metadata":{}},{"cell_type":"code","source":"# Elastic Transform\n#https://www.kaggle.com/bguberfain/elastic-transform-for-data-augmentation\n\nfrom scipy.ndimage.interpolation import map_coordinates\nfrom scipy.ndimage.filters import gaussian_filter\n\n# Function to distort image\ndef elastic_transform(image, alpha, sigma, alpha_affine, random_state=None):\n    \"\"\"Elastic deformation of images as described in [Simard2003]_ (with modifications).\n    .. [Simard2003] Simard, Steinkraus and Platt, \"Best Practices for\n         Convolutional Neural Networks applied to Visual Document Analysis\", in\n         Proc. of the International Conference on Document Analysis and\n         Recognition, 2003.\n\n     Based on https://gist.github.com/erniejunior/601cdf56d2b424757de5\n    \"\"\"\n    if random_state is None:\n        random_state = np.random.RandomState(None)\n\n    shape = image.shape\n    shape_size = shape[:2]\n    \n    # Random affine\n    center_square = np.float32(shape_size) // 2\n    square_size = min(shape_size) // 3\n    pts1 = np.float32([center_square + square_size, [center_square[0]+square_size, center_square[1]-square_size], center_square - square_size])\n    pts2 = pts1 + random_state.uniform(-alpha_affine, alpha_affine, size=pts1.shape).astype(np.float32)\n    M = cv2.getAffineTransform(pts1, pts2)\n    image = cv2.warpAffine(image, M, shape_size[::-1], borderMode=cv2.BORDER_REFLECT_101)\n\n    dx = gaussian_filter((random_state.rand(*shape) * 2 - 1), sigma) * alpha\n    dy = gaussian_filter((random_state.rand(*shape) * 2 - 1), sigma) * alpha\n    dz = np.zeros_like(dx)\n\n    x, y, z = np.meshgrid(np.arange(shape[1]), np.arange(shape[0]), np.arange(shape[2]))\n    indices = np.reshape(y+dy, (-1, 1)), np.reshape(x+dx, (-1, 1)), np.reshape(z, (-1, 1))\n\n    return map_coordinates(image, indices, order=1, mode='reflect').reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.652829Z","iopub.execute_input":"2021-09-28T16:36:02.653004Z","iopub.status.idle":"2021-09-28T16:36:02.664293Z","shell.execute_reply.started":"2021-09-28T16:36:02.652985Z","shell.execute_reply":"2021-09-28T16:36:02.663602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://www.kaggle.com/sreevishnudamodaran/rsna-3d-clahe-voxels-tpu-3d-augmentations\n\ndef random_resized_crop2D(image, min_width, min_height, p=0.5):\n    if np.random.uniform(()) < p:\n        image_shape = image.shape\n        assert image_shape[0] >= min_height\n        assert image_shape[1] >= min_width\n        \n        width = np.random.uniform((), minval=min_width, maxval=image_shape[1],\n                                  dtype=tf.int32)\n        height = np.random.uniform((), minval=min_height, maxval=image_shape[0],\n                                   dtype=tf.int32)\n        x = np.random.uniform((), minval=0, maxval=image_shape[1] - width,\n                                  dtype=tf.int32)\n        y = np.random.uniform((), minval=0, maxval=image_shape[0] - height,\n                                  dtype=tf.int32)\n        image = image[:, y:y+height, x:x+width, :]\n        image = cv2.resize(image, image_shape[0:2], interpolation =cv2.INTER_LANCZOS4)\n        image = np.nan_to_num(image)\n        \n    return image","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.665565Z","iopub.execute_input":"2021-09-28T16:36:02.666185Z","iopub.status.idle":"2021-09-28T16:36:02.676192Z","shell.execute_reply.started":"2021-09-28T16:36:02.666152Z","shell.execute_reply":"2021-09-28T16:36:02.675518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flip(image):\n    flip_direction = random.randint(0,2)\n    if flip_direction == 0:\n        return image\n    elif flip_direction == 1:\n        return np.fliplr(image)\n    else:\n        return np.flipud(image)\n\ndef rotate_image(image):\n    angle = random.randrange(-90, 90)\n    for i in range(CHANNEL):\n        image[:,:,i]\n        image_center = tuple(np.array(image[:,:,i].shape[1::-1]) / 2)\n        rot_mat = cv2.getRotationMatrix2D(image_center, angle, 1.0)\n        image[:,:,i] = cv2.warpAffine(image[:,:,i], rot_mat, image[:,:,i].shape[1::-1], flags=cv2.INTER_LINEAR, borderMode=cv2.BORDER_REFLECT_101)\n    return image  \n    \ndef data_preprocessing(volume, training):\n    \"\"\"Process training data by rotating and adding a channel.\"\"\"\n    # Rotate volume\n    if training:\n        volume = flip(volume)\n        volume = rotate_image(volume)\n        \n        # Elastic deformation\n        deform = random.randint(0,1)\n        magnitude = random.uniform(0.07, 0.09)\n        if deform:\n            volume = elastic_transform(volume, SIZE*2, SIZE*magnitude, SIZE*magnitude, random_state=None)\n        \n        #min_val = random.randint(180,256)\n        #prob = random.uniform(0.5, 1)\n        RANDOM_CROP = (180, 180, 1) # @params: (min_width, min_height, probability)\n        volume = random_resized_crop2D(volume, RANDOM_CROP[0], RANDOM_CROP[1], RANDOM_CROP[2])      \n    return volume\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, data, labels, batch_size=BATCH_SIZE, dim=(SIZE,SIZE), n_channels=CHANNEL, shuffle=True, training=True):\n        'Initialization'\n        self.data = data\n        self.dim = dim\n        self.batch_size = batch_size\n        self.labels = labels\n        self.n_channels = n_channels\n        self.shuffle = shuffle\n        self.training = training\n        self.on_epoch_end()\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(self.data.shape[0] / self.batch_size))\n\n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        # Generate data\n        X, y = self.__data_generation(indexes)\n\n        return X, y\n\n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(self.data.shape[0])\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n\n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples' # X : (n_samples, *dim, n_channels)\n        # Initialization\n        X = np.empty((self.batch_size, *self.dim, self.n_channels), dtype=np.float32)\n        y = np.empty((self.batch_size), dtype=np.int8)\n\n        # Generate data\n        for i, index in enumerate (indexes):\n            # Store sample\n            X[i,] = data_preprocessing(self.data[index,:,:,:], self.training)\n\n            # Store class\n            y[i] = self.labels[index]\n\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.677441Z","iopub.execute_input":"2021-09-28T16:36:02.677832Z","iopub.status.idle":"2021-09-28T16:36:02.698002Z","shell.execute_reply.started":"2021-09-28T16:36:02.677796Z","shell.execute_reply":"2021-09-28T16:36:02.697325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Test IDs Split","metadata":{}},{"cell_type":"markdown","source":"cohort = 'train'\n\nfrom sklearn.model_selection import train_test_split\n\ntrain, val = train_test_split(train_data, train_size = TRAIN_SIZE, random_state = 42, stratify=train_data[\"MGMT_value\"])\n\n# Create list of ids and dictionary of labels for Data Generator input\ntrain_ids = train['BraTS21ID'].values.tolist()\ny_train = train['MGMT_value'].values.astype('int8')\nval_ids = val['BraTS21ID'].values.tolist()\ny_val = val['MGMT_value'].values.astype('int8')","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:14.425437Z","iopub.execute_input":"2021-09-18T16:12:14.426108Z","iopub.status.idle":"2021-09-18T16:12:15.155557Z","shell.execute_reply.started":"2021-09-18T16:12:14.426065Z","shell.execute_reply":"2021-09-18T16:12:15.154381Z"}}},{"cell_type":"markdown","source":"del train_data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:15.159336Z","iopub.execute_input":"2021-09-18T16:12:15.159638Z","iopub.status.idle":"2021-09-18T16:12:15.340989Z","shell.execute_reply.started":"2021-09-18T16:12:15.159604Z","shell.execute_reply":"2021-09-18T16:12:15.340031Z"}}},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"markdown","source":"import time\nstart = time.time()\nx_train_1 = np.array([get_dcm_data(ID, cohort) for ID in train_ids[:5]], dtype=np.float32)\nend = time.time()\nprint(end - start)","metadata":{"execution":{"iopub.status.busy":"2021-09-16T03:12:18.19827Z","iopub.execute_input":"2021-09-16T03:12:18.198625Z","iopub.status.idle":"2021-09-16T03:12:33.452405Z","shell.execute_reply.started":"2021-09-16T03:12:18.198594Z","shell.execute_reply":"2021-09-16T03:12:33.451176Z"}}},{"cell_type":"markdown","source":"plt.imshow(x_train_1[4,:,:,1], cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2021-09-16T03:14:17.056592Z","iopub.execute_input":"2021-09-16T03:14:17.056914Z","iopub.status.idle":"2021-09-16T03:14:17.6389Z","shell.execute_reply.started":"2021-09-16T03:14:17.056885Z","shell.execute_reply":"2021-09-16T03:14:17.638194Z"}}},{"cell_type":"markdown","source":"plt.hist(x_train_1[4,:,:,1])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-15T02:57:09.364618Z","iopub.execute_input":"2021-09-15T02:57:09.364888Z","iopub.status.idle":"2021-09-15T02:57:13.896917Z","shell.execute_reply.started":"2021-09-15T02:57:09.364858Z","shell.execute_reply":"2021-09-15T02:57:13.895965Z"}}},{"cell_type":"markdown","source":"train_dataset = DataGenerator(x_train_1.copy(), y_train[:5])","metadata":{"execution":{"iopub.status.busy":"2021-09-16T03:15:51.891299Z","iopub.execute_input":"2021-09-16T03:15:51.892305Z","iopub.status.idle":"2021-09-16T03:15:51.898487Z","shell.execute_reply.started":"2021-09-16T03:15:51.892259Z","shell.execute_reply":"2021-09-16T03:15:51.897293Z"}}},{"cell_type":"markdown","source":"X = train_dataset[0][0]","metadata":{"execution":{"iopub.status.busy":"2021-09-16T03:17:41.42983Z","iopub.execute_input":"2021-09-16T03:17:41.430154Z","iopub.status.idle":"2021-09-16T03:17:41.451068Z","shell.execute_reply.started":"2021-09-16T03:17:41.430124Z","shell.execute_reply":"2021-09-16T03:17:41.44986Z"}}},{"cell_type":"markdown","source":"plt.imshow(X[4,:,:,0], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2021-09-16T03:19:01.946797Z","iopub.execute_input":"2021-09-16T03:19:01.947132Z","iopub.status.idle":"2021-09-16T03:19:02.190071Z","shell.execute_reply.started":"2021-09-16T03:19:01.947084Z","shell.execute_reply":"2021-09-16T03:19:02.189005Z"}}},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"markdown","source":"import time\nstart = time.time()\n# Generate Data\nx_train = np.empty((len(train_ids),SIZE,SIZE,CHANNEL), dtype=np.float32)\n\nfor i, ID in enumerate(train_ids):\n    x_train[i] = get_dcm_data(ID, cohort)","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:12:15.342286Z","iopub.execute_input":"2021-09-18T16:12:15.34253Z","iopub.status.idle":"2021-09-18T16:39:12.457525Z","shell.execute_reply.started":"2021-09-18T16:12:15.342504Z","shell.execute_reply":"2021-09-18T16:39:12.456144Z"}}},{"cell_type":"markdown","source":"x_val = np.array([get_dcm_data(ID, cohort) for ID in val_ids], dtype=np.float32)\nend = time.time()\nprint(end - start)","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:39:12.461673Z","iopub.execute_input":"2021-09-18T16:39:12.46207Z","iopub.status.idle":"2021-09-18T16:49:40.021363Z","shell.execute_reply.started":"2021-09-18T16:39:12.461994Z","shell.execute_reply":"2021-09-18T16:49:40.020365Z"}}},{"cell_type":"markdown","source":"del train_ids, val_ids\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-09-18T16:49:40.022998Z","iopub.execute_input":"2021-09-18T16:49:40.023342Z","iopub.status.idle":"2021-09-18T16:49:40.213915Z","shell.execute_reply.started":"2021-09-18T16:49:40.023284Z","shell.execute_reply":"2021-09-18T16:49:40.213038Z"}}},{"cell_type":"code","source":"cohort = 'train'\n\nfrom sklearn.model_selection import train_test_split\n\nx = np.load(\"../input/rsna-btrc-xception2d-imagenet-data/x.npy\")\ny = np.load(\"../input/rsna-btrc-xception2d-imagenet-data/y.npy\")\n\nx_train, x_val, y_train, y_val = train_test_split(x, y, train_size = TRAIN_SIZE, random_state = 42, stratify=train_data[\"MGMT_value\"])\n","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:02.699118Z","iopub.execute_input":"2021-09-28T16:36:02.699957Z","iopub.status.idle":"2021-09-28T16:36:07.390045Z","shell.execute_reply.started":"2021-09-28T16:36:02.699923Z","shell.execute_reply":"2021-09-28T16:36:07.389287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use copy so that original array doesn't get augmented\ntrain_dataset = DataGenerator(x_train.copy(), y_train)\nval_dataset = DataGenerator(x_val, y_val, training = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:07.392011Z","iopub.execute_input":"2021-09-28T16:36:07.392537Z","iopub.status.idle":"2021-09-28T16:36:07.52378Z","shell.execute_reply.started":"2021-09-28T16:36:07.392501Z","shell.execute_reply":"2021-09-28T16:36:07.523017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"markdown","source":"Xception 2D model paper reference: https://arxiv.org/abs/1610.02357\n3D depthwise implementation source (with modification):\nhttps://github.com/danganea/keras-DepthwiseConv3D (MIT License, Copyright (c) 2019 Alexandros G. Stergiou)\nhttps://github.com/tensorflow/tensorflow/pull/31492#issuecomment-520063961 (loop variation)","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Input,Dense\nfrom tensorflow.keras.layers import ReLU as ReLU\nfrom tensorflow.keras.layers import BatchNormalization\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras import Model","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:07.525092Z","iopub.execute_input":"2021-09-28T16:36:07.525343Z","iopub.status.idle":"2021-09-28T16:36:07.529954Z","shell.execute_reply.started":"2021-09-28T16:36:07.525311Z","shell.execute_reply":"2021-09-28T16:36:07.529211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xception = tf.keras.applications.Xception(\n    include_top=False,\n    weights=None,\n    input_tensor=None,\n    input_shape=(256, 256, 3),\n    pooling='avg'\n)\nxception.load_weights('../input/xception-imagenet/xception_imagenet.h5')\nxception.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:07.531309Z","iopub.execute_input":"2021-09-28T16:36:07.531805Z","iopub.status.idle":"2021-09-28T16:36:11.870172Z","shell.execute_reply.started":"2021-09-28T16:36:07.53177Z","shell.execute_reply":"2021-09-28T16:36:11.869429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fcn_relu(X, drop_prob=0, initializer='he_normal', regularizer=None):\n    X = Dense(2048, activation='relu', kernel_initializer = initializer, kernel_regularizer=regularizer)(X)\n    #X = BatchNormalization()(X)\n    X = Dropout(drop_prob)(X)\n    \n    return X","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:11.871314Z","iopub.execute_input":"2021-09-28T16:36:11.871589Z","iopub.status.idle":"2021-09-28T16:36:11.878051Z","shell.execute_reply.started":"2021-09-28T16:36:11.871557Z","shell.execute_reply":"2021-09-28T16:36:11.877251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def exit_flow(X, drop_prob=0, fcn = False, initializer='he_normal'):\n    if fcn:\n        #X = fcn_relu(X, drop_prob=0.5, initializer=initializer, regularizer=tf.keras.regularizers.l2(l=0.01))\n        X = fcn_relu(X, drop_prob=0.5, initializer=initializer, regularizer=tf.keras.regularizers.l2(l=0.01))\n        X = fcn_relu(X, drop_prob=0.5, initializer=initializer)\n    \n    X = Dense(1, activation='sigmoid')(X)\n    \n    return X","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:11.879547Z","iopub.execute_input":"2021-09-28T16:36:11.879976Z","iopub.status.idle":"2021-09-28T16:36:11.888695Z","shell.execute_reply.started":"2021-09-28T16:36:11.87983Z","shell.execute_reply":"2021-09-28T16:36:11.88793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def xception_2D(input_size=(SIZE,SIZE,CHANNEL), drop_prob=0, fcn = False):\n    inputs = Input(input_size)\n    X = tf.keras.applications.xception.preprocess_input(inputs)\n    X = xception(X)\n    outputs = exit_flow(X, drop_prob, fcn)\n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:11.891797Z","iopub.execute_input":"2021-09-28T16:36:11.892053Z","iopub.status.idle":"2021-09-28T16:36:11.898336Z","shell.execute_reply.started":"2021-09-28T16:36:11.892022Z","shell.execute_reply":"2021-09-28T16:36:11.897243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compile Model","metadata":{}},{"cell_type":"code","source":"xception = xception_2D(input_size=(SIZE,SIZE,CHANNEL), drop_prob = 0.3, fcn=True)\nxception.summary()","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:11.899565Z","iopub.execute_input":"2021-09-28T16:36:11.899902Z","iopub.status.idle":"2021-09-28T16:36:12.266828Z","shell.execute_reply.started":"2021-09-28T16:36:11.899775Z","shell.execute_reply":"2021-09-28T16:36:12.266071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auc = tf.keras.metrics.AUC(\n    num_thresholds=200, curve='ROC',\n    summation_method='interpolation')","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:12.267939Z","iopub.execute_input":"2021-09-28T16:36:12.268172Z","iopub.status.idle":"2021-09-28T16:36:12.282386Z","shell.execute_reply.started":"2021-09-28T16:36:12.268141Z","shell.execute_reply":"2021-09-28T16:36:12.281609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Model","metadata":{}},{"cell_type":"code","source":"import tensorflow_addons as tfa","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:12.285107Z","iopub.execute_input":"2021-09-28T16:36:12.285283Z","iopub.status.idle":"2021-09-28T16:36:12.395256Z","shell.execute_reply.started":"2021-09-28T16:36:12.285263Z","shell.execute_reply":"2021-09-28T16:36:12.394611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile Model\nEPOCHS = 500 #set 1 epoch 1 sample\n\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    0.001, decay_steps=280, decay_rate=0.90, staircase=False)\n#reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='loss', factor=0.2,\n#                              patience=100, min_lr=0.00001)\n#lr_schedule = tfa.optimizers.CyclicalLearningRate(\n#    initial_learning_rate=0.00001,\n#    maximal_learning_rate=0.01,\n#    step_size=930,\n#    scale_fn=lambda x: 1/(2.**(x-1))\n#)\n\nxception.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule), \n                 loss='binary_crossentropy', \n                 metrics=[auc, 'accuracy'])\n\n# Fit Model\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    'checkpoint.h5',\n    save_weights_only=True,\n    monitor='val_loss',\n    mode='min',\n    save_best_only=True, verbose = 1)\nearly_stop = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss', patience=30, verbose=0,\n    mode='min', restore_best_weights=True)\n\ncallbacks = [model_checkpoint_callback, early_stop]\n\nhistory = xception.fit(train_dataset, validation_data=val_dataset, \n         epochs = EPOCHS, shuffle=True, callbacks=callbacks, workers=5)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T16:36:12.396498Z","iopub.execute_input":"2021-09-28T16:36:12.397017Z","iopub.status.idle":"2021-09-28T17:25:01.502175Z","shell.execute_reply.started":"2021-09-28T16:36:12.39698Z","shell.execute_reply":"2021-09-28T17:25:01.501454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_table = pd.DataFrame(history.history)\nhistory_table.to_csv('history.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-28T17:25:01.506478Z","iopub.execute_input":"2021-09-28T17:25:01.507658Z","iopub.status.idle":"2021-09-28T17:25:01.548855Z","shell.execute_reply.started":"2021-09-28T17:25:01.507621Z","shell.execute_reply":"2021-09-28T17:25:01.54823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nhistory_table.plot(figsize=(8, 5))\nplt.grid(True)\nplt.gca().set_ylim(0, 1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-28T17:25:01.549874Z","iopub.execute_input":"2021-09-28T17:25:01.550109Z","iopub.status.idle":"2021-09-28T17:25:02.714952Z","shell.execute_reply.started":"2021-09-28T17:25:01.550076Z","shell.execute_reply":"2021-09-28T17:25:02.714286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"markdown","source":"PATH = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\ncohort = 'test'\nmpMRIs = ['FLAIR','T2w']\ntest_ids = sorted(os.listdir(os.path.join(PATH,'test'))) \nx_test = np.array([get_dcm_data(ID, cohort) for ID in test_ids], dtype=np.float32)","metadata":{}},{"cell_type":"markdown","source":"pred = xception.predict(x_test).reshape((-1,)).tolist()","metadata":{}},{"cell_type":"markdown","source":"submission = pd.DataFrame({'BraTS21ID': test_ids, 'MGMT_value': pred})\nsubmission.head(10)","metadata":{}},{"cell_type":"markdown","source":"filename = 'submission.csv'\nsubmission.to_csv(filename,index=False)\nprint('Saved file: ' + filename)","metadata":{}}]}