{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport glob\nimport random\nimport collections\n\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nfrom tqdm.notebook import tqdm\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\n","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:08.719099Z","iopub.execute_input":"2022-06-15T15:12:08.719766Z","iopub.status.idle":"2022-06-15T15:12:08.725932Z","shell.execute_reply.started":"2022-06-15T15:12:08.719722Z","shell.execute_reply":"2022-06-15T15:12:08.725018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef Laplacian(data):\n    \n    if data is None:\n        return -1\n    \n    kernel = np.array(([1, 1, 1],\n                       [1, -8, 1],\n                      [1, 1, 1]), dtype=\"float32\")\n    kernel1 = np.array(([0, -1, 0],\n                       [-1, 5, -1],\n                        [0, -1, 0]), dtype=\"float32\")\n  \n    result = cv2.filter2D(data, -1, kernel)\n    result1 = cv2.filter2D(data, -1, kernel1)\n        \n    \n#     fig = plt.figure(figsize=(12,8))\n#     ax1 = plt.subplot(1,2,1)\n#     ax1.imshow(data, cmap=\"gray\")\n#     ax1.set_title(f\"Original image shape = {result.shape}\")\n#     ax2 = plt.subplot(1,2,2)\n#     ax2.imshow(result1, cmap=\"gray\")\n#     ax2.set_title(f\"Preproc image shape = {result1.shape}\")\n#     plt.show()\n    \n    return result1","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:08.772322Z","iopub.execute_input":"2022-06-15T15:12:08.772517Z","iopub.status.idle":"2022-06-15T15:12:08.778758Z","shell.execute_reply.started":"2022-06-15T15:12:08.772494Z","shell.execute_reply":"2022-06-15T15:12:08.777888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TYPES = [\"FLAIR\", \"T1w\", \"T2w\", \"T1wCE\"]\nWHITE_THRESHOLD = 10 # out of 255\nEXCLUDE = [109, 123, 709]\n\n\ntrain_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ntest_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\ntrain_df = train_df[~train_df.BraTS21ID.isin(EXCLUDE)]\n\ndef load_dicom(path, size = 224):\n    ''' \n    Reads a DICOM image, standardizes so that the pixel values are between 0 and 1, then rescales to 0 and 255\n    \n    Note super sure if this kind of scaling is appropriate, but everyone seems to do it. \n    '''\n    \n     # Load single image\n    img = pydicom.read_file(path).pixel_array\n    # Crop image\n    center_x, center_y = img.shape[1] / 2, img.shape[0] / 2\n    width_scaled, height_scaled = img.shape[1] * 0.8, img.shape[0] * 0.8\n    left_x, right_x = center_x - width_scaled / 2, center_x + width_scaled / 2\n    top_y, bottom_y = center_y - height_scaled / 2, center_y + height_scaled / 2\n    img = img[int(top_y):int(bottom_y), int(left_x):int(right_x)]\n    # Resize image\n    img = cv2.resize(img, (img_size, img_size))\n    \n    img = Laplacian(img)\n    return img\n\ndef get_all_image_paths(brats21id, image_type, folder='train'): \n    '''\n    Returns an arry of all the images of a particular type for a particular patient ID\n    '''\n    assert(image_type in TYPES)\n    \n    patient_path = os.path.join(\n        \"./%s/\" % folder, \n        str(brats21id).zfill(5),\n    )\n\n    paths = sorted(\n        glob.glob(os.path.join(patient_path, image_type, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    \n    num_images = len(paths)\n    \n    start = int(num_images * 0.25)\n    end = int(num_images * 0.75)\n\n    interval = 3\n    \n    if num_images < 10: \n        interval = 1\n    return np.array(paths[start:end:interval])\n\ndef get_all_images(brats21id, image_type, folder='train', size=225):\n    return [load_dicom(path, size) for path in get_all_image_paths(brats21id, image_type, folder)]\nIMAGE_SIZE = 128\n\ndef get_all_data_for_train(image_type):\n    global train_df\n    \n    X = []\n    y = []\n    train_ids = []\n\n    for i in tqdm(train_df.index):\n        x = train_df.loc[i]\n        images = get_all_images(int(x['BraTS21ID']), image_type, 'train', IMAGE_SIZE)\n        label = x['MGMT_value']\n\n        X += images\n        y += [label] * len(images)\n\n        train_ids += [int(x['BraTS21ID'])] * len(images)\n        assert(len(X) == len(y))\n    return np.array(X), np.array(y), np.array(train_ids)\n\ndef get_all_data_for_test(image_type):\n    global test_df\n    \n    X = []\n    test_ids = []\n\n    for i in tqdm(test_df.index):\n        x = test_df.loc[i]\n        images = get_all_images(int(x['BraTS21ID']), image_type, 'test', IMAGE_SIZE)\n        X += images\n        test_ids += [int(x['BraTS21ID'])] * len(images)\n\n    return np.array(X), np.array(test_ids)\nX, y, trainidt = get_all_data_for_train('FLAIR')\nX_test, testidt = get_all_data_for_test('FLAIR')","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:08.793796Z","iopub.execute_input":"2022-06-15T15:12:08.794008Z","iopub.status.idle":"2022-06-15T15:12:09.059565Z","shell.execute_reply.started":"2022-06-15T15:12:08.793986Z","shell.execute_reply":"2022-06-15T15:12:09.058841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid, trainidt_train, trainidt_valid = train_test_split(X, y, trainidt, test_size=0.2, random_state=40)\n\nsplit = int(X.shape[0] * 0.8)\n\n# X_train = X[:split]\n# X_valid = X[split:]\n\n# y_train = y[:split]\n# y_valid = y[split:]\n\n# trainidt_train = trainidt[:split]\n# trainidt_valid = trainidt[split:]\n\nX_train = tf.expand_dims(X_train, axis=-1)\nX_valid = tf.expand_dims(X_valid, axis=-1)\n\ny_train = to_categorical(y_train)\ny_valid = to_categorical(y_valid)\n\nX_train.shape, y_train.shape, X_valid.shape, y_valid.shape, trainidt_train.shape, trainidt_valid.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:09.061289Z","iopub.execute_input":"2022-06-15T15:12:09.061674Z","iopub.status.idle":"2022-06-15T15:12:09.091188Z","shell.execute_reply.started":"2022-06-15T15:12:09.06164Z","shell.execute_reply":"2022-06-15T15:12:09.089826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_average_layer = tf.keras.layers.GlobalAveragePooling2D()\ndata_augmentation = tf.keras.Sequential([\n  tf.keras.layers.experimental.preprocessing.RandomFlip('horizontal'),\n  tf.keras.layers.experimental.preprocessing.RandomRotation(0.2),\n])","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:09.092401Z","iopub.status.idle":"2022-06-15T15:12:09.093143Z","shell.execute_reply.started":"2022-06-15T15:12:09.092894Z","shell.execute_reply":"2022-06-15T15:12:09.092935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(0)\nrandom.seed(12)\ntf.random.set_seed(12)\n\ninpt = keras.Input(shape=(128, 128, 1))\n\nh = keras.layers.experimental.preprocessing.Rescaling(1./255)(inpt)\n# h = data_augmentation(h)\n\n# convolutional layer!\nh = keras.layers.Conv2D(32, kernel_size=(3, 3),strides=(1,1), activation=\"relu\", name=\"Conv_1\", padding=\"valid\")(h) \nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\nh = keras.layers.Conv2D(64, kernel_size=(3, 3),strides=(1,1), activation=\"relu\", name=\"Conv_2\", padding=\"same\")(h) \nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\n# pooling layer\nh = keras.layers.MaxPool2D(pool_size=(2,2))(h) \nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\n# convolutional layer!\nh = keras.layers.Conv2D(64, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_3\",padding =\"same\")(h)\n# h = tf.keras.layers.BatchNormalization(axis=-1)(h)\n# pooling layer\n# h = keras.layers.MaxPool2D(pool_size=(1,1))(h)\nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\nh = keras.layers.Conv2D(128, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_4\",padding =\"valid\")(h)\nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\nh = keras.layers.Conv2D(128, kernel_size=(3, 3), activation=\"relu\", name=\"Conv_5\",padding =\"same\")(h)\nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\nh = keras.layers.MaxPool2D(pool_size=(2,2))(h)\nh = tf.keras.layers.BatchNormalization(axis=-1)(h)\nh = keras.layers.Dropout(0.3)(h)   \n\nh = keras.layers.Flatten()(h) \n# h = global_average_layer(h)\nh = keras.layers.Dense(128, activation='relu')(h)   \n\noutput = keras.layers.Dense(2, activation=\"softmax\")(h)\n\nmodel = keras.Model(inpt, output)\n\nfrom tensorflow.keras.optimizers import SGD\n# opt = SGD(lr=0.1)\n\ncheckpoint_filepath = 'best_model.h5'\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\nfilepath=checkpoint_filepath,\nsave_weights_only=False,\nmonitor='val_auc',\nmode='max',\nsave_best_only=True,\nsave_freq='epoch')\n\nmodel.compile(loss='categorical_crossentropy',\n             optimizer=tf.keras.optimizers.SGD(learning_rate =0.0001),\n             metrics=[tf.keras.metrics.AUC()])\n\nhistory = model.fit(x=X_train, y = y_train, epochs=50, callbacks=[model_checkpoint_callback], validation_data= (X_valid, y_valid))","metadata":{"execution":{"iopub.status.busy":"2022-06-15T15:12:09.094693Z","iopub.status.idle":"2022-06-15T15:12:09.095114Z","shell.execute_reply.started":"2022-06-15T15:12:09.094883Z","shell.execute_reply":"2022-06-15T15:12:09.094917Z"},"trusted":true},"execution_count":null,"outputs":[]}]}