{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I'm reading through several existing notebooks and trying to distill down the information into a new notebook to help me understand the project.  All help appreciated!\n\n# References\n\n- [Advanced EDA - Brain Tumor Data](https://www.kaggle.com/smoschou55/advanced-eda-brain-tumor-data)\n- [Team 9 Second Week](https://www.kaggle.com/evanyao27/team-9-second-week)\n  - The only model that is working. get_model02()\n- [Dataset to Model with Tensorflow](https://www.kaggle.com/ohbewise/dataset-to-model-with-tensorflow)\n- [Brain Tumer Train Class Flair](https://www.kaggle.com/lucamtb/brain-tumer-train-class-flair)\n  - Uses TPU\n  - Generates a Tensorflow model: Brain_flair_model_effect_3e-05_0.0001.h5\n- [Brain Tumor very basic inference](https://www.kaggle.com/lucamtb/brain-tumor-very-basice-inference)\n  - Uses the above mentioned model: Brain_flair_model_effect_3e-05_0.0001.h5\n  - Add this Kaggle Dataset: https://www.kaggle.com/lucamtb/effect0-brain","metadata":{}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\n\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\nimport random\nfrom tqdm.notebook import tqdm\nimport pydicom # Handle MRI images\n\nimport cv2  # OpenCV - https://docs.opencv.org/master/d6/d00/tutorial_py_root.html\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\n\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\n","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:08.707103Z","iopub.execute_input":"2021-10-05T02:14:08.707514Z","iopub.status.idle":"2021-10-05T02:14:10.690317Z","shell.execute_reply.started":"2021-10-05T02:14:08.707454Z","shell.execute_reply":"2021-10-05T02:14:10.689407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration, Constants, Setup","metadata":{}},{"cell_type":"code","source":"data_dir = Path('../input/rsna-miccai-brain-tumor-radiogenomic-classification/')\n\nmri_types = [\"FLAIR\", \"T1w\", \"T2w\", \"T1wCE\"]\nexcluded_images = [109, 123, 709] # Bad images","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.695274Z","iopub.execute_input":"2021-10-05T02:14:10.695562Z","iopub.status.idle":"2021-10-05T02:14:10.700332Z","shell.execute_reply.started":"2021-10-05T02:14:10.695527Z","shell.execute_reply":"2021-10-05T02:14:10.699529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Datasets","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(data_dir / \"train_labels.csv\",\n#                        index='id',\n#                       nrows=100000\n                      )\ntest_df = pd.read_csv(data_dir / \"sample_submission.csv\")\nsample_submission = pd.read_csv(data_dir / \"sample_submission.csv\")\n\ntrain_df = train_df[~train_df.BraTS21ID.isin(excluded_images)]\n\nprint(f\"train data: Rows={train_df.shape[0]}, Columns={train_df.shape[1]}\")\n# print(f\"test data : Rows={test_df.shape[0]}, Columns={test_df.shape[1]}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.701598Z","iopub.execute_input":"2021-10-05T02:14:10.702035Z","iopub.status.idle":"2021-10-05T02:14:10.723264Z","shell.execute_reply.started":"2021-10-05T02:14:10.701993Z","shell.execute_reply":"2021-10-05T02:14:10.722518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"markdown","source":"### There's a version that converts into grayscale: \n\n- https://www.kaggle.com/smoschou55/advanced-eda-brain-tumor-data\n","metadata":{}},{"cell_type":"code","source":"def load_dicom(path, size = 224):\n    ''' \n    Reads a DICOM image, standardizes so that the pixel values are between 0 and 1, then rescales to 0 and 255\n    \n    Not super sure if this kind of scaling is appropriate, but everyone seems to do it. \n    '''\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    # transform data into black and white scale / grayscale\n#     data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return cv2.resize(data, (size, size))","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.725401Z","iopub.execute_input":"2021-10-05T02:14:10.725873Z","iopub.status.idle":"2021-10-05T02:14:10.733359Z","shell.execute_reply.started":"2021-10-05T02:14:10.725836Z","shell.execute_reply":"2021-10-05T02:14:10.732631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_image_paths(brats21id, image_type, folder='train'): \n    '''\n    Returns an arry of all the images of a particular type for a particular patient ID\n    '''\n    assert(image_type in mri_types)\n    \n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/%s/\" % folder, \n        str(brats21id).zfill(5),\n    )\n\n    paths = sorted(\n        glob.glob(os.path.join(patient_path, image_type, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    \n    num_images = len(paths)\n    \n    start = int(num_images * 0.25)\n    end = int(num_images * 0.75)\n\n    interval = 3\n    \n    if num_images < 10: \n        interval = 1\n    \n    return np.array(paths[start:end:interval])\n\ndef get_all_images(brats21id, image_type, folder='train', size=225):\n    return [load_dicom(path, size) for path in get_all_image_paths(brats21id, image_type, folder)]","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.734653Z","iopub.execute_input":"2021-10-05T02:14:10.735074Z","iopub.status.idle":"2021-10-05T02:14:10.74455Z","shell.execute_reply.started":"2021-10-05T02:14:10.735038Z","shell.execute_reply":"2021-10-05T02:14:10.743612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Images We Will Need","metadata":{}},{"cell_type":"code","source":"def get_all_data_for_train(image_type, image_size=32):\n    global train_df\n    \n    X = []\n    y = []\n    train_ids = []\n\n    for i in tqdm(train_df.index):\n        x = train_df.loc[i]\n        images = get_all_images(int(x['BraTS21ID']), image_type, 'train', image_size)\n        label = x['MGMT_value']\n\n        X += images\n        y += [label] * len(images)\n        train_ids += [int(x['BraTS21ID'])] * len(images)\n        assert(len(X) == len(y))\n    return np.array(X), np.array(y), np.array(train_ids)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.745696Z","iopub.execute_input":"2021-10-05T02:14:10.746615Z","iopub.status.idle":"2021-10-05T02:14:10.754761Z","shell.execute_reply.started":"2021-10-05T02:14:10.74658Z","shell.execute_reply":"2021-10-05T02:14:10.754094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_data_for_test(image_type, image_size=32):\n    global test_df\n    \n    X = []\n    test_ids = []\n\n    for i in tqdm(test_df.index):\n        x = test_df.loc[i]\n        images = get_all_images(int(x['BraTS21ID']), image_type, 'test', image_size)\n        X += images\n        test_ids += [int(x['BraTS21ID'])] * len(images)\n\n    return np.array(X), np.array(test_ids)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.757007Z","iopub.execute_input":"2021-10-05T02:14:10.75793Z","iopub.status.idle":"2021-10-05T02:14:10.767287Z","shell.execute_reply.started":"2021-10-05T02:14:10.757827Z","shell.execute_reply":"2021-10-05T02:14:10.766471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y, trainidt = get_all_data_for_train('T1wCE', image_size=32)\nX_test, testidt = get_all_data_for_test('T1wCE', image_size=32)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:14:10.768846Z","iopub.execute_input":"2021-10-05T02:14:10.769353Z","iopub.status.idle":"2021-10-05T02:15:15.501478Z","shell.execute_reply.started":"2021-10-05T02:14:10.769295Z","shell.execute_reply":"2021-10-05T02:15:15.500643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape, trainidt.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:15.502741Z","iopub.execute_input":"2021-10-05T02:15:15.503088Z","iopub.status.idle":"2021-10-05T02:15:15.510695Z","shell.execute_reply.started":"2021-10-05T02:15:15.503049Z","shell.execute_reply":"2021-10-05T02:15:15.509939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train/Validation Split","metadata":{}},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid, trainidt_train, trainidt_valid = train_test_split(X, y, trainidt, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:15.511914Z","iopub.execute_input":"2021-10-05T02:15:15.512857Z","iopub.status.idle":"2021-10-05T02:15:15.529773Z","shell.execute_reply.started":"2021-10-05T02:15:15.512819Z","shell.execute_reply":"2021-10-05T02:15:15.528934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Adding a Dimension","metadata":{}},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:15.530848Z","iopub.execute_input":"2021-10-05T02:15:15.531634Z","iopub.status.idle":"2021-10-05T02:15:15.537917Z","shell.execute_reply.started":"2021-10-05T02:15:15.531595Z","shell.execute_reply":"2021-10-05T02:15:15.537119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = tf.expand_dims(X_train, axis=-1)\nX_valid = tf.expand_dims(X_valid, axis=-1)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:15.539335Z","iopub.execute_input":"2021-10-05T02:15:15.539609Z","iopub.status.idle":"2021-10-05T02:15:16.476831Z","shell.execute_reply.started":"2021-10-05T02:15:15.539575Z","shell.execute_reply":"2021-10-05T02:15:16.476092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.481236Z","iopub.execute_input":"2021-10-05T02:15:16.481842Z","iopub.status.idle":"2021-10-05T02:15:16.48964Z","shell.execute_reply.started":"2021-10-05T02:15:16.481808Z","shell.execute_reply":"2021-10-05T02:15:16.488866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One-hot encode labels","metadata":{}},{"cell_type":"code","source":"y_train = to_categorical(y_train)\ny_valid = to_categorical(y_valid)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.491067Z","iopub.execute_input":"2021-10-05T02:15:16.491404Z","iopub.status.idle":"2021-10-05T02:15:16.497435Z","shell.execute_reply.started":"2021-10-05T02:15:16.491366Z","shell.execute_reply":"2021-10-05T02:15:16.496597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Models","metadata":{}},{"cell_type":"markdown","source":"## Model 1\n- from:  https://www.kaggle.com/ohbewise/dataset-to-model-with-tensorflow\n- Keras code from here: https://keras.io/examples/vision/3D_image_classification/","metadata":{}},{"cell_type":"markdown","source":"## Define a 3D convolutional neural network","metadata":{}},{"cell_type":"code","source":"# Define, train, and evaluate model\n# source: https://keras.io/examples/vision/3D_image_classification/\ndef get_model01(width=128, height=128, depth=64, name='3dcnn'):\n    \"\"\"Build a 3D convolutional neural network model.\"\"\"\n\n    inputs = tf.keras.Input((width, height, depth, 1))\n\n    x = tf.keras.layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(inputs)\n    x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    x = tf.keras.layers.Conv3D(filters=64, kernel_size=3, activation=\"relu\")(x)\n    x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    x = tf.keras.layers.Conv3D(filters=128, kernel_size=3, activation=\"relu\")(x)\n    x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    x = tf.keras.layers.Conv3D(filters=256, kernel_size=3, activation=\"relu\")(x)\n    x = tf.keras.layers.MaxPool3D(pool_size=2)(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    x = tf.keras.layers.GlobalAveragePooling3D()(x)\n    x = tf.keras.layers.Dense(units=512, activation=\"relu\")(x)\n    x = tf.keras.layers.Dropout(0.3)(x)\n\n    outputs = tf.keras.layers.Dense(units=1, activation=\"sigmoid\")(x)\n\n    # Define the model.\n    model = tf.keras.Model(inputs, outputs, name=name)\n    \n    # Compile model.\n    initial_learning_rate = 0.0001\n    lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n        initial_learning_rate, decay_steps=100000, decay_rate=0.96, staircase=True\n    )\n    model.compile(\n        loss=\"binary_crossentropy\",\n        optimizer=tf.keras.optimizers.Adam(learning_rate=lr_schedule),\n        metrics=[\"acc\"],\n    )\n    \n    return model\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.49909Z","iopub.execute_input":"2021-10-05T02:15:16.499379Z","iopub.status.idle":"2021-10-05T02:15:16.512933Z","shell.execute_reply.started":"2021-10-05T02:15:16.499347Z","shell.execute_reply":"2021-10-05T02:15:16.512023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 2\n\n- from: https://www.kaggle.com/evanyao27/team-9-second-week/notebook\n- Validation AUC=0.9148664856146349","metadata":{}},{"cell_type":"code","source":"def get_model02():\n    np.random.seed(0)\n    random.seed(12)\n    tf.random.set_seed(12)\n\n    inpt = keras.Input(shape=X_train.shape[1:])\n\n    h = keras.layers.experimental.preprocessing.Rescaling(1.0 / 255)(inpt)\n\n    h = keras.layers.Conv2D(64, kernel_size=(4, 4), activation=\"relu\", name=\"Conv_1\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(2, 2))(h)\n\n    h = keras.layers.Conv2D(32, kernel_size=(2, 2), activation=\"relu\", name=\"Conv_2\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(1, 1))(h)\n\n    h = keras.layers.Dropout(0.1)(h)\n\n    h = keras.layers.Flatten()(h)\n    h = keras.layers.Dense(32, activation=\"relu\")(h)\n\n    output = keras.layers.Dense(2, activation=\"softmax\")(h)\n\n    model = keras.Model(inpt, output)\n\n    roc_auc = tf.keras.metrics.AUC(name='roc_auc', curve='ROC')\n\n    model.compile(\n        loss=\"categorical_crossentropy\", optimizer=\"adam\", metrics=[roc_auc]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.514053Z","iopub.execute_input":"2021-10-05T02:15:16.51476Z","iopub.status.idle":"2021-10-05T02:15:16.526255Z","shell.execute_reply.started":"2021-10-05T02:15:16.514726Z","shell.execute_reply":"2021-10-05T02:15:16.525447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model 3\n\nAdding LR scheduler, early stopping, etc","metadata":{}},{"cell_type":"code","source":"def get_model03():\n    np.random.seed(0)\n    random.seed(12)\n    tf.random.set_seed(12)\n\n    inpt = keras.Input(shape=X_train.shape[1:])\n\n    h = keras.layers.experimental.preprocessing.Rescaling(1.0 / 255)(inpt)\n\n    h = keras.layers.Conv2D(64, kernel_size=(4, 4), activation=\"relu\", name=\"Conv_1\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(2, 2))(h)\n\n    h = keras.layers.Conv2D(32, kernel_size=(2, 2), activation=\"relu\", name=\"Conv_2\")(h)\n    h = keras.layers.MaxPool2D(pool_size=(1, 1))(h)\n\n    h = keras.layers.Dropout(0.1)(h)\n\n    h = keras.layers.Flatten()(h)\n    h = keras.layers.Dense(32, activation=\"relu\")(h)\n\n    output = keras.layers.Dense(2, activation=\"softmax\")(h)\n\n    model = keras.Model(inpt, output)\n\n    # https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/schedules/ExponentialDecay\n    \n    initial_learning_rate =  0.0001\n    lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n        initial_learning_rate,\n        decay_steps=100000,\n        decay_rate=0.96, \n        staircase=True\n    )\n  \n    roc_auc = tf.keras.metrics.AUC(name='roc_auc', curve='ROC')\n\n    model.compile(\n        loss=\"categorical_crossentropy\", \n#         loss=\"binary_crossentropy\", \n        \n#         optimizer=keras.optimizers.Adam(learning_rate=lr_schedule),\n        optimizer=keras.optimizers.Adam(),\n\n        metrics=[roc_auc],\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.527318Z","iopub.execute_input":"2021-10-05T02:15:16.527641Z","iopub.status.idle":"2021-10-05T02:15:16.539953Z","shell.execute_reply.started":"2021-10-05T02:15:16.527604Z","shell.execute_reply":"2021-10-05T02:15:16.539199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set up Model Checkpoint","metadata":{}},{"cell_type":"code","source":"checkpoint_filepath = \"best_model.h5\"\n\nmodel_checkpoint_cb = tf.keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=False,\n    monitor=\"val_roc_auc\",\n    mode=\"max\",\n    save_best_only=True,\n    save_freq=\"epoch\",\n    verbose=1,\n)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.541301Z","iopub.execute_input":"2021-10-05T02:15:16.54164Z","iopub.status.idle":"2021-10-05T02:15:16.551716Z","shell.execute_reply.started":"2021-10-05T02:15:16.541605Z","shell.execute_reply":"2021-10-05T02:15:16.550945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Early Stopping Callback\n\n- https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/EarlyStopping","metadata":{}},{"cell_type":"code","source":"# early_stopping_cb = tf.keras.callbacks.EarlyStopping(monitor=tf.keras.metrics.AUC(), mode='auto', verbose=1, patience=5)\n# early_stopping_cb = tf.keras.callbacks.EarlyStopping(monitor=\"val_acc\", patience=15)\nearly_stopping_cb = tf.keras.callbacks.EarlyStopping(monitor=\"val_roc_auc\", mode='max', patience=3)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.553794Z","iopub.execute_input":"2021-10-05T02:15:16.554374Z","iopub.status.idle":"2021-10-05T02:15:16.56139Z","shell.execute_reply.started":"2021-10-05T02:15:16.554338Z","shell.execute_reply":"2021-10-05T02:15:16.560679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Note that rerunning the cell below will change val_acc to val_acc_N and the model will not be saved.\n\nForce name:  https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/discussion/276230\n\n    roc_auc = tf.keras.metrics.AUC(name='roc_auc', curve='ROC')\n    model.compile(optimizer=..., loss=..., metrics=[roc_auc, ...])\n","metadata":{}},{"cell_type":"code","source":"# model = get_model02() # LB score 0.676\nmodel = get_model03() # LB score 0.5\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.562798Z","iopub.execute_input":"2021-10-05T02:15:16.563087Z","iopub.status.idle":"2021-10-05T02:15:16.645733Z","shell.execute_reply.started":"2021-10-05T02:15:16.563055Z","shell.execute_reply":"2021-10-05T02:15:16.644496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = model.fit(x=X_train, y = y_train, epochs=20, \n#                     callbacks=[model_checkpoint_cb], \n#                     validation_data=(X_valid, y_valid))\n\nhistory = model.fit(x=X_train, y = y_train, epochs=40, \n                    callbacks=[model_checkpoint_cb, early_stopping_cb],\n                    validation_data=(X_valid, y_valid))","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:15:16.64705Z","iopub.execute_input":"2021-10-05T02:15:16.64736Z","iopub.status.idle":"2021-10-05T02:16:00.214433Z","shell.execute_reply.started":"2021-10-05T02:15:16.647327Z","shell.execute_reply":"2021-10-05T02:16:00.213606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Our Best Model","metadata":{}},{"cell_type":"code","source":"model_best = tf.keras.models.load_model(filepath=checkpoint_filepath)","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:16:00.2176Z","iopub.execute_input":"2021-10-05T02:16:00.217847Z","iopub.status.idle":"2021-10-05T02:16:00.315394Z","shell.execute_reply.started":"2021-10-05T02:16:00.217818Z","shell.execute_reply":"2021-10-05T02:16:00.314555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions on Validation Set","metadata":{}},{"cell_type":"code","source":"y_pred = model_best.predict(X_valid)\n\npred = np.argmax(y_pred, axis=1)\n\nresult = pd.DataFrame(trainidt_valid)\nresult[1] = pred\n\nresult.columns = [\"BraTS21ID\", \"MGMT_value\"]\nresult2 = result.groupby(\"BraTS21ID\", as_index=False).mean()\n\nresult2 = result2.merge(train_df, on=\"BraTS21ID\")\nauc = roc_auc_score(\n    result2.MGMT_value_y,\n    result2.MGMT_value_x,\n)\nprint(f\"Validation AUC={auc}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:16:00.316524Z","iopub.execute_input":"2021-10-05T02:16:00.31732Z","iopub.status.idle":"2021-10-05T02:16:00.530986Z","shell.execute_reply.started":"2021-10-05T02:16:00.317275Z","shell.execute_reply":"2021-10-05T02:16:00.530206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions on the Test Set","metadata":{}},{"cell_type":"code","source":"y_pred = model_best.predict(X_test)\n\npred = np.argmax(y_pred, axis=1) #\n\nresult = pd.DataFrame(testidt)\nresult[1] = pred\npred","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:16:00.532389Z","iopub.execute_input":"2021-10-05T02:16:00.532654Z","iopub.status.idle":"2021-10-05T02:16:00.730614Z","shell.execute_reply.started":"2021-10-05T02:16:00.532619Z","shell.execute_reply":"2021-10-05T02:16:00.729872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission File","metadata":{}},{"cell_type":"code","source":"result.columns=['BraTS21ID','MGMT_value']\n\nresult2 = result.groupby('BraTS21ID',as_index=False).mean()\nresult2['BraTS21ID'] = sample_submission['BraTS21ID']\n\n# Rounding... 0.907866 -> 0.9\nresult2['MGMT_value'] = result2['MGMT_value'].apply(lambda x:round(x*10)/10)\n# result2['MGMT_value'] = result2['MGMT_value'] # No rounding\nresult2.to_csv('submission.csv',index=False)\nresult2","metadata":{"execution":{"iopub.status.busy":"2021-10-05T02:16:00.731783Z","iopub.execute_input":"2021-10-05T02:16:00.732041Z","iopub.status.idle":"2021-10-05T02:16:00.75923Z","shell.execute_reply.started":"2021-10-05T02:16:00.732005Z","shell.execute_reply":"2021-10-05T02:16:00.758379Z"},"trusted":true},"execution_count":null,"outputs":[]}]}