{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport random\n\nimport pydicom #DICOM file\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nimport cv2 #openCV\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm #progress bar\n\nimport glob #glob\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.optimizers import SGD","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-08T04:06:39.489457Z","iopub.execute_input":"2021-10-08T04:06:39.490160Z","iopub.status.idle":"2021-10-08T04:06:44.609503Z","shell.execute_reply.started":"2021-10-08T04:06:39.490042Z","shell.execute_reply":"2021-10-08T04:06:44.608712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ntest_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.611032Z","iopub.execute_input":"2021-10-08T04:06:44.611328Z","iopub.status.idle":"2021-10-08T04:06:44.634805Z","shell.execute_reply.started":"2021-10-08T04:06:44.611291Z","shell.execute_reply":"2021-10-08T04:06:44.634220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this [discussion](https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/discussion/262046) a competition host has notified that there are some issues with these 3 cases   \nPatient IDs - \n\n1. 00109 (FLAIR images are blank)\n2. 00123 (T1w images are blank)\n3. 00709 (FLAIR images are blank)    \n\nHence these can be excluded\n","metadata":{}},{"cell_type":"code","source":"#refer: https://www.kaggle.com/arnabs007/part-1-rsna-miccai-btrc-understanding-the-data\nEXCLUDE = [109, 123, 709]\ntrain_df = train_df[~train_df.BraTS21ID.isin(EXCLUDE)]","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.635856Z","iopub.execute_input":"2021-10-08T04:06:44.636095Z","iopub.status.idle":"2021-10-08T04:06:44.648655Z","shell.execute_reply.started":"2021-10-08T04:06:44.636061Z","shell.execute_reply":"2021-10-08T04:06:44.647899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.650559Z","iopub.execute_input":"2021-10-08T04:06:44.651134Z","iopub.status.idle":"2021-10-08T04:06:44.666912Z","shell.execute_reply.started":"2021-10-08T04:06:44.651100Z","shell.execute_reply":"2021-10-08T04:06:44.666334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.669448Z","iopub.execute_input":"2021-10-08T04:06:44.669623Z","iopub.status.idle":"2021-10-08T04:06:44.679085Z","shell.execute_reply.started":"2021-10-08T04:06:44.669602Z","shell.execute_reply":"2021-10-08T04:06:44.677560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TYPES = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"] #mpMRI scans","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.680578Z","iopub.execute_input":"2021-10-08T04:06:44.680874Z","iopub.status.idle":"2021-10-08T04:06:44.685938Z","shell.execute_reply.started":"2021-10-08T04:06:44.680840Z","shell.execute_reply":"2021-10-08T04:06:44.685121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path, size = 224): #load DICOM files\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array #returns a numpy.ndarray containing the pixel data\n    if np.max(data) != 0:\n        data = data / np.max(data) #standardizes so that the pixel values are between 0 and 1\n    data = (data * 255).astype(np.uint8) #rescales to 0 and 255\n    return cv2.resize(data, (size, size))","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.687384Z","iopub.execute_input":"2021-10-08T04:06:44.688036Z","iopub.status.idle":"2021-10-08T04:06:44.695597Z","shell.execute_reply.started":"2021-10-08T04:06:44.688000Z","shell.execute_reply":"2021-10-08T04:06:44.694818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_image_paths(BraTS21ID, image_type, folder=\"train\"): #get an array of all the images of a particular type or a particular patient id\n    assert(image_type in TYPES) #only in types\n    patient_path = os.path.join(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/%s/\" % folder, str(BraTS21ID).zfill(5)) #다른 폴더일 수도 있음\n    #print(lambda x: int(x[:-4].split(\"-\")[-1]))\n    \n    paths = sorted(glob.glob(os.path.join(patient_path, image_type, \"*\")), key=lambda x: int(x[:-4].split(\"-\")[-1])) #sort\n    #print(paths)\n    \n    num_images = len(paths)\n    \n    start = int(num_images * 0.25)\n    end = int(num_images * 0.75)\n    if num_images < 10:\n        jump = 1\n    else:\n        jump = 3\n        \n    return np.array(paths[start:end:jump])","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.697022Z","iopub.execute_input":"2021-10-08T04:06:44.697463Z","iopub.status.idle":"2021-10-08T04:06:44.706120Z","shell.execute_reply.started":"2021-10-08T04:06:44.697428Z","shell.execute_reply":"2021-10-08T04:06:44.705381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_all_images(BraTS21ID, image_type, folder=\"train\", size=225):\n    return [load_dicom(path, size) for path in get_all_image_paths(BraTS21ID, image_type, folder)]","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:06:44.707523Z","iopub.execute_input":"2021-10-08T04:06:44.707784Z","iopub.status.idle":"2021-10-08T04:06:44.715327Z","shell.execute_reply.started":"2021-10-08T04:06:44.707752Z","shell.execute_reply":"2021-10-08T04:06:44.714634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 128\n\ndef get_all_data_train(image_type):\n    global train_df\n    \n    X = []\n    y = []\n    train_ids = []\n\n    for i in tqdm(train_df.index):\n        tmp_x = train_df.loc[i]\n        images = get_all_images(int(tmp_x[\"BraTS21ID\"]), image_type, \"train\", IMAGE_SIZE)\n        label = tmp_x[\"MGMT_value\"]\n\n        X += images\n        y += [label] * len(images)\n        train_ids += [int(tmp_x[\"BraTS21ID\"])] * len(images)\n        assert(len(X) == len(y))\n    return np.array(X), np.array(y), np.array(train_ids)\n\ndef get_all_data_test(image_type):\n    global test_df\n    \n    X = []\n    test_ids = []\n\n    for i in tqdm(test_df.index):\n        tmp_x = test_df.loc[i]\n        images = get_all_images(int(tmp_x[\"BraTS21ID\"]), image_type, \"test\", IMAGE_SIZE)\n        X += images\n        test_ids += [int(tmp_x[\"BraTS21ID\"])] * len(images)\n\n    return np.array(X), np.array(test_ids)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:14:00.399357Z","iopub.execute_input":"2021-10-08T04:14:00.400037Z","iopub.status.idle":"2021-10-08T04:14:00.408342Z","shell.execute_reply.started":"2021-10-08T04:14:00.400000Z","shell.execute_reply":"2021-10-08T04:14:00.407643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y, train_idt = get_all_data_train(\"T1wCE\")\nX_test, test_idt = get_all_data_test(\"T1wCE\")\nX.shape, y.shape, train_idt.shape","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-10-08T04:14:03.456816Z","iopub.execute_input":"2021-10-08T04:14:03.457068Z","iopub.status.idle":"2021-10-08T04:14:51.953412Z","shell.execute_reply.started":"2021-10-08T04:14:03.457040Z","shell.execute_reply":"2021-10-08T04:14:51.952522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:14:51.955014Z","iopub.execute_input":"2021-10-08T04:14:51.955449Z","iopub.status.idle":"2021-10-08T04:14:51.960836Z","shell.execute_reply.started":"2021-10-08T04:14:51.955410Z","shell.execute_reply":"2021-10-08T04:14:51.960145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid, train_idt_train, train_idt_valid = train_test_split(X, y, train_idt, test_size=0.2, random_state=42)\n\nsplit = int(X.shape[0] * 0.8) #8:2 split\n\nX_train = tf.expand_dims(X_train, axis=-1) #expand the dimension at the end of the array\nX_valid = tf.expand_dims(X_valid, axis=-1)\n\ny_train = to_categorical(y_train) #one-hot incoding\ny_valid = to_categorical(y_valid)\n\nX_train.shape, y_train.shape, X_valid.shape, y_valid.shape, train_idt_train.shape, train_idt_valid.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:27:25.121288Z","iopub.execute_input":"2021-10-08T04:27:25.121861Z","iopub.status.idle":"2021-10-08T04:27:25.474212Z","shell.execute_reply.started":"2021-10-08T04:27:25.121823Z","shell.execute_reply":"2021-10-08T04:27:25.473560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path1 = \"../input/rsna-model-2/rsna_model_data_augment_best_model_3.h5\" #shape=128\n#file_path2 = \"../input/rsna-best-model-training2/best_model (2).h5\"\nfile_path2 = \"../input/rsna-model-model-1/rsna_model_data_augment_best_model_2.h5\" #shape=128\nfile_path3 = \"../input/best-model-trainingver3/best_model_trainingVer3.h5\" #shape=32","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:09:08.729890Z","iopub.execute_input":"2021-10-08T04:09:08.730540Z","iopub.status.idle":"2021-10-08T04:09:08.738444Z","shell.execute_reply.started":"2021-10-08T04:09:08.730505Z","shell.execute_reply":"2021-10-08T04:09:08.737775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import tensorflow_hub as tfhub\n#import tensorflow_addons as tfa\nfrom tensorflow.keras import layers","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:13:02.401751Z","iopub.execute_input":"2021-10-08T04:13:02.402029Z","iopub.status.idle":"2021-10-08T04:13:02.406395Z","shell.execute_reply.started":"2021-10-08T04:13:02.401999Z","shell.execute_reply":"2021-10-08T04:13:02.405402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#best_model = tf.keras.models.load_model(filepath = file_path, custom_objects={'KerasLayer': tfhub.KerasLayer})\nmodel1 = tf.keras.models.load_model(filepath = file_path1)\nmodel2 = tf.keras.models.load_model(filepath = file_path2)\n#model3 = tf.keras.models.load_model(filepath = file_path3)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:22:59.011977Z","iopub.execute_input":"2021-10-08T04:22:59.012257Z","iopub.status.idle":"2021-10-08T04:23:03.998983Z","shell.execute_reply.started":"2021-10-08T04:22:59.012228Z","shell.execute_reply":"2021-10-08T04:23:03.998215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred1 = model1.predict(X_valid) #pedict on X_valid\ny_pred2 = model2.predict(X_valid) #pedict on X_valid\n\npred1 = np.argmax(y_pred1, axis = 1)\npred2 = np.argmax(y_pred2, axis = 1)\n\nresult = pd.DataFrame(train_idt_valid)\nresult[1] = pred1*0.3+pred2*0.7\nresult.columns=[\"BraTS21ID\",\"MGMT_value\"]\n\n#Group by BraTS21ID and average + do not use index\nresult_temp = result.groupby(\"BraTS21ID\", as_index = False).mean()\nresult_temp = result_temp.merge(train_df, on = \"BraTS21ID\") #merge train_df\nresult_temp","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:31.914700Z","iopub.execute_input":"2021-10-08T04:28:31.915523Z","iopub.status.idle":"2021-10-08T04:28:34.499896Z","shell.execute_reply.started":"2021-10-08T04:28:31.915484Z","shell.execute_reply":"2021-10-08T04:28:34.499252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"auc = roc_auc_score(result_temp.MGMT_value_y, result_temp.MGMT_value_x)\n\nprint(f\"Validation AUC={auc}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:39.887022Z","iopub.execute_input":"2021-10-08T04:28:39.887298Z","iopub.status.idle":"2021-10-08T04:28:39.898921Z","shell.execute_reply.started":"2021-10-08T04:28:39.887270Z","shell.execute_reply":"2021-10-08T04:28:39.897930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission\nsample_sub = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\")\n\ny_pred1 = model1.predict(X_test) #predict test\ny_pred2 = model2.predict(X_test)\n#y_pred3 = model3.predict(X_test)\n\npred1 = np.argmax(y_pred1, axis = 1)\npred2 = np.argmax(y_pred2, axis = 1)\n#pred3 = np.argmax(y_pred3, axis = 1)\n\nresult = pd.DataFrame(test_idt)\nresult[1] = pred1*0.3+pred2*0.7\n\nresult.columns=[\"BraTS21ID\",\"MGMT_value\"]\nresult_final = result.groupby(\"BraTS21ID\",as_index = False).mean()\n\nresult_final[\"BraTS21ID\"] = sample_sub[\"BraTS21ID\"]\nresult_final[\"MGMT_value\"] = result_final[\"MGMT_value\"]\nresult_final","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:29:54.665300Z","iopub.execute_input":"2021-10-08T04:29:54.665840Z","iopub.status.idle":"2021-10-08T04:29:56.423669Z","shell.execute_reply.started":"2021-10-08T04:29:54.665807Z","shell.execute_reply":"2021-10-08T04:29:56.422987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_final.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:30:49.032726Z","iopub.execute_input":"2021-10-08T04:30:49.033040Z","iopub.status.idle":"2021-10-08T04:30:49.039417Z","shell.execute_reply.started":"2021-10-08T04:30:49.032996Z","shell.execute_reply":"2021-10-08T04:30:49.038577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.hist(result_final[\"MGMT_value\"]);","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:30:47.174859Z","iopub.execute_input":"2021-10-08T04:30:47.175130Z","iopub.status.idle":"2021-10-08T04:30:47.419431Z","shell.execute_reply.started":"2021-10-08T04:30:47.175102Z","shell.execute_reply":"2021-10-08T04:30:47.418666Z"},"trusted":true},"execution_count":null,"outputs":[]}]}