{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport glob\nimport random\nimport pydicom\n\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:38.883927Z","iopub.execute_input":"2022-07-07T19:30:38.884535Z","iopub.status.idle":"2022-07-07T19:30:39.94642Z","shell.execute_reply.started":"2022-07-07T19:30:38.884439Z","shell.execute_reply":"2022-07-07T19:30:39.945525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Data and Creating DataFrame","metadata":{}},{"cell_type":"code","source":"data_dir = '../input/rsna-miccai-png/train'\npatients = sorted(os.listdir(data_dir))\npatients[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:39.951781Z","iopub.execute_input":"2022-07-07T19:30:39.952052Z","iopub.status.idle":"2022-07-07T19:30:40.054729Z","shell.execute_reply.started":"2022-07-07T19:30:39.952017Z","shell.execute_reply":"2022-07-07T19:30:40.053986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/rsna-miccai-png/test'\npatients_test = sorted(os.listdir(data_dir))\npatients_test[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.055846Z","iopub.execute_input":"2022-07-07T19:30:40.056305Z","iopub.status.idle":"2022-07-07T19:30:40.079459Z","shell.execute_reply.started":"2022-07-07T19:30:40.056271Z","shell.execute_reply":"2022-07-07T19:30:40.078764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id = []\nfor i in patients:\n    id.append(int(i))\nid[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.081539Z","iopub.execute_input":"2022-07-07T19:30:40.082004Z","iopub.status.idle":"2022-07-07T19:30:40.089131Z","shell.execute_reply.started":"2022-07-07T19:30:40.08197Z","shell.execute_reply":"2022-07-07T19:30:40.088337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.090718Z","iopub.execute_input":"2022-07-07T19:30:40.091276Z","iopub.status.idle":"2022-07-07T19:30:40.127402Z","shell.execute_reply.started":"2022-07-07T19:30:40.09124Z","shell.execute_reply":"2022-07-07T19:30:40.126796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.MGMT_value.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.132034Z","iopub.execute_input":"2022-07-07T19:30:40.133878Z","iopub.status.idle":"2022-07-07T19:30:40.147755Z","shell.execute_reply.started":"2022-07-07T19:30:40.133845Z","shell.execute_reply":"2022-07-07T19:30:40.14697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nsns.countplot(data=df, x=\"MGMT_value\");","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.15128Z","iopub.execute_input":"2022-07-07T19:30:40.153307Z","iopub.status.idle":"2022-07-07T19:30:40.362675Z","shell.execute_reply.started":"2022-07-07T19:30:40.153274Z","shell.execute_reply":"2022-07-07T19:30:40.361993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(patients)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.363769Z","iopub.execute_input":"2022-07-07T19:30:40.364006Z","iopub.status.idle":"2022-07-07T19:30:40.372623Z","shell.execute_reply.started":"2022-07-07T19:30:40.363974Z","shell.execute_reply":"2022-07-07T19:30:40.371886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\n\ndef visualize_sample(\n    brats21id, \n    slice_i,\n    mgmt_value,\n    types=(\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\")\n):\n    plt.figure(figsize=(16, 5))\n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/\", \n        str(brats21id).zfill(5),\n    )\n    for i, t in enumerate(types, 1):\n        t_paths = sorted(\n            glob.glob(os.path.join(patient_path, t, \"*\")), \n            key=lambda x: int(x[:-4].split(\"-\")[-1]),\n        )\n        data = load_dicom(t_paths[int(len(t_paths) * slice_i)])\n        plt.subplot(1, 4, i)\n        plt.imshow(data, cmap=\"gray\")\n        plt.title(f\"{t}\", fontsize=16)\n        plt.axis(\"off\")\n\n    plt.suptitle(f\"MGMT_value: {mgmt_value}\", fontsize=16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.373905Z","iopub.execute_input":"2022-07-07T19:30:40.374588Z","iopub.status.idle":"2022-07-07T19:30:40.385106Z","shell.execute_reply.started":"2022-07-07T19:30:40.374555Z","shell.execute_reply":"2022-07-07T19:30:40.384492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in random.sample(range(df.shape[0]), 10):\n    _brats21id = df.iloc[i][\"BraTS21ID\"]\n    _mgmt_value = df.iloc[i][\"MGMT_value\"]\n    visualize_sample(brats21id=_brats21id, mgmt_value=_mgmt_value, slice_i=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:40.388095Z","iopub.execute_input":"2022-07-07T19:30:40.388494Z","iopub.status.idle":"2022-07-07T19:30:45.724536Z","shell.execute_reply.started":"2022-07-07T19:30:40.38845Z","shell.execute_reply":"2022-07-07T19:30:45.723848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = pd.DataFrame({'path' : glob.glob('../input/rsna-miccai-png/train/00000/FLAIR/*.png'),\n                              'label' : 1})\n\nfor i, patient in enumerate(patients[1:]):\n    data_slice = pd.DataFrame({'path' : glob.glob('../input/rsna-miccai-png/train/' + patient +'/FLAIR/*.png'),\n                               'label' : df['MGMT_value'][i+1]})\n    train_dataset = pd.concat([train_dataset, data_slice])\n    \ntrain_dataset = train_dataset.reset_index(drop = True)\ntrain_dataset['label'] = train_dataset['label'].astype(str)\ntrain_dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:30:45.725721Z","iopub.execute_input":"2022-07-07T19:30:45.72609Z","iopub.status.idle":"2022-07-07T19:31:01.486736Z","shell.execute_reply.started":"2022-07-07T19:30:45.726055Z","shell.execute_reply":"2022-07-07T19:31:01.486039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:01.488048Z","iopub.execute_input":"2022-07-07T19:31:01.488312Z","iopub.status.idle":"2022-07-07T19:31:01.509076Z","shell.execute_reply.started":"2022-07-07T19:31:01.48828Z","shell.execute_reply":"2022-07-07T19:31:01.508262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = pd.DataFrame({'path' : glob.glob('../input/rsna-miccai-png/test/00001/FLAIR/*.png')})\n\nfor i, patient in enumerate(patients_test[1:]):\n    data_slice = pd.DataFrame({'path' : glob.glob('../input/rsna-miccai-png/test/' + patient +'/FLAIR/*.png')})\n    test_dataset = pd.concat([test_dataset, data_slice])\n    \ntest_dataset = test_dataset.reset_index(drop = True)\n\ntest_dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:01.511747Z","iopub.execute_input":"2022-07-07T19:31:01.511993Z","iopub.status.idle":"2022-07-07T19:31:04.630749Z","shell.execute_reply.started":"2022-07-07T19:31:01.511965Z","shell.execute_reply":"2022-07-07T19:31:04.630078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val = train_test_split(train_dataset, stratify = train_dataset['label'], random_state = 42, test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.632073Z","iopub.execute_input":"2022-07-07T19:31:04.632355Z","iopub.status.idle":"2022-07-07T19:31:04.738482Z","shell.execute_reply.started":"2022-07-07T19:31:04.632321Z","shell.execute_reply":"2022-07-07T19:31:04.737807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.739623Z","iopub.execute_input":"2022-07-07T19:31:04.739869Z","iopub.status.idle":"2022-07-07T19:31:04.74855Z","shell.execute_reply.started":"2022-07-07T19:31:04.739835Z","shell.execute_reply":"2022-07-07T19:31:04.74776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_val.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.750552Z","iopub.execute_input":"2022-07-07T19:31:04.750727Z","iopub.status.idle":"2022-07-07T19:31:04.757883Z","shell.execute_reply.started":"2022-07-07T19:31:04.750706Z","shell.execute_reply":"2022-07-07T19:31:04.757067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.759183Z","iopub.execute_input":"2022-07-07T19:31:04.759684Z","iopub.status.idle":"2022-07-07T19:31:04.768427Z","shell.execute_reply.started":"2022-07-07T19:31:04.759647Z","shell.execute_reply":"2022-07-07T19:31:04.767671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nx_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.770504Z","iopub.execute_input":"2022-07-07T19:31:04.770826Z","iopub.status.idle":"2022-07-07T19:31:04.778224Z","shell.execute_reply.started":"2022-07-07T19:31:04.770796Z","shell.execute_reply":"2022-07-07T19:31:04.777538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EffiecientNet B3","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom tensorflow.keras.callbacks import *\n\nidg = ImageDataGenerator(horizontal_flip = True)\nidg2 = ImageDataGenerator()\n\ntrain_dataset = idg.flow_from_dataframe(x_train, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\nvalid_dataset = idg2.flow_from_dataframe(x_val, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\n\nefn = EfficientNetB3(include_top = False, pooling = 'avg', input_shape=(150, 150, 3), weights='../input/keras-pretrained-models/EfficientNetB3_NoTop_ImageNet.h5')\nes = EarlyStopping(patience = 2, restore_best_weights = True)\nrl = ReduceLROnPlateau(patience = 1, factor = 0.2, verbose = 1)\n\nmodel = Sequential()\nmodel.add(efn)\n\nmodel.add(Dense(2, activation = 'softmax'))\n\nmodel.compile(metrics = ['acc'], loss = 'categorical_crossentropy', optimizer = 'adam')\n\nmodel.fit(train_dataset, epochs = 10, callbacks = [es, rl])\n# test_generator = idg2.flow_from_dataframe(x_test, x_col = 'path', y_col = None, target_size = (150, 150), batch_size = 64, class_mode = None, shuffle = False)\n\n# result = model.predict(test_generator, verbose = True, workers = 2)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T19:31:04.779543Z","iopub.execute_input":"2022-07-07T19:31:04.779949Z","iopub.status.idle":"2022-07-07T20:15:48.152672Z","shell.execute_reply.started":"2022-07-07T19:31:04.779916Z","shell.execute_reply":"2022-07-07T20:15:48.151892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = idg2.flow_from_dataframe(test_dataset, x_col = 'path', y_col = None, target_size = (150, 150), batch_size = 64, class_mode = None, shuffle = False)\n\nresult = model.predict(test_generator, verbose = True, workers = 2)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T20:15:48.154595Z","iopub.execute_input":"2022-07-07T20:15:48.154866Z","iopub.status.idle":"2022-07-07T20:16:27.166805Z","shell.execute_reply.started":"2022-07-07T20:15:48.154829Z","shell.execute_reply":"2022-07-07T20:16:27.165995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_result = model.predict(valid_dataset, verbose = True, workers = 2)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T20:58:40.410828Z","iopub.execute_input":"2022-07-07T20:58:40.411097Z","iopub.status.idle":"2022-07-07T20:59:46.820039Z","shell.execute_reply.started":"2022-07-07T20:58:40.411069Z","shell.execute_reply":"2022-07-07T20:59:46.819266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_result","metadata":{"execution":{"iopub.status.busy":"2022-07-07T21:01:47.957729Z","iopub.execute_input":"2022-07-07T21:01:47.958295Z","iopub.status.idle":"2022-07-07T21:01:47.963967Z","shell.execute_reply.started":"2022-07-07T21:01:47.958256Z","shell.execute_reply":"2022-07-07T21:01:47.963294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ntn, fp, fn, tp = confusion_matrix(x_val['label'], new_result).ravel()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T21:01:29.449093Z","iopub.execute_input":"2022-07-07T21:01:29.449721Z","iopub.status.idle":"2022-07-07T21:01:29.491582Z","shell.execute_reply.started":"2022-07-07T21:01:29.449685Z","shell.execute_reply":"2022-07-07T21:01:29.49056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EffiecientNet B5valid_dataset","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.applications import EfficientNetB1\nfrom tensorflow.keras.callbacks import *\n\nidg = ImageDataGenerator(horizontal_flip = True)\nidg2 = ImageDataGenerator()\n\ntrain_dataset = idg.flow_from_dataframe(x_train, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\nvalid_dataset = idg2.flow_from_dataframe(x_valid, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\n\nefn = EfficientNetB5(include_top = False, pooling = 'avg', input_shape=(150, 150, 3), weights='../input/keras-pretrained-models/EfficientNetB5_NoTop_ImageNet.h5',)\nes = EarlyStopping(patience = 2, restore_best_weights = True)\nrl = ReduceLROnPlateau(patience = 1, factor = 0.2, verbose = 1)\n\nmodel = Sequential()\nmodel.add(efn)\nmodel.add(Dense(2, activation = 'softmax'))\n\nmodel.compile(metrics = ['acc'], loss = 'categorical_crossentropy', optimizer = 'adam')\n\nmodel.fit(train_dataset, validation_data = valid_dataset, epochs = 2, callbacks = [es, rl])\n\ntest_generator = idg2.flow_from_dataframe(test_dataset, x_col = 'path', y_col = None, target_size = (150, 150), batch_size = 64, class_mode = None, shuffle = False)\n\nresult = model.predict(test_generator, verbose = True, workers = 2)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.463145Z","iopub.status.idle":"2022-05-23T08:24:55.463991Z","shell.execute_reply.started":"2022-05-23T08:24:55.463666Z","shell.execute_reply":"2022-05-23T08:24:55.463694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EffiecientNetB3 with Image Augmentations","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.applications import EfficientNetB1\nfrom tensorflow.keras.callbacks import *\n\nidg = ImageDataGenerator(horizontal_flip = True, rotation_range=30, width_shift_range=0.2, height_shift_range=0.2, brightness_range=[0.4,1.5])\nidg2 = ImageDataGenerator()\n\ntrain_dataset = idg.flow_from_dataframe(x_train, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\nvalid_dataset = idg2.flow_from_dataframe(x_valid, x_col = 'path', y_col = 'label', target_size=(150, 150), batch_size = 64)\n\nefn = EfficientNetB3(include_top = False, pooling = 'avg', input_shape=(150, 150, 3), weights='../input/keras-pretrained-models/EfficientNetB3_NoTop_ImageNet.h5',)\nes = EarlyStopping(patience = 2, restore_best_weights = True)\nrl = ReduceLROnPlateau(patience = 1, factor = 0.2, verbose = 1)\n\nmodel = Sequential()\nmodel.add(efn)\nmodel.add(Dense(2, activation = 'softmax'))\n\nmodel.compile(metrics = ['acc'], loss = 'categorical_crossentropy', optimizer = 'adam')\n\nmodel.fit(train_dataset, validation_data = valid_dataset, epochs = 2, callbacks = [es, rl])\n\ntest_generator = idg2.flow_from_dataframe(test_dataset, x_col = 'path', y_col = None, target_size = (150, 150), batch_size = 64, class_mode = None, shuffle = False)\n\nresult = model.predict(test_generator, verbose = True, workers = 2)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.4656Z","iopub.status.idle":"2022-05-23T08:24:55.466381Z","shell.execute_reply.started":"2022-05-23T08:24:55.466102Z","shell.execute_reply":"2022-05-23T08:24:55.46613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(result)","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.467959Z","iopub.status.idle":"2022-05-23T08:24:55.46887Z","shell.execute_reply.started":"2022-05-23T08:24:55.46856Z","shell.execute_reply":"2022-05-23T08:24:55.46859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('eff-b6')","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.470517Z","iopub.status.idle":"2022-05-23T08:24:55.471448Z","shell.execute_reply.started":"2022-05-23T08:24:55.471129Z","shell.execute_reply":"2022-05-23T08:24:55.471161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.472964Z","iopub.status.idle":"2022-05-23T08:24:55.47384Z","shell.execute_reply.started":"2022-05-23T08:24:55.473531Z","shell.execute_reply":"2022-05-23T08:24:55.473562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_model = keras.models.load_model(\"eff-b6\")","metadata":{"execution":{"iopub.status.busy":"2022-05-23T08:24:55.475368Z","iopub.status.idle":"2022-05-23T08:24:55.476203Z","shell.execute_reply.started":"2022-05-23T08:24:55.475873Z","shell.execute_reply":"2022-05-23T08:24:55.475901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}