{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# importing the necessary libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n#from PIL import Image \nimport seaborn as sns\nimport os\nimport pydicom as dicom\n#from pympler import asizeof\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D, MaxPool2D, Flatten, Dropout, BatchNormalization\nfrom keras import layers\nfrom keras.layers.experimental import preprocessing\nfrom keras.preprocessing.image import ImageDataGenerator\nimport cv2\nfrom random import randrange\nfrom skimage.transform import resize\nfrom keras.applications.inception_v3 import InceptionV3\nfrom sklearn.metrics import roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:25:42.19659Z","iopub.execute_input":"2021-10-08T04:25:42.197011Z","iopub.status.idle":"2021-10-08T04:25:48.287281Z","shell.execute_reply.started":"2021-10-08T04:25:42.196929Z","shell.execute_reply":"2021-10-08T04:25:48.286245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ntest = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/test\"","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:25:48.288936Z","iopub.execute_input":"2021-10-08T04:25:48.289341Z","iopub.status.idle":"2021-10-08T04:25:48.302971Z","shell.execute_reply.started":"2021-10-08T04:25:48.289294Z","shell.execute_reply":"2021-10-08T04:25:48.301988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining a function to read images for training\n\ndef load_train_images(path_train):\n    array = []          \n    label = []\n    IMG_PX_SIZE = 150\n    cases_to_be_excluded = ['../input/rsna-miccai-png/train/00109', \n                            '../input/rsna-miccai-png/train/00123', \n                            '../input/rsna-miccai-png/train/00709']\n    path_cases = sorted([f.path for f in os.scandir(path_train)])\n    for i in range(len(path_cases)):\n        if (path_cases!=cases_to_be_excluded[0] or  path_cases!=cases_to_be_excluded[1] or path_cases!=cases_to_be_excluded[2]):\n            mri_type = sorted([f.path for f in os.scandir(path_cases[i])])\n            img_path = sorted([f.path for f in os.scandir(mri_type[0])]) # 0 for flair images\n            for k in range(len(img_path)//9):  #10 for 6.4k images\n                img = dicom.dcmread(img_path[k])\n                if (img.pixel_array.sum()>100000):\n                    resized_img = resize(img.pixel_array, (IMG_PX_SIZE, IMG_PX_SIZE))\n                    img = np.array(resized_img)\n                    stacked_img = np.stack((img,)*3, axis=-1)\n                    stacked_img_normalize = stacked_img/np.max(stacked_img)\n                    if stacked_img_normalize.sum()>3000:\n                        array.append(stacked_img_normalize)\n                        label.append(train_labels.MGMT_value[i])\n                        #print(img_path[k])\n    array = array/np.max(array)\n    return array, label","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:25:48.305161Z","iopub.execute_input":"2021-10-08T04:25:48.305849Z","iopub.status.idle":"2021-10-08T04:25:48.316715Z","shell.execute_reply.started":"2021-10-08T04:25:48.305793Z","shell.execute_reply":"2021-10-08T04:25:48.315643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading about 7 thousand images.\n\ntrain_path = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train\"\npixels, y = load_train_images(train_path)\nprint(\"Number of images loaded are \", len(pixels))\nprint(\"Number of labels loaded are \", len(y))","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:25:48.318172Z","iopub.execute_input":"2021-10-08T04:25:48.318515Z","iopub.status.idle":"2021-10-08T04:28:44.597627Z","shell.execute_reply.started":"2021-10-08T04:25:48.318479Z","shell.execute_reply":"2021-10-08T04:28:44.596815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking whether our data is imbalanced\n\nplt.figure(figsize=(15,7))\nsns.countplot(x = y)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:44.599342Z","iopub.execute_input":"2021-10-08T04:28:44.599683Z","iopub.status.idle":"2021-10-08T04:28:44.786968Z","shell.execute_reply.started":"2021-10-08T04:28:44.599648Z","shell.execute_reply":"2021-10-08T04:28:44.785945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing random 10 images.\n\nplt.figure(figsize=(18,12))\nfor i in range(6):\n    plt.subplot(3,2,i+1)\n    random_number = randrange(500)\n    plt.imshow(pixels[random_number])\n    plt.title(y[random_number])\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:44.788755Z","iopub.execute_input":"2021-10-08T04:28:44.789158Z","iopub.status.idle":"2021-10-08T04:28:45.195669Z","shell.execute_reply.started":"2021-10-08T04:28:44.789119Z","shell.execute_reply":"2021-10-08T04:28:45.194694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data augmentation to prevent overfitting and handling the imbalance in dataset\n\ndatagen = ImageDataGenerator(\n        #featurewise_center=True,  # set input mean to 0 over the dataset\n        #samplewise_center=True,  # set each sample mean to 0\n        #featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        #samplewise_std_normalization=False,  # divide each input by its std\n        #zca_whitening=True,  # apply ZCA whitening\n        rotation_range = 10,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.30, # Randomly zoom image \n        #width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        #height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        #horizontal_flip = True, #randomly flip images\n        #vertical_flip=True  # randomly flip images\n        )\n\n\ndatagen.fit(pixels)\n\n# Location plays a major role in detecting the promoter. Therefore minor manipulations are done.","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:45.196789Z","iopub.execute_input":"2021-10-08T04:28:45.197182Z","iopub.status.idle":"2021-10-08T04:28:46.471077Z","shell.execute_reply.started":"2021-10-08T04:28:45.19714Z","shell.execute_reply":"2021-10-08T04:28:46.470201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x, test_x, train_y, test_y = train_test_split(pixels, y, test_size = 0.20,\n                                                    stratify = y)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:46.47317Z","iopub.execute_input":"2021-10-08T04:28:46.473528Z","iopub.status.idle":"2021-10-08T04:28:47.536679Z","shell.execute_reply.started":"2021-10-08T04:28:46.47349Z","shell.execute_reply":"2021-10-08T04:28:47.535845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Size of train_x\", len(train_x))\nprint(\"Size of train_y\", len(train_y))\nprint(\"Size of test_x\", len(test_x))\nprint(\"Size of test_y\", len(test_y))","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:47.538195Z","iopub.execute_input":"2021-10-08T04:28:47.538555Z","iopub.status.idle":"2021-10-08T04:28:47.545425Z","shell.execute_reply.started":"2021-10-08T04:28:47.538517Z","shell.execute_reply":"2021-10-08T04:28:47.544416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining a function to create a model\n\ndef RSNA_model():\n    # Designing our model\n    model = keras.Sequential([\n\n        #base,\n\n        layers.Conv2D(filters=16,kernel_size=2,padding=\"same\",activation=\"relu\",input_shape=(150, 150, 3)),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=32,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=64,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=64,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=128,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=128,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Conv2D(filters=512,kernel_size=2,padding=\"same\",activation=\"relu\"),\n        layers.MaxPooling2D(),\n\n        layers.Flatten(),\n    \n        layers.Dense(20, activation='relu'),\n\n        layers.Dense(10, activation='relu'),\n\n        #layers.Dense(4, activation='relu'),\n\n        layers.Dense(2, activation='sigmoid')\n    ])\n\n    # Compiling our model\n    model.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['accuracy']\n    )\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:47.546705Z","iopub.execute_input":"2021-10-08T04:28:47.547113Z","iopub.status.idle":"2021-10-08T04:28:47.559322Z","shell.execute_reply.started":"2021-10-08T04:28:47.547073Z","shell.execute_reply":"2021-10-08T04:28:47.55791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RSNA_model()\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:47.560707Z","iopub.execute_input":"2021-10-08T04:28:47.561081Z","iopub.status.idle":"2021-10-08T04:28:49.399553Z","shell.execute_reply.started":"2021-10-08T04:28:47.561046Z","shell.execute_reply":"2021-10-08T04:28:49.398742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#callback = keras.callbacks.EarlyStopping(monitor = \"loss\", patience=3, restore_best_weights=True)\n\n# saving the training details in history variable\nhistory = model.fit(datagen.flow(train_x,train_y, batch_size =600), \n                    epochs=10,\n                    #callbacks=[callback]\n                   )","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:28:49.400884Z","iopub.execute_input":"2021-10-08T04:28:49.401245Z","iopub.status.idle":"2021-10-08T04:38:27.484506Z","shell.execute_reply.started":"2021-10-08T04:28:49.401199Z","shell.execute_reply":"2021-10-08T04:38:27.483533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_x)\nprediction = preds[:,1]\n\nroc_auc_score(test_y, prediction)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:39:09.929719Z","iopub.execute_input":"2021-10-08T04:39:09.930164Z","iopub.status.idle":"2021-10-08T04:39:11.173961Z","shell.execute_reply.started":"2021-10-08T04:39:09.930122Z","shell.execute_reply":"2021-10-08T04:39:11.173064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,8))\nsns.displot(prediction)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:39:13.57569Z","iopub.execute_input":"2021-10-08T04:39:13.576051Z","iopub.status.idle":"2021-10-08T04:39:13.968023Z","shell.execute_reply.started":"2021-10-08T04:39:13.576018Z","shell.execute_reply":"2021-10-08T04:39:13.967101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A quick check with test dataset \n\ndef load_test_images(path_test):\n    array_1 = []\n    IMG_PX_SIZE = 150\n    path_cases = sorted([f.path for f in os.scandir(path_test)])\n    for i in range(len(path_cases)):\n        mri_type = sorted([f.path for f in os.scandir(path_cases[i])])\n        img_path = sorted([f.path for f in os.scandir(mri_type[0])])\n        for k in range(len(img_path)): \n            img = dicom.dcmread(img_path[k])\n            if (img.pixel_array.sum()>100000):\n                    resized_img = resize(img.pixel_array, (IMG_PX_SIZE, IMG_PX_SIZE))\n                    img = np.array(resized_img)\n                    stacked_img = np.stack((img,)*3, axis=-1)\n                    stacked_img_normalize = stacked_img/np.max(stacked_img)\n                    if stacked_img_normalize.sum()>2000:\n                        array_1.append(stacked_img_normalize)\n                        #array_1.append(img_path[k])\n                        break    \n                            \n    array_1 = array_1/np.max(array_1)\n    \n    print(\"Number of t1wce images loaded are \", len(array_1))\n    \n    \n    return array_1","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:39:32.655995Z","iopub.execute_input":"2021-10-08T04:39:32.656347Z","iopub.status.idle":"2021-10-08T04:39:32.665754Z","shell.execute_reply.started":"2021-10-08T04:39:32.656315Z","shell.execute_reply":"2021-10-08T04:39:32.664474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pixels_1 = load_test_images(test)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:39:34.750552Z","iopub.execute_input":"2021-10-08T04:39:34.750944Z","iopub.status.idle":"2021-10-08T04:39:47.700022Z","shell.execute_reply.started":"2021-10-08T04:39:34.750906Z","shell.execute_reply":"2021-10-08T04:39:47.698995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_1 = model.predict(pixels_1)\nprediction_1 = preds_1[:,1]\n\nplt.figure(figsize=(12,8))\nsns.displot(prediction_1)","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:39:47.701618Z","iopub.execute_input":"2021-10-08T04:39:47.702154Z","iopub.status.idle":"2021-10-08T04:39:48.125293Z","shell.execute_reply.started":"2021-10-08T04:39:47.702112Z","shell.execute_reply":"2021-10-08T04:39:48.124479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving the model\n\n#model.save(\"rsna_miccai_10_b600_flair_7k_0.70auc_imgs.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-10-08T04:40:37.382419Z","iopub.execute_input":"2021-10-08T04:40:37.382775Z","iopub.status.idle":"2021-10-08T04:40:37.464307Z","shell.execute_reply.started":"2021-10-08T04:40:37.382745Z","shell.execute_reply":"2021-10-08T04:40:37.463386Z"},"trusted":true},"execution_count":null,"outputs":[]}]}