{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align=\"center\">\n    <h2>RSNA-MICCAI Brain Tumor Radiogenomic Classification</h2>\n    <img src=\"https://user-images.githubusercontent.com/48846576/126234744-eda092a0-dcfa-4d9c-896b-aef858806298.png\"  width=\"700\" height=\"200\">\n</div>    \n\n### What is the competition?\n\nThis is a binary image classification problem where in we will predict the genetic subtype of glioblastoma using MRI (magnetic resonance imaging) scans to detect for the presence of MGMT promoter methylation.\n\nThere are MRI scans for 585 cases given and each case has been classified as MGMT_value 0 or 1. Each case has several images in the following MRI sequence types\n\n* Fluid Attenuated Inversion Recovery (FLAIR)\n* T1-weighted pre-contrast (T1w)\n* T1-weighted post-contrast (T1wCE)\n* T2-weighted (T2)\n\nThe MRI scan images are given in DICOM format. \n\n### What is DICOM?\n\nDICOM® — [Digital Imaging and Communications in Medicine](https://www.dicomstandard.org) — is the ISO recognized international standard for medical images and related information. It defines the formats for medical images that can be exchanged with the data and quality necessary for clinical use.\n\n> This notebook aims to do exploratory data analysis from a beginner standpoint!","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":false}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nimport plotly.graph_objects as go\nfrom plotly.offline import iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport plotly_express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nimport plotly.io as pio\nfrom plotly.subplots import make_subplots\n# setting default template to plotly_white for all visualizations\npio.templates.default = \"plotly_white\"\n%matplotlib inline\nimport gc\n\n\nfrom colorama import Fore, Back, Style\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nres = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\n#/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\n#/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\n!pip install python-gdcm\n","metadata":{"_kg_hide-output":true,"_kg_hide-input":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        all_files.append(os.path.join(dirname, filename))\n        \nlbl_df = pd.read_csv('/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv', index_col=None)\ntrain_files = [file for file in all_files if 'train' in file and 'train_labels.csv' not in file]\n\nimport re\ntrain_cases = []\nfor file in train_files:\n    train_cases.append(re.findall(r\"\\D(\\d{5})\\D\", file)[0])\ntrain_cases=list(set(train_cases))\ntrain_cases={ int(num) : num for num in train_cases } ","metadata":{"execution":{"iopub.status.busy":"2021-07-19T18:58:41.507534Z","iopub.execute_input":"2021-07-19T18:58:41.507933Z","iopub.status.idle":"2021-07-19T18:58:41.693874Z","shell.execute_reply.started":"2021-07-19T18:58:41.507902Z","shell.execute_reply":"2021-07-19T18:58:41.692488Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculat_num_imgs(case_id, img_type='Total'):\n    if img_type == 'Total':\n        return len([file for file in train_files if train_cases[case_id] in file])\n    else:\n        return len([file for file in train_files if train_cases[case_id] in file and '/' + img_type + '/' in file])\n    \nfor index, row in lbl_df.iterrows():\n    lbl_df.at[index, 'total'] = calculat_num_imgs(row['BraTS21ID'])\n    lbl_df.at[index, 'FLAIR'] = calculat_num_imgs(row['BraTS21ID'],'FLAIR')\n    lbl_df.at[index, 'T1w'] = calculat_num_imgs(row['BraTS21ID'],'T1w')\n    lbl_df.at[index, 'T1wCE'] = calculat_num_imgs(row['BraTS21ID'],'T1wCE')\n    lbl_df.at[index, 'T2w'] = calculat_num_imgs(row['BraTS21ID'],'T2w')\n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-19T20:04:17.970744Z","iopub.execute_input":"2021-07-19T20:04:17.971092Z","iopub.status.idle":"2021-07-19T20:10:15.82259Z","shell.execute_reply.started":"2021-07-19T20:04:17.971063Z","shell.execute_reply":"2021-07-19T20:10:15.821461Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbl_df['total'] = lbl_df['total'].astype(int)\nlbl_df['FLAIR'] = lbl_df['FLAIR'].astype(int)\nlbl_df['T1w'] = lbl_df['T1w'].astype(int)\nlbl_df['T1wCE'] = lbl_df['T1wCE'].astype(int)\nlbl_df['T2w'] = lbl_df['T2w'].astype(int)\nlbl_df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-19T20:11:45.347599Z","iopub.execute_input":"2021-07-19T20:11:45.347949Z","iopub.status.idle":"2021-07-19T20:11:45.368241Z","shell.execute_reply.started":"2021-07-19T20:11:45.347915Z","shell.execute_reply":"2021-07-19T20:11:45.367119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = {'0' : '#DCD427',\n'1' : '#0092CC'\n         }\ncount_df = lbl_df.groupby(['MGMT_value'])['BraTS21ID'].count().reset_index()\ncount_df.rename(columns={'BraTS21ID':'count'},inplace=True)\ncount_df['color'] = count_df['MGMT_value'].astype(str).apply(lambda x: colors[x])\ncount_df['MGMT_value'] = count_df['MGMT_value'].astype(str)\ncount_df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-19T20:14:03.309461Z","iopub.execute_input":"2021-07-19T20:14:03.310227Z","iopub.status.idle":"2021-07-19T20:14:03.326754Z","shell.execute_reply.started":"2021-07-19T20:14:03.310182Z","shell.execute_reply":"2021-07-19T20:14:03.325625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"{y_}Total cases in training data  : {lbl_df.shape[0]}{res}\\n{g_}Total images in training data : {lbl_df['total'].sum()}{res}\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2021-07-19T20:16:17.856962Z","iopub.execute_input":"2021-07-19T20:16:17.857484Z","iopub.status.idle":"2021-07-19T20:16:17.863638Z","shell.execute_reply.started":"2021-07-19T20:16:17.85745Z","shell.execute_reply":"2021-07-19T20:16:17.862368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Color Palettes ","metadata":{"_kg_hide-input":true,"_kg_hide-output":true}},{"cell_type":"code","source":"colors1 = ['#FC6238', '#FFD872','#F2D4CC','#E77577','#0065A2','#74737A']\ncolors2 = ['#3E7DCC', '#8F9CB3','#00C8C8','#F9D84A','#8CC0FF','#4D525A']\ncolors3 = ['#B29476', '#E3D6C9','#1F5C70','#FBA01D','#FCBC49','#393B45']\nsns.palplot(sns.color_palette(colors1),size=0.9)\nsns.palplot(sns.color_palette(colors2),size=0.9)\nsns.palplot(sns.color_palette(colors3),size=0.9)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-19T20:34:44.346227Z","iopub.execute_input":"2021-07-19T20:34:44.346575Z","iopub.status.idle":"2021-07-19T20:34:44.54011Z","shell.execute_reply.started":"2021-07-19T20:34:44.346546Z","shell.execute_reply":"2021-07-19T20:34:44.539134Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_mgmt():\n    pio.templates.default = \"plotly_dark\"\n    fig = px.bar(count_df, x='MGMT_value', y='count',\n           hover_data=['MGMT_value', 'count'], color='MGMT_value',\n           #labels={column: label},\n           color_discrete_map=colors,\n           text='count')\n    fig.update_layout(xaxis={'categoryorder':'array', 'categoryarray': count_df['MGMT_value'],\n                           'title' : None, \n                           'showgrid':False},\n                    yaxis={'showgrid':False,\n                          'title' : 'Count'},\n                    showlegend=True,\n                   title = 'Cases in train data')\n    fig.update_traces(textfont_size=16)\n    fig.show()\n\nplot_mgmt()","metadata":{"execution":{"iopub.status.busy":"2021-07-19T20:14:04.726487Z","iopub.execute_input":"2021-07-19T20:14:04.727033Z","iopub.status.idle":"2021-07-19T20:14:04.817146Z","shell.execute_reply.started":"2021-07-19T20:14:04.726963Z","shell.execute_reply":"2021-07-19T20:14:04.816355Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_count = { 'type' : ['FLAIR','T1w','T1wCE','T2w'], 'count' : [lbl_df['FLAIR'].sum(),lbl_df['T1w'].sum(),lbl_df['T1wCE'].sum(), lbl_df['T2w'].sum()]}\nimg_count_df = pd.DataFrame.from_dict(img_count)\ncolors1 = ['#FC6238', '#FFD872','#F2D4CC','#E77577','#0065A2','#74737A']\n\nimg_type_colors = {'FLAIR' : colors1[0],\n'T1w' : colors1[1],\n'T1wCE' : colors1[-1],\n'T2w' : colors1[-2]}\n\nimg_count_df['color'] = img_count_df['type'].apply(lambda x: img_type_colors[x])\nimg_count_df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-19T20:35:45.79068Z","iopub.execute_input":"2021-07-19T20:35:45.791072Z","iopub.status.idle":"2021-07-19T20:35:45.811046Z","shell.execute_reply.started":"2021-07-19T20:35:45.791039Z","shell.execute_reply":"2021-07-19T20:35:45.809949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_image_type():\n    pio.templates.default = \"plotly_dark\"\n    fig = px.bar(img_count_df, x='type', y='count',\n             hover_data=['type', 'count'], color='type',\n             #labels={column: label},\n             color_discrete_map=img_type_colors,\n             text='count')\n    fig.update_layout(xaxis={'categoryorder':'array', 'categoryarray':img_count_df['type'],\n                             'title' : None, \n                             'showgrid':False},\n                      yaxis={'showgrid':False,\n                            'title' : 'Count'},\n                      showlegend=False,\n                     title = 'Types of MRI Sequences in training data')\n    fig.update_traces(textfont_size=16)\n    fig.show()\n\nplot_image_type()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-19T20:35:46.236297Z","iopub.execute_input":"2021-07-19T20:35:46.236639Z","iopub.status.idle":"2021-07-19T20:35:46.343112Z","shell.execute_reply.started":"2021-07-19T20:35:46.236611Z","shell.execute_reply":"2021-07-19T20:35:46.341874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace(go.Box(y=lbl_df['FLAIR'], \n                         name='FLAIR', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor=colors1[0],\n                         marker_size=5,\n                         line_width=1))\nfig.add_trace(go.Box(y=lbl_df['T1w'], \n                         name='T1w', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor=colors1[1],\n                         marker_size=5,\n                         line_width=1))\nfig.add_trace(go.Box(y=lbl_df['T1wCE'], \n                         name='T1wCE', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor=colors1[-1],\n                         marker_size=5,\n                         line_width=1))\nfig.add_trace(go.Box(y=lbl_df['T2w'], \n                         name='T2w', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor=colors1[-2],\n                         marker_size=5,\n                         line_width=1))\n\nfig.update_layout(xaxis={'title' : None,'showgrid' :False},\n                  yaxis=dict(title='Number of images',showgrid=False,zeroline=False),\n                  showlegend=False,\n                 title = 'Number of images per case/MRI Sequence - Interquartile range (IQR)')    \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-19T20:50:03.825926Z","iopub.execute_input":"2021-07-19T20:50:03.826271Z","iopub.status.idle":"2021-07-19T20:50:03.858406Z","shell.execute_reply.started":"2021-07-19T20:50:03.826242Z","shell.execute_reply":"2021-07-19T20:50:03.857101Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display Sample Images\n> Each case has several images in of the four different MRI scan sequences. Let's look at the sample across each category.","metadata":{}},{"cell_type":"code","source":"from ast import literal_eval\nimport matplotlib.patches as patches\nfrom pydicom import dcmread, read_file\nfrom pydicom.data import get_testdata_file\n\ndef display_sample(case_id = 0, mgmt_value = 1, cmap='gray', axis_off='on'):\n    fig1, ax1 = plt.subplots(1,4, figsize=(18, 5), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for img_type, ax in zip(list(img_count_df['type']), axs):\n        sample_files = [file for file in train_files if train_cases[case_id] in file and img_type in file]\n        dicom = read_file(sample_files[0], stop_before_pixels=False)\n        ax.imshow(dicom.pixel_array, cmap=cmap)    \n        ax.set_title('{}'.format(img_type),fontsize = 16)    \n        ax.axis(axis_off)\n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.81)\n    plt.suptitle('Samples Images, Case : {}, MGMT_value = {}'.format(train_cases[case_id],mgmt_value),fontsize = 20)\n    plt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-19T21:26:54.528528Z","iopub.execute_input":"2021-07-19T21:26:54.528912Z","iopub.status.idle":"2021-07-19T21:26:54.538811Z","shell.execute_reply.started":"2021-07-19T21:26:54.528875Z","shell.execute_reply":"2021-07-19T21:26:54.537369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_sample(case_id = 0,mgmt_value = 1, cmap='viridis')\ndisplay_sample(case_id = 2,mgmt_value = 1, cmap='viridis')\ndisplay_sample(case_id = 1005,mgmt_value = 1, cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2021-07-19T21:29:00.655251Z","iopub.execute_input":"2021-07-19T21:29:00.655653Z","iopub.status.idle":"2021-07-19T21:29:03.029367Z","shell.execute_reply.started":"2021-07-19T21:29:00.655622Z","shell.execute_reply":"2021-07-19T21:29:03.028401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_sample(case_id = 3,mgmt_value = 0, cmap='viridis')\ndisplay_sample(case_id = 17,mgmt_value = 0, cmap='viridis')\ndisplay_sample(case_id = 1009,mgmt_value = 0, cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2021-07-19T21:32:35.073687Z","iopub.execute_input":"2021-07-19T21:32:35.074042Z","iopub.status.idle":"2021-07-19T21:32:37.451126Z","shell.execute_reply.started":"2021-07-19T21:32:35.074008Z","shell.execute_reply":"2021-07-19T21:32:37.450041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display all images of a case\n> Let's take the case which has lesser number of images and display those","metadata":{}},{"cell_type":"code","source":"lbl_df.loc[lbl_df['FLAIR'] == lbl_df['FLAIR'].min()]","metadata":{"execution":{"iopub.status.busy":"2021-07-19T22:08:00.443072Z","iopub.execute_input":"2021-07-19T22:08:00.443736Z","iopub.status.idle":"2021-07-19T22:08:00.460828Z","shell.execute_reply.started":"2021-07-19T22:08:00.443679Z","shell.execute_reply":"2021-07-19T22:08:00.459334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_all_images(case_id, rows = 10, cols = 3, mgmt_value = 0, mri_type='FLAIR'):\n    fig1, ax1 = plt.subplots(rows, cols, figsize=(5, 18), facecolor='w', edgecolor='b')\n    axs = ax1.ravel()\n    sample_files = [file for file in train_files if train_cases[case_id] in file and '/' + mri_type +'/' in file]\n    for file, ax in zip(sample_files, axs):\n        dicom = read_file(file, stop_before_pixels=False)\n        ax.imshow(dicom.pixel_array, cmap='viridis')    \n        ax.axis('off')\n    if mri_type != 'FLAIR':\n        ax1[9,2].set_axis_off()    \n    plt.tight_layout(pad=1)\n    plt.subplots_adjust(top=0.91)\n    plt.suptitle('Samples Images, Case : {}, {}, MGMT_value = {}'.format(train_cases[case_id], mri_type, mgmt_value),fontsize = 20)\n    plt.show()\n\ndisplay_all_images(818, 5, 3, 0, 'FLAIR')    \ndisplay_all_images(818, 10, 3, 0, 'T1w')    \ndisplay_all_images(818, 10, 3, 0, 'T1wCE')    \ndisplay_all_images(818, 10, 3, 0, 'T2w')    \n","metadata":{"execution":{"iopub.status.busy":"2021-07-19T22:11:03.199518Z","iopub.execute_input":"2021-07-19T22:11:03.19994Z","iopub.status.idle":"2021-07-19T22:11:09.263655Z","shell.execute_reply.started":"2021-07-19T22:11:03.199904Z","shell.execute_reply":"2021-07-19T22:11:09.262232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Work in progress!","metadata":{}},{"cell_type":"code","source":"from ast import literal_eval\nimport matplotlib.patches as patches\nfrom pydicom import dcmread, read_file\nfrom pydicom.data import get_testdata_file\n\ndef display_sample(case_id = 0, mgmt_value = 1, cmap='gray', axis_off='on'):\n    fig1, ax1 = plt.subplots(1,4, figsize=(18, 5), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for img_type, ax in zip(list(img_count_df['type']), axs):\n        sample_files = [file for file in train_files if train_cases[0] in file and img_type in file]\n        dicom = read_file(sample_files[0], stop_before_pixels=False)\n        ax.imshow(dicom.pixel_array, cmap=cmap)    \n        ax.set_title('{}'.format(img_type),fontsize = 16)    \n        ax.axis('off')\n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.81)\n    plt.suptitle('Case : {}, MGMT_value = {}'.format(train_cases[0],1),fontsize = 20)\n    plt.show()\n    \n    fig1, ax1 = plt.subplots(1,4, figsize=(18, 5), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for img_type, ax in zip(list(img_count_df['type']), axs):\n        sample_files = [file for file in train_files if train_cases[3] in file and img_type in file]\n        dicom = read_file(sample_files[0], stop_before_pixels=False)\n        ax.imshow(dicom.pixel_array, cmap=cmap)    \n        ax.set_title('{}'.format(img_type),fontsize = 16)    \n        ax.axis('off')\n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.81)\n    plt.suptitle('Case : {}, MGMT_value = {}'.format(train_cases[3],0),fontsize = 20)\n    plt.show()\n    \n    \ndisplay_sample(case_id = 3,mgmt_value = 0, cmap='viridis')\n\n\n#display_sample(case_id = 0,mgmt_value = 1, cmap='viridis')","metadata":{"execution":{"iopub.status.busy":"2021-07-19T22:17:12.803632Z","iopub.execute_input":"2021-07-19T22:17:12.804058Z","iopub.status.idle":"2021-07-19T22:17:14.321597Z","shell.execute_reply.started":"2021-07-19T22:17:12.804007Z","shell.execute_reply":"2021-07-19T22:17:14.320255Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}