{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **RSNA-MICCAI Brain Tumor Radiogenomic Classification**\n# DATA EXPLORATION","metadata":{}},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:40:54.995559Z","iopub.execute_input":"2021-07-14T08:40:54.996161Z","iopub.status.idle":"2021-07-14T08:41:57.539596Z","shell.execute_reply.started":"2021-07-14T08:40:54.996045Z","shell.execute_reply":"2021-07-14T08:41:57.538405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport pydicom as dicom\nimport cv2\nimport ast\nimport warnings\nfrom collections import Counter\nimport seaborn as sns\nimport pydicom\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:01.577757Z","iopub.execute_input":"2021-07-14T08:42:01.578200Z","iopub.status.idle":"2021-07-14T08:42:03.081858Z","shell.execute_reply.started":"2021-07-14T08:42:01.578157Z","shell.execute_reply":"2021-07-14T08:42:03.081006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **1. DATA PROCESSING**","metadata":{}},{"cell_type":"markdown","source":"First of all, we define the path variable and check files and folders :","metadata":{}},{"cell_type":"code","source":"path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/'\ntrain_path = os.path.join(path, 'train')\ntest_path = os.path.join(path, 'test')\n\nflair_dir = 'FLAIR'\nt1w_dir = 'T1w'\nt1wce_dir = 'T1wCE'\nt2w_dir = 'T2w'\n\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:06.517455Z","iopub.execute_input":"2021-07-14T08:42:06.518049Z","iopub.status.idle":"2021-07-14T08:42:06.536514Z","shell.execute_reply.started":"2021-07-14T08:42:06.517995Z","shell.execute_reply":"2021-07-14T08:42:06.535129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's time to import data :","metadata":{}},{"cell_type":"code","source":"ldf = pd.read_csv(path + 'train_labels.csv')\nldf.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:10.562501Z","iopub.execute_input":"2021-07-14T08:42:10.562893Z","iopub.status.idle":"2021-07-14T08:42:10.596989Z","shell.execute_reply.started":"2021-07-14T08:42:10.562861Z","shell.execute_reply":"2021-07-14T08:42:10.595877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's display some stats :","metadata":{}},{"cell_type":"code","source":"ldf.groupby('MGMT_value').count()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:13.377605Z","iopub.execute_input":"2021-07-14T08:42:13.378028Z","iopub.status.idle":"2021-07-14T08:42:13.393999Z","shell.execute_reply.started":"2021-07-14T08:42:13.377989Z","shell.execute_reply":"2021-07-14T08:42:13.392785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the data set is relatively well balanced.","metadata":{}},{"cell_type":"markdown","source":"Let's define some functions to retreive full path from Id.<br>\nWe take this opportunity to check is subdirectories are created for all Id : ","metadata":{}},{"cell_type":"code","source":"def getFullId(id):\n    return str(id).zfill(5)\n    \ndef getFlairPath(id):\n    flair_path = os.path.join(train_path, getFullId(id), flair_dir)\n    return flair_path if os.path.isdir(flair_path) else False\n\ndef getT1wPath(id):\n    t1w_path = os.path.join(train_path, getFullId(id), t1w_dir)\n    return t1w_path if os.path.isdir(t1w_path) else False\n\ndef getT1wcePath(id):\n    t1wce_path = os.path.join(train_path, getFullId(id), t1wce_dir)\n    return t1wce_path if os.path.isdir(t1wce_path) else False\n\ndef getT2wPath(id):\n    t2w_path = os.path.join(train_path, getFullId(id), t2w_dir)\n    return t2w_path if os.path.isdir(t2w_path) else False\n\nprint('Missing FLAIR directories number = ', ldf['BraTS21ID'].apply(lambda x: getFlairPath(x)).tolist().count(False))\nprint('Missing T1w directories number = ', ldf['BraTS21ID'].apply(lambda x: getT1wPath(x)).tolist().count(False))\nprint('Missing T1wCE directories number = ', ldf['BraTS21ID'].apply(lambda x: getT1wcePath(x)).tolist().count(False))\nprint('Missing T2w directories number = ', ldf['BraTS21ID'].apply(lambda x: getT2wPath(x)).tolist().count(False))","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:15.917909Z","iopub.execute_input":"2021-07-14T08:42:15.918341Z","iopub.status.idle":"2021-07-14T08:42:18.641957Z","shell.execute_reply.started":"2021-07-14T08:42:15.918304Z","shell.execute_reply":"2021-07-14T08:42:18.640907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Everything seems to be ok, let's count files for each subdirectory :","metadata":{}},{"cell_type":"code","source":"def countFlairFiles(id):\n    path = getFlairPath(id)\n    return len([file for file in os.listdir(path)])\n\ndef countT1wFiles(id):\n    path = getT1wPath(id)\n    return len([file for file in os.listdir(path)])\n\ndef countT1wceFiles(id):\n    path = getT1wcePath(id)\n    return len([file for file in os.listdir(path)])\n\ndef countT2wFiles(id):\n    path = getT2wPath(id)\n    return len([file for file in os.listdir(path)])\n\nldf['FLAIR'] = ldf['BraTS21ID'].apply(lambda x: countFlairFiles(x))\nldf['T1w'] = ldf['BraTS21ID'].apply(lambda x: countT1wFiles(x))\nldf['T1wCE'] = ldf['BraTS21ID'].apply(lambda x: countT1wceFiles(x))\nldf['T2w'] = ldf['BraTS21ID'].apply(lambda x: countT2wFiles(x))","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:42:21.727674Z","iopub.execute_input":"2021-07-14T08:42:21.728084Z","iopub.status.idle":"2021-07-14T08:43:55.593124Z","shell.execute_reply.started":"2021-07-14T08:42:21.728046Z","shell.execute_reply":"2021-07-14T08:43:55.592170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x=\"variable\", y=\"value\", data=pd.melt(ldf[['FLAIR', 'T1w', 'T1wCE', 'T2w']]))\nplt.title('Number of images files by structural multi-parametric MRI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:44:00.477103Z","iopub.execute_input":"2021-07-14T08:44:00.477519Z","iopub.status.idle":"2021-07-14T08:44:00.691125Z","shell.execute_reply.started":"2021-07-14T08:44:00.477484Z","shell.execute_reply":"2021-07-14T08:44:00.689811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It seems that there is some disparity between the Ids, it is perhaps on this point that we will have problems inherent to unbalanced datasets.","metadata":{}},{"cell_type":"markdown","source":"## **2. IMAGES / VIDEO**","metadata":{}},{"cell_type":"markdown","source":"We first create a function to convert dicom file to 3 channel image","metadata":{}},{"cell_type":"code","source":"def get3ScaledImage(path):\n\n    dicom = pydicom.read_file(path)\n    img = dicom.pixel_array\n\n    r, c = img.shape\n    img_conv = np.empty((c, r, 3), dtype=img.dtype)\n    img_conv[:,:,2] = img_conv[:,:,1] = img_conv[:,:,0] = img\n\n    ## Step 1. Convert to float to avoid overflow or underflow losses.\n    img_2d = img_conv.astype(float)\n\n    ## Step 2. Rescaling grey scale between 0-255\n    img_2d_scaled = (np.maximum(img_2d,0) / img_2d.max()) * 255.0\n\n    ## Step 3. Convert to uint\n    img_2d_scaled = np.uint8(img_2d_scaled)\n    img_2d_scaled.reshape([img_2d_scaled.shape[0], img_2d_scaled.shape[1], 3])\n    \n    return img_2d_scaled, (c, r)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:44:04.542717Z","iopub.execute_input":"2021-07-14T08:44:04.543130Z","iopub.status.idle":"2021-07-14T08:44:04.550453Z","shell.execute_reply.started":"2021-07-14T08:44:04.543094Z","shell.execute_reply":"2021-07-14T08:44:04.549479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then we generate a video clip by reading file in sequence.\nHere is an example for the first Id (=0) and FLAIR structure.\n\nGenerated video is saved as mp4 file in th working directory. Feel free to download it to visualize the sequence in action.","metadata":{}},{"cell_type":"code","source":"id = 0\n\nnb = countFlairFiles(id)\npath = getFlairPath(id)\nframes =[]\n\nfor i in range(nb):\n    file_name = 'Image-' + str(i+1) + '.dcm'\n    img_path = os.path.join(path, file_name)\n    img_2d_scaled, size = get3ScaledImage(img_path)\n    frames.append(img_2d_scaled)\n\n\nout = cv2.VideoWriter('/kaggle/working/video.mp4', 0x7634706d, 15, size)\nfor i in range(len(frames)):\n    out.write(frames[i])\n    \nout.release()\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:47:24.268071Z","iopub.execute_input":"2021-07-14T08:47:24.268477Z","iopub.status.idle":"2021-07-14T08:47:28.422859Z","shell.execute_reply.started":"2021-07-14T08:47:24.268444Z","shell.execute_reply":"2021-07-14T08:47:28.421690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can also visualize the sequence as below :","metadata":{}},{"cell_type":"code","source":"import matplotlib.animation as animation\n\nfrom matplotlib import animation, rc\n\nrc('animation', html='jshtml')\n\ndef create_animation(ims):\n    ims = ims\n    fps = 1\n    nSeconds = 10\n\n    fig = plt.figure( figsize=(9,9) )\n\n    a = ims[0]\n    im = plt.imshow(a)\n\n    def animate_func(i):\n        im.set_array(ims[i])\n        return [im]\n\n    anim = animation.FuncAnimation(fig, animate_func, frames = len(ims), interval = 1000//24)\n    \n    return anim\n\nvideo = create_animation(frames)\nvideo","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:53:23.757469Z","iopub.execute_input":"2021-07-14T08:53:23.757875Z","iopub.status.idle":"2021-07-14T08:54:14.280139Z","shell.execute_reply.started":"2021-07-14T08:53:23.757836Z","shell.execute_reply":"2021-07-14T08:54:14.279282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's plot somme slices :","metadata":{}},{"cell_type":"code","source":"def plot_slices(num_rows, num_columns, size, data):\n\n    fig_width = 20\n    fig_height = (fig_width / num_columns) / (size[1] / size[0]) * num_rows\n    \n    fig, ax = plt.subplots(nrows=num_rows, ncols=num_columns, figsize=(fig_width, fig_height))\n    for i in range(num_rows):\n        for j in range(num_columns):\n            ax[i, j].imshow(data[i*num_columns+j], cmap=\"gray\")\n            ax[i, j].axis(\"off\")\n    plt.subplots_adjust(wspace=0, hspace=0, left=0, right=1, bottom=0, top=1)\n    plt.show()\n\nid = 5\n\nnb_flair = countFlairFiles(id)\npath_flair = getFlairPath(id)\nframes_flair = []\n\nfor i in range(nb_flair):\n    file_name = 'Image-' + str(i+1) + '.dcm'\n    img_path = os.path.join(path_flair, file_name)\n    img_2d_scaled, size = get3ScaledImage(img_path)\n    frames_flair.append(img_2d_scaled)\n    \nplot_slices(5, 6, size, frames[90:131])","metadata":{"execution":{"iopub.status.busy":"2021-07-14T08:54:55.457697Z","iopub.execute_input":"2021-07-14T08:54:55.458124Z","iopub.status.idle":"2021-07-14T08:55:10.038327Z","shell.execute_reply.started":"2021-07-14T08:54:55.458084Z","shell.execute_reply":"2021-07-14T08:55:10.036998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feel free to test this notebook, and upvote ...","metadata":{}}]}