{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Predicting Genetic Biomarker in Brain Tumor. \n\n## This Notebook only contains EDA and training data prep\n\n#### Problem \nIn this competition you will predict the genetic subtype of glioblastoma using MRI (magnetic resonance imaging) scans to train and test your model to detect for the presence of MGMT promoter methylation.\n\n#### Glossary \n\n- MGMT promoter methylation  - The presence of a specific genetic sequence in the tumor known as MGMT promoter methylation has been shown to be a favorable predictive factor and a strong predictor of responsiveness to chemotherapy.\n- Radio genomics - the field of predicting the genetics of the cancer through imaging\n- Types of mpMRI scans:\n    - Fluid Attenuated Inversion Recovery (FLAIR)\n    - T1-weighted pre-contrast (T1w)\n    - T1-weighted post-contrast (T1Gd)\n    - T2-weighted (T2)\n\n\n\n#### Notebooks Referred. \n- https://www.kaggle.com/ayuraj/train-brain-tumor-as-video-classification-w-b\n- https://www.kaggle.com/ihelon/brain-tumor-eda-with-animations-and-modeling\n- https://www.kaggle.com/smoschou55/advanced-eda-brain-tumor-data/comments#Main-Competition-Workflow","metadata":{}},{"cell_type":"code","source":"import os\nimport re \nimport glob\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport seaborn as sns\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# Pydicom related imports\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport SimpleITK as sitk\n\n# Deep learning packages\nimport tensorflow as tf\n\n# For gif creation\nimport imageio\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-11T13:28:21.995659Z","iopub.execute_input":"2021-09-11T13:28:21.996122Z","iopub.status.idle":"2021-09-11T13:28:29.728263Z","shell.execute_reply.started":"2021-09-11T13:28:21.99603Z","shell.execute_reply":"2021-09-11T13:28:29.727127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Visualization\n\nThe training data contains 585 values each corresponding to a patient/subject. \nEach row is marked with target MGMT_value for each subject (BraTS21ID) in the training data (e.g. the presence of MGMT promoter methylation).\nFrom the training set 307 subjects reported presence of MGMT promoter, and 278 reported absence. \nThe imbalance in the training data set is acceptable. \n","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nprint('Number of rows: ', len(train_df))\ntrain_df['MGMT_value'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:28:29.729615Z","iopub.execute_input":"2021-09-11T13:28:29.729933Z","iopub.status.idle":"2021-09-11T13:28:29.765961Z","shell.execute_reply.started":"2021-09-11T13:28:29.7299Z","shell.execute_reply":"2021-09-11T13:28:29.764907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nprint(train_df.MGMT_value.value_counts())\nsns.countplot(data=train_df, x=\"MGMT_value\");","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:28:29.767655Z","iopub.execute_input":"2021-09-11T13:28:29.767963Z","iopub.status.idle":"2021-09-11T13:28:30.116423Z","shell.execute_reply.started":"2021-09-11T13:28:29.767934Z","shell.execute_reply":"2021-09-11T13:28:30.115063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us look at the volume of training data.","metadata":{}},{"cell_type":"code","source":"train_files = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*/*/*')\nprint(f'There are {len(train_files)} dicom files in the training data')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:28:30.117978Z","iopub.execute_input":"2021-09-11T13:28:30.118273Z","iopub.status.idle":"2021-09-11T13:29:46.811354Z","shell.execute_reply.started":"2021-09-11T13:28:30.118243Z","shell.execute_reply":"2021-09-11T13:29:46.810342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_files = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/*/*/*')\nprint(f'There are {len(test_files)} dicom files in the test data')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:46.812613Z","iopub.execute_input":"2021-09-11T13:29:46.812922Z","iopub.status.idle":"2021-09-11T13:29:56.919819Z","shell.execute_reply.started":"2021-09-11T13:29:46.812889Z","shell.execute_reply":"2021-09-11T13:29:56.918686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\ndf_train_labels = df_train_labels.rename(columns={'BraTS21ID': 'PatientId'})\ndf_train_labels['PatientId'] = [format(x, '05d') for x in df_train_labels.PatientId]\ndf_train_labels['PatientId'] = df_train_labels['PatientId'].astype(str)\ndf_train_labels.describe()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:56.921329Z","iopub.execute_input":"2021-09-11T13:29:56.921656Z","iopub.status.idle":"2021-09-11T13:29:56.965646Z","shell.execute_reply.started":"2021-09-11T13:29:56.921623Z","shell.execute_reply":"2021-09-11T13:29:56.964368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*')\nprint(f'There are {len(patients)} patients in the training data')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:56.967125Z","iopub.execute_input":"2021-09-11T13:29:56.967521Z","iopub.status.idle":"2021-09-11T13:29:56.977127Z","shell.execute_reply.started":"2021-09-11T13:29:56.967486Z","shell.execute_reply":"2021-09-11T13:29:56.976017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = glob.glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/*')\nprint(f'There are {len(patients)} patients in the test data')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:56.981594Z","iopub.execute_input":"2021-09-11T13:29:56.981963Z","iopub.status.idle":"2021-09-11T13:29:56.988843Z","shell.execute_reply.started":"2021-09-11T13:29:56.981928Z","shell.execute_reply":"2021-09-11T13:29:56.987803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keys = ['FLAIR', 'T1w', 'T1wCE', 'T2w']\n\nlabel_dict = {\n    'FLAIR': [],\n    'T1w': [],\n    'T1wCE': [],\n    'T2w': []\n}\n\nlabel_dict_counts = {}\n\nfor filename in tqdm(train_files):\n    \n    scan = filename.split('/')[-2]\n    \n    if scan=='FLAIR':\n        label_dict['FLAIR'].append(filename)\n        \n    elif scan=='T1w':\n        label_dict['T1w'].append(filename)\n\n    elif scan=='T1wCE':\n        label_dict['T1wCE'].append(filename)\n\n    else:\n        label_dict['T2w'].append(filename)\n    \nfor key in keys:\n    label_dict_counts[key] = len(label_dict[key])\n\nvalues = label_dict_counts.values()\nsns.barplot(x=keys, y=list(values))","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:56.991366Z","iopub.execute_input":"2021-09-11T13:29:56.992176Z","iopub.status.idle":"2021-09-11T13:29:57.656917Z","shell.execute_reply.started":"2021-09-11T13:29:56.992125Z","shell.execute_reply":"2021-09-11T13:29:57.655885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of files per patient per Key.\ntrain_folders = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/'\ndf_patient_records_train = pd.DataFrame(columns=['PatientId'] + keys)\ndf_patient_records_train.set_index('PatientId')\nfor f in tqdm(os.listdir(train_folders)):\n    patientId = f\n    df_patient_records_train = df_patient_records_train.append({'PatientId': patientId, 'FLAIR': 0, 'T1w': 0, 'T1wCE': 0, 'T2w' : 0}, ignore_index=True)\n    for key in keys:\n        patientId_key_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{patientId}/{key}/*.dcm'\n        df_patient_records_train.loc[df_patient_records_train['PatientId'] == patientId, [key]] = len(glob.glob(patientId_key_path))\ndf_patient_records_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:29:57.658452Z","iopub.execute_input":"2021-09-11T13:29:57.65892Z","iopub.status.idle":"2021-09-11T13:30:06.965035Z","shell.execute_reply.started":"2021-09-11T13:29:57.658858Z","shell.execute_reply":"2021-09-11T13:30:06.96366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of files per patient per Key.\ntest_folders = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/'\ndf_patient_records_test = pd.DataFrame(columns=['PatientId'] + keys)\ndf_patient_records_test.set_index('PatientId')\nfor f in tqdm(os.listdir(test_folders)):\n    patientId = f\n    df_patient_records_test = df_patient_records_test.append({'PatientId': patientId, 'FLAIR': 0, 'T1w': 0, 'T1wCE': 0, 'T2w' : 0}, ignore_index=True)\n    for key in keys:\n        patientId_key_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/{patientId}/{key}/*.dcm'\n        df_patient_records_test.loc[df_patient_records_test['PatientId'] == patientId, [key]] = len(glob.glob(patientId_key_path))\ndf_patient_records_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:06.96766Z","iopub.execute_input":"2021-09-11T13:30:06.968024Z","iopub.status.idle":"2021-09-11T13:30:08.315654Z","shell.execute_reply.started":"2021-09-11T13:30:06.967987Z","shell.execute_reply":"2021-09-11T13:30:08.314418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in keys:\n    df_patient_records_train[key] = df_patient_records_train[key].astype(int)\ndf_patient_records_train['PatientId'] = df_patient_records_train['PatientId'].astype(str)\ndf_patient_records_train[\"TotalFiles\"] = df_patient_records_train[keys].sum(axis=1)\nassert df_patient_records_train.TotalFiles.sum() == len(train_files)\ndf_patient_records_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:08.317061Z","iopub.execute_input":"2021-09-11T13:30:08.31738Z","iopub.status.idle":"2021-09-11T13:30:08.33854Z","shell.execute_reply.started":"2021-09-11T13:30:08.317341Z","shell.execute_reply":"2021-09-11T13:30:08.336439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for key in keys:\n    df_patient_records_test[key] = df_patient_records_test[key].astype(int)\ndf_patient_records_test['PatientId'] = df_patient_records_test['PatientId'].astype(str)\ndf_patient_records_test[\"TotalFiles\"] = df_patient_records_test[keys].sum(axis=1)\nassert df_patient_records_test.TotalFiles.sum() == len(test_files)\ndf_patient_records_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:08.341671Z","iopub.execute_input":"2021-09-11T13:30:08.342029Z","iopub.status.idle":"2021-09-11T13:30:08.365354Z","shell.execute_reply.started":"2021-09-11T13:30:08.341982Z","shell.execute_reply":"2021-09-11T13:30:08.363747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_train = pd.merge(df_patient_records_train, df_train_labels, on=['PatientId'])","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:08.366687Z","iopub.execute_input":"2021-09-11T13:30:08.367137Z","iopub.status.idle":"2021-09-11T13:30:08.387742Z","shell.execute_reply.started":"2021-09-11T13:30:08.367089Z","shell.execute_reply":"2021-09-11T13:30:08.386471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:08.389195Z","iopub.execute_input":"2021-09-11T13:30:08.389831Z","iopub.status.idle":"2021-09-11T13:30:08.406884Z","shell.execute_reply.started":"2021-09-11T13:30:08.389791Z","shell.execute_reply":"2021-09-11T13:30:08.405678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_train.sort_values(by='TotalFiles', ascending=False).head(50)[keys].plot(kind='bar',figsize=(20, 8), stacked=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:08.408264Z","iopub.execute_input":"2021-09-11T13:30:08.408582Z","iopub.status.idle":"2021-09-11T13:30:09.584023Z","shell.execute_reply.started":"2021-09-11T13:30:08.40855Z","shell.execute_reply":"2021-09-11T13:30:09.58285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_test.sort_values(by='TotalFiles', ascending=False).head(50)[keys].plot(kind='bar',figsize=(20, 8), stacked=True)","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:09.585347Z","iopub.execute_input":"2021-09-11T13:30:09.585675Z","iopub.status.idle":"2021-09-11T13:30:10.744235Z","shell.execute_reply.started":"2021-09-11T13:30:09.585642Z","shell.execute_reply":"2021-09-11T13:30:10.743497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxprops = dict(linestyle='-', linewidth=4, color='r')\nmedianprops = dict(linestyle='-', linewidth=4, color='b')\ndf_patient_records_train[keys].plot(kind='box', figsize=(10, 4), showfliers=True, showmeans=True,\n                boxprops=boxprops,\n                medianprops=medianprops)\nplt.suptitle(\"Distribution of files per patient\")\nplt.xlabel(\"Types\")\nplt.ylabel(\"Count of files\")","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:10.74535Z","iopub.execute_input":"2021-09-11T13:30:10.745807Z","iopub.status.idle":"2021-09-11T13:30:10.970727Z","shell.execute_reply.started":"2021-09-11T13:30:10.745762Z","shell.execute_reply":"2021-09-11T13:30:10.969967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The images that belong to T2w are higher in number, the images that belong to T1wcE are lowest in number\n- More outliers observed for T1wCE Kind","metadata":{}},{"cell_type":"code","source":"boxprops = dict(linestyle='-', linewidth=4, color='r')\nmedianprops = dict(linestyle='-', linewidth=4, color='b')\ndf_patient_records_test[keys].plot(kind='box', figsize=(10, 4), showfliers=True, showmeans=True,\n                boxprops=boxprops,\n                medianprops=medianprops)\nplt.suptitle(\"Distribution of files per patient\")\nplt.xlabel(\"Types\")\nplt.ylabel(\"Count of files\")","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:10.971878Z","iopub.execute_input":"2021-09-11T13:30:10.97232Z","iopub.status.idle":"2021-09-11T13:30:11.179467Z","shell.execute_reply.started":"2021-09-11T13:30:10.972279Z","shell.execute_reply":"2021-09-11T13:30:11.178628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxprops = dict(linestyle='-', linewidth=4, color='r')\nmedianprops = dict(linestyle='-', linewidth=4, color='b')\ndf_patient_records_train['TotalFiles'].plot(kind='box', figsize=(8, 5), showfliers=True, showmeans=True,\n                boxprops=boxprops,\n                medianprops=medianprops)\nplt.suptitle(\"Distribution of Total files per patient\")","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.180599Z","iopub.execute_input":"2021-09-11T13:30:11.181079Z","iopub.status.idle":"2021-09-11T13:30:11.336941Z","shell.execute_reply.started":"2021-09-11T13:30:11.181032Z","shell.execute_reply":"2021-09-11T13:30:11.335855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxprops = dict(linestyle='-', linewidth=4, color='r')\nmedianprops = dict(linestyle='-', linewidth=4, color='b')\ndf_patient_records_test['TotalFiles'].plot(kind='box', figsize=(8, 5), showfliers=True, showmeans=True,\n                boxprops=boxprops,\n                medianprops=medianprops)\nplt.suptitle(\"Distribution of Total files per patient\")","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.338243Z","iopub.execute_input":"2021-09-11T13:30:11.338565Z","iopub.status.idle":"2021-09-11T13:30:11.498142Z","shell.execute_reply.started":"2021-09-11T13:30:11.338532Z","shell.execute_reply":"2021-09-11T13:30:11.497023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(pd.DataFrame.from_dict(label_dict_counts, orient='index')/len(train_files)*100, 2).plot(kind='bar')\nplt.suptitle(\"Percentage Data by Type\")\nplt.xlabel(\"Types\")\nplt.ylabel(\"Percentage\")","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.499524Z","iopub.execute_input":"2021-09-11T13:30:11.49995Z","iopub.status.idle":"2021-09-11T13:30:11.704553Z","shell.execute_reply.started":"2021-09-11T13:30:11.499915Z","shell.execute_reply":"2021-09-11T13:30:11.703414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_train.describe()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.708155Z","iopub.execute_input":"2021-09-11T13:30:11.70868Z","iopub.status.idle":"2021-09-11T13:30:11.745128Z","shell.execute_reply.started":"2021-09-11T13:30:11.708638Z","shell.execute_reply":"2021-09-11T13:30:11.744084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read DICOM images","metadata":{}},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px\ndef ReadMRI(path, voi_lut = True, fix_monochrome = True):\n    \n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        if (im.size != (size, size)):\n            im = im.resize((size, size), resample)\n    return im","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.746716Z","iopub.execute_input":"2021-09-11T13:30:11.747015Z","iopub.status.idle":"2021-09-11T13:30:11.756262Z","shell.execute_reply.started":"2021-09-11T13:30:11.746985Z","shell.execute_reply":"2021-09-11T13:30:11.754812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = ReadMRI(train_files[1])\nprint('Shape of data: ', data.shape)\nplt.rcdefaults()\nplt.figure(figsize=(5, 5))\nplt.imshow(data, cmap='gray');","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:30:11.758106Z","iopub.execute_input":"2021-09-11T13:30:11.758562Z","iopub.status.idle":"2021-09-11T13:30:12.042057Z","shell.execute_reply.started":"2021-09-11T13:30:11.758514Z","shell.execute_reply":"2021-09-11T13:30:12.040864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Animate MRI images for a patient","metadata":{}},{"cell_type":"code","source":"patientIds = os.listdir('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train')\npatientId = np.random.choice(patientIds)\nkey = np.random.choice(keys)\n\noutput_dir_path_train = '/kaggle/working/output/images/train' \nos.makedirs(output_dir_path_train, exist_ok=True)\n\noutput_dir_path_test = '/kaggle/working/output/images/test' \nos.makedirs(output_dir_path_test, exist_ok=True)\n\ndef convert_dicom_to_png(patientId, key, ds_type = 'train'):\n    if ds_type == 'train':\n        mgmt_value = df_patient_records_train.loc[df_patient_records_train['PatientId'] == patientId][\"MGMT_value\"].item()\n    files_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/{ds_type}/{patientId}/{key}/*.dcm'\n#     print(len(files_path))\n    for file in glob.glob(files_path):\n        file_name = file.split('/')[-1].split('.')[0]\n        img_data = ReadMRI(file)\n        # skipping blank images\n        if (np.count_nonzero(img_data) > 0):\n            img_data = resize(img_data, size=224)\n            if \"train\" == ds_type:\n                os.makedirs(f'{output_dir_path_train}/{patientId}/{key}', exist_ok=True)\n                img_data.save(f'{output_dir_path_train}/{patientId}/{key}/{file_name}-{mgmt_value}.png')\n            else:\n                os.makedirs(f'{output_dir_path_test}/{patientId}/{key}', exist_ok=True)\n                img_data.save(f'{output_dir_path_test}/{patientId}/{key}/{file_name}.png')\n\nconvert_dicom_to_png(patientId, key)","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:34:11.237784Z","iopub.execute_input":"2021-09-11T13:34:11.238175Z","iopub.status.idle":"2021-09-11T13:34:11.991869Z","shell.execute_reply.started":"2021-09-11T13:34:11.238138Z","shell.execute_reply":"2021-09-11T13:34:11.99098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"anim_file = 'brain_scan.gif'\nwith imageio.get_writer(anim_file, mode='I') as writer:\n    filenames = glob.glob(f'{output_dir_path_train}/{patientId}/{key}/Image*.png')\n    filenames = sorted(filenames)\n    for filename in filenames:\n        image = imageio.imread(filename)\n        writer.append_data(image)","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:33:23.406913Z","iopub.execute_input":"2021-09-11T13:33:23.407262Z","iopub.status.idle":"2021-09-11T13:33:25.212824Z","shell.execute_reply.started":"2021-09-11T13:33:23.407222Z","shell.execute_reply":"2021-09-11T13:33:25.211847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/tensorflow/docs","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:33:30.383521Z","iopub.execute_input":"2021-09-11T13:33:30.383953Z","iopub.status.idle":"2021-09-11T13:33:50.503514Z","shell.execute_reply.started":"2021-09-11T13:33:30.383913Z","shell.execute_reply":"2021-09-11T13:33:50.502206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_docs.vis.embed as embed\nprint(f'Showing Animated gif for patient: {patientId}, for key: {key}')\nembed.embed_file(anim_file)","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:34:38.904399Z","iopub.execute_input":"2021-09-11T13:34:38.904821Z","iopub.status.idle":"2021-09-11T13:34:39.008891Z","shell.execute_reply.started":"2021-09-11T13:34:38.904787Z","shell.execute_reply":"2021-09-11T13:34:39.007572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize Images per type","metadata":{}},{"cell_type":"code","source":"patient_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{patientId}/{key}'\nfor p in list(df_patient_records_train.sample(n=5).PatientId):\n    for i, key in enumerate(keys, 1):\n        patient_path = f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{p}/'\n        t_paths = sorted(glob.glob(os.path.join(patient_path, key, \"*\")), key=lambda x: int(x[:-4].split(\"-\")[-1]))\n        data = ReadMRI(t_paths[int(len(t_paths)*0.5)])\n        plt.subplot(1, 4, i)\n        plt.imshow(data, cmap=\"gray\")\n        plt.title(f\"{key}\", fontsize=12)\n        plt.axis(\"off\")\n    mgmt_value = df_patient_records_train.loc[df_patient_records_train['PatientId'] == p][\"MGMT_value\"]\n    plt.suptitle(f\"MGMT_value: {mgmt_value.item()}, patient Id: {p}\", fontsize=12)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:34:58.385861Z","iopub.execute_input":"2021-09-11T13:34:58.386249Z","iopub.status.idle":"2021-09-11T13:35:00.066941Z","shell.execute_reply.started":"2021-09-11T13:34:58.386215Z","shell.execute_reply":"2021-09-11T13:35:00.065821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert DICOM to Images","metadata":{}},{"cell_type":"code","source":"for patientId in tqdm(list(df_patient_records_train.PatientId)[:100]):\n    for key in keys:\n        convert_dicom_to_png(patientId, key)\n        \n        \nfor patientId in tqdm(list(df_patient_records_test.PatientId)[:100]):\n    for key in keys:\n        convert_dicom_to_png(patientId, key, 'test')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:35:04.705943Z","iopub.execute_input":"2021-09-11T13:35:04.706315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_patient_records_train.to_csv(f'{output_dir_path_train}/train.csv')\ndf_patient_records_test.to_csv(f'{output_dir_path_test}/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:33:50.552099Z","iopub.status.idle":"2021-09-11T13:33:50.55251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!mkdir /kaggle/tmp\n!tar -zcf train.tar.gz -C \"./output/images/train\" .\n!tar -zcf test.tar.gz -C \"./output/images/test\" .\n!rm -r ./output","metadata":{"execution":{"iopub.status.busy":"2021-09-11T13:33:50.553326Z","iopub.status.idle":"2021-09-11T13:33:50.553752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}