{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Prepare Enviroment","metadata":{}},{"cell_type":"code","source":"pip install pydicom","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:32.415099Z","iopub.execute_input":"2022-02-17T04:22:32.415471Z","iopub.status.idle":"2022-02-17T04:22:43.607367Z","shell.execute_reply.started":"2022-02-17T04:22:32.415433Z","shell.execute_reply":"2022-02-17T04:22:43.606354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\n\nimport pydicom\nfrom pydicom import dcmread","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.609863Z","iopub.execute_input":"2022-02-17T04:22:43.610178Z","iopub.status.idle":"2022-02-17T04:22:43.806993Z","shell.execute_reply.started":"2022-02-17T04:22:43.610139Z","shell.execute_reply":"2022-02-17T04:22:43.805875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load DataFrame","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')\nprint('The shape of the training data is:', train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.808410Z","iopub.execute_input":"2022-02-17T04:22:43.808660Z","iopub.status.idle":"2022-02-17T04:22:43.831584Z","shell.execute_reply.started":"2022-02-17T04:22:43.808632Z","shell.execute_reply":"2022-02-17T04:22:43.830605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.834495Z","iopub.execute_input":"2022-02-17T04:22:43.834837Z","iopub.status.idle":"2022-02-17T04:22:43.856037Z","shell.execute_reply.started":"2022-02-17T04:22:43.834789Z","shell.execute_reply":"2022-02-17T04:22:43.855380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['dir'] = train.BraTS21ID.map(lambda x : f'{x:05}')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.857587Z","iopub.execute_input":"2022-02-17T04:22:43.857819Z","iopub.status.idle":"2022-02-17T04:22:43.873329Z","shell.execute_reply.started":"2022-02-17T04:22:43.857791Z","shell.execute_reply":"2022-02-17T04:22:43.872564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Distribution","metadata":{}},{"cell_type":"code","source":"#Dataframe showing the proportion of observations with each target variable\nprint((train.MGMT_value.value_counts() /len(train)).to_frame().T)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.874673Z","iopub.execute_input":"2022-02-17T04:22:43.874944Z","iopub.status.idle":"2022-02-17T04:22:43.899300Z","shell.execute_reply.started":"2022-02-17T04:22:43.874898Z","shell.execute_reply":"2022-02-17T04:22:43.898349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Add semicolon to end of code to get rid of header description\ntrain.MGMT_value.value_counts().plot(kind='bar', xlabel='MGMT_value', ylabel='Count', \n                                     color=['#1E90FF', '#00C957'], edgecolor='black');","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:43.900933Z","iopub.execute_input":"2022-02-17T04:22:43.901412Z","iopub.status.idle":"2022-02-17T04:22:44.330065Z","shell.execute_reply.started":"2022-02-17T04:22:43.901355Z","shell.execute_reply":"2022-02-17T04:22:44.329133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Determining the Number of Images for Each Modality Type","metadata":{}},{"cell_type":"code","source":"train_path = ('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train')\n\nmodes = ['FLAIR', 'T1w', 'T1wCE', 'T2w']","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:44.331557Z","iopub.execute_input":"2022-02-17T04:22:44.331815Z","iopub.status.idle":"2022-02-17T04:22:44.336903Z","shell.execute_reply.started":"2022-02-17T04:22:44.331783Z","shell.execute_reply":"2022-02-17T04:22:44.336092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_counts = {m : [] for m in modes}\n\nfor i in tqdm(train.index):\n    row = train.loc[i, :]\n    folder = row['dir']\n    \n    for m in modes:\n        temp = os.listdir(f'{train_path}/{folder}/{m}')\n        print(temp)\n        image_counts[m].append(len(temp))\n        \nfor m in modes:\n    train[m] = image_counts[m]\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T04:22:44.338253Z","iopub.execute_input":"2022-02-17T04:22:44.338966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Histogram of Distributions for Each Modality","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}