{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport os\nimport matplotlib.pyplot as plt\nplt.style.use('dark_background')\n%matplotlib inline","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-07-23T10:51:03.087575Z","iopub.status.busy":"2021-07-23T10:51:03.085787Z","iopub.status.idle":"2021-07-23T10:51:03.100390Z","shell.execute_reply":"2021-07-23T10:51:03.100889Z","shell.execute_reply.started":"2021-07-23T09:44:53.111027Z"},"papermill":{"duration":0.058799,"end_time":"2021-07-23T10:51:03.101198","exception":false,"start_time":"2021-07-23T10:51:03.042399","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combining to a Single DataFrame\n\nOkay Yeah!    \nLet me be pretty honest about this : I want **ONE** DataFrame.  \nJust One Good Big DF that has everything in it and I don't have to fumble accross all dfs to lookup again.    \nSo let's just get on with it!","metadata":{"papermill":{"duration":0.036615,"end_time":"2021-07-23T10:51:03.175468","exception":false,"start_time":"2021-07-23T10:51:03.138853","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sub_pth = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\"\ndf_sub = pd.read_csv(sub_pth)\ndf_sub.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:51:03.255820Z","iopub.status.busy":"2021-07-23T10:51:03.254864Z","iopub.status.idle":"2021-07-23T10:51:03.387042Z","shell.execute_reply":"2021-07-23T10:51:03.387522Z","shell.execute_reply.started":"2021-07-23T08:58:40.719324Z"},"papermill":{"duration":0.173879,"end_time":"2021-07-23T10:51:03.387703","exception":false,"start_time":"2021-07-23T10:51:03.213824","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get Paths : Directory Structure\n\n```\nTest/Train Directory\n    Study_ID_Number - 00001, ....\n        FLAIR\n        T1w\n        T1wCE\n        T2w\n            Image-{Img_ID}.dcm\n```","metadata":{"papermill":{"duration":0.037783,"end_time":"2021-07-23T10:51:03.463757","exception":false,"start_time":"2021-07-23T10:51:03.425974","status":"completed"},"tags":[]}},{"cell_type":"code","source":"PARENT_DIRS = [\"test\",\"train\"]\nCHILD_DIRS = [\"FLAIR\", \"T1w\",\"T1wCE\",\"T2w\"]\n\nSplit_Types = []\nStudy_IDs = []\nTumour_Types = []\nImage_IDs = []\nAbsolute_Paths = []\nIMG_FORMAT = \".dcm\"","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:51:03.546076Z","iopub.status.busy":"2021-07-23T10:51:03.545381Z","iopub.status.idle":"2021-07-23T10:51:03.547253Z","shell.execute_reply":"2021-07-23T10:51:03.547749Z","shell.execute_reply.started":"2021-07-23T08:58:40.838763Z"},"papermill":{"duration":0.045702,"end_time":"2021-07-23T10:51:03.547915","exception":false,"start_time":"2021-07-23T10:51:03.502213","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"File Absolute Paths will be of the following format :\n```\nSplit_Type/Study_ID/Tumour_Type/Image_ID.dcm\n```","metadata":{"papermill":{"duration":0.037672,"end_time":"2021-07-23T10:51:03.624108","exception":false,"start_time":"2021-07-23T10:51:03.586436","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def splitall(path):\n    # https://www.oreilly.com/library/view/python-cookbook/0596001673/ch04s16.html\n    allparts = []\n    while 1:\n        parts = os.path.split(path)\n        if parts[0] == path:  # sentinel for absolute paths\n            allparts.insert(0, parts[0])\n            break\n        elif parts[1] == path: # sentinel for relative paths\n            allparts.insert(0, parts[1])\n            break\n        else:\n            path = parts[0]\n            allparts.insert(0, parts[1])\n    return allparts\nsplitall(\"Split_Type/Study_ID/Tumour_Type/Image_ID.dcm\")","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:51:03.702981Z","iopub.status.busy":"2021-07-23T10:51:03.702337Z","iopub.status.idle":"2021-07-23T10:51:03.711897Z","shell.execute_reply":"2021-07-23T10:51:03.711243Z","shell.execute_reply.started":"2021-07-23T08:58:40.848012Z"},"papermill":{"duration":0.05025,"end_time":"2021-07-23T10:51:03.712041","exception":false,"start_time":"2021-07-23T10:51:03.661791","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_count = 400*1000 # from kaggle files counter\nDATA_FOLDER = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\n\nwith tqdm(total=file_count) as pbar:\n    for path, directories, files in os.walk(DATA_FOLDER):\n         for file in files:\n                if file.endswith(IMG_FORMAT):\n                    pbar.update(1)\n                    abs_path = os.path.join(path, file)\n                    Image_ID = os.path.basename(abs_path)\n                    Splitted_Path = splitall(abs_path)\n                    Tumour_Type =  Splitted_Path[-2]\n                    Study_ID = Splitted_Path[-3]\n                    Split_Type = Splitted_Path[-4]\n\n                    Split_Types.append(Split_Type)\n                    Study_IDs.append(Study_ID)\n                    Tumour_Types.append(Tumour_Type)\n                    Image_IDs.append(Image_ID)\n                    Absolute_Paths.append(abs_path)","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:51:03.793141Z","iopub.status.busy":"2021-07-23T10:51:03.792448Z","iopub.status.idle":"2021-07-23T10:52:28.165197Z","shell.execute_reply":"2021-07-23T10:52:28.165716Z","shell.execute_reply.started":"2021-07-23T08:58:40.864447Z"},"papermill":{"duration":84.414558,"end_time":"2021-07-23T10:52:28.165899","exception":false,"start_time":"2021-07-23T10:51:03.751341","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext = pd.DataFrame.from_dict({\"Split_Type\":Split_Types,\n                             \"Study_ID\":Study_IDs,\n                             \"Tumour_Type\":Tumour_Types,\n                             \"Image_ID\": Image_IDs,\n                             \"Absolute_Path\":Absolute_Paths})\ndf_ext.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:28.714460Z","iopub.status.busy":"2021-07-23T10:52:28.669448Z","iopub.status.idle":"2021-07-23T10:52:29.115010Z","shell.execute_reply":"2021-07-23T10:52:29.114444Z","shell.execute_reply.started":"2021-07-23T08:58:52.967542Z"},"papermill":{"duration":0.704628,"end_time":"2021-07-23T10:52:29.115148","exception":false,"start_time":"2021-07-23T10:52:28.410520","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pth = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\"\ndf_train = pd.read_csv(train_pth)\ndf_train.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:29.609470Z","iopub.status.busy":"2021-07-23T10:52:29.608861Z","iopub.status.idle":"2021-07-23T10:52:29.624158Z","shell.execute_reply":"2021-07-23T10:52:29.623555Z","shell.execute_reply.started":"2021-07-23T08:58:53.406683Z"},"papermill":{"duration":0.263454,"end_time":"2021-07-23T10:52:29.624292","exception":false,"start_time":"2021-07-23T10:52:29.360838","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:30.122924Z","iopub.status.busy":"2021-07-23T10:52:30.122243Z","iopub.status.idle":"2021-07-23T10:52:30.126348Z","shell.execute_reply":"2021-07-23T10:52:30.125812Z","shell.execute_reply.started":"2021-07-23T08:58:53.418887Z"},"papermill":{"duration":0.256757,"end_time":"2021-07-23T10:52:30.126484","exception":false,"start_time":"2021-07-23T10:52:29.869727","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df_ext[df_ext[\"Split_Type\"]==\"train\"]\ntrain_df.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:30.656078Z","iopub.status.busy":"2021-07-23T10:52:30.655335Z","iopub.status.idle":"2021-07-23T10:52:30.789262Z","shell.execute_reply":"2021-07-23T10:52:30.788702Z","shell.execute_reply.started":"2021-07-23T08:58:53.426926Z"},"papermill":{"duration":0.382746,"end_time":"2021-07-23T10:52:30.789402","exception":false,"start_time":"2021-07-23T10:52:30.406656","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_1 = set(list(df_train[\"BraTS21ID\"]))\nset_2 = set(list(map(int,train_df[\"Study_ID\"])))\n\nset_1==set_2","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:31.415415Z","iopub.status.busy":"2021-07-23T10:52:31.414784Z","iopub.status.idle":"2021-07-23T10:52:31.418495Z","shell.execute_reply":"2021-07-23T10:52:31.418954Z","shell.execute_reply.started":"2021-07-23T08:58:53.564250Z"},"papermill":{"duration":0.382032,"end_time":"2021-07-23T10:52:31.419123","exception":false,"start_time":"2021-07-23T10:52:31.037091","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insert Ground Truth Values into the DataFrame","metadata":{"papermill":{"duration":0.247154,"end_time":"2021-07-23T10:52:31.913755","exception":false,"start_time":"2021-07-23T10:52:31.666601","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_ext[\"Ground_Truth_Class\"] = -1","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:32.413996Z","iopub.status.busy":"2021-07-23T10:52:32.413339Z","iopub.status.idle":"2021-07-23T10:52:32.417796Z","shell.execute_reply":"2021-07-23T10:52:32.417177Z","shell.execute_reply.started":"2021-07-23T08:58:53.691478Z"},"papermill":{"duration":0.255934,"end_time":"2021-07-23T10:52:32.417931","exception":false,"start_time":"2021-07-23T10:52:32.161997","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in tqdm(df_ext.iterrows(),total=400114):\n    i = 0\n    # print(int(row['Study_ID']))\n    study_id = int(row['Study_ID'])\n    single_df = df_train[df_train[\"BraTS21ID\"] == study_id]\n    # print(single_df.shape)\n    if not single_df.shape[0]==0:\n        label_val = single_df[\"MGMT_value\"].values.flatten()[0]\n        df_ext.loc[index,\"Ground_Truth_Class\"] = label_val","metadata":{"execution":{"iopub.execute_input":"2021-07-23T10:52:32.918167Z","iopub.status.busy":"2021-07-23T10:52:32.917543Z","iopub.status.idle":"2021-07-23T11:05:10.334395Z","shell.execute_reply":"2021-07-23T11:05:10.334928Z","shell.execute_reply.started":"2021-07-23T09:14:30.568960Z"},"papermill":{"duration":757.671276,"end_time":"2021-07-23T11:05:10.335139","exception":false,"start_time":"2021-07-23T10:52:32.663863","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"MGMT_value\"].value_counts()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:15.242069Z","iopub.status.busy":"2021-07-23T11:05:15.241315Z","iopub.status.idle":"2021-07-23T11:05:15.245174Z","shell.execute_reply":"2021-07-23T11:05:15.244488Z","shell.execute_reply.started":"2021-07-23T09:27:19.113044Z"},"papermill":{"duration":2.488432,"end_time":"2021-07-23T11:05:15.245315","exception":false,"start_time":"2021-07-23T11:05:12.756883","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"Ground_Truth_Class\"].value_counts()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:20.117652Z","iopub.status.busy":"2021-07-23T11:05:20.116997Z","iopub.status.idle":"2021-07-23T11:05:20.126126Z","shell.execute_reply":"2021-07-23T11:05:20.126666Z","shell.execute_reply.started":"2021-07-23T09:27:19.126107Z"},"papermill":{"duration":2.447813,"end_time":"2021-07-23T11:05:20.126831","exception":false,"start_time":"2021-07-23T11:05:17.679018","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `len(Images)` per Study","metadata":{"papermill":{"duration":2.424711,"end_time":"2021-07-23T11:05:25.050118","exception":false,"start_time":"2021-07-23T11:05:22.625407","status":"completed"},"tags":[]}},{"cell_type":"code","source":"images_per_study = df_ext.groupby(['Study_ID']).size()\nimages_per_study","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:29.980785Z","iopub.status.busy":"2021-07-23T11:05:29.957789Z","iopub.status.idle":"2021-07-23T11:05:30.016030Z","shell.execute_reply":"2021-07-23T11:05:30.015462Z","shell.execute_reply.started":"2021-07-23T09:41:11.738955Z"},"papermill":{"duration":2.553422,"end_time":"2021-07-23T11:05:30.016183","exception":false,"start_time":"2021-07-23T11:05:27.462761","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Insights & Visualizations","metadata":{"papermill":{"duration":2.45975,"end_time":"2021-07-23T11:05:34.934759","exception":false,"start_time":"2021-07-23T11:05:32.475009","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_ext.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:39.794089Z","iopub.status.busy":"2021-07-23T11:05:39.793183Z","iopub.status.idle":"2021-07-23T11:05:39.797194Z","shell.execute_reply":"2021-07-23T11:05:39.797677Z","shell.execute_reply.started":"2021-07-23T09:41:14.437772Z"},"papermill":{"duration":2.447644,"end_time":"2021-07-23T11:05:39.797849","exception":false,"start_time":"2021-07-23T11:05:37.350205","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.describe()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:44.861262Z","iopub.status.busy":"2021-07-23T11:05:44.860549Z","iopub.status.idle":"2021-07-23T11:05:44.886439Z","shell.execute_reply":"2021-07-23T11:05:44.886986Z","shell.execute_reply.started":"2021-07-23T09:41:15.367831Z"},"papermill":{"duration":2.527432,"end_time":"2021-07-23T11:05:44.887158","exception":false,"start_time":"2021-07-23T11:05:42.359726","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.info()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:49.760238Z","iopub.status.busy":"2021-07-23T11:05:49.759580Z","iopub.status.idle":"2021-07-23T11:05:49.993577Z","shell.execute_reply":"2021-07-23T11:05:49.994067Z","shell.execute_reply.started":"2021-07-23T09:41:16.637731Z"},"papermill":{"duration":2.709257,"end_time":"2021-07-23T11:05:49.994259","exception":false,"start_time":"2021-07-23T11:05:47.285002","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### More About Columns","metadata":{"papermill":{"duration":2.431069,"end_time":"2021-07-23T11:05:54.841190","exception":false,"start_time":"2021-07-23T11:05:52.410121","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_ext.head()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:05:59.721784Z","iopub.status.busy":"2021-07-23T11:05:59.720881Z","iopub.status.idle":"2021-07-23T11:05:59.724622Z","shell.execute_reply":"2021-07-23T11:05:59.725108Z","shell.execute_reply.started":"2021-07-23T09:41:18.128912Z"},"papermill":{"duration":2.418439,"end_time":"2021-07-23T11:05:59.725274","exception":false,"start_time":"2021-07-23T11:05:57.306835","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df_ext[df_ext[\"Split_Type\"]==\"train\"]\ntrain_df.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:04.644292Z","iopub.status.busy":"2021-07-23T11:06:04.643300Z","iopub.status.idle":"2021-07-23T11:06:04.790678Z","shell.execute_reply":"2021-07-23T11:06:04.791163Z","shell.execute_reply.started":"2021-07-23T09:42:28.044532Z"},"papermill":{"duration":2.644585,"end_time":"2021-07-23T11:06:04.791328","exception":false,"start_time":"2021-07-23T11:06:02.146743","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:09.637400Z","iopub.status.busy":"2021-07-23T11:06:09.636710Z","iopub.status.idle":"2021-07-23T11:06:10.791610Z","shell.execute_reply":"2021-07-23T11:06:10.790909Z"},"papermill":{"duration":3.569636,"end_time":"2021-07-23T11:06:10.791775","exception":false,"start_time":"2021-07-23T11:06:07.222139","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_count_viz(data, column,title = \"Distribution Count\",figure_size= (20,4)):\n    print(\"Absolute Value Counts :\")\n    print(data[column].value_counts())\n    print(\"Normalized (Percentage) Value Counts :\")\n    print(data[column].value_counts(normalize=True))\n    plt.figure(figsize=figure_size)\n    ax = sns.countplot(data=data, y=column)\n    ax.set_title(title)\n    plt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:15.782982Z","iopub.status.busy":"2021-07-23T11:06:15.782323Z","iopub.status.idle":"2021-07-23T11:06:15.785425Z","shell.execute_reply":"2021-07-23T11:06:15.784916Z","shell.execute_reply.started":"2021-07-23T09:54:49.429975Z"},"papermill":{"duration":2.466482,"end_time":"2021-07-23T11:06:15.785586","exception":false,"start_time":"2021-07-23T11:06:13.319104","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_count_viz(data = train_df, column = \"Ground_Truth_Class\",title = \"Label Distribution Count\")","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:20.700895Z","iopub.status.busy":"2021-07-23T11:06:20.700205Z","iopub.status.idle":"2021-07-23T11:06:20.963486Z","shell.execute_reply":"2021-07-23T11:06:20.962910Z","shell.execute_reply.started":"2021-07-23T09:54:49.819688Z"},"papermill":{"duration":2.744685,"end_time":"2021-07-23T11:06:20.963648","exception":false,"start_time":"2021-07-23T11:06:18.218963","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_count_viz(data = df_ext, column = \"Split_Type\",title = \"Train/Test Distribution Count\")","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:25.995274Z","iopub.status.busy":"2021-07-23T11:06:25.994538Z","iopub.status.idle":"2021-07-23T11:06:26.896476Z","shell.execute_reply":"2021-07-23T11:06:26.895920Z","shell.execute_reply.started":"2021-07-23T09:54:52.950623Z"},"papermill":{"duration":3.485165,"end_time":"2021-07-23T11:06:26.896643","exception":false,"start_time":"2021-07-23T11:06:23.411478","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize How Many Images Per Study","metadata":{"papermill":{"duration":2.437758,"end_time":"2021-07-23T11:06:31.773764","exception":false,"start_time":"2021-07-23T11:06:29.336006","status":"completed"},"tags":[]}},{"cell_type":"code","source":"per_study = dict(images_per_study)","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:36.691404Z","iopub.status.busy":"2021-07-23T11:06:36.690723Z","iopub.status.idle":"2021-07-23T11:06:36.694129Z","shell.execute_reply":"2021-07-23T11:06:36.693455Z","shell.execute_reply.started":"2021-07-23T09:45:13.951450Z"},"papermill":{"duration":2.445901,"end_time":"2021-07-23T11:06:36.694269","exception":false,"start_time":"2021-07-23T11:06:34.248368","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,12))\nplt.plot(list(per_study.keys()), list(per_study.values()))\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:41.967274Z","iopub.status.busy":"2021-07-23T11:06:41.783951Z","iopub.status.idle":"2021-07-23T11:06:48.428786Z","shell.execute_reply":"2021-07-23T11:06:48.428154Z","shell.execute_reply.started":"2021-07-23T09:45:20.431087Z"},"papermill":{"duration":9.284503,"end_time":"2021-07-23T11:06:48.428931","exception":false,"start_time":"2021-07-23T11:06:39.144428","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution Plot\nWell, That's a bit too ragged, let's try breaking it down to a frequency-range plot","metadata":{"papermill":{"duration":2.42412,"end_time":"2021-07-23T11:06:53.286410","exception":false,"start_time":"2021-07-23T11:06:50.862290","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:06:58.194782Z","iopub.status.busy":"2021-07-23T11:06:58.193791Z","iopub.status.idle":"2021-07-23T11:06:58.197095Z","shell.execute_reply":"2021-07-23T11:06:58.196574Z","shell.execute_reply.started":"2021-07-23T09:45:27.201806Z"},"papermill":{"duration":2.43249,"end_time":"2021-07-23T11:06:58.197240","exception":false,"start_time":"2021-07-23T11:06:55.764750","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = list(per_study.values())\nplt.figure(figsize=(20,12))\n\nsns.distplot(data,bins=\"doane\",kde=True,hist_kws={\"align\" : \"left\"})\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:03.163658Z","iopub.status.busy":"2021-07-23T11:07:03.151460Z","iopub.status.idle":"2021-07-23T11:07:03.583874Z","shell.execute_reply":"2021-07-23T11:07:03.583202Z","shell.execute_reply.started":"2021-07-23T09:45:27.208331Z"},"papermill":{"duration":2.931168,"end_time":"2021-07-23T11:07:03.584014","exception":false,"start_time":"2021-07-23T11:07:00.652846","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Helper Functions\n\nWe need these functions to preprocess the dcm images to the array because of which we use the `pydicom API`. This uses the `gdcm` library, which has to be installed beforehand.  \nHowever using this tool isn't easy because\n1. Pydicom isn't previously installed on Kaggle Notebooks.\n2. For Submissions, we need notebooks that run offline, so we can't use the internet to perform `pip install`.\n\n**The Solution** - Using Offline Installation as Dataset from a different Notebook. This approach is being used here.","metadata":{"papermill":{"duration":2.434161,"end_time":"2021-07-23T11:07:08.479792","exception":false,"start_time":"2021-07-23T11:07:06.045631","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from tqdm import tqdm\n\n# Pydicom related imports\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n# Reference: https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px\n# and https://www.kaggle.com/ayuraj/brain-tumor-eda-and-interactive-viz-with-w-b\ndef ReadMRI(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:13.434483Z","iopub.status.busy":"2021-07-23T11:07:13.433807Z","iopub.status.idle":"2021-07-23T11:07:13.602855Z","shell.execute_reply":"2021-07-23T11:07:13.603334Z","shell.execute_reply.started":"2021-07-23T09:55:57.154412Z"},"papermill":{"duration":2.609674,"end_time":"2021-07-23T11:07:13.603533","exception":false,"start_time":"2021-07-23T11:07:10.993859","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(42)","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:18.599026Z","iopub.status.busy":"2021-07-23T11:07:18.598343Z","iopub.status.idle":"2021-07-23T11:07:18.603416Z","shell.execute_reply":"2021-07-23T11:07:18.603895Z","shell.execute_reply.started":"2021-07-23T10:22:34.201284Z"},"papermill":{"duration":2.520854,"end_time":"2021-07-23T11:07:18.604073","exception":false,"start_time":"2021-07-23T11:07:16.083219","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def props(img):\n    print(\"Shape :\",img.shape,\"Maximum :\",img.max(),\"Minimum :\",img.min())\n\ndef view_data(sample_set,label_col=\"Study_ID\",path_col = \"Absolute_Path\",figure_size = (20,20),ht = 5,wd = 4):\n    n = ht * wd\n    fig, axs = plt.subplots(wd, ht, figsize=figure_size)\n    fig.subplots_adjust(hspace=.2, wspace=.2)\n    axs = axs.ravel()\n    sample_set = sample_set.reindex(np.random.permutation(sample_set.index))\n    sample_set.reset_index(drop=True, inplace=True)\n    i = 0\n    plots_done = 0\n    while plots_done<20:\n    # for i in range(n):\n        img_path = sample_set.loc[i,path_col]\n        img =  ReadMRI(img_path)  \n        # props(img)\n        if not img.max()==0:\n            axs[plots_done].imshow(img,cmap=plt.cm.gist_ncar)\n            axs[plots_done].set_title(sample_set.loc[i,label_col])\n            plots_done+=1\n        i+=1","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:23.555147Z","iopub.status.busy":"2021-07-23T11:07:23.554111Z","iopub.status.idle":"2021-07-23T11:07:23.557167Z","shell.execute_reply":"2021-07-23T11:07:23.556441Z","shell.execute_reply.started":"2021-07-23T10:30:20.573342Z"},"papermill":{"duration":2.519106,"end_time":"2021-07-23T11:07:23.557324","exception":false,"start_time":"2021-07-23T11:07:21.038218","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df_ext[df_ext[\"Split_Type\"]==\"train\"]\ntrain_df.shape","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:28.462742Z","iopub.status.busy":"2021-07-23T11:07:28.461579Z","iopub.status.idle":"2021-07-23T11:07:28.548079Z","shell.execute_reply":"2021-07-23T11:07:28.547405Z","shell.execute_reply.started":"2021-07-23T10:30:03.858664Z"},"papermill":{"duration":2.578787,"end_time":"2021-07-23T11:07:28.548237","exception":false,"start_time":"2021-07-23T11:07:25.969450","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View MRI Scans","metadata":{"papermill":{"duration":2.463218,"end_time":"2021-07-23T11:07:33.489460","exception":false,"start_time":"2021-07-23T11:07:31.026242","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Train Set - Positive Classes","metadata":{"papermill":{"duration":2.479709,"end_time":"2021-07-23T11:07:38.396979","exception":false,"start_time":"2021-07-23T11:07:35.917270","status":"completed"},"tags":[]}},{"cell_type":"code","source":"view_data(train_df[train_df[\"Ground_Truth_Class\"]==1])","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:43.277340Z","iopub.status.busy":"2021-07-23T11:07:43.276624Z","iopub.status.idle":"2021-07-23T11:07:47.491135Z","shell.execute_reply":"2021-07-23T11:07:47.491696Z","shell.execute_reply.started":"2021-07-23T10:30:05.190797Z"},"papermill":{"duration":6.66513,"end_time":"2021-07-23T11:07:47.491873","exception":false,"start_time":"2021-07-23T11:07:40.826743","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Set - Negative Classes","metadata":{"papermill":{"duration":2.486687,"end_time":"2021-07-23T11:07:52.551266","exception":false,"start_time":"2021-07-23T11:07:50.064579","status":"completed"},"tags":[]}},{"cell_type":"code","source":"view_data(train_df[train_df[\"Ground_Truth_Class\"]==0])","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:07:57.542323Z","iopub.status.busy":"2021-07-23T11:07:57.541615Z","iopub.status.idle":"2021-07-23T11:08:01.719773Z","shell.execute_reply":"2021-07-23T11:08:01.720314Z","shell.execute_reply.started":"2021-07-23T10:43:01.944328Z"},"papermill":{"duration":6.690468,"end_time":"2021-07-23T11:08:01.720494","exception":false,"start_time":"2021-07-23T11:07:55.030026","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test Set","metadata":{"papermill":{"duration":2.501462,"end_time":"2021-07-23T11:08:06.701218","exception":false,"start_time":"2021-07-23T11:08:04.199756","status":"completed"},"tags":[]}},{"cell_type":"code","source":"view_data(df_ext[df_ext[\"Ground_Truth_Class\"]==-1])","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:08:11.720955Z","iopub.status.busy":"2021-07-23T11:08:11.720272Z","iopub.status.idle":"2021-07-23T11:08:15.612865Z","shell.execute_reply":"2021-07-23T11:08:15.613368Z","shell.execute_reply.started":"2021-07-23T10:43:51.869188Z"},"papermill":{"duration":6.390399,"end_time":"2021-07-23T11:08:15.613535","exception":false,"start_time":"2021-07-23T11:08:09.223136","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.columns","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:08:20.674234Z","iopub.status.busy":"2021-07-23T11:08:20.673491Z","iopub.status.idle":"2021-07-23T11:08:20.677890Z","shell.execute_reply":"2021-07-23T11:08:20.677352Z","shell.execute_reply.started":"2021-07-23T09:57:28.888696Z"},"papermill":{"duration":2.533255,"end_time":"2021-07-23T11:08:20.678028","exception":false,"start_time":"2021-07-23T11:08:18.144773","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = df_ext.loc[0,'Absolute_Path']\ndata_path","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:08:25.855737Z","iopub.status.busy":"2021-07-23T11:08:25.854680Z","iopub.status.idle":"2021-07-23T11:08:25.873500Z","shell.execute_reply":"2021-07-23T11:08:25.872905Z","shell.execute_reply.started":"2021-07-23T09:58:14.119205Z"},"papermill":{"duration":2.61536,"end_time":"2021-07-23T11:08:25.873661","exception":false,"start_time":"2021-07-23T11:08:23.258301","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = ReadMRI(data_path)\nprint('Shape of data: ', data.shape)\n\nplt.figure(figsize=(5, 5))\nplt.imshow(data, cmap=plt.cm.gist_ncar);","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:08:30.900595Z","iopub.status.busy":"2021-07-23T11:08:30.899834Z","iopub.status.idle":"2021-07-23T11:08:31.081069Z","shell.execute_reply":"2021-07-23T11:08:31.081516Z","shell.execute_reply.started":"2021-07-23T09:58:21.029421Z"},"papermill":{"duration":2.663913,"end_time":"2021-07-23T11:08:31.081723","exception":false,"start_time":"2021-07-23T11:08:28.417810","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.to_csv('Extracted_Study_Series_Img.csv',index=False)","metadata":{"execution":{"iopub.execute_input":"2021-07-23T11:08:36.091539Z","iopub.status.busy":"2021-07-23T11:08:36.090881Z","iopub.status.idle":"2021-07-23T11:08:39.013415Z","shell.execute_reply":"2021-07-23T11:08:39.014150Z"},"papermill":{"duration":5.400916,"end_time":"2021-07-23T11:08:39.014350","exception":false,"start_time":"2021-07-23T11:08:33.613434","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}