{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\ntest_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv')\ntrain_df = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:05:55.654179Z","iopub.execute_input":"2021-08-31T03:05:55.654686Z","iopub.status.idle":"2021-08-31T03:05:56.981410Z","shell.execute_reply.started":"2021-08-31T03:05:55.654642Z","shell.execute_reply":"2021-08-31T03:05:56.980368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['caseID'] = test_df['BraTS21ID'].astype(str).str.zfill(5)   \ntest_df","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:06:00.856485Z","iopub.execute_input":"2021-08-31T03:06:00.856949Z","iopub.status.idle":"2021-08-31T03:06:00.887618Z","shell.execute_reply.started":"2021-08-31T03:06:00.856913Z","shell.execute_reply":"2021-08-31T03:06:00.886572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folders = ['T1w', 'T1wCE', 'T2w', 'FLAIR']\nt1w = []\nt1wce = []\nt2w = []\nflair = []\n\nfor case_id in test_df.caseID:\n    for folder in folders:\n        for dirname, _, filenames in os.walk(f'../input/rsna-miccai-png/test/{case_id}/{folder}/'):\n            max_nonblack = 0\n            FILENAME = ''\n            for filename in filenames:\n                img = plt.imread(f'{dirname}{filename}')\n                if cv2.countNonZero(img) > max_nonblack:\n                #if np.mean(img) > max_nonblack:\n                    max_nonblack = cv2.countNonZero(img)\n                    FILENAME = filename\n\n            if folder == 'T1w':\n                t1w.append(FILENAME)\n            elif folder == 'T1wCE':\n                t1wce.append(FILENAME)\n            elif folder == 'T2w':\n                t2w.append(FILENAME)\n            elif folder == 'FLAIR':\n                flair.append(FILENAME)\n        \ntest_df['T1w'] = t1w \ntest_df['T1wCE'] = t1wce\ntest_df['T2w'] = t2w\ntest_df['FLAIR'] = flair\n\ntest_df","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:07:24.861551Z","iopub.execute_input":"2021-08-31T03:07:24.861952Z","iopub.status.idle":"2021-08-31T03:12:40.803185Z","shell.execute_reply.started":"2021-08-31T03:07:24.861920Z","shell.execute_reply":"2021-08-31T03:12:40.801995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 1\ncase_id = test_df.caseID.iloc[idx]\nfilename = test_df.T1w.iloc[idx]\nt1w = plt.imread(f'../input/rsna-miccai-png/test/{case_id}/T1w/{filename}')\nfilename = test_df.T1wCE.iloc[idx]\nt1wce = plt.imread(f'../input/rsna-miccai-png/test/{case_id}/T1wCE/{filename}')\nfilename = test_df.T2w.iloc[idx]\nt2w = plt.imread(f'../input/rsna-miccai-png/test/{case_id}/T2w/{filename}')\nfilename = test_df.FLAIR.iloc[idx]\nflair = plt.imread(f'../input/rsna-miccai-png/test/{case_id}/FLAIR/{filename}')\n\nfig = plt.figure(figsize=(26,6))\nplt.gray()\nax1 = fig.add_subplot(141)\nplt.imshow(t1w, aspect='auto')\nax2 = fig.add_subplot(142)\nplt.imshow(t1wce, aspect='auto')\nax3 = fig.add_subplot(143)\nplt.imshow(t2w, aspect='auto')\nax4 = fig.add_subplot(144)\nplt.imshow(flair, aspect='auto')","metadata":{"execution":{"iopub.status.busy":"2021-08-31T04:02:34.548958Z","iopub.execute_input":"2021-08-31T04:02:34.549395Z","iopub.status.idle":"2021-08-31T04:02:35.429683Z","shell.execute_reply.started":"2021-08-31T04:02:34.549357Z","shell.execute_reply":"2021-08-31T04:02:35.428476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['caseID'] = train_df['BraTS21ID'].astype(str).str.zfill(5)   \ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:12:40.805355Z","iopub.execute_input":"2021-08-31T03:12:40.805723Z","iopub.status.idle":"2021-08-31T03:12:40.825333Z","shell.execute_reply.started":"2021-08-31T03:12:40.805681Z","shell.execute_reply":"2021-08-31T03:12:40.824235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note that according to the competition host ([discussion thread](https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/discussion/262046)), there are three case ids (`00109`, `00123`, `00709`) in the train set that should be excluded because they contain unexpected errors (e.g. missing images). \n\nDrop rows with the following caseID:\n* 00109\n* 00123\n* 00709","metadata":{}},{"cell_type":"code","source":"train_df = train_df[(train_df.caseID != \"00109\") & (train_df.caseID != \"00123\") & (train_df.caseID != \"00709\")]\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:53:52.449179Z","iopub.execute_input":"2021-08-31T03:53:52.449700Z","iopub.status.idle":"2021-08-31T03:53:52.472752Z","shell.execute_reply.started":"2021-08-31T03:53:52.449657Z","shell.execute_reply":"2021-08-31T03:53:52.471595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folders = ['T1w', 'T1wCE', 'T2w', 'FLAIR']\nt1w = []\nt1wce = []\nt2w = []\nflair = []\n\nfor case_id in train_df.caseID:\n    for folder in folders:\n        for dirname, _, filenames in os.walk(f'../input/rsna-miccai-png/train/{case_id}/{folder}/'):\n            max_nonblack = 0\n            FILENAME = ''\n            for filename in filenames:\n                img = plt.imread(f'{dirname}{filename}')\n                if cv2.countNonZero(img) > max_nonblack:\n                #if np.mean(img) > max_nonblack:\n                    max_nonblack = cv2.countNonZero(img)\n                    FILENAME = filename\n\n            if folder == 'T1w':\n                t1w.append(FILENAME)\n            elif folder == 'T1wCE':\n                t1wce.append(FILENAME)\n            elif folder == 'T2w':\n                t2w.append(FILENAME)\n            elif folder == 'FLAIR':\n                flair.append(FILENAME)\n        \ntrain_df['T1w'] = t1w \ntrain_df['T1wCE'] = t1wce\ntrain_df['T2w'] = t2w\ntrain_df['FLAIR'] = flair\n\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2021-08-31T03:54:03.593695Z","iopub.execute_input":"2021-08-31T03:54:03.594163Z","iopub.status.idle":"2021-08-31T03:54:08.110479Z","shell.execute_reply.started":"2021-08-31T03:54:03.594119Z","shell.execute_reply":"2021-08-31T03:54:08.108673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 2\ncase_id = train_df.caseID.iloc[idx]\nfilename = train_df.T1w.iloc[idx]\nt1w = plt.imread(f'../input/rsna-miccai-png/train/{case_id}/T1w/{filename}')\nfilename = train_df.T1wCE.iloc[idx]\nt1wce = plt.imread(f'../input/rsna-miccai-png/train/{case_id}/T1wCE/{filename}')\nfilename = train_df.T2w.iloc[idx]\nt2w = plt.imread(f'../input/rsna-miccai-png/train/{case_id}/T2w/{filename}')\nfilename = train_df.FLAIR.iloc[idx]\nflair = plt.imread(f'../input/rsna-miccai-png/train/{case_id}/FLAIR/{filename}')\n\nfig = plt.figure(figsize=(26,6))\nplt.gray()\nax1 = fig.add_subplot(141)\nplt.imshow(t1w, aspect='auto')\nax2 = fig.add_subplot(142)\nplt.imshow(t1wce, aspect='auto')\nax3 = fig.add_subplot(143)\nplt.imshow(t2w, aspect='auto')\nax4 = fig.add_subplot(144)\nplt.imshow(flair, aspect='auto')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.to_csv('test_df.csv', index=False)\ntrain_df.to_csv('train_df.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}