{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom os import listdir\nfrom os.path import isfile, join","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-09T09:10:54.533824Z","iopub.execute_input":"2021-08-09T09:10:54.534189Z","iopub.status.idle":"2021-08-09T09:10:54.540037Z","shell.execute_reply.started":"2021-08-09T09:10:54.534161Z","shell.execute_reply":"2021-08-09T09:10:54.538765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification\"\ntrain = pd.read_csv(\"{}/train_labels.csv\".format(root_path))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:11:00.645279Z","iopub.execute_input":"2021-08-09T09:11:00.645661Z","iopub.status.idle":"2021-08-09T09:11:00.684543Z","shell.execute_reply.started":"2021-08-09T09:11:00.645630Z","shell.execute_reply":"2021-08-09T09:11:00.683541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of files in each dirs\np_ids = listdir(\"{}/train/\".format(root_path))\nmri_kinds = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\np_counts = []\nfor p_id in tqdm(p_ids):\n    p_dir = \"{}/train/{}\".format(root_path, p_id)\n    p_count = {\"BraTS21ID\": p_id, \"FLAIR\": np.nan, \"T1w\": np.nan, \"T1wCE\": np.nan, \"T2w\": np.nan}\n    for kind in mri_kinds:\n        each_kind_dir = \"{}/{}\".format(p_dir, kind)\n        files = [f for f in listdir(each_kind_dir) if isfile(join(each_kind_dir, f))]\n        p_count[kind] = len(files)\n    p_counts.append(p_count)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:11:00.859612Z","iopub.execute_input":"2021-08-09T09:11:00.859981Z","iopub.status.idle":"2021-08-09T09:11:55.453648Z","shell.execute_reply.started":"2021-08-09T09:11:00.859948Z","shell.execute_reply":"2021-08-09T09:11:55.452636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_count = pd.DataFrame(p_counts)\ndf_count","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:11:55.455047Z","iopub.execute_input":"2021-08-09T09:11:55.455343Z","iopub.status.idle":"2021-08-09T09:11:55.474041Z","shell.execute_reply.started":"2021-08-09T09:11:55.455314Z","shell.execute_reply":"2021-08-09T09:11:55.472869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create flag if each patient has the same length of MRI images\ndf_count[\"is_length_same\"] = df_count.apply(lambda x: x[\"FLAIR\"] == x[\"T1w\"] == x[\"T1wCE\"] == x[\"T2w\"], axis=1)\ndf_count.head(20)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:11:55.475991Z","iopub.execute_input":"2021-08-09T09:11:55.476287Z","iopub.status.idle":"2021-08-09T09:11:55.514318Z","shell.execute_reply.started":"2021-08-09T09:11:55.476258Z","shell.execute_reply":"2021-08-09T09:11:55.513639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"same length / all = {} / {}\".format(df_count[\"is_length_same\"].sum(), df_count[\"BraTS21ID\"].count()))","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:11:55.515549Z","iopub.execute_input":"2021-08-09T09:11:55.516032Z","iopub.status.idle":"2021-08-09T09:11:55.521658Z","shell.execute_reply.started":"2021-08-09T09:11:55.516000Z","shell.execute_reply":"2021-08-09T09:11:55.520713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2, ax3, ax4) = plt.subplots(1, 4, figsize=[12, 4])\nax1.hist(df_count[\"FLAIR\"])\nax1.set_title(\"FLAIR\")\nax2.hist(df_count[\"T1w\"])\nax2.set_title(\"T1w\")\nax3.hist(df_count[\"T1wCE\"])\nax3.set_title(\"T1wCE\")\nax4.hist(df_count[\"T2w\"])\nax4.set_title(\"T2w\")","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:25:50.542703Z","iopub.execute_input":"2021-08-09T09:25:50.543086Z","iopub.status.idle":"2021-08-09T09:25:51.134536Z","shell.execute_reply.started":"2021-08-09T09:25:50.543051Z","shell.execute_reply":"2021-08-09T09:25:51.133287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n","metadata":{}}]}