{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nimport re\nimport cv2\nimport glob\nimport time\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nimport dask as dd\nimport dask.array as da\nfrom dask.distributed import Client, progress\n\nseed = 111\nnp.random.seed(seed)\n%config IPCompleter.use_jedi = False","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-14T19:34:09.474293Z","iopub.execute_input":"2021-07-14T19:34:09.474944Z","iopub.status.idle":"2021-07-14T19:34:10.434396Z","shell.execute_reply.started":"2021-07-14T19:34:09.474827Z","shell.execute_reply":"2021-07-14T19:34:10.433182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"data_path = Path(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/\")\ntrain_data = data_path / \"train\"\ntest_data = data_path / \"test\"","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:34:10.436568Z","iopub.execute_input":"2021-07-14T19:34:10.436888Z","iopub.status.idle":"2021-07-14T19:34:10.442484Z","shell.execute_reply.started":"2021-07-14T19:34:10.436858Z","shell.execute_reply":"2021-07-14T19:34:10.441284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(data_path / \"train_labels.csv\")\nprint(\"Total number of training samples:\", len(train_df))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:37:02.080605Z","iopub.execute_input":"2021-07-14T19:37:02.08107Z","iopub.status.idle":"2021-07-14T19:37:02.101704Z","shell.execute_reply.started":"2021-07-14T19:37:02.081009Z","shell.execute_reply":"2021-07-14T19:37:02.099879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Because each directory in the training images have 5 digits in its name\n# we will make a small change in our df\ntrain_df[\"BraTS21ID\"] = train_df[\"BraTS21ID\"].apply(lambda x: str(x).zfill(5))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:37:02.706Z","iopub.execute_input":"2021-07-14T19:37:02.706491Z","iopub.status.idle":"2021-07-14T19:37:02.725128Z","shell.execute_reply.started":"2021-07-14T19:37:02.706445Z","shell.execute_reply":"2021-07-14T19:37:02.723944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizations","metadata":{}},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\nrandom_idx = np.random.randint(len(train_df))\ncase = train_df[\"BraTS21ID\"][random_idx]\nlabel = train_df[\"MGMT_value\"][random_idx]\nscans = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\n\nall_slices = list(map(str, list((train_data / case / scans[0]).glob(\"*.dcm\"))))\n\n# Sort the slices\nall_slices.sort(key=lambda x: int(re.sub('\\D', '', x)))\n\nnrows = len(all_slices) // 5\nncols = 5\n\nf, ax = plt.subplots(nrows, ncols, figsize=(20, 40))\n\nfor i in range(len(all_slices)):\n    img = dicom2array(all_slices[i])\n    ax[i // ncols, i % ncols].imshow(img, cmap=\"gray\")\n    ax[i // ncols, i % ncols].axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:34:40.594183Z","iopub.execute_input":"2021-07-14T19:34:40.594558Z","iopub.status.idle":"2021-07-14T19:34:48.410523Z","shell.execute_reply.started":"2021-07-14T19:34:40.594525Z","shell.execute_reply":"2021-07-14T19:34:48.409094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process DICOM in parallel","metadata":{}},{"cell_type":"code","source":"@dd.delayed\ndef read_dicom_file(filepath):\n    return pydicom.read_file(filepath)\n\n@dd.delayed\ndef apply_sequence(dicom):\n    return apply_voi_lut(dicom.pixel_array, dicom)\n\n@dd.delayed\ndef fix_monochrome(data):\n    data = np.amax(data) - data\n    return data\n\n@dd.delayed\ndef preprocess_data(data):\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \ndef process_dicom(slice_path):\n    dicom = read_dicom_file(slice_path)\n    data = apply_sequence(dicom)\n    data = fix_monochrome(data)\n    data = preprocess_data(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:35:38.024437Z","iopub.execute_input":"2021-07-14T19:35:38.02484Z","iopub.status.idle":"2021-07-14T19:35:38.037468Z","shell.execute_reply.started":"2021-07-14T19:35:38.024796Z","shell.execute_reply":"2021-07-14T19:35:38.036278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"client = Client()\nclient","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:35:46.196109Z","iopub.execute_input":"2021-07-14T19:35:46.196511Z","iopub.status.idle":"2021-07-14T19:35:47.868702Z","shell.execute_reply.started":"2021-07-14T19:35:46.196468Z","shell.execute_reply":"2021-07-14T19:35:47.867912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processed_slice_data = []\nscans = [\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"]\ncount = 0\n\nstart_time = time.time()\n\n# I will process only first 10 samples to keep it simple!\nfor i in range(len(train_df.iloc[:10, :])):\n    case = train_df[\"BraTS21ID\"][i]\n    outputs = []\n    \n    for scan in scans:\n        all_slices = list((train_data / case / scan).glob(\"*.dcm\"))\n        all_slices.sort(key=lambda x: int(re.sub('\\D', '', str(x))))\n        outputs.append(process_dicom(slice_path) for slice_path in all_slices)\n    \n    processed_slice_data.append(dd.compute(outputs)[0])\n    \n    \n    for data in processed_slice_data[i:]:\n        count += len(data[0]) + len(data[1]) + len(data[2]) + len(data[3])\n    print(\"Number of DICOM files processed: \", count)\n    \n    \nprint(f\"\\nTime taken to read all slices for 10 samples: {time.time()-start_time:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:46:57.424395Z","iopub.execute_input":"2021-07-14T19:46:57.424781Z","iopub.status.idle":"2021-07-14T19:48:32.113755Z","shell.execute_reply.started":"2021-07-14T19:46:57.424746Z","shell.execute_reply":"2021-07-14T19:48:32.106435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"client.close()","metadata":{"execution":{"iopub.status.busy":"2021-07-14T19:48:40.590121Z","iopub.execute_input":"2021-07-14T19:48:40.590615Z","iopub.status.idle":"2021-07-14T19:48:41.177524Z","shell.execute_reply.started":"2021-07-14T19:48:40.590577Z","shell.execute_reply":"2021-07-14T19:48:41.176148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}