{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"!pip install pydicom -q","metadata":{"execution":{"iopub.status.busy":"2021-09-15T01:35:52.496735Z","iopub.execute_input":"2021-09-15T01:35:52.497151Z","iopub.status.idle":"2021-09-15T01:35:59.484456Z","shell.execute_reply.started":"2021-09-15T01:35:52.497117Z","shell.execute_reply":"2021-09-15T01:35:59.483364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# General imports.\nimport os\nimport pydicom\n\nimport cv2\nimport pandas as pd\nimport numpy as np\n\n# Specific imports.\nfrom multiprocessing import Pool\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-09-15T01:35:59.486288Z","iopub.execute_input":"2021-09-15T01:35:59.486606Z","iopub.status.idle":"2021-09-15T01:35:59.491914Z","shell.execute_reply.started":"2021-09-15T01:35:59.486571Z","shell.execute_reply":"2021-09-15T01:35:59.490713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting to PNGs and Extracting Meta DataFrames","metadata":{}},{"cell_type":"code","source":"# Data is here: https://www.kaggle.com/c/rsna-miccai-brain-tumor-radiogenomic-classification/data.\n\nmode = \"test\"\nmeta_df_name = \"test_meta\"\npng_image_path_root = \"./images/\"\ncomp_data_root = \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/\"\nmeta_df_root = \"./\"\n\nos.makedirs(png_image_path_root, exist_ok=True)\nos.makedirs(meta_df_root, exist_ok=True)\n\nclass ME:\n    def __init__(self, file_path, ImageID, PatientID, mpMRI_type):\n        self.file_path = file_path\n        self.ImageID = ImageID\n        self.PatientID = PatientID\n        self.mpMRI_type = mpMRI_type\n\n        \ndef dicom2image(ele):\n    dcm_file = pydicom.read_file(ele.file_path)\n    \n    PatientID = dcm_file.PatientID\n    StudyInstanceUID = dcm_file.StudyInstanceUID\n    SeriesInstanceUID = dcm_file.SeriesInstanceUID\n    SeriesDescription = dcm_file.SeriesDescription  # This is the mpMRI scan type.\n\n    assert PatientID == ele.PatientID, \"DCM Image patientid and file path patientid do not match!\"\n    assert SeriesDescription == ele.mpMRI_type, \"SeriesDescription and mpMRI scan type do not match!\"\n\n    data = apply_voi_lut(dcm_file.pixel_array, dcm_file)\n\n    if dcm_file.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n\n    image_path = os.path.join(png_image_path_root, f\"{PatientID}_{SeriesDescription}_{ele.ImageID}.png\")\n    cv2.imwrite(image_path, data)\n    \n    return [ele.file_path, image_path, PatientID, SeriesDescription, ele.ImageID, StudyInstanceUID, SeriesInstanceUID]\n\nimages_meta = []\nfor root, dirs, files in os.walk(os.path.join(comp_data_root, f\"{mode}/\")):\n    if len(files) != 0 and (\".dcm\" in files[0] or \".dicom\" in files[0]):\n        split = root.split(\"/\")\n        patientid = split[-2]\n        mpMRI_type = split[-1]\n        for file in files:\n            full_path = os.path.join(root, file)\n            ImageID = file.split(\".\")[0]  # Get the image file name.\n            \n            dcm_file = pydicom.read_file(full_path)\n            PatientID = dcm_file.PatientID\n            SeriesDescription = dcm_file.SeriesDescription  # This is the mpMRI scan type.\n            \n            images_meta.append(ME(full_path, ImageID, PatientID, SeriesDescription))\n    \np = Pool(16)\nresults = p.map(func=dicom2image, iterable=images_meta)\nmeta_df = pd.DataFrame(\n        data=np.array(results), \n        columns=[\"dicom_filepath\", \"png_filepath\", \"PatientID\", \"SeriesDescription\", \"ImageID\", \"StudyInstanceUID\", \"SeriesInstanceUID\"])\n\n# This part is for when the PatientIDs are turned into ints (for some weird reason).\npatientids = [x.split(\"/\")[-3] for x in meta_df.dicom_filepath.values]\nmeta_df.PatientID = patientids\n\nmeta_df.to_csv(os.path.join(meta_df_root, f\"{meta_df_name}.csv\"), index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink, FileLinks\nFileLink(\"test_meta.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = pydicom.dcmread(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/FLAIR/Image-1.dcm\")\na","metadata":{"execution":{"iopub.status.busy":"2021-09-15T01:35:59.493779Z","iopub.execute_input":"2021-09-15T01:35:59.494101Z","iopub.status.idle":"2021-09-15T01:35:59.52676Z","shell.execute_reply.started":"2021-09-15T01:35:59.494072Z","shell.execute_reply":"2021-09-15T01:35:59.525792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}