{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The DICOM Standard\n> DICOM (Digital Imaging and Communications in Medicine) is a standard protocol for the management and transmission of medical images and related data.\n\nFrom [searchhealthit](https://searchhealthit.techtarget.com/definition/DICOM-Digital-Imaging-and-Communications-in-Medicine)\n\n**The standard keeps information about:**\n - Acquisition method\n - Medical images\n - Patient information\n \n## Why DICOM\nIt was developed by the American College of Radiologists in 1993 to allow for interoperability accross hospitals.\n\nThis way one hospital can work as usual with the files from another hospital using the same software they always use.\n\nBy using DICOM, one can get derived structured documents and manage the related workflow more conveniently.\n\n—\n\n**Protected Health Information (PHI)** is part of DICOM and clinical data and radiologist report are not part of DICOM\n\n**PHI**: *Individually identifiable* healt information, including demohraphic information, and other information used to identify a patient or provide healtcare services.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport pandas as pd\n\nfrom tqdm import tqdm\nfrom multiprocessing import Pool\nargs={}\nargs['input'] = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\nargs['output'] = './'\nargs['dataset'] = 'test'\nargs['n_jobs'] = 20\nargs['debug'] = 0\n\n\nFIELDS = [\n    'AccessionNumber',\n    'AcquisitionMatrix',\n#    'B1rms',  # Empty\n#    'BitsAllocated',  # = 16\n#    'BitsStored',  # = 16\n    'Columns',\n    'ConversionType',\n#    'DiffusionBValue',  # 0 or empty\n#    'DiffusionGradientOrientation',  # [0.0, 0.0, 0.0] or empty\n    'EchoNumbers',\n#    'EchoTime',  # empty\n    'EchoTrainLength',\n    'FlipAngle',\n#    'HighBit',  # = 15\n#    'HighRRValue',  #  0 or empty\n    'ImageDimensions',  # 2 or epty\n    'ImageFormat',\n    'ImageGeometryType',\n    'ImageLocation',\n    'ImageOrientation',\n    'ImageOrientationPatient',\n    'ImagePosition',\n    'ImagePositionPatient',\n#    'ImageType',  # ['DERIVED', 'SECONDARY']\n    'ImagedNucleus',\n    'ImagingFrequency',\n    'InPlanePhaseEncodingDirection',\n    'InStackPositionNumber',\n    'InstanceNumber',\n#    'InversionTime',   # empty\n#    'Laterality',  # empty\n#    'LowRRValue',  # empty\n    'MRAcquisitionType',\n    'MagneticFieldStrength',\n#    'Modality',  # MR\n    'NumberOfAverages',\n    'NumberOfPhaseEncodingSteps',\n    'PatientID',\n    'PatientName',\n#    'PatientPosition',  # HFS\n    'PercentPhaseFieldOfView',\n    'PercentSampling',\n#    'PhotometricInterpretation',  # MONOCHROME2\n    'PixelBandwidth',\n#    'PixelPaddingValue',  # empty or 0\n    'PixelRepresentation',\n    'PixelSpacing',\n#    'PlanarConfiguration',  # 0 or empty\n#    'PositionReferenceIndicator',  # 'NA' or empty\n    'PresentationLUTShape',\n    'ReconstructionDiameter',\n#    'RescaleIntercept',  # = 0\n#    'RescaleSlope',  # = 1\n#    'RescaleType',  # = US\n    'Rows',\n    'SAR',\n    'SOPClassUID',\n    'SOPInstanceUID',\n#    'SamplesPerPixel',  # = 1\n    'SeriesDescription',\n    'SeriesInstanceUID',\n    'SeriesNumber',\n    'SliceLocation',\n    'SliceThickness',\n    'SpacingBetweenSlices',\n    'SpatialResolution',\n    'SpecificCharacterSet',\n    'StudyInstanceUID',\n#    'TemporalResolution',  # 0 or empty\n#    'TransferSyntaxUID',  # = 1.2.840.10008.1.2\n#    'TriggerWindow',  # = 0\n    'WindowCenter',\n    'WindowWidth'\n]\n\n# All of the FM fields are empty\nFM_FIELDS = [\n    'FileMetaInformationGroupLength',\n    'FileMetaInformationVersion',\n    'ImplementationClassUID',\n    'ImplementationVersionName',\n    'MediaStorageSOPClassUID',\n    'MediaStorageSOPInstanceUID',\n    'SourceApplicationEntityTitle',\n    'TransferSyntaxUID',\n]\n\nfinal = []\n\n\ndef get_meta_info(dicom):\n    row = {f: dicom.get(f) for f in FIELDS}\n    row_fm = {f: dicom.file_meta.get(f) for f in FM_FIELDS}\n    row_other = {\n#  This are all True\n#        'is_original_encoding': dicom.is_original_encoding,\n#        'is_implicit_VR': dicom.is_implicit_VR,\n#        'is_little_endian': dicom.is_little_endian,\n        'timestamp': dicom.timestamp,\n    }\n    return {**row, **row_fm, **row_other}\n\n\ndef get_dicom_files(input_dir, ds='train'):\n    dicoms = []\n\n    for subdir, dirs, files in os.walk(f\"{input_dir}/{ds}\"):\n        for filename in files:\n            filepath = subdir + os.sep + filename\n\n            if filepath.endswith(\".dcm\"):\n                dicoms.append(filepath)\n\n    return dicoms\n\n\ndef process_dicom(dicom_src, _x):\n    dicom = pydicom.dcmread(dicom_src)\n    file_data = dicom_src.split(\"/\")\n    file_src = \"/\".join(file_data[-4:])\n\n    tmp = {\"BraTS21ID\": file_data[-3], \"dataset\": file_data[-4], \"type\": file_data[-2], \"dicom_src\": f\"./{file_src}\"}\n    tmp.update(get_meta_info(dicom))\n\n    return tmp\n\n\ndef update(res):\n    if res is not None:\n        final.append(res)\n\n    pbar.update()\n\n\ndef error(e):\n    print(e)\n\n\nif __name__ == \"__main__\":\n    dicom_files = get_dicom_files(args[\"input\"], args[\"dataset\"])\n\n    if args[\"debug\"]:\n        dicom_files = dicom_files[:1000]\n\n    pool = Pool(processes=args[\"n_jobs\"])\n    pbar = tqdm(total=len(dicom_files))\n\n    for dicom_file in dicom_files:\n        pool.apply_async(\n            process_dicom,\n            args=(dicom_file, ''),\n            callback=update,\n            error_callback=error,\n        )\n\n    pool.close()\n    pool.join()\n    pbar.close()\n\n    final = pd.DataFrame(final)\n    final.to_csv(f\"{args['output']}/dicom_meta_{args['dataset']}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-30T20:55:41.180036Z","iopub.execute_input":"2021-07-30T20:55:41.180315Z","iopub.status.idle":"2021-07-30T20:57:08.35937Z","shell.execute_reply.started":"2021-07-30T20:55:41.18028Z","shell.execute_reply":"2021-07-30T20:57:08.358305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dicom file structure\n\nThe file includes two sections:\n\n - Header: Contains the details of the image.\n - Images: The actually image captured.\n\n### Header Attributes\n* Patient: Name, ID, Birth date, sex.\n* Study: UID, Date, Time, Physician.\n* Series: UID, Number, Type.\n* Equipment: Manufacturer, Instructions.\n* Image metainformation: Settings, size, resolution...\n\n> Clinical diagnosis or diagnosis must not be included.\n","metadata":{}},{"cell_type":"markdown","source":"## Abbreviations:\n\n* HGG = high grade gliomas\n* T1Gd = T1-weighted gadolinium contrasted\n* T2-FLAIR = T2 weighted fluid attenuated inversion recovery\n* ROI = region of interest\n* CC = contralateral control\n* CE = contrast enhancing\n* MRI = magnetic resonance imaging\n* MRE = magnetic resonance elastography\n* MNO = Mathematical Neuro-Oncology Lab\n* PNT = Precision Neurotherapeutics Innovation Program\n* D = tumor diffusiveness\n* ⍴ = tumor proliferation\n* D/⍴ = tumor invasiveness","metadata":{}},{"cell_type":"markdown","source":"# Set up","metadata":{}},{"cell_type":"markdown","source":"This kernel is defined by the kaggle/python [Docker image](https://github.com/kaggle/docker-python)\n```python\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n```\n - Input data files are available in the read-only `../input/ directory`.\n - You can write up to 20GB to the current directory `/kaggle/working/` that gets preserved as output when you create a version using \"Save & Run All\"\n - You can also write temporary files to `/kaggle/temp/`, but they won't be saved outside of the current session","metadata":{}},{"cell_type":"markdown","source":"## Installing missing libraries","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install --upgrade tables\n!pip install --upgrade pip\n!pip install tf-nightly\n!pip install -q tensorflow-io","metadata":{"execution":{"iopub.status.busy":"2021-08-01T00:15:34.170114Z","iopub.execute_input":"2021-08-01T00:15:34.170531Z","iopub.status.idle":"2021-08-01T00:16:03.863239Z","shell.execute_reply.started":"2021-08-01T00:15:34.170492Z","shell.execute_reply":"2021-08-01T00:16:03.861479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport random\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport pydicom\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport scipy.misc\n\ndata = {\n    'test_path' : '../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/',\n    'train_path' : '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/',\n    'train_labels': \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\",\n    'types': (\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"),\n    'ds_path': \"../input/rsna-miccai-brain-tumor-radiogenomic-classification\",\n    'metadata_train': \"./dicom_meta_train.csv\"\n}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-02T04:50:48.225547Z","iopub.execute_input":"2021-08-02T04:50:48.22593Z","iopub.status.idle":"2021-08-02T04:50:54.307426Z","shell.execute_reply.started":"2021-08-02T04:50:48.2259Z","shell.execute_reply":"2021-08-02T04:50:54.306545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## DICOM file","metadata":{}},{"cell_type":"code","source":"path_example = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00610/T2w/Image-103.dcm'\ndef avg_intensity(path):\n    return pydicom.read_file(path)\navg_intensity(path_example)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:29:40.672763Z","iopub.execute_input":"2021-08-02T06:29:40.673185Z","iopub.status.idle":"2021-08-02T06:29:40.709032Z","shell.execute_reply.started":"2021-08-02T06:29:40.673148Z","shell.execute_reply":"2021-08-02T06:29:40.708025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(data['train_labels'], index_col='BraTS21ID')\nprint(f'mean {train_df.MGMT_value.mean()}')\nprint(train_df.MGMT_value.value_counts())\ntrain_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:32:19.877211Z","iopub.execute_input":"2021-08-02T06:32:19.877629Z","iopub.status.idle":"2021-08-02T06:32:19.899025Z","shell.execute_reply.started":"2021-08-02T06:32:19.877588Z","shell.execute_reply":"2021-08-02T06:32:19.898048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"At first I tried to find the features we could obtain from the images to try with some kind of classical algorithm. \nWe need to take a look to think of some features.\n","metadata":{}},{"cell_type":"markdown","source":"Let's take a glimps of the pictures with some functions inspired by [@ihelon](https://www.kaggle.com/ihelon)'s [notebook](https://www.kaggle.com/ihelon/brain-tumor-eda-with-animations-and-modeling)\n","metadata":{}},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef get_type_paths(dataframe = train_df, *,sample_size=5, slice_percent=None, types=data['types']):\n    if sample_size is not None:\n        dataframe = dataframe.sample(sample_size, random_state = 42).to_dict()\n    \n    def helper_f(idx, type_, path=data['train_path']):\n        patient_path = path+str(idx).zfill(5)\n        t_paths = sorted(glob.glob(os.path.join(patient_path, type_, \"*\")), \n           key=lambda x: int(x[:-4].split(\"-\")[-1]))\n        \n        l=len(t_paths)\n        i=int(l * slice_percent)\n        if slice_percent is None: return t_paths\n        else: return t_paths[i]\n    return ((idx, MGMT,{type_:helper_f(idx, type_) for type_ in types}) for idx, MGMT in dataframe['MGMT_value'].items())\n\ndef visualize_sample(data):\n    plt.figure(figsize=(16, 5))\n    \n    if type(data) != tuple: data = next(data)\n    \n    def helper_g(i, idx=data[0]):\n        print(idx)\n        plt.subplot(1, 4, i)\n        plt.imshow(load_dicom(path), cmap=\"gray\")\n        plt.title(f\"{type_}\", fontsize=16)\n        plt.axis(\"off\")\n    \n    for i, (type_, path) in enumerate(data[2].items(), 1):\n        helper_g(i, idx=data[0])\n\n    healt_condition = ('Healty','Tumor')[data[1]]\n    print(f\"MGMT_value: {healt_condition}\")\n    plt.suptitle(f\"MGMT_value: {healt_condition}\", fontsize=16)\n    plt.show()\n\nwfi = get_type_paths(train_df, sample_size=5, slice_percent=.5)\nfor i in wfi:\n    visualize_sample(i)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:30:33.891643Z","iopub.execute_input":"2021-08-02T06:30:33.892056Z","iopub.status.idle":"2021-08-02T06:30:36.218004Z","shell.execute_reply.started":"2021-08-02T06:30:33.892021Z","shell.execute_reply":"2021-08-02T06:30:36.217283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* These images seem to be clean because they have no foreign bodies\n* The scaners seem to be well calibrated.\n* The images are clear which means that there's no aparent movement in this pictures.\n* There's a lot of black space.","metadata":{}},{"cell_type":"markdown","source":"#### Questions\nI tought of a few things:\n - Do images with the same label have similar intensity profiles?\n - Would applaying the Ostyr method help?","metadata":{}},{"cell_type":"code","source":"def avg_intensity(path):\n    pix_val = pydicom.read_file(path)\\\n                    .pixel_array\\\n                    .astype(np.int64)\n    try: return sum(sum(pix_val))/sum(sum(pix_val != 0))\n    except: return np.nan\n\navg_intensity(path_example)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:30:01.160512Z","iopub.execute_input":"2021-08-02T06:30:01.161144Z","iopub.status.idle":"2021-08-02T06:30:01.17884Z","shell.execute_reply.started":"2021-08-02T06:30:01.161108Z","shell.execute_reply":"2021-08-02T06:30:01.178134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(load_dicom(path_example)).sum().sum()","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:30:20.883954Z","iopub.execute_input":"2021-08-02T06:30:20.884304Z","iopub.status.idle":"2021-08-02T06:30:20.903228Z","shell.execute_reply.started":"2021-08-02T06:30:20.884274Z","shell.execute_reply":"2021-08-02T06:30:20.902211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, I don't know what to ask a picture of a tumor to learn more about it. So I went out to explore characteristics of both types of tumors and I got convinced that 3D processing is the way to go here, but we only have 16 GB of ram.","metadata":{}},{"cell_type":"markdown","source":"## Metadta","metadata":{}},{"cell_type":"markdown","source":"### Extract","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nimport pandas as pd\nfrom tqdm import tqdm\nfrom multiprocessing import Pool\nargs={}\nargs['input'] = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\nargs['output'] = './'\nargs['dataset'] = 'train'\nargs['n_jobs'] = 20\nargs['debug'] = 0\n\n\nFIELDS = [\n    'AccessionNumber',\n    'AcquisitionMatrix',\n#    'B1rms',  # Empty\n#    'BitsAllocated',  # = 16\n#    'BitsStored',  # = 16\n    'Columns',\n    'ConversionType',\n#    'DiffusionBValue',  # 0 or empty\n#    'DiffusionGradientOrientation',  # [0.0, 0.0, 0.0] or empty\n    'EchoNumbers',\n#    'EchoTime',  # empty\n    'EchoTrainLength',\n    'FlipAngle',\n#    'HighBit',  # = 15\n#    'HighRRValue',  #  0 or empty\n    'ImageDimensions',  # 2 or epty\n    'ImageFormat',\n    'ImageGeometryType',\n    'ImageLocation',\n    'ImageOrientation',\n    'ImageOrientationPatient',\n    'ImagePosition',\n    'ImagePositionPatient',\n#    'ImageType',  # ['DERIVED', 'SECONDARY']\n    'ImagedNucleus',\n    'ImagingFrequency',\n    'InPlanePhaseEncodingDirection',\n    'InStackPositionNumber',\n    'InstanceNumber',\n#    'InversionTime',   # empty\n#    'Laterality',  # empty\n#    'LowRRValue',  # empty\n    'MRAcquisitionType',\n    'MagneticFieldStrength',\n#    'Modality',  # MR\n    'NumberOfAverages',\n    'NumberOfPhaseEncodingSteps',\n    'PatientID',\n    'PatientName',\n#    'PatientPosition',  # HFS\n    'PercentPhaseFieldOfView',\n    'PercentSampling',\n#    'PhotometricInterpretation',  # MONOCHROME2\n    'PixelBandwidth',\n#    'PixelPaddingValue',  # empty or 0\n    'PixelRepresentation',\n    'PixelSpacing',\n#    'PlanarConfiguration',  # 0 or empty\n#    'PositionReferenceIndicator',  # 'NA' or empty\n    'PresentationLUTShape',\n    'ReconstructionDiameter',\n#    'RescaleIntercept',  # = 0\n#    'RescaleSlope',  # = 1\n#    'RescaleType',  # = US\n    'Rows',\n    'SAR',\n    'SOPClassUID',\n    'SOPInstanceUID',\n#    'SamplesPerPixel',  # = 1\n    'SeriesDescription',\n    'SeriesInstanceUID',\n    'SeriesNumber',\n    'SliceLocation',\n    'SliceThickness',\n    'SpacingBetweenSlices',\n    'SpatialResolution',\n    'SpecificCharacterSet',\n    'StudyInstanceUID',\n#    'TemporalResolution',  # 0 or empty\n#    'TransferSyntaxUID',  # = 1.2.840.10008.1.2\n#    'TriggerWindow',  # = 0\n    'WindowCenter',\n    'WindowWidth'\n]\n\n# All of the FM fields are empty\nFM_FIELDS = [\n    'FileMetaInformationGroupLength',\n    'FileMetaInformationVersion',\n    'ImplementationClassUID',\n    'ImplementationVersionName',\n    'MediaStorageSOPClassUID',\n    'MediaStorageSOPInstanceUID',\n    'SourceApplicationEntityTitle',\n    'TransferSyntaxUID',\n]\n\nfinal = []\n\n\ndef get_meta_info(dicom):\n    row = {f: dicom.get(f) for f in FIELDS}\n    row_fm = {f: dicom.file_meta.get(f) for f in FM_FIELDS}\n    row_other = {\n#        'is_original_encoding': dicom.is_original_encoding,  # = True\n#        'is_implicit_VR': dicom.is_implicit_VR,  # = True\n#        'is_little_endian': dicom.is_little_endian, # = True\n        'timestamp': dicom.timestamp,\n    }\n    return {**row,\n            #**row_fm,  # All are emtpy\n            **row_other}\n\n\ndef get_dicom_files(input_dir, ds='train'):\n    dicoms = []\n\n    for subdir, dirs, files in os.walk(f\"{input_dir}/{ds}\"):\n        for filename in files:\n            filepath = subdir + os.sep + filename\n\n            if filepath.endswith(\".dcm\"):\n                dicoms.append(filepath)\n\n    return dicoms\n\n\ndef process_dicom(dicom_src, _x):\n    dicom = pydicom.dcmread(dicom_src)\n    file_data = dicom_src.split(\"/\")\n    file_src = \"/\".join(file_data[-4:])\n\n    tmp = {\"BraTS21ID\": file_data[-3], \"dataset\": file_data[-4], \"type\": file_data[-2], \"dicom_src\": f\"./{file_src}\"}\n    tmp.update(get_meta_info(dicom))\n\n    return tmp\n\n\ndef update(res):\n    if res is not None:\n        final.append(res)\n\n    pbar.update()\n\n\ndef error(e):\n    print(e)\n\n\nif __name__ == \"__main__\":\n    dicom_files = get_dicom_files(args[\"input\"], args[\"dataset\"])\n\n    if args[\"debug\"]:\n        dicom_files = dicom_files[:1000]\n\n    pool = Pool(processes=args[\"n_jobs\"])\n    pbar = tqdm(total=len(dicom_files))\n\n    for dicom_file in dicom_files:\n        pool.apply_async(\n            process_dicom,\n            args=(dicom_file, ''),\n            callback=update,\n            error_callback=error,\n        )\n\n    pool.close()\n    pool.join()\n    pbar.close()\n\n    final = pd.DataFrame(final)\n    final.to_csv(f\"{args['output']}/dicom_meta_{args['dataset']}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T04:52:25.667905Z","iopub.execute_input":"2021-08-02T04:52:25.668506Z","iopub.status.idle":"2021-08-02T05:02:57.080147Z","shell.execute_reply.started":"2021-08-02T04:52:25.668464Z","shell.execute_reply":"2021-08-02T05:02:57.079049Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ","metadata":{}},{"cell_type":"code","source":"meta_train = pd.read_csv(data['metadata_train'])  # Read metadata\npatient_ids = meta_train.BraTS21ID.sort_values().unique()  # Unique patient ids\nmeta_train.dicom_src = meta_train.dicom_src.apply(lambda x: x.replace('.', data['ds_path'],1))","metadata":{"execution":{"iopub.status.busy":"2021-08-02T05:02:57.082215Z","iopub.execute_input":"2021-08-02T05:02:57.082657Z","iopub.status.idle":"2021-08-02T05:03:01.74198Z","shell.execute_reply.started":"2021-08-02T05:02:57.082609Z","shell.execute_reply":"2021-08-02T05:03:01.741057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocess Images","metadata":{}},{"cell_type":"markdown","source":"### Create the trimmed 3D model","metadata":{}},{"cell_type":"code","source":"def patient_paths_series(patient_id, technique=\"'FLAIR'\"):\n    return meta_train.query(f\"BraTS21ID == {patient_id} & type=={technique}\")\\\n                 .sort_values(by='InstanceNumber')\\\n                 .dicom_src\n\ndef load_pixels(path):\n    dicom = pydicom.read_file(path)\n    return dicom.pixel_array\n\ndef trim(arr, patient_id):\n    mask = arr != 0\n    try:\n        bounding_box = tuple(\n            slice(np.min(indexes), np.max(indexes) + 1)\n            for indexes in np.where(mask))\n        return arr[bounding_box]\n    except:\n        print(patient_id)\n        return np.nan\n\n# Do all above\ndef load_and_reduce(patient_id, technique=\"'FLAIR'\"):\n    patient_paths = patient_paths_series(patient_id,technique)\n    pixels = patient_paths.apply(load_pixels) #.to_list()  # load_pixels\n    stacked_pixels = np.stack(pixels, axis=2)\n    return trim(stacked_pixels, patient_id)\n\n\ndef patient_metadata(patient_id, technique):\n    meta_example = meta_train.query(f\"BraTS21ID == {patient_id} & type=='{technique}'\")\n    df = pd.DataFrame(meta_example.nunique() <= 1).reset_index().rename(columns={'index':'col_name', 0:'unique'})\n    simplified_df = meta_example[df[df['unique']==True]['col_name']].head(1).set_index('BraTS21ID')\n\n    extended_info = meta_example[df[df['unique']==False]['col_name']]\n    extended_info.set_index(['InstanceNumber']).sort_index()\n    \n    return {\n        'simple': simplified_df,\n        'extended': extended_info\n    }\n","metadata":{"execution":{"iopub.status.busy":"2021-08-02T05:03:14.322561Z","iopub.execute_input":"2021-08-02T05:03:14.322952Z","iopub.status.idle":"2021-08-02T05:03:14.334462Z","shell.execute_reply.started":"2021-08-02T05:03:14.322918Z","shell.execute_reply":"2021-08-02T05:03:14.33372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Example** Let's take a look different perspectives of the same image","metadata":{}},{"cell_type":"code","source":"patient_id=0\n\ntrimmed_array = load_and_reduce(patient_id, technique=\"'FLAIR'\")\nprint(trimmed_array.shape)\nplt.figure(figsize=(20, 5))\n\nplt.subplot(1, 3, 1)\nsns.heatmap(trimmed_array[146,:,:])\nplt.title(f\"Axial\", fontsize=16)\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 2)\nsns.heatmap(trimmed_array[:,174,:])\nplt.title(f\"Coronal\", fontsize=16)\nplt.axis(\"off\")\n\nplt.subplot(1, 3, 3)\nsns.heatmap(trimmed_array[:,:,70])\nplt.title(f\"Sagital\", fontsize=16)\nplt.axis(\"off\")\n","metadata":{"execution":{"iopub.status.busy":"2021-08-02T04:48:25.170213Z","iopub.execute_input":"2021-08-02T04:48:25.170886Z","iopub.status.idle":"2021-08-02T04:48:31.743629Z","shell.execute_reply.started":"2021-08-02T04:48:25.170846Z","shell.execute_reply":"2021-08-02T04:48:31.742535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Time to perform this transformation on all of the images.\nBecause of the limited memory of this notebook I'll use split the transformation in 3 sections and store their resoults in an HDF5 file.","metadata":{}},{"cell_type":"markdown","source":"## Data transformed","metadata":{}},{"cell_type":"markdown","source":"### Metadata","metadata":{}},{"cell_type":"code","source":"%%capture\n# To hide the warnings\npath_to_hdf = 'metadata.h5'\nfor technique in data['types']:\n    with pd.HDFStore(path_to_hdf) as store:\n        for patient_id in patient_ids:\n            metadata = patient_metadata(patient_id, f\"{technique}\")\n            store[f'simple/{patient_id}/{technique}'] = metadata['simple']\n            store[f'extended/{patient_id}/{technique}'] = metadata['extended']","metadata":{"execution":{"iopub.status.busy":"2021-08-02T06:23:07.265152Z","iopub.execute_input":"2021-08-02T06:23:07.265599Z","iopub.status.idle":"2021-08-02T06:25:54.232864Z","shell.execute_reply.started":"2021-08-02T06:23:07.26554Z","shell.execute_reply":"2021-08-02T06:25:54.231792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Images","metadata":{}},{"cell_type":"markdown","source":"<p>At first I tried to save the images as part of the HDF but it forgets the shape of the array, so it would  be easier if we export them as npz's are files containing numpy arrays conservating the [savez_compressed](https://numpy.org/doc/stable/reference/generated/numpy.savez_compressed.html#numpy.savez_compressed) compreses the trimed images.</p>\n<p>After the transformations we reduce significatively the storage space it takes for the whole training data.</p>\n\n","metadata":{}},{"cell_type":"code","source":"splits = np.split(patient_ids,9)\ngenerator = (np.savez_compressed(f'part_{e}', **{str(patient_id):load_and_reduce(patient_id, technique=\"'FLAIR'\") for patient_id in split}) for e, split in enumerate(splits))\nfor part in generator:\n    next(part)","metadata":{"execution":{"iopub.status.busy":"2021-08-02T05:04:33.305446Z","iopub.execute_input":"2021-08-02T05:04:33.306009Z","iopub.status.idle":"2021-08-02T05:04:33.310829Z","shell.execute_reply.started":"2021-08-02T05:04:33.305974Z","shell.execute_reply":"2021-08-02T05:04:33.310143Z"},"trusted":true},"execution_count":null,"outputs":[]}]}