{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport os\nfrom tensorflow import keras\nimport sys\nimport random\nimport warnings\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\nfrom matplotlib import rcParams\nfrom IPython.display import IFrame\nfrom IPython.core.display import display, HTML\nimport imageio\n\nfrom mpl_toolkits import mplot3d\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom itertools import chain\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import losses\nfrom tensorflow.keras import models\nfrom tensorflow.keras import callbacks\n\nimport matplotlib\nmatplotlib.rcParams['animation.html'] = 'jshtml'\nfrom PIL import Image\nimport cv2\nimport glob\nimport re\nimport random\nfrom scipy import ndimage, misc\n\nimport time\nimport cv2\nimport pydicom\nfrom multiprocessing import Pool\nfrom matplotlib.animation import FuncAnimation\nprint(tf.__version__)\nprint(keras.__version__)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-23T03:56:48.011691Z","iopub.execute_input":"2021-10-23T03:56:48.012053Z","iopub.status.idle":"2021-10-23T03:56:56.332082Z","shell.execute_reply.started":"2021-10-23T03:56:48.011961Z","shell.execute_reply":"2021-10-23T03:56:56.330951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"allInputs = os.walk('/kaggle/input')\n# print(type(walking)) #<class 'generator'>\ncount = 0\nfor root, dirs, filenames in allInputs:\n    for filename in filenames:\n        count+=1\n        if(count > 10):\n            break\n        print(os.path.join(root, filename))\n    if(count > 10):\n        break","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:56.333784Z","iopub.execute_input":"2021-10-23T03:56:56.334021Z","iopub.status.idle":"2021-10-23T03:56:56.36349Z","shell.execute_reply.started":"2021-10-23T03:56:56.333994Z","shell.execute_reply":"2021-10-23T03:56:56.362648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = {\n    'data_path': '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification',\n    'train_data_path': '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train',\n    'test_data_path': '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/test',\n    'input_path': '/kaggle/input'\n}","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:56.364923Z","iopub.execute_input":"2021-10-23T03:56:56.365356Z","iopub.status.idle":"2021-10-23T03:56:56.370688Z","shell.execute_reply.started":"2021-10-23T03:56:56.36532Z","shell.execute_reply":"2021-10-23T03:56:56.369795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def list_files(startpath, num):\n    masterCount = 0\n    for root, dirs, files in os.walk(startpath):\n        level = root.replace(startpath, '').count(os.sep)\n        indent = ' ' * 4 * (level)\n        masterCount += 1\n        if(masterCount >= 12):\n            break\n        print('{}{}/'.format(indent, os.path.basename(root)))\n        subindent = ' ' * 4 * (level + 1)\n        count = 0\n        for f in files:\n            count += 1\n            if(count> num):\n                break\n            print('{}{}'.format(subindent, f))","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:56.372539Z","iopub.execute_input":"2021-10-23T03:56:56.372925Z","iopub.status.idle":"2021-10-23T03:56:56.383105Z","shell.execute_reply.started":"2021-10-23T03:56:56.372882Z","shell.execute_reply":"2021-10-23T03:56:56.382344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_files(config['train_data_path'],3)","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:56.384719Z","iopub.execute_input":"2021-10-23T03:56:56.385067Z","iopub.status.idle":"2021-10-23T03:56:56.974609Z","shell.execute_reply.started":"2021-10-23T03:56:56.385024Z","shell.execute_reply":"2021-10-23T03:56:56.973647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(config['data_path'],'train_labels.csv'))\nprint(\"Total number of training data points (number of patient cases): \", len(train_df))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:56.975787Z","iopub.execute_input":"2021-10-23T03:56:56.976029Z","iopub.status.idle":"2021-10-23T03:56:57.009527Z","shell.execute_reply.started":"2021-10-23T03:56:56.976001Z","shell.execute_reply":"2021-10-23T03:56:57.008665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nsns.countplot(data=train_df, x=\"MGMT_value\");","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:57.011247Z","iopub.execute_input":"2021-10-23T03:56:57.01157Z","iopub.status.idle":"2021-10-23T03:56:57.250209Z","shell.execute_reply.started":"2021-10-23T03:56:57.011527Z","shell.execute_reply":"2021-10-23T03:56:57.249288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def natural_sort(li): \n    convert = lambda text: int(text) if text.isdigit() else text.lower()\n    alphanum_key = lambda key: [convert(c) for c in re.split('([0-9]+)', key)]\n    return sorted(li, key=alphanum_key)","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:57.251635Z","iopub.execute_input":"2021-10-23T03:56:57.251948Z","iopub.status.idle":"2021-10-23T03:56:57.258787Z","shell.execute_reply.started":"2021-10-23T03:56:57.251905Z","shell.execute_reply":"2021-10-23T03:56:57.257625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgFolders = glob.glob(config['train_data_path']+\"*/*/*\")\nfor i in range(5):\n    print(imgFolders[i])\n\nprint(\"Do we have four sub-folders per case?:\")\nprint(len(imgFolders) == 4*len(train_df))\nfor folder in tqdm(imgFolders):\n    filenames = glob.glob(folder+\"/*\")\n    flag = True\n    for filename in natural_sort(filenames):\n        assert \"Image-\" in filename and \".dcm\" in filename\n        if(flag):\n            count = int(filename[7+len(folder):-4])\n            flag = False\n#         print(filename[7+len(folder):-4])\n        if(count == int(filename[7+len(folder):-4])):\n            count += 1\n        else:\n            print(\"there was a discontinuity, no file: \"+filename)\n            break","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:56:57.260518Z","iopub.execute_input":"2021-10-23T03:56:57.260861Z","iopub.status.idle":"2021-10-23T03:58:15.134834Z","shell.execute_reply.started":"2021-10-23T03:56:57.260817Z","shell.execute_reply":"2021-10-23T03:58:15.133789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#There are indeed discontinuities in the filenames!!","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:58:15.13827Z","iopub.execute_input":"2021-10-23T03:58:15.138511Z","iopub.status.idle":"2021-10-23T03:58:15.142863Z","shell.execute_reply.started":"2021-10-23T03:58:15.138482Z","shell.execute_reply":"2021-10-23T03:58:15.141946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"args={}\nargs['input'] = '../input/rsna-miccai-brain-tumor-radiogenomic-classification'\nargs['output'] = './'\nargs['dataset'] = 'train'\nargs['n_jobs'] = 20\nargs['debug'] = 0\n\n\nFIELDS = [\n    'AccessionNumber',\n    'AcquisitionMatrix',\n#    'B1rms',  # Empty\n#    'BitsAllocated',  # = 16\n#    'BitsStored',  # = 16\n    'Columns',\n    'ConversionType',\n#    'DiffusionBValue',  # 0 or empty\n#    'DiffusionGradientOrientation',  # [0.0, 0.0, 0.0] or empty\n    'EchoNumbers',\n#    'EchoTime',  # empty\n    'EchoTrainLength',\n    'FlipAngle',\n#    'HighBit',  # = 15\n#    'HighRRValue',  #  0 or empty\n    'ImageDimensions',  # 2 or epty\n    'ImageFormat',\n    'ImageGeometryType',\n    'ImageLocation',\n    'ImageOrientation',\n    'ImageOrientationPatient',\n    'ImagePosition',\n    'ImagePositionPatient',\n#    'ImageType',  # ['DERIVED', 'SECONDARY']\n    'ImagedNucleus',\n    'ImagingFrequency',\n    'InPlanePhaseEncodingDirection',\n    'InStackPositionNumber',\n    'InstanceNumber',\n#    'InversionTime',   # empty\n#    'Laterality',  # empty\n#    'LowRRValue',  # empty\n    'MRAcquisitionType',\n    'MagneticFieldStrength',\n#    'Modality',  # MR\n    'NumberOfAverages',\n    'NumberOfPhaseEncodingSteps',\n    'PatientID',\n    'PatientName',\n#    'PatientPosition',  # HFS\n    'PercentPhaseFieldOfView',\n    'PercentSampling',\n#    'PhotometricInterpretation',  # MONOCHROME2\n    'PixelBandwidth',\n#    'PixelPaddingValue',  # empty or 0\n    'PixelRepresentation',\n    'PixelSpacing',\n#    'PlanarConfiguration',  # 0 or empty\n#    'PositionReferenceIndicator',  # 'NA' or empty\n    'PresentationLUTShape',\n    'ReconstructionDiameter',\n#    'RescaleIntercept',  # = 0\n#    'RescaleSlope',  # = 1\n#    'RescaleType',  # = US\n    'Rows',\n    'SAR',\n    'SOPClassUID',\n    'SOPInstanceUID',\n#    'SamplesPerPixel',  # = 1\n    'SeriesDescription',\n    'SeriesInstanceUID',\n    'SeriesNumber',\n    'SliceLocation',\n    'SliceThickness',\n    'SpacingBetweenSlices',\n    'SpatialResolution',\n    'SpecificCharacterSet',\n    'StudyInstanceUID',\n#    'TemporalResolution',  # 0 or empty\n#    'TransferSyntaxUID',  # = 1.2.840.10008.1.2\n#    'TriggerWindow',  # = 0\n    'WindowCenter',\n    'WindowWidth'\n]\n\n# All of the FM fields are empty\nFM_FIELDS = [\n    'FileMetaInformationGroupLength',\n    'FileMetaInformationVersion',\n    'ImplementationClassUID',\n    'ImplementationVersionName',\n    'MediaStorageSOPClassUID',\n    'MediaStorageSOPInstanceUID',\n    'SourceApplicationEntityTitle',\n    'TransferSyntaxUID',\n]\n\nfinal = []\n\n\ndef get_meta_info(dicom):\n    row = {f: dicom.get(f) for f in FIELDS}\n    row_fm = {f: dicom.file_meta.get(f) for f in FM_FIELDS}\n    row_other = {\n#        'is_original_encoding': dicom.is_original_encoding,  # = True\n#        'is_implicit_VR': dicom.is_implicit_VR,  # = True\n#        'is_little_endian': dicom.is_little_endian, # = True\n        'timestamp': dicom.timestamp,\n    }\n    return {**row,\n            #**row_fm,  # All are emtpy\n            **row_other}\n\n\ndef get_dicom_files(input_dir, ds='train'):\n    dicoms = []\n\n    for subdir, dirs, files in os.walk(f\"{input_dir}/{ds}\"):\n        for filename in files:\n            filepath = subdir + os.sep + filename\n\n            if filepath.endswith(\".dcm\"):\n                dicoms.append(filepath)\n\n    return dicoms\n\n\ndef process_dicom(dicom_src, _x):\n    dicom = pydicom.dcmread(dicom_src)\n    file_data = dicom_src.split(\"/\")\n    file_src = \"/\".join(file_data[-4:])\n\n    tmp = {\"BraTS21ID\": file_data[-3], \"dataset\": file_data[-4], \"type\": file_data[-2], \"dicom_src\": f\"./{file_src}\"}\n    tmp.update(get_meta_info(dicom))\n\n    return tmp\n\n\ndef update(res):\n    if res is not None:\n        final.append(res)\n\n    pbar.update()\n\n\ndef error(e):\n    print(e)\n\n\n# Actually run this file\ndicom_files = get_dicom_files(args[\"input\"], args[\"dataset\"])\n\nif args[\"debug\"]:\n     dicom_files = dicom_files[:1000]\n\npool = Pool(processes=args[\"n_jobs\"])\npbar = tqdm(total=len(dicom_files))\n\nfor dicom_file in dicom_files:\n    pool.apply_async(\n         process_dicom,\n       args=(dicom_file, ''),\n         callback=update,\n        error_callback=error,\n    )\n\npool.close()\npool.join()\npbar.close()\n\nfinal = pd.DataFrame(final)\n#final.to_csv(f\"{args['output']}/dicom_meta_{args['dataset']}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-23T03:58:15.144082Z","iopub.execute_input":"2021-10-23T03:58:15.144364Z","iopub.status.idle":"2021-10-23T04:08:50.062787Z","shell.execute_reply.started":"2021-10-23T03:58:15.144334Z","shell.execute_reply":"2021-10-23T04:08:50.061052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.064883Z","iopub.execute_input":"2021-10-23T04:08:50.065207Z","iopub.status.idle":"2021-10-23T04:08:50.105988Z","shell.execute_reply.started":"2021-10-23T04:08:50.065152Z","shell.execute_reply":"2021-10-23T04:08:50.105108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def images_paths_from_patientID(BraTS21ID, imgs_per_folder = 1):\n    patientFolderPath = os.path.join(config['train_data_path'],str(BraTS21ID).zfill(5))\n    imgList = []\n    \n    # get the name of the four sub-folders\n    for dir in glob.glob(os.path.join(patientFolderPath,\"*\")):\n        for i in range(imgs_per_folder):\n            imgFileName = random.choice(glob.glob(os.path.join(dir,\"*\")))\n            imgList.append(imgFileName)\n    return imgList\n\nprint(images_paths_from_patientID(324, imgs_per_folder = 2))","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.107393Z","iopub.execute_input":"2021-10-23T04:08:50.107655Z","iopub.status.idle":"2021-10-23T04:08:50.16655Z","shell.execute_reply.started":"2021-10-23T04:08:50.107622Z","shell.execute_reply":"2021-10-23T04:08:50.165896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A function to get the image plane from DICOM file","metadata":{}},{"cell_type":"code","source":"def get_image_plane(dicomFile):\n    '''\n    Returns the MRI's plane from the dicom data.\n    \n    '''\n#     print(dicomFile.get(\"ImageOrientationPatient\"))\n    x1,y1,_,x2,y2,_ = [round(j) for j in dicomFile.get(\"ImageOrientationPatient\")]\n    cords = [x1,y1,x2,y2]\n\n    if cords == [1,0,0,0]:\n        return 'coronal'\n    if cords == [1,0,0,1]:\n        return 'axial'\n    if cords == [0,1,0,0]:\n        return 'sagittal'","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.167683Z","iopub.execute_input":"2021-10-23T04:08:50.168341Z","iopub.status.idle":"2021-10-23T04:08:50.174848Z","shell.execute_reply.started":"2021-10-23T04:08:50.168289Z","shell.execute_reply":"2021-10-23T04:08:50.17424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_img_from_file(fileNames, resize = True):\n    num = len(fileNames)\n    print(num)\n    fig, axs = plt.subplots(num//4, 4, sharex=True, sharey=True, figsize=(15,15))\n    count = 0\n    for img in fileNames:\n        print(img)\n        image = pydicom.dcmread(img)\n        plane = get_image_plane(image)\n        \n        if resize:\n            imageArr = cv2.resize(image.pixel_array, (256,256))\n        else:\n            imageArr = image.pixel_array\n\n        ser = img.split(\"/\")\n        title = ser[-2] + \" - \" + plane + \" | size: \" + str(imageArr.shape)\n        axs[count//4, count%4].title.set_text(title)\n        axs[count//4, count%4].imshow(imageArr, cmap='gray')\n        axs[count//4, count%4].axis(\"off\")\n        count += 1","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.176069Z","iopub.execute_input":"2021-10-23T04:08:50.176711Z","iopub.status.idle":"2021-10-23T04:08:50.190069Z","shell.execute_reply.started":"2021-10-23T04:08:50.176676Z","shell.execute_reply":"2021-10-23T04:08:50.189322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patientIDs = train_df[\"BraTS21ID\"].sample(n=3).to_numpy()\nprint(patientIDs)\nfileNames = []\n\nfor patientID in patientIDs:\n    for fileName in images_paths_from_patientID(patientID, 2):\n        fileNames.append(fileName)\n        \nprint(len(fileNames))","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.19161Z","iopub.execute_input":"2021-10-23T04:08:50.192384Z","iopub.status.idle":"2021-10-23T04:08:50.477243Z","shell.execute_reply.started":"2021-10-23T04:08:50.192353Z","shell.execute_reply":"2021-10-23T04:08:50.476592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_img_from_file(fileNames, showHist = False):\n    num = len(fileNames)\n    print(num)\n    fig, axs = plt.subplots(num//4, 4, sharex=True, sharey=True, figsize=(15,15))\n    fig.subplots_adjust(left=None, bottom=None, right=None, top=1.25, wspace=None, hspace=None)\n    count = 0\n    \n    for img in fileNames:\n        \n        image = pydicom.dcmread(img)\n        plane = get_image_plane(image)\n        image = image.pixel_array\n        \n        # display metadata in title\n        ser = img.split(\"/\")\n        title = ser[-2] + \" - \" + plane + \" | size: \" + str(image.shape)\n        title += \"\\n\" + \"data type:\" + str(image.dtype)\n        title += \"\\n\" + \"range: \" + str(np.amax(image) - np.amin(image))\n        title +=  \" mean: \" + \"{:.2f}\".format(np.mean(image))\n        \n        axs[count//4, count%4].title.set_text(title)\n        if(not showHist):\n            axs[count//4, count%4].imshow(image, cmap='gray')\n        else:\n            histogram, bin_edges = np.histogram(image, bins=256)\n            axs[count//4, count%4].plot(bin_edges[:-1], histogram)\n            \n        axs[count//4, count%4].axis(\"off\")\n        count += 1      ","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.478365Z","iopub.execute_input":"2021-10-23T04:08:50.478939Z","iopub.status.idle":"2021-10-23T04:08:50.490462Z","shell.execute_reply.started":"2021-10-23T04:08:50.478909Z","shell.execute_reply":"2021-10-23T04:08:50.489451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_img_from_file(fileNames)","metadata":{"execution":{"iopub.status.busy":"2021-10-23T04:08:50.492413Z","iopub.execute_input":"2021-10-23T04:08:50.493215Z","iopub.status.idle":"2021-10-23T04:08:53.790652Z","shell.execute_reply.started":"2021-10-23T04:08:50.49316Z","shell.execute_reply":"2021-10-23T04:08:53.789286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These pictures need to be normalized and correctly aligned!\n\nOtherwise, we can see that there range, mean, and even size vary a lot","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}