{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport glob\nimport scipy.ndimage\nfrom skimage import morphology\nfrom skimage import measure\nimport pydicom\nimport time\npath_input = '/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/train/'\n\noutput_path = '/kaggle/working/rsna-miccai/train'","metadata":{"execution":{"iopub.status.busy":"2023-02-02T04:27:04.331495Z","iopub.execute_input":"2023-02-02T04:27:04.332672Z","iopub.status.idle":"2023-02-02T04:27:04.338683Z","shell.execute_reply.started":"2023-02-02T04:27:04.332625Z","shell.execute_reply":"2023-02-02T04:27:04.337245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_scan(path):\n    slices = [pydicom.dcmread(path + '/' + s) for s in os.listdir(path)]\n    slices.sort(key = lambda x: int(x.InstanceNumber))\n    try:\n        slice_thickness = np.abs(slices[0].ImagePositionPatient[2] - slices[1].ImagePositionPatient[2])\n    except:\n        slice_thickness = np.abs(slices[0].SliceLocation - slices[1].SliceLocation)\n        \n    for s in slices:\n        s.SliceThickness = slice_thickness\n        \n    return slices\n\ndef get_pixels_hu(scans):\n    image = np.stack([s.pixel_array for s in scans])\n    # Convert to int16 (from sometimes int16), \n    # should be possible as values should always be low enough (<32k)\n    image = image.astype(np.int16)\n\n    # Set outside-of-scan pixels to 1\n    # The intercept is usually -1024, so air is approximately 0\n    image[image == -2000] = 0\n    \n    # Convert to Hounsfield units (HU)\n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    \n    if slope != 1:\n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n        \n    image += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)\n\n# id=0\n# patient = load_scan(data_path)\n# imgs = get_pixels_hu(patient)\n\n# np.save(output_path + \"fullimages_%d.npy\" % (id), imgs)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T04:27:05.878864Z","iopub.execute_input":"2023-02-02T04:27:05.879333Z","iopub.status.idle":"2023-02-02T04:27:05.89393Z","shell.execute_reply.started":"2023-02-02T04:27:05.879292Z","shell.execute_reply":"2023-02-02T04:27:05.892734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def createFolder(path):\n    if not os.path.exists(path):\n        os.makedirs(path)\n\n# Clear output folder\ndef remove_folder_contents(folder):\n    for the_file in os.listdir(folder):\n        file_path = os.path.join(folder, the_file)\n        try:\n            if os.path.isfile(file_path):\n                os.unlink(file_path)\n            elif os.path.isdir(file_path):\n                remove_folder_contents(file_path)\n                os.rmdir(file_path)\n        except Exception as e:\n            print(e)\n            \n# remove_folder_contents(output_path)\n        \ndef convert(path_input,output_path):\n    i = 0\n    for filename in glob.glob(path_input +'/*'):\n        createFolder(output_path + '/' + os.path.basename(filename))\n        i += 1\n        print(i)\n        for filename_ in glob.glob(filename +'/T1wCE'):\n            createFolder(output_path + '/' + os.path.basename(filename) + '/' + os.path.basename(filename_))\n            print(filename_)\n            name = os.path.basename(filename)+'_'+os.path.basename(filename_)\n            patient = load_scan(filename_)\n            imgs = get_pixels_hu(patient)\n            np.save(output_path + '/' + os.path.basename(filename) + '/' + os.path.basename(filename_) + '/' +\"{}.npy\".format(name), imgs)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T06:01:53.542956Z","iopub.execute_input":"2023-02-02T06:01:53.549992Z","iopub.status.idle":"2023-02-02T06:01:56.690298Z","shell.execute_reply.started":"2023-02-02T06:01:53.54983Z","shell.execute_reply":"2023-02-02T06:01:56.688897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convert(path_input,output_path)","metadata":{"execution":{"iopub.status.busy":"2023-02-02T06:02:06.422754Z","iopub.execute_input":"2023-02-02T06:02:06.423163Z","iopub.status.idle":"2023-02-02T06:19:32.45471Z","shell.execute_reply.started":"2023-02-02T06:02:06.423131Z","shell.execute_reply":"2023-02-02T06:19:32.452975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}