{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\n\nimport os\nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-08-17T01:00:27.275162Z","iopub.execute_input":"2021-08-17T01:00:27.275552Z","iopub.status.idle":"2021-08-17T01:00:27.624341Z","shell.execute_reply.started":"2021-08-17T01:00:27.275470Z","shell.execute_reply":"2021-08-17T01:00:27.623318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/'\nOUT_FOLDER = 'train'\nMRI_TYPES = ['T1w', 'T1wCE','T2w', 'FLAIR']\nEXT = 'jpg'\n\nDEBUG = False","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.625960Z","iopub.execute_input":"2021-08-17T01:00:27.626266Z","iopub.status.idle":"2021-08-17T01:00:27.631461Z","shell.execute_reply.started":"2021-08-17T01:00:27.626234Z","shell.execute_reply":"2021-08-17T01:00:27.630389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.633622Z","iopub.execute_input":"2021-08-17T01:00:27.634024Z","iopub.status.idle":"2021-08-17T01:00:27.652290Z","shell.execute_reply.started":"2021-08-17T01:00:27.633990Z","shell.execute_reply":"2021-08-17T01:00:27.651204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.654153Z","iopub.execute_input":"2021-08-17T01:00:27.654576Z","iopub.status.idle":"2021-08-17T01:00:27.890206Z","shell.execute_reply.started":"2021-08-17T01:00:27.654525Z","shell.execute_reply":"2021-08-17T01:00:27.889226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dicom = pydicom.dcmread('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/00000/FLAIR/Image-1.dcm')","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.891698Z","iopub.execute_input":"2021-08-17T01:00:27.892156Z","iopub.status.idle":"2021-08-17T01:00:27.896300Z","shell.execute_reply.started":"2021-08-17T01:00:27.892111Z","shell.execute_reply":"2021-08-17T01:00:27.895105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_meta_info_v2(dicom):\n    ret_dict = {dicom.get(k).name.replace(' ',''): dicom.get(k).value for k in dicom.keys() if dicom.get(k).name != 'Pixel Data'}\n    ret_dict['timestamp'] = dicom.timestamp\n    return ret_dict","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.897933Z","iopub.execute_input":"2021-08-17T01:00:27.898338Z","iopub.status.idle":"2021-08-17T01:00:27.909352Z","shell.execute_reply.started":"2021-08-17T01:00:27.898295Z","shell.execute_reply":"2021-08-17T01:00:27.908484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# meta = get_meta_info_v2(dicom)\n# meta2 = meta.copy()\n# meta2['abc'] = 111","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.910925Z","iopub.execute_input":"2021-08-17T01:00:27.911414Z","iopub.status.idle":"2021-08-17T01:00:27.918630Z","shell.execute_reply.started":"2021-08-17T01:00:27.911370Z","shell.execute_reply":"2021-08-17T01:00:27.917875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.DataFrame([meta, meta2])","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.921536Z","iopub.execute_input":"2021-08-17T01:00:27.921855Z","iopub.status.idle":"2021-08-17T01:00:27.929523Z","shell.execute_reply.started":"2021-08-17T01:00:27.921828Z","shell.execute_reply":"2021-08-17T01:00:27.928452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_dicom(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n    # read other meta\n    meta = get_meta_info_v2(dicom)\n    \n    return data, meta","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.931054Z","iopub.execute_input":"2021-08-17T01:00:27.931377Z","iopub.status.idle":"2021-08-17T01:00:27.940509Z","shell.execute_reply.started":"2021-08-17T01:00:27.931346Z","shell.execute_reply":"2021-08-17T01:00:27.939448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_ids = []\nimage_names = []\nmri_types = []\nmetas = []\n\niterations = tqdm(os.listdir(TRAIN_DIR)) if not DEBUG else tqdm(os.listdir(TRAIN_DIR)[:3])\n\nfor patient_id in iterations:\n    patient_dir = os.path.join(TRAIN_DIR, patient_id) \n    for mri_type in MRI_TYPES:\n        type_dir = os.path.join(patient_dir, mri_type)\n        out_dir = os.path.join(OUT_FOLDER, patient_id, mri_type)\n        os.makedirs(out_dir, exist_ok=True)\n        for image_name in os.listdir(type_dir):\n            # read dicom\n            try:\n                path = os.path.join(type_dir, image_name)\n                image, meta = process_dicom(path)\n\n                cv2.imwrite(os.path.join(out_dir, image_name.replace('dcm', EXT)), image)\n\n                image_names.append(image_name.replace('dcm',EXT))\n                patient_ids.append(patient_id)\n                mri_types.append(mri_type)\n                metas.append(meta)\n            except Exception as ex:\n                print(ex)\n                \n#             break\n#         break\n#     break","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:00:27.941833Z","iopub.execute_input":"2021-08-17T01:00:27.942143Z","iopub.status.idle":"2021-08-17T01:01:06.746149Z","shell.execute_reply.started":"2021-08-17T01:00:27.942114Z","shell.execute_reply":"2021-08-17T01:01:06.745145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.imshow(image, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:06.747570Z","iopub.execute_input":"2021-08-17T01:01:06.748165Z","iopub.status.idle":"2021-08-17T01:01:06.752707Z","shell.execute_reply.started":"2021-08-17T01:01:06.748119Z","shell.execute_reply":"2021-08-17T01:01:06.751641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame({'image_name':image_names, 'patient_id':patient_ids, 'mri_type':mri_types})\ntrain_df = pd.concat([train_df, pd.DataFrame(metas)], axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:06.754061Z","iopub.execute_input":"2021-08-17T01:01:06.754457Z","iopub.status.idle":"2021-08-17T01:01:06.862388Z","shell.execute_reply.started":"2021-08-17T01:01:06.754361Z","shell.execute_reply":"2021-08-17T01:01:06.861361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:06.863991Z","iopub.execute_input":"2021-08-17T01:01:06.864416Z","iopub.status.idle":"2021-08-17T01:01:06.913266Z","shell.execute_reply.started":"2021-08-17T01:01:06.864372Z","shell.execute_reply":"2021-08-17T01:01:06.912249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r train.zip train >> log.txt","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:06.914700Z","iopub.execute_input":"2021-08-17T01:01:06.915018Z","iopub.status.idle":"2021-08-17T01:01:08.534273Z","shell.execute_reply.started":"2021-08-17T01:01:06.914989Z","shell.execute_reply":"2021-08-17T01:01:08.533066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf train","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:08.536070Z","iopub.execute_input":"2021-08-17T01:01:08.538555Z","iopub.status.idle":"2021-08-17T01:01:09.332671Z","shell.execute_reply.started":"2021-08-17T01:01:08.538508Z","shell.execute_reply":"2021-08-17T01:01:09.331453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-17T01:01:09.334149Z","iopub.execute_input":"2021-08-17T01:01:09.334443Z","iopub.status.idle":"2021-08-17T01:01:09.561729Z","shell.execute_reply.started":"2021-08-17T01:01:09.334412Z","shell.execute_reply":"2021-08-17T01:01:09.560690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}