{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Source:** https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport ast\nfrom tqdm.auto import tqdm\nimport cv2\nimport os\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-07-29T20:14:42.090753Z","iopub.execute_input":"2021-07-29T20:14:42.091125Z","iopub.status.idle":"2021-07-29T20:14:42.096243Z","shell.execute_reply.started":"2021-07-29T20:14:42.091094Z","shell.execute_reply":"2021-07-29T20:14:42.095453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-29T20:14:44.616997Z","iopub.execute_input":"2021-07-29T20:14:44.617339Z","iopub.status.idle":"2021-07-29T20:14:44.623817Z","shell.execute_reply.started":"2021-07-29T20:14:44.617311Z","shell.execute_reply":"2021-07-29T20:14:44.623129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\ncount = 0\nfor data in ['test', 'train']:\n    save_dir = f'/kaggle/tmp/{data}/'\n    os.makedirs(save_dir, exist_ok=True)\n    for dirname, _, filenames in tqdm(os.walk(f'../input/rsna-miccai-brain-tumor-radiogenomic-classification/{data}')):\n        for file in filenames:\n            image = dicom2array(os.path.join(dirname, file))\n            dim0.append(image.shape[0])\n            dim1.append(image.shape[1])\n            image = cv2.resize(image, (256, 256))\n            im = Image.fromarray(image)\n            im.save(os.path.join(save_dir, dirname.split(\"/\")[4]+'-'+dirname.split(\"/\")[5] + file.replace('dcm', 'jpg')))\n            image_id.append(dirname.split(\"/\")[4]+'-'+dirname.split(\"/\")[5] +'-'+ file.replace('dcm', 'jpg'))\n            splits.append(data)","metadata":{"execution":{"iopub.status.busy":"2021-07-29T20:14:52.778919Z","iopub.execute_input":"2021-07-29T20:14:52.779435Z","iopub.status.idle":"2021-07-29T20:14:58.023807Z","shell.execute_reply.started":"2021-07-29T20:14:52.779396Z","shell.execute_reply":"2021-07-29T20:14:58.022029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf train.tar.gz -C \"/kaggle/tmp/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/tmp/test/\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-14T00:01:57.561425Z","iopub.status.idle":"2021-07-14T00:01:57.561947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame.from_dict({'id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})\ndf.to_csv('size.csv', index=False)\ndf","metadata":{},"execution_count":null,"outputs":[]}]}