{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"IMG_SIZE = 640","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:34:39.838774Z","iopub.execute_input":"2021-06-25T07:34:39.839497Z","iopub.status.idle":"2021-06-25T07:34:39.850027Z","shell.execute_reply.started":"2021-06-25T07:34:39.839353Z","shell.execute_reply":"2021-06-25T07:34:39.849085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-06-25T07:34:39.851746Z","iopub.execute_input":"2021-06-25T07:34:39.85235Z","iopub.status.idle":"2021-06-25T07:35:49.918321Z","shell.execute_reply.started":"2021-06-25T07:34:39.852303Z","shell.execute_reply":"2021-06-25T07:35:49.91702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-25T07:35:49.920478Z","iopub.execute_input":"2021-06-25T07:35:49.920786Z","iopub.status.idle":"2021-06-25T07:35:49.931519Z","shell.execute_reply.started":"2021-06-25T07:35:49.920754Z","shell.execute_reply":"2021-06-25T07:35:49.930513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:35:49.936285Z","iopub.execute_input":"2021-06-25T07:35:49.936933Z","iopub.status.idle":"2021-06-25T07:35:50.274075Z","shell.execute_reply.started":"2021-06-25T07:35:49.936893Z","shell.execute_reply":"2021-06-25T07:35:50.272972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:35:50.275568Z","iopub.execute_input":"2021-06-25T07:35:50.275954Z","iopub.status.idle":"2021-06-25T07:35:50.281078Z","shell.execute_reply.started":"2021-06-25T07:35:50.275913Z","shell.execute_reply":"2021-06-25T07:35:50.280245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:35:50.282341Z","iopub.execute_input":"2021-06-25T07:35:50.282797Z","iopub.status.idle":"2021-06-25T07:35:50.343457Z","shell.execute_reply.started":"2021-06-25T07:35:50.282752Z","shell.execute_reply":"2021-06-25T07:35:50.342248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/siim-covid19-detection/train/ae3e63d94c13/288554eb6182/e00f9fe0cce5.dcm'\ndicom = pydicom.read_file(path)","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:35:50.344735Z","iopub.execute_input":"2021-06-25T07:35:50.345131Z","iopub.status.idle":"2021-06-25T07:35:50.678394Z","shell.execute_reply.started":"2021-06-25T07:35:50.345093Z","shell.execute_reply":"2021-06-25T07:35:50.677326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\n\ntuples = []\nfor split in ['train', 'test']:\n    save_dir = f'/kaggle/tmp/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n    \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            tuples.append((file, dirname, save_dir, split))","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:35:50.680825Z","iopub.execute_input":"2021-06-25T07:35:50.681181Z","iopub.status.idle":"2021-06-25T07:49:22.098069Z","shell.execute_reply.started":"2021-06-25T07:35:50.681147Z","shell.execute_reply":"2021-06-25T07:49:22.09701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\n\ndef resize_and_save(t):\n    file, dirname, save_dir, split = t\n    # set keep_ratio=True to have original aspect ratio\n    xray = read_xray(os.path.join(dirname, file))\n    im = resize(xray, size=IMG_SIZE) \n#     print(os.path.join(dirname, file), os.path.join(save_dir, file.replace('dcm', 'png')))\n    im.save(os.path.join(save_dir, file.replace('dcm', 'png')))\n\n    image_id.append(file.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)\n    \n\npool = multiprocessing.Pool(processes=multiprocessing.cpu_count())\n# for t in tqdm(tuples):\nfor _ in tqdm(pool.imap_unordered(resize_and_save, tuples), total=len(tuples)):\n    pass","metadata":{"execution":{"iopub.status.busy":"2021-06-25T07:49:22.100166Z","iopub.execute_input":"2021-06-25T07:49:22.100637Z","iopub.status.idle":"2021-06-25T08:02:28.17151Z","shell.execute_reply.started":"2021-06-25T07:49:22.100592Z","shell.execute_reply":"2021-06-25T08:02:28.170564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/tmp/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/tmp/test/\" .","metadata":{"execution":{"iopub.status.busy":"2021-06-25T08:02:28.173407Z","iopub.execute_input":"2021-06-25T08:02:28.173885Z","iopub.status.idle":"2021-06-25T08:02:36.550062Z","shell.execute_reply.started":"2021-06-25T08:02:28.173821Z","shell.execute_reply":"2021-06-25T08:02:36.548796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})\ndf.to_csv('meta.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-25T08:02:36.551641Z","iopub.execute_input":"2021-06-25T08:02:36.551978Z","iopub.status.idle":"2021-06-25T08:02:36.56559Z","shell.execute_reply.started":"2021-06-25T08:02:36.551932Z","shell.execute_reply":"2021-06-25T08:02:36.563985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}