{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-22T17:04:37.635915Z","iopub.execute_input":"2021-07-22T17:04:37.636360Z","iopub.status.idle":"2021-07-22T17:05:53.773113Z","shell.execute_reply.started":"2021-07-22T17:04:37.636326Z","shell.execute_reply":"2021-07-22T17:05:53.771975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## References\n- https://www.kaggle.com/ammarnassanalhajali/siim-covid-19-convert-dcm-to-jpg-384-512-and-640px/execution \n- https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px\n","metadata":{}},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nfrom tqdm.contrib.concurrent import process_map\nimport glob\nfrom collections import namedtuple","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-22T17:05:59.791864Z","iopub.execute_input":"2021-07-22T17:05:59.792222Z","iopub.status.idle":"2021-07-22T17:05:59.806897Z","shell.execute_reply.started":"2021-07-22T17:05:59.792189Z","shell.execute_reply":"2021-07-22T17:05:59.805882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-22T17:06:01.352610Z","iopub.execute_input":"2021-07-22T17:06:01.352952Z","iopub.status.idle":"2021-07-22T17:06:01.682444Z","shell.execute_reply.started":"2021-07-22T17:06:01.352922Z","shell.execute_reply":"2021-07-22T17:06:01.681602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, shape, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail(shape, resample)\n    else:\n        im = im.resize(shape, resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-22T17:06:02.566022Z","iopub.execute_input":"2021-07-22T17:06:02.566672Z","iopub.status.idle":"2021-07-22T17:06:02.573672Z","shell.execute_reply.started":"2021-07-22T17:06:02.566621Z","shell.execute_reply":"2021-07-22T17:06:02.572209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ImageMeta = namedtuple(\"ImageMeta\", (\"height\", \"width\", \"fname\"))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T17:06:06.143503Z","iopub.execute_input":"2021-07-22T17:06:06.143897Z","iopub.status.idle":"2021-07-22T17:06:06.149545Z","shell.execute_reply.started":"2021-07-22T17:06:06.143866Z","shell.execute_reply":"2021-07-22T17:06:06.148207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nSHAPE = (512, 512)\nfor split in [\"train\", \"test\"]:\n    filenames = glob.glob(\"/kaggle/input/siim-covid19-detection/{}/*/*/*.dcm\".format(split))\n    SAVE_DIR = f\"/kaggle/tmp/{split}\"\n    os.makedirs(SAVE_DIR, exist_ok=True)\n    def persist_image(path):\n        xray = read_xray(path)\n        height = xray.shape[0]\n        width = xray.shape[1]\n        im = resize(xray, shape=SHAPE)\n        fname = os.path.basename(os.path.splitext(path)[-2])\n        jpg_fname = os.path.join(SAVE_DIR, \"{}.jpg\".format(fname))\n        im.save(jpg_fname)\n        return ImageMeta(height, width, fname)\n    split_imgs = process_map(persist_image, filenames, max_workers=8, chunksize=1)\n    pd.DataFrame.from_records(split_imgs, columns=ImageMeta._fields).to_csv(\"/kaggle/working/{}_meta.csv\".format(split), index=False)\n    print(\"No. of Images in split {}: {}\".format(split, len(split_imgs)))","metadata":{"execution":{"iopub.status.busy":"2021-07-22T17:06:06.463427Z","iopub.execute_input":"2021-07-22T17:06:06.463856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf train_{SHAPE[0]}x{SHAPE[1]}.tar.gz -C \"/kaggle/tmp/train\" .\n!tar -zcf test_{SHAPE[0]}x{SHAPE[1]}.tar.gz -C \"/kaggle/tmp/test\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-22T16:48:26.749587Z","iopub.status.idle":"2021-07-22T16:48:26.750248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}