{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-02T04:51:45.274684Z","iopub.execute_input":"2021-07-02T04:51:45.275403Z","iopub.status.idle":"2021-07-02T04:52:45.618478Z","shell.execute_reply.started":"2021-07-02T04:51:45.275356Z","shell.execute_reply":"2021-07-02T04:52:45.617656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-02T04:52:45.621113Z","iopub.execute_input":"2021-07-02T04:52:45.621397Z","iopub.status.idle":"2021-07-02T04:52:45.631664Z","shell.execute_reply.started":"2021-07-02T04:52:45.621355Z","shell.execute_reply":"2021-07-02T04:52:45.630843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:45.633075Z","iopub.execute_input":"2021-07-02T04:52:45.633491Z","iopub.status.idle":"2021-07-02T04:52:45.982530Z","shell.execute_reply.started":"2021-07-02T04:52:45.633460Z","shell.execute_reply":"2021-07-02T04:52:45.981682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    '''\n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    '''\n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:45.983781Z","iopub.execute_input":"2021-07-02T04:52:45.984177Z","iopub.status.idle":"2021-07-02T04:52:45.988154Z","shell.execute_reply.started":"2021-07-02T04:52:45.984147Z","shell.execute_reply":"2021-07-02T04:52:45.987524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''image_id = []\ndim0 = []\ndim1 = []\nsplits = []\n\nfor split in ['test', 'train']:\n    i = 0\n    save_dir = f'/kaggle/tmp/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n    \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            c=Counter()\n            c.update(xray.reshape(1,-1)[0])\n            ee = np.array(sorted(list(zip(c.keys(),c.values()))))\n            plt.bar(ee[:,0],ee[:,1])\n            plt.title(split)\n            plt.show()\n            #im = resize(xray, size=256)  \n            #im.save(os.path.join(save_dir, file.replace('dcm', 'jpg')))\n\n            #image_id.append(file.replace('.dcm', ''))\n            #dim0.append(xray.shape[0])\n            #dim1.append(xray.shape[1])\n            #splits.append(split)\n            i+=1\n        if i == 10:\n            break'''","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:45.989225Z","iopub.execute_input":"2021-07-02T04:52:45.989682Z","iopub.status.idle":"2021-07-02T04:52:46.004934Z","shell.execute_reply.started":"2021-07-02T04:52:45.989643Z","shell.execute_reply":"2021-07-02T04:52:46.004031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:46.006010Z","iopub.execute_input":"2021-07-02T04:52:46.006480Z","iopub.status.idle":"2021-07-02T04:52:46.074378Z","shell.execute_reply.started":"2021-07-02T04:52:46.006448Z","shell.execute_reply":"2021-07-02T04:52:46.073415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#path = '../input/siim-covid19-detection/train/000c9c05fd14/e555410bd2cd/51759b5579bc.dcm'\n#dicom = pydicom.read_file(path)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:46.075587Z","iopub.execute_input":"2021-07-02T04:52:46.075874Z","iopub.status.idle":"2021-07-02T04:52:46.079789Z","shell.execute_reply.started":"2021-07-02T04:52:46.075847Z","shell.execute_reply":"2021-07-02T04:52:46.078575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\ndef save_png(dirname, file,save_dir):\n    xray = read_xray(os.path.join(dirname, file))\n    im = resize(xray, size=512) \n    \n    study = dirname.split('/')[-2] + '_study.png'\n    im.save(os.path.join(save_dir, study))\n    return file.replace('.dcm', '') ,xray.shape[0], xray.shape[1],dirname.split('/')[3]\n\nar = []\nsplit = 'train'\npath = f'../input/siim-covid19-detection/{split}'\nnamess = list(os.walk(path))\n\nfor i in range(1):\n    save_dir = f'/kaggle/tmp/{split}/'\n    os.makedirs(save_dir, exist_ok=True)\n    outputs = Parallel(n_jobs=os.cpu_count())(delayed(save_png)(d, f, save_dir) \n              for d, _, fnames  in tqdm(namess[:len(namess)//2 ]) for f in fnames )\n    ar.append(outputs)\noutputs = np.concatenate(ar)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:55:04.434776Z","iopub.execute_input":"2021-07-02T04:55:04.435123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf train_first_half.tar.gz -C \"/kaggle/tmp/train/\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:46.199874Z","iopub.execute_input":"2021-07-02T04:52:46.200330Z","iopub.status.idle":"2021-07-02T04:52:47.029068Z","shell.execute_reply.started":"2021-07-02T04:52:46.200294Z","shell.execute_reply":"2021-07-02T04:52:47.027743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!tar -zcf test.tar.gz -C \"/kaggle/tmp/test/\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:47.031158Z","iopub.execute_input":"2021-07-02T04:52:47.031600Z","iopub.status.idle":"2021-07-02T04:52:47.035518Z","shell.execute_reply.started":"2021-07-02T04:52:47.031542Z","shell.execute_reply":"2021-07-02T04:52:47.034781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame.from_dict({'image_id': outputs[:,0], 'dim0': outputs[:,1], 'dim1': outputs[:,2],\n                             'split': outputs[:,3]})\ndf.to_csv('meta.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-07-02T04:52:47.036781Z","iopub.execute_input":"2021-07-02T04:52:47.037046Z","iopub.status.idle":"2021-07-02T04:52:47.114590Z","shell.execute_reply.started":"2021-07-02T04:52:47.037020Z","shell.execute_reply":"2021-07-02T04:52:47.113542Z"},"trusted":true},"execution_count":null,"outputs":[]}]}