{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2023-01-30T02:29:25.010577Z","iopub.execute_input":"2023-01-30T02:29:25.010961Z","iopub.status.idle":"2023-01-30T02:30:30.517815Z","shell.execute_reply.started":"2023-01-30T02:29:25.010881Z","shell.execute_reply":"2023-01-30T02:30:30.516707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport ast\nimport cv2\nimport pandas as pd\nimport numpy as np\nimport pydicom as dicom\nfrom tqdm import tqdm\nfrom glob import glob\nfrom PIL import Image\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2023-01-30T02:30:36.286068Z","iopub.execute_input":"2023-01-30T02:30:36.286411Z","iopub.status.idle":"2023-01-30T02:30:36.762006Z","shell.execute_reply.started":"2023-01-30T02:30:36.286377Z","shell.execute_reply":"2023-01-30T02:30:36.760746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    w, h = im.size\n    \n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    return im, size/w, size/h\n\n\ndef robust_read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom_ = dicom.read_file(path)\n\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom_.pixel_array, dicom_)\n    else:\n        data = dicom_.pixel_array\n    \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom_.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    # 近似3个标准差内数据\n    q1,q2,q3 = np.quantile(data, [.15865,.5,.84135]) \n    iqr = q3 - q1\n    multiplier = 1.5\n    # 原始方法:http://www.kaggle.com/yukiszk/robust-pixel-array-scaling\n\n    mask = ((q2 - multiplier * iqr) < data) & (data < (q2 + multiplier * iqr))\n    \n    if data[mask].size != 0:\n        p = .001\n        data = data.astype(np.float32) - np.quantile(data[mask], p)\n        data = data / np.quantile(data[mask], 1-p)\n    else:\n        data = data - np.min(data)\n        data = data / np.max(data)\n\n    data = np.clip(data, 0, 1)\n    data = (data * 255).astype(np.uint8)\n    # 3-channel\n#     img_equ = cv2.equalizeHist(data)\n#     img_edge = cv2.Canny(img_equ, 50, 130)\n#     img = np.concatenate([\n#         data[:, :, None],\n#         img_equ[:,:,None],\n#         img_edge[:,:,None],\n#     ], axis=-1)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2023-01-30T02:48:14.839572Z","iopub.execute_input":"2023-01-30T02:48:14.839904Z","iopub.status.idle":"2023-01-30T02:48:14.849349Z","shell.execute_reply.started":"2023-01-30T02:48:14.839874Z","shell.execute_reply":"2023-01-30T02:48:14.848402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits = ['image']\nremove_id = [\n    '61f3ac249c50',\n    'a39667fe9a81',\n    '267a250932bc',\n    'b97c6b32105e',\n    '869476b0763a',\n    '49664f078f0e',\n    'c3a09e8a600d',\n    '35e398a5a431',\n    '0bd6cd815ba9',\n    '6f54e9cbd180',\n    'c4b68b29a072',\n    '9872a8a48f23',\n    'e738c549fe8e',\n    'c636ac67c19a',\n    '8d4b3609ed92',\n]\nimg_size = 512\n#os.makedirs('/kaggle/working/train-csv/', exist_ok=True)\n\nbbox_dict = {'Image Index':[], 'Finding Label':[], 'x1':[], 'y1':[], 'x2':[], 'y2':[]}\n\nfor split in splits:\n    save_dir = f'/kaggle/tmp/{split}/'\n    os.makedirs(save_dir, exist_ok=True)\n    if split == 'image':   \n        df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv')\n        none = []\n        for _, row in tqdm(df.iterrows()):\n            study_id = row['StudyInstanceUID']\n            image_id = row['id'].split('_')[0]\n            if image_id not in remove_id:\n                \n                img_path = glob(f'/kaggle/input/siim-covid19-detection/train/{study_id}/*/{image_id}.dcm')\n                xray = robust_read_xray(img_path[0])\n                #h, w = xray.shape\n                img, wr, hr = resize(xray, size=img_size)\n                \n                lb = list(row['label'].split(' '))\n                if lb[0] == 'opacity':\n                    for stri in range(0,len(lb),6):\n                        bbox_dict['Image Index'].append(f'{image_id}_image.png')\n                        bbox_dict['Finding Label'].append(lb[stri])\n                        conf = float(lb[stri+1])\n\n                        x_min = float(lb[stri+2]) * wr\n                        y_min = float(lb[stri+3]) * hr\n                        x_max = float(lb[stri+4]) * wr\n                        y_max = float(lb[stri+5]) * hr\n\n                        bbox_dict['x1'].append(x_min)\n                        bbox_dict['y1'].append(y_min)\n                        bbox_dict['x2'].append(x_max)\n                        bbox_dict['y2'].append(y_max)\n                    \n                \n                \n                img.save(f'{save_dir}/{image_id}_image.png')\n            #if 'none' in str(row['label']):\n            #    none.append('1')\n            #else:\n            #    none.append('0')\n        #df['none'] = none\n        #df.to_csv('/kaggle/working/train-csv/train.csv')\n    elif split == 'study':\n        df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv')\n        for _, row in df.iterrows():\n            study_id = row['id'].split('_')[0]\n            img_path = glob(f'/kaggle/input/siim-covid19-detection/train/{study_id}/*/*.dcm')\n            for img_p in img_path:\n                xray = robust_read_xray(img_p)\n                #h, w = xray.shape\n                img = resize(xray, size=512)\n                img.save(f'{save_dir}/{study_id}_study.png')\n\nbbox_df = pd.DataFrame(bbox_dict)\nbbox_df.to_csv('image_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-30T02:48:17.31981Z","iopub.execute_input":"2023-01-30T02:48:17.320156Z","iopub.status.idle":"2023-01-30T02:48:24.916713Z","shell.execute_reply.started":"2023-01-30T02:48:17.320121Z","shell.execute_reply":"2023-01-30T02:48:24.914899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf data.tar.gz -C \"/kaggle/tmp/\" .","metadata":{"execution":{"iopub.status.busy":"2021-07-07T06:15:58.314524Z","iopub.execute_input":"2021-07-07T06:15:58.314944Z","iopub.status.idle":"2021-07-07T06:16:29.337885Z","shell.execute_reply.started":"2021-07-07T06:15:58.314903Z","shell.execute_reply":"2021-07-07T06:16:29.336981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}