{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-08-11T06:11:01.122483Z","iopub.execute_input":"2021-08-11T06:11:01.123041Z","iopub.status.idle":"2021-08-11T06:12:15.371052Z","shell.execute_reply.started":"2021-08-11T06:11:01.122925Z","shell.execute_reply":"2021-08-11T06:12:15.369724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import os\n\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pydicom\nimport glob","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:56.553124Z","iopub.execute_input":"2021-08-11T06:42:56.553494Z","iopub.status.idle":"2021-08-11T06:42:56.558369Z","shell.execute_reply.started":"2021-08-11T06:42:56.553463Z","shell.execute_reply":"2021-08-11T06:42:56.557160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:15:26.150635Z","iopub.execute_input":"2021-08-11T06:15:26.151034Z","iopub.status.idle":"2021-08-11T06:15:26.158261Z","shell.execute_reply.started":"2021-08-11T06:15:26.151002Z","shell.execute_reply":"2021-08-11T06:15:26.157123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:15:34.221050Z","iopub.execute_input":"2021-08-11T06:15:34.221432Z","iopub.status.idle":"2021-08-11T06:15:34.227219Z","shell.execute_reply.started":"2021-08-11T06:15:34.221401Z","shell.execute_reply":"2021-08-11T06:15:34.226121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ntrain_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:18:58.531083Z","iopub.execute_input":"2021-08-11T06:18:58.531464Z","iopub.status.idle":"2021-08-11T06:18:58.570554Z","shell.execute_reply.started":"2021-08-11T06:18:58.531429Z","shell.execute_reply":"2021-08-11T06:18:58.569500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the identified images to be dropped from previous notebook ljmu-2-preprocessing\ndrop_df = pd.read_csv('../input/images-to-drop/images_to_drop.csv')\ndrop_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:33:18.512104Z","iopub.execute_input":"2021-08-11T06:33:18.512531Z","iopub.status.idle":"2021-08-11T06:33:18.532424Z","shell.execute_reply.started":"2021-08-11T06:33:18.512492Z","shell.execute_reply":"2021-08-11T06:33:18.530020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mark the rows for dropping in train_df if that image is present in drop_df\n# and convert remaining dicoms to jpg\n#train_df['drop'] = train_df.apply(lambda row: 'Y' if (row['id'] in drop_df['p1'].values) else 'N', axis=1)\n#train_df[train_df['drop']=='Y'].head(3)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:10.874626Z","iopub.execute_input":"2021-08-11T06:42:10.875032Z","iopub.status.idle":"2021-08-11T06:42:10.879401Z","shell.execute_reply.started":"2021-08-11T06:42:10.874996Z","shell.execute_reply":"2021-08-11T06:42:10.878135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm'\ndicom = pydicom.read_file(path)\nprint(dicom)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:32:30.137391Z","iopub.execute_input":"2021-08-11T06:32:30.137930Z","iopub.status.idle":"2021-08-11T06:32:30.651301Z","shell.execute_reply.started":"2021-08-11T06:32:30.137783Z","shell.execute_reply":"2021-08-11T06:32:30.650215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = '../input/siim-covid19-detection'\ntrain_files = sorted(glob.glob(os.path.join(base_path, 'train/*/*/*.dcm')))\nprint(f'Number of training files: {len(train_files)}')","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:30.907600Z","iopub.execute_input":"2021-08-11T06:43:30.908017Z","iopub.status.idle":"2021-08-11T06:44:00.038436Z","shell.execute_reply.started":"2021-08-11T06:43:30.907981Z","shell.execute_reply":"2021-08-11T06:44:00.037422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\nsplits = []\nimages_to_drop = drop_df['p1'].tolist()\n\nsave_dir = f'/kaggle/tmp/train/'\nsplit='train'\nos.makedirs(save_dir, exist_ok=True)\n#os.walk(f'../input/siim-covid19-detection/train')\nfor path in tqdm(train_files):\n    # set keep_ratio=True to have original aspect ratio\n    file = path.split('/')[-1]\n    if(file.replace('.dcm', '_image') in images_to_drop):\n        continue\n    xray = read_xray(path)\n    im = resize(xray, size=256)  \n    im.save(os.path.join(save_dir, file.replace('.dcm', '.jpg')))\n\n    image_id.append(file.replace('.dcm', ''))\n    dim0.append(xray.shape[0])\n    dim1.append(xray.shape[1])\n    splits.append(split)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T07:00:09.287170Z","iopub.execute_input":"2021-08-11T07:00:09.287592Z","iopub.status.idle":"2021-08-11T07:47:53.965130Z","shell.execute_reply.started":"2021-08-11T07:00:09.287555Z","shell.execute_reply":"2021-08-11T07:47:53.962203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/tmp/train/\" .\n","metadata":{"execution":{"iopub.status.busy":"2021-08-11T07:59:57.339438Z","iopub.execute_input":"2021-08-11T07:59:57.339918Z","iopub.status.idle":"2021-08-11T08:00:00.527554Z","shell.execute_reply.started":"2021-08-11T07:59:57.339876Z","shell.execute_reply":"2021-08-11T08:00:00.526540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1, 'split': splits})\ndf.to_csv('meta.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T08:00:17.882843Z","iopub.execute_input":"2021-08-11T08:00:17.883250Z","iopub.status.idle":"2021-08-11T08:00:17.930201Z","shell.execute_reply.started":"2021-08-11T08:00:17.883210Z","shell.execute_reply":"2021-08-11T08:00:17.929305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check duplicates between \nimport pandas as pd\nimg_df = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\nstu_df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nimg_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:46:34.698086Z","iopub.execute_input":"2021-08-11T10:46:34.698414Z","iopub.status.idle":"2021-08-11T10:46:34.745395Z","shell.execute_reply.started":"2021-08-11T10:46:34.698382Z","shell.execute_reply":"2021-08-11T10:46:34.744489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stu_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:46:38.173042Z","iopub.execute_input":"2021-08-11T10:46:38.173387Z","iopub.status.idle":"2021-08-11T10:46:38.182012Z","shell.execute_reply.started":"2021-08-11T10:46:38.173359Z","shell.execute_reply":"2021-08-11T10:46:38.181433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lis1 = img_df['StudyInstanceUID'].tolist()\nlis2 = [l.replace(\"_study\",\"\") for l in stu_df['id'].tolist() ]\nprint( lis1[0] + \"  \" + lis2[0])","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:46:40.749099Z","iopub.execute_input":"2021-08-11T10:46:40.749625Z","iopub.status.idle":"2021-08-11T10:46:40.757603Z","shell.execute_reply.started":"2021-08-11T10:46:40.749580Z","shell.execute_reply":"2021-08-11T10:46:40.756884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s1 = set(lis1)\ns2 = set(lis2)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:42:33.814582Z","iopub.execute_input":"2021-08-11T10:42:33.815012Z","iopub.status.idle":"2021-08-11T10:42:33.821208Z","shell.execute_reply.started":"2021-08-11T10:42:33.814981Z","shell.execute_reply":"2021-08-11T10:42:33.820388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique entries in image set not in study set\nprint(s1.difference(s2))","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:43:15.590306Z","iopub.execute_input":"2021-08-11T10:43:15.590623Z","iopub.status.idle":"2021-08-11T10:43:15.595571Z","shell.execute_reply.started":"2021-08-11T10:43:15.590597Z","shell.execute_reply":"2021-08-11T10:43:15.594872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique entries in study set not in image set\nprint(s2.difference(s1))","metadata":{"execution":{"iopub.status.busy":"2021-08-11T10:43:36.253298Z","iopub.execute_input":"2021-08-11T10:43:36.253639Z","iopub.status.idle":"2021-08-11T10:43:36.258573Z","shell.execute_reply.started":"2021-08-11T10:43:36.253608Z","shell.execute_reply":"2021-08-11T10:43:36.257639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images are converted to jpg from dicom.","metadata":{}}]}