{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q efficientnet >> /dev/null","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:42:30.857650Z","iopub.execute_input":"2021-08-13T08:42:30.858012Z","iopub.status.idle":"2021-08-13T08:42:36.740747Z","shell.execute_reply.started":"2021-08-13T08:42:30.857976Z","shell.execute_reply":"2021-08-13T08:42:36.739781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:53:47.151631Z","iopub.execute_input":"2021-08-13T08:53:47.152201Z","iopub.status.idle":"2021-08-13T08:54:49.993426Z","shell.execute_reply.started":"2021-08-13T08:53:47.152077Z","shell.execute_reply":"2021-08-13T08:54:49.992306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport pandas as pd\n","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:55:03.585045Z","iopub.execute_input":"2021-08-13T08:55:03.585421Z","iopub.status.idle":"2021-08-13T08:55:04.025213Z","shell.execute_reply.started":"2021-08-13T08:55:03.585386Z","shell.execute_reply":"2021-08-13T08:55:04.024250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os, shutil\nfrom glob import glob\nfrom sklearn.cluster import KMeans\nfrom tqdm.notebook import tqdm\nfrom sklearn.preprocessing import LabelEncoder\nimport random\ntqdm.pandas()","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:55:08.160462Z","iopub.execute_input":"2021-08-13T08:55:08.160797Z","iopub.status.idle":"2021-08-13T08:55:09.082413Z","shell.execute_reply.started":"2021-08-13T08:55:08.160767Z","shell.execute_reply":"2021-08-13T08:55:09.081586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_level = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")  \nstudy_level = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\n\nstudy_level['id'] = study_level['id'].str.replace('_study','')\nstudy_level.rename(columns={\"id\":\"StudyID\"},inplace=True)\nimage_level.rename(columns={\"StudyInstanceUID\":\"StudyID\"},inplace=True)\nmerged_df = pd.merge(image_level,study_level,on=['StudyID'])\nmerged_df['id'] = merged_df['id'].str.replace('_image','')\nmerged_df.rename(columns={\"id\":\"ImageID\"},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:55:12.630733Z","iopub.execute_input":"2021-08-13T08:55:12.631266Z","iopub.status.idle":"2021-08-13T08:55:12.735974Z","shell.execute_reply.started":"2021-08-13T08:55:12.631232Z","shell.execute_reply":"2021-08-13T08:55:12.735286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:55:15.390956Z","iopub.execute_input":"2021-08-13T08:55:15.391470Z","iopub.status.idle":"2021-08-13T08:55:15.399229Z","shell.execute_reply.started":"2021-08-13T08:55:15.391421Z","shell.execute_reply":"2021-08-13T08:55:15.398395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for split in ['test', 'train']:\n    save_dir = f'/kaggle/tmp/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n    \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size=2048)  \n            im.save(os.path.join(save_dir, file.replace('dcm', 'jpg')))\n            merged_df.loc[merged_df.index[merged_df['ImageID']==file],\"ImagePath\"] = os.path.join(save_dir, file.replace('dcm', 'jpg'))\n            merged_df.loc[merged_df.index[merged_df['ImageID']==file],\"Split\"] = split","metadata":{"execution":{"iopub.status.busy":"2021-08-13T08:55:20.927092Z","iopub.execute_input":"2021-08-13T08:55:20.927618Z","iopub.status.idle":"2021-08-13T10:12:08.154757Z","shell.execute_reply.started":"2021-08-13T08:55:20.927578Z","shell.execute_reply":"2021-08-13T10:12:08.152741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df.to_csv(\"/kaggle/working/merged.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-08-13T10:12:08.158656Z","iopub.execute_input":"2021-08-13T10:12:08.159030Z","iopub.status.idle":"2021-08-13T10:12:08.252123Z","shell.execute_reply.started":"2021-08-13T10:12:08.158980Z","shell.execute_reply":"2021-08-13T10:12:08.251248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf dataset.tar.gz -C \"/kaggle/tmp/\" .","metadata":{"execution":{"iopub.status.busy":"2021-08-13T10:12:08.254147Z","iopub.execute_input":"2021-08-13T10:12:08.254588Z","iopub.status.idle":"2021-08-13T10:13:56.177125Z","shell.execute_reply.started":"2021-08-13T10:12:08.254544Z","shell.execute_reply":"2021-08-13T10:13:56.176206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}