{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:30:24.615882Z","iopub.execute_input":"2021-07-14T11:30:24.616397Z","iopub.status.idle":"2021-07-14T11:31:03.267815Z","shell.execute_reply.started":"2021-07-14T11:30:24.616359Z","shell.execute_reply":"2021-07-14T11:31:03.266647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pydicom\nimport glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nwarnings.filterwarnings('ignore')\nfrom glob import iglob\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:31:03.272058Z","iopub.execute_input":"2021-07-14T11:31:03.272393Z","iopub.status.idle":"2021-07-14T11:31:04.81388Z","shell.execute_reply.started":"2021-07-14T11:31:03.272358Z","shell.execute_reply":"2021-07-14T11:31:04.812884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# config params\nclass CFG:\n    data_path = '../input/rsna-miccai-brain-tumor-radiogenomic-classification/'\n    size = 192\n    seed = 2021\n    working_dir = '/kaggle/working/'","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:31:04.815851Z","iopub.execute_input":"2021-07-14T11:31:04.816166Z","iopub.status.idle":"2021-07-14T11:31:04.821524Z","shell.execute_reply.started":"2021-07-14T11:31:04.816134Z","shell.execute_reply":"2021-07-14T11:31:04.820494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im\n\ndef load_and_preprocess(fn):\n    # read the image\n    img = dicom2array(fn)\n    im = resize(img, size = CFG.size)  \n    # new filename\n    filename = fn.split('/')[-1].replace('dcm', 'jpg')\n    # folder to store in\n    fn2 = '/kaggle/working/' + '/'.join(fn.split('/')[3:-1])#.replace('train', 'train2')\n    os.makedirs(fn2, exist_ok=True)\n\n    im.save(os.path.join(fn2, filename))\n","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:31:04.823404Z","iopub.execute_input":"2021-07-14T11:31:04.823827Z","iopub.status.idle":"2021-07-14T11:31:04.842712Z","shell.execute_reply.started":"2021-07-14T11:31:04.823781Z","shell.execute_reply":"2021-07-14T11:31:04.841279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# list the files\nfilelist = []\n\nlevel3 = iglob(CFG.data_path + 'train/*/*/*')\nfor file in level3:\n    filelist.append(file)\n    \nprint(len(filelist))","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:31:04.844264Z","iopub.execute_input":"2021-07-14T11:31:04.844718Z","iopub.status.idle":"2021-07-14T11:31:15.327079Z","shell.execute_reply.started":"2021-07-14T11:31:04.844672Z","shell.execute_reply":"2021-07-14T11:31:15.326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = Parallel(n_jobs = 4, prefer=\"threads\")(delayed(load_and_preprocess)(fn) for fn in tqdm(filelist))    ","metadata":{"execution":{"iopub.status.busy":"2021-07-14T11:31:59.669072Z","iopub.execute_input":"2021-07-14T11:31:59.66943Z","iopub.status.idle":"2021-07-14T11:32:42.983555Z","shell.execute_reply.started":"2021-07-14T11:31:59.669399Z","shell.execute_reply":"2021-07-14T11:32:42.981095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -rf /kaggle/tmp/train2/\n!tar -zcf train.tar.gz -C \"/kaggle/working/train/\" .\n!rm -rf /kaggle/working/train","metadata":{},"execution_count":null,"outputs":[]}]}