{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# ~400 .dcm files have Transfer Syntax UID : JPEG Lossless, Nonhierarchical, First- Order Prediction \n# therefore, GDCM must be installed beforehand, in order to decode it.\n# The other files have 'Explicit VR Little Endian', which is supported by pydicom alone.\n# GDCM package available: +Add data -> gdcm-conda-forge -> Add\n!tar -xvf ../input/gdcm-conda-install/gdcm.tar\n!conda install ../working/gdcm/conda-4.8.4-py37hc8dfbb8_2.tar.bz2\n!conda install ../working/gdcm/gdcm-2.8.9-py37h71b2a6d_0.tar.bz2\n!conda install ../working/gdcm/libjpeg-turbo-2.0.3-h516909a_1.tar.bz2","metadata":{"execution":{"iopub.status.busy":"2021-05-21T09:53:58.572291Z","iopub.execute_input":"2021-05-21T09:53:58.573359Z","iopub.status.idle":"2021-05-21T09:54:50.673060Z","shell.execute_reply.started":"2021-05-21T09:53:58.573272Z","shell.execute_reply":"2021-05-21T09:54:50.671907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nfrom tqdm import tqdm\nfrom pathlib import Path\n\nimport pandas as pd\nimport numpy as np\n\n# from pydicom.pixel_data_handlers.util import apply_voi_lut\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom collections import Counter\n\nfrom PIL import Image\nimport gdcm\nimport pydicom\n\nimport cv2\nimport random","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-21T12:26:24.697413Z","iopub.execute_input":"2021-05-21T12:26:24.697822Z","iopub.status.idle":"2021-05-21T12:26:24.703962Z","shell.execute_reply.started":"2021-05-21T12:26:24.697787Z","shell.execute_reply":"2021-05-21T12:26:24.702557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOME = Path('/kaggle/input/siim-covid19-detection/')\nSTUDY_ANN = Path('train_study_level.csv')\nIMG_ANN = Path('train_image_level.csv')\n\nTRAIN = '/kaggle/input/siim-covid19-detection/train/'","metadata":{"execution":{"iopub.status.busy":"2021-05-21T09:57:24.446413Z","iopub.execute_input":"2021-05-21T09:57:24.446802Z","iopub.status.idle":"2021-05-21T09:57:24.453856Z","shell.execute_reply.started":"2021-05-21T09:57:24.446766Z","shell.execute_reply":"2021-05-21T09:57:24.452104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ann = pd.read_csv(HOME/IMG_ANN)\nimg_ann.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-05-21T09:57:26.481765Z","iopub.execute_input":"2021-05-21T09:57:26.482170Z","iopub.status.idle":"2021-05-21T09:57:26.575736Z","shell.execute_reply.started":"2021-05-21T09:57:26.482136Z","shell.execute_reply":"2021-05-21T09:57:26.574567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The equivalent of utils.py\ndef rescale_dcm(dcm_px):\n    zero_one = (dcm_px - dcm_px.min())/((dcm_px.max() - dcm_px.min()))\n    rescaled = (zero_one * 255).astype(np.uint8)\n    return rescaled\n\ndef apply_hist_equalization(array):\n    clahe = cv2.createCLAHE(clipLimit = 2, tileGridSize = (8,8))\n    cl_array = clahe.apply(array)\n    return cl_array","metadata":{"execution":{"iopub.status.busy":"2021-05-21T11:14:14.050221Z","iopub.execute_input":"2021-05-21T11:14:14.050755Z","iopub.status.idle":"2021-05-21T11:14:14.056688Z","shell.execute_reply.started":"2021-05-21T11:14:14.050719Z","shell.execute_reply":"2021-05-21T11:14:14.055746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The overall mean and standard deviation were calculated:\nSAMPLE_MEAN = 134.0\nSAMPLE_STD = 56.0","metadata":{"execution":{"iopub.status.busy":"2021-05-21T12:54:05.945638Z","iopub.execute_input":"2021-05-21T12:54:05.946094Z","iopub.status.idle":"2021-05-21T12:54:05.951226Z","shell.execute_reply.started":"2021-05-21T12:54:05.946057Z","shell.execute_reply":"2021-05-21T12:54:05.950186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visual inspection of the dcm\n\nThe chosen transformations of the raw pixel arrays are:\n- harmonize the meaning of each pixel value by chosing the baseline Photometric Interpretation;\n- scale the pixel values in the 0-255 range;\n- CLAHE for histogram equalization;\n    The purpose is trying to remove some of the noise in the images, and obtaining a better view of the X-rays. Applying histogram equalization generates an improvement in how the images look like. \n- Standardize the values to have mean 0 and unit standard deviation.","metadata":{}},{"cell_type":"code","source":"# Sample 6 images from the annotation file\nsample_df = img_ann.sample(6)\n\n# Plot\nfig, ax = plt.subplots(nrows = 6, ncols = 4, figsize = (25,30))\nr = 0\nc = 0\n\nfor idx, image in tqdm(sample_df.iterrows(), total = len(sample_df)):\n    dcm_path = glob.glob(os.path.join(TRAIN, \n                                      image.StudyInstanceUID, \n                                      \"*\",\n                                      image.id.split(\"_\")[0]+\".dcm\"))[0]\n    \n    # Read the .dcm metadata\n    dcm = pydicom.dcmread(dcm_path)\n      \n    # Get the pixel array\n    dcm_px = dcm.pixel_array\n\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        dcm_px = np.amax(dcm_px) - dcm_px\n    \n    # Rescale the values in the 0-255 range\n    dcm_rescaled = rescale_dcm(dcm_px)\n    \n    #Histogram equalization\n    cl_array = apply_hist_equalization(dcm_rescaled)\n    \n#     std_array = (cl_array - SAMPLE_MEAN)/ SAMPLE_STD\n#     std_array = cv2.resize(std_array, (512,512))\n    \n\n    ax[r, c].imshow(dcm_rescaled)\n    ax[r, 0].set_ylabel(\"Rescaled pixel array\")\n    ax[r, c+1].hist(dcm_rescaled.flatten(), bins = 100)\n    \n    ax[r+1, c].imshow(cl_array)\n    ax[r+1, 0].set_ylabel(\"CLAHE\")\n    ax[r+1, c+1].hist(cl_array.flatten(), bins = 100)\n    \n    \n    c = c + 2\n    if c%4 == 0:\n        c = 0\n        r = r+2","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:11:15.096579Z","iopub.execute_input":"2021-05-21T13:11:15.096972Z","iopub.status.idle":"2021-05-21T13:11:29.443969Z","shell.execute_reply.started":"2021-05-21T13:11:15.096937Z","shell.execute_reply":"2021-05-21T13:11:29.442886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transform data","metadata":{}},{"cell_type":"code","source":"DESTINATION_RESIZED = 'resized_512_train'\nif os.path.exists(DESTINATION_RESIZED):\n    print(\"{} folder exists.\".format(DESTINATION_RESIZED))\nelse:\n    os.makedirs(DESTINATION_RESIZED)","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:22:17.756365Z","iopub.execute_input":"2021-05-21T13:22:17.757052Z","iopub.status.idle":"2021-05-21T13:22:17.761364Z","shell.execute_reply.started":"2021-05-21T13:22:17.757009Z","shell.execute_reply":"2021-05-21T13:22:17.760593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.rmdir(DESTINATION_RESIZED)\n# import shutil\n# shutil.rmtree(DESTINATION_RESIZED)","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:45:09.638604Z","iopub.execute_input":"2021-05-21T13:45:09.639010Z","iopub.status.idle":"2021-05-21T13:45:09.646004Z","shell.execute_reply.started":"2021-05-21T13:45:09.638977Z","shell.execute_reply":"2021-05-21T13:45:09.643564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx, image in tqdm(img_ann[0:5].iterrows(), total = len(img_ann[0:5])):\n    dcm_path = glob.glob(os.path.join(TRAIN, \n                                      image.StudyInstanceUID, \n                                      \"*\",\n                                      image.id.split(\"_\")[0]+\".dcm\"))[0]\n    \n    # Read the .dcm metadata\n    dcm = pydicom.dcmread(dcm_path)\n      \n    # Get the pixel array\n    dcm_px = dcm.pixel_array\n\n    # Harmonize the images to match MONOCHROME2\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        dcm_px = np.amax(dcm_px) - dcm_px\n    \n    # Rescale the values in the 0-255 range\n    dcm_rescaled = rescale_dcm(dcm_px)\n    \n    # Apply histogram equalization as a transformation step\n    cl_array = apply_hist_equalization(dcm_rescaled)\n    \n    # Standardize to 0 mean and 1 standard deviation?\n    # For this purpose, first determine the mean and standard deviation of the overall train data\n#     std_array = (cl_array - SAMPLE_MEAN)/ SAMPLE_STD\n    \n#     # Maybe resize the images\n    resized_array = cv2.resize(cl_array, (512,512))\n\n    Image.fromarray(resized_array).save(os.path.join(DESTINATION_RESIZED, str(image.id) + \".jpg\"))","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:28:32.406089Z","iopub.execute_input":"2021-05-21T13:28:32.406519Z","iopub.status.idle":"2021-05-21T13:28:33.164561Z","shell.execute_reply.started":"2021-05-21T13:28:32.406483Z","shell.execute_reply":"2021-05-21T13:28:33.163547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(DESTINATION_RESIZED)","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:31:41.876368Z","iopub.execute_input":"2021-05-21T13:31:41.877188Z","iopub.status.idle":"2021-05-21T13:31:41.888362Z","shell.execute_reply.started":"2021-05-21T13:31:41.877126Z","shell.execute_reply":"2021-05-21T13:31:41.885707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -czf train_images_512.tar.gz resized_512_train\n!du -h train_images_512.tar.gz","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:31:12.125844Z","iopub.execute_input":"2021-05-21T13:31:12.126525Z","iopub.status.idle":"2021-05-21T13:31:12.931264Z","shell.execute_reply.started":"2021-05-21T13:31:12.126482Z","shell.execute_reply":"2021-05-21T13:31:12.929883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'train_images_512.tar.gz')","metadata":{"execution":{"iopub.status.busy":"2021-05-21T13:31:34.254392Z","iopub.execute_input":"2021-05-21T13:31:34.254896Z","iopub.status.idle":"2021-05-21T13:31:35.030563Z","shell.execute_reply.started":"2021-05-21T13:31:34.254856Z","shell.execute_reply":"2021-05-21T13:31:35.028860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sources:\n\n1. CLAHE: https://towardsdatascience.com/clahe-and-thresholding-in-python-3bf690303e40\n2. @avinashrai for saving the transformed images","metadata":{}}]}