{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## High-level info","metadata":{}},{"cell_type":"markdown","source":"* Ideal image resolution: https://pubs.rsna.org/doi/10.1148/ryai.2019190177\n    * Plateau's around 300 x 300 -- use higher resolution (512 x 512) for nodule pathologies\n* Outputs:\n    * We are going to output the label of the study\n        * One of: Atypical, Typical, Undeterminate, Negative\n    * Also the coordinates of the bounding box if possible\n    ","metadata":{}},{"cell_type":"markdown","source":"## Module Imports","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pydicom as dicom # Read .dicom medical image files\nimport matplotlib.pyplot as plt  # Visualize image files\nfrom matplotlib import cm\nimport os","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install python-gdcm -q","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/asvcode/fmi.git","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fmi.fmi.explore import *\nfrom fmi.fmi.preprocessing import *\nfrom fmi.fmi.pipeline import *\nfrom fmi.fmi.retinanet import *\nimport gdcm","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring dataset","metadata":{}},{"cell_type":"code","source":"## Reading data\ntrain = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What do we know about DICOM files\nIt seems to contain not only the image, but the image metadata:\n* Image type\n* Study date and time\n* Patient UID - name, ID, sex, \n* Body part examined\n* samples per pixel\n* Bit information (high bit)\n* Image size (number of rows and columns)\n* Number of elements (pixel data)","metadata":{}},{"cell_type":"markdown","source":"Information about Pydicom (full user guide is [here](https://pydicom.github.io/pydicom/stable/old/pydicom_user_guide.html)):\n\nImportant methods and properties:\n***FileDataset* object**\n* `Dataset.pixel_array` -- converts pydicom object into pixel array\n* `Dataset.PatientName` -- returns the value stored in the file_meta file with the variable name `PatientName` (works for other variables)\n* `Dataset.Rows` & `Dataset.Columns` -- returns the number of rows and columns, respectively","metadata":{}},{"cell_type":"markdown","source":"## Plotting a sample image","metadata":{"execution":{"iopub.status.busy":"2021-06-15T04:09:03.292941Z","iopub.execute_input":"2021-06-15T04:09:03.293504Z","iopub.status.idle":"2021-06-15T04:09:03.297745Z","shell.execute_reply.started":"2021-06-15T04:09:03.293437Z","shell.execute_reply":"2021-06-15T04:09:03.296726Z"}}},{"cell_type":"code","source":"sample_img = dicom.read_file(\"../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(sample_img.pixel_array, cmap=cm.bone)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Consolidating the datasets\nI now need to figure out how to link all the images in the train dataset with the data from `train_study_level.csv`.\n1) I notice that the `id` column in `train` has the format of \"hashcode\"+ \"_image\". So I will need to extract these hash values and map those to the filepath\n2) We can walk through the `train` file tree structure and grab the series UID, study UID, and image UID and put that into a dataframe\n3) We can then join the two dataframes so that we can link the labels from `train_study_level.csv` to the image identifiers in `/train/`.","metadata":{}},{"cell_type":"code","source":"from pathlib import Path","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the image UID from the ID column using regex\ntrain[\"image\"] = train.id.str.extract(r\"(.*)_image\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Organize data identifiers into a dataframe (by study UID, series UID, image UID, and path)","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm_notebook","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# identifier_df = pd.DataFrame()\n# for dirname, _, filenames in tqdm_notebook(os.walk('../input/siim-covid19-detection/train'), total=6054):\n#     for filename in filenames:\n#         study = str(Path(dirname).parent.stem)\n#         series = str(Path(dirname).stem)\n#         image = str(Path(filename).stem)\n#         path = os.path.join(dirname, filename)\n#         img = dicom.dcmread(path)\n#         identifier_df = identifier_df.append({\"study\": study, \"series\": series, \"image\": image, \"path\": path, \"img\": img}, ignore_index=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"identifier_df = pd.DataFrame()\nfor dirname, _, filenames in tqdm_notebook(os.walk('../input/siim-covid19-detection/train'), total=12386):\n    for filename in filenames:\n        study = str(Path(dirname).parent.stem)\n        series = str(Path(dirname).stem)\n        image = str(Path(filename).stem)\n        path = os.path.join(dirname, filename)\n        identifier_df = identifier_df.append({\"study\": study, \"series\": series, \"image\": image, \"path\": path}, ignore_index=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create consolidated dataset to link images and labels\nimage_linker = pd.merge(identifier_df, train, on=\"image\", how=\"inner\")\n\n# I check the shape to make sure that the join produced the expected result (there should be one image for\n# every row in the label dataframe)\nprint(f\"Identifier shape: {identifier_df.shape}\\nLabel shape: {train.shape}\\nJoined dataframe shape: {image_linker.shape}\")\nimage_linker.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bingo!","metadata":{"execution":{"iopub.status.busy":"2021-06-15T04:53:07.274265Z","iopub.execute_input":"2021-06-15T04:53:07.274645Z","iopub.status.idle":"2021-06-15T04:53:07.279141Z","shell.execute_reply.started":"2021-06-15T04:53:07.274613Z","shell.execute_reply":"2021-06-15T04:53:07.277699Z"}}},{"cell_type":"code","source":"class LabeledImage:\n    def __init__(self, image_linker):\n        self.df = image_linker\n    def get_info(self, image_id, *args, **kwargs):\n        path, label, boxes = self.df.loc[self.df.image == image_id, ['path', 'label', 'boxes']].iloc[0]\n        img = dicom.read_file(path)\n        return self.get_labeled_image(path, boxes, *args, **kwargs)\n    \n    def get_labeled_image(self, path, boxes, show_plot=True, show_binary=True):\n        import json\n        import matplotlib.patches as patches\n        \n        # Create figure and axes\n        fig, ax = plt.subplots()\n\n#         Display the image\n        dicom_file = dicom.read_file(path)\n        img = dicom_file.pixel_array\n        \n        rect = None\n        if isinstance(boxes, str):\n            print('Diagnosis: Positive')\n            labels = json.loads(boxes.replace('\\'', '\"'))\n\n            for label in labels:\n                x = label['x']\n                y = label['y']\n                width = label['width']\n                height = label['height']\n\n                # Create a Rectangle patch\n                rect = patches.Rectangle((x, y), width, height, linewidth=1, edgecolor='r', facecolor='none')\n\n                # Add the patch to the Axes\n                ax.add_patch(rect)\n        else:\n            print('Diagnosis: Negative')\n        \n        if show_plot:\n            ax.imshow(img, cmap=cm.bone)\n            plt.show()\n        if show_binary:\n            self.plot_binary_threshold(img, rect)\n        else:\n            plt.close()\n            \n        return dicom_file, path\n        \n    def get_random_image_id(self):\n        return self.df.sample(1).image.iloc[0]\n    \n    def get_random_image(self, *args, **kwargs):\n        img_id = self.get_random_image_id()\n        return self.get_info(img_id, *args, **kwargs)\n    \n    def plot_binary_threshold(self, arr, rect):\n        fig, ax = plt.subplots()\n        \n        mini = arr.flatten().min()\n        maxi = arr.flatten().max()\n        norm_arr = (arr.flatten() - mini)/(maxi - mini)\n        norm_arr[norm_arr < 0.5] = 0\n        norm_arr[norm_arr >= 0.5] = 1\n            \n        ax.imshow(norm_arr.reshape(arr.shape), cmap=cm.bone)\n\n        plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dash = LabeledImage(image_linker)\nimg, path = dash.get_random_image(show_plot=True, show_binary=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ----\n## PART TWO","metadata":{}},{"cell_type":"code","source":"from fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\n\nimport pydicom\n\nimport pandas as pd","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = get_dicom_files(\"../input/siim-covid19-detection/train\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient = 7\nxray_sample = items[patient].dcmread()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xray_sample.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get labels\ndf = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    df['Negative for Pneumonia'] == 1,\n    df['Typical Appearance'] == 1,\n    df['Indeterminate Appearance'] == 1,\n    df['Atypical Appearance'] == 1\n]\nchoices = [\n    'Negative', 'Typical', 'Indeterminate', 'Atypical'\n]\ndf['label'] = np.select(conditions, choices, None)\ndf['study'] = df.id.str.extract(r'(.*)_study')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef get_boxes(boxes):\n    bbx = []\n    for i, box in enumerate(boxes):\n        if box == '': continue\n        elif box == 'none 1 0 0 1 1': return [0, 0, 1, 1]\n        else:\n            nums = re.findall('\\d+\\.\\d+', box)\n            bbx.append([float(num) for num in nums])\n            \n    return bbx\n\ntrain['bbox1'] = train.label.str.split('opacity 1')\ntrain['bbox2'] = train['bbox1'].apply(get_boxes)\ntrain['study'] = train['StudyInstanceUID']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Getting DataFrame with box labels\nbbox_df = pd.merge(train[['study', 'bbox2']], df[['label', 'study']], on='study', how='inner')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Combine dataframe with image path, label, and bounding box as 2D array\nlabeled_df = pd.merge(bbox_df[['study', 'label', 'bbox2']], identifier_df[['study', 'path']], on='study', how='inner')[['path', 'label', 'bbox2']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Extract rows that have a non-zero bounding box\n# bbox_labeled_df = labeled_df[labeled_df.bbox2.str.len()>0]\nbbox_labeled_df = labeled_df.copy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Put data into format required by dataloader\nrecords = bbox_labeled_df.to_dict('records')\nnew_record = {}\nfor i in records:\n    num_labels = len(i['bbox2'])\n    new_record[i['path']] = {'bbox': i['bbox2'], 'label': [i['label']]*num_labels}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Prepare other data loader arguments\ngetters = [lambda o: str(o), lambda o: new_record[str(o)]['bbox'], lambda o: new_record[str(o)]['label']]\nitem_tfms = [Resize(128, method='pad'), ]\nbatch_tfms = [Normalize.from_stats(*imagenet_stats)]\nimgs = bbox_labeled_df.path.tolist()\ndef get_train_imgs(noop): return imgs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class PILDicom2(PILBase):\n#     _open_args,_tensor_cls,_show_args = {},TensorDicom,TensorDicom._show_args\n#     @classmethod\n#     def create(cls, fn:(Path,str,bytes), mode=None)->None:\n#         \"Open a `DICOM file` from path `fn` or bytes `fn` and load it as a `PIL Image`\"\n#         if isinstance(fn,bytes): im = Image.fromarray(dicom.dcmread(dicom.filebase.DicomBytesIO(fn)).pixel_array)\n#         if isinstance(fn,(Path,str)): im = Image.fromarray(dicom.dcmread(file).pixel_array.astype(np.int32))\n#         im.load()\n#         im = im._new(im.im)\n#         return cls(im.convert(mode) if mode else im)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"siim = DataBlock(blocks=(ImageBlock(cls=DicomView), BBoxBlock, BBoxLblBlock),\n                 get_items=get_train_imgs,\n                 splitter=RandomSplitter(),\n                 getters=getters,\n                 item_tfms=Resize(512, method=ResizeMethod.Pad),\n                 n_inp=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = pascal.dataloaders(\"../input/siim-covid19-detection/train\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch(bs=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_record.get('../input/siim-covid19-detection/train/80432401db2b/34abc8a0073f/3976de768891.dcm')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nim1 = Image.fromarray(dicom.dcmread(bbox_labeled_df.iloc[0].path).pixel_array)\nwidth = 50\nheight = 42\nim2 = im1.resize((width, height), Image.NEAREST)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}