{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \nimport pydicom\nimport glob, pylab\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nplt.style.use('seaborn-white')\nimport seaborn as sns\nsns.set_style(\"white\")\n\n\nimport os\nprint(os.listdir(\"../input\"))\n\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# number of images in datasets\nprint(len(os.listdir(\"../input/stage_2_test_images\")), 'imgs in a test set')\nprint(len(os.listdir(\"../input/stage_2_train_images\")), 'imgs in a train set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"886f9e9116b7fe7f1b22d5bd3569857223a02b36"},"cell_type":"code","source":"pd.set_option('display.max_columns',None)\nclass_info = pd.read_csv('../input/stage_2_detailed_class_info.csv', index_col='patientId')\ntrain_labeles = pd.read_csv('../input/stage_2_train_labels.csv', index_col='patientId')\nprint(len(class_info))\nprint(class_info.head(7))\nprint(len(train_labeles))\nprint(train_labeles.head(7))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1aec977c60f932cb22e701b0f960430eb2538e54"},"cell_type":"code","source":"#some patients have more than one bounding box\ntrain_labeles[train_labeles.index.values == '00436515-870c-4b36-a041-de91049b9ab4']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5270ab8e3bcfdd81de603ff7638e1b8f0aef474"},"cell_type":"code","source":"#a vast majority of images though have 1 bounding box, rarely 2, and almost never 3 or 4\nnum_of_boxes_per_patient = train_labeles.index.value_counts()\nsns.distplot(num_of_boxes_per_patient, kde=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db48641a0340b710e7ae2a4863f6d4dd9ae3dddf"},"cell_type":"markdown","source":"# Different types of classes"},{"metadata":{"trusted":true,"_uuid":"8acde128371464bf3e3828b4e75ec17d0a52c45a"},"cell_type":"code","source":"classes = class_info['class'].value_counts()\nsns.barplot(y=classes.index, x=classes.values, alpha=0.5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5807e2ea8c3c0d62021e399d82d4680928eaf3dc"},"cell_type":"markdown","source":"# Distribution of bounding boxes"},{"metadata":{"trusted":true,"_uuid":"122cf468c79566437eebe693b5e1c848b924c35d"},"cell_type":"code","source":"train_labeles['BoundBoxArea'] = train_labeles.width*train_labeles.height\nprint(train_labeles.head(10))\ntrain_labeles.BoundBoxArea.fillna(0, inplace=True)\nboxes = train_labeles[train_labeles.BoundBoxArea > 0]\n\nfig, axs = plt.subplots(1, 2, figsize=(13, 7))\nsns.scatterplot(x='x', y='y', hue='BoundBoxArea', data=boxes, ax=axs[0])\nsns.distplot(boxes.BoundBoxArea.values, ax=axs[1], label='BoundBoxArea')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5881624401127b626d86bd492f446d80a25fd62a"},"cell_type":"markdown","source":"# Some images with and without bounding boxes"},{"metadata":{"trusted":true,"_uuid":"6ad3f008da163d46f5759674542db2f67a240488"},"cell_type":"code","source":"#https://www.kaggle.com/peterchang77/exploratory-data-analysis\n\ndef parse_data(df):\n    \"\"\"\n    Method to read a CSV file (Pandas dataframe) and parse the \n    data into the following nested dictionary:\n\n      parsed = {\n        \n        'patientId-00': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        },\n        'patientId-01': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        }, ...\n\n      }\n\n    \"\"\"\n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '../input/stage_1_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed\n\ndef draw(data):\n    \"\"\"\n    Method to draw single patient with bounding box(es) if present \n\n    \"\"\"\n    # --- Open DICOM file\n    d = pydicom.read_file(data['dicom'])\n    im = d.pixel_array\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=6)\n\n    pylab.imshow(im, cmap=pylab.cm.gist_gray)\n    pylab.axis('off')\n\ndef overlay_box(im, box, rgb, stroke=1):\n    \"\"\"\n    Method to overlay single box on image\n\n    \"\"\"\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c95e7f8531a25b8736232ab23adc8461c11a3981"},"cell_type":"code","source":"df = pd.read_csv('../input/stage_1_train_labels.csv')\nparsed = parse_data(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"457a5f76cb69c6587fedbd3d8176b7b826a1df22"},"cell_type":"code","source":"for i in range(5,10):\n    fig, axes = plt.subplots(1, 2, figsize=(10,10))\n    patientId = train_labeles.index.unique()[i]\n    draw(parsed[patientId])\n    dcm_file = '../input/stage_1_train_images/%s.dcm' % patientId\n    dcm_data = pydicom.read_file(dcm_file)\n    im = dcm_data.pixel_array\n    axes[0].imshow(im, cmap=pylab.cm.gist_gray)\n    axes[0].set_yticklabels([])\n    axes[0].set_xticklabels([])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4703a63ed5c1a84e06427e0b7d5fcc2a207d700d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}