{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import glob, pylab, pandas as pd\nimport pydicom, numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"723538b8e289d0ebc4c819e138d5d8423178de74"},"cell_type":"code","source":"!ls ../input\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5bbe6fa03923e1c6489b6842e06e1f274673f1a1"},"cell_type":"code","source":"df = pd.read_csv('../input/stage_1_train_labels.csv')\nprint(df.iloc[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29de0266ce260085d7ebc946a760858b0dae9ea5"},"cell_type":"code","source":"print(df.iloc[5])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d68fed9e4b1682082dd76e5af0f6166f4f2e2aa2"},"cell_type":"code","source":"patientId = df['patientId'][0]\ndcm_file = '../input/stage_1_train_images/%s.dcm' % patientId\ndcm_data = pydicom.read_file(dcm_file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8a3dbec2cb08d1d0aa5dbed0d32ceb39ad46ee4"},"cell_type":"code","source":"print(dcm_data)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e715c47335e78f6fb3b201317b444024d924cd1d"},"cell_type":"code","source":"im = dcm_data.pixel_array\nprint(type(im))\nprint(im.dtype)\nprint(im.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"904c1bcb9c1e6b60d5c9445829ca6e560e13a2ec"},"cell_type":"code","source":"pylab.imshow(im, cmap=pylab.cm.gist_gray)\npylab.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b48d58ee6e10188ce2571a17158d58bcc579a954"},"cell_type":"code","source":"def parse_data(df):\n    \"\"\"\n    Method to read a CSV file (Pandas dataframe) and parse the \n    data into the following nested dictionary:\n\n      parsed = {\n        \n        'patientId-00': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        },\n        'patientId-01': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        }, ...\n\n      }\n\n    \"\"\"\n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '../input/stage_1_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21527848f9c79dfefe093bec7a08b4e0ea80e63a"},"cell_type":"code","source":"parsed = parse_data(df)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c54bdeed77a1163be01d3f4698b10def105c3518"},"cell_type":"code","source":"print(parsed['00436515-870c-4b36-a041-de91049b9ab4'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16b81bc2e61176e97581b3cb385295931a8a927a"},"cell_type":"code","source":"def draw(data):\n    \"\"\"\n    Method to draw single patient with bounding box(es) if present \n\n    \"\"\"\n    # --- Open DICOM file\n    d = pydicom.read_file(data['dicom'])\n    im = d.pixel_array\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=6)\n\n    pylab.imshow(im, cmap=pylab.cm.gist_gray)\n    pylab.axis('off')\n\ndef overlay_box(im, box, rgb, stroke=1):\n    \"\"\"\n    Method to overlay single box on image\n\n    \"\"\"\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7dfd1b2f1108136eff3a7e444dc2196a2eef9b00"},"cell_type":"code","source":"draw(parsed['00436515-870c-4b36-a041-de91049b9ab4'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a8069e62287a27e5bf7650c6361f9ee02e6349e6"},"cell_type":"code","source":"df_detailed = pd.read_csv('../input/stage_1_detailed_class_info.csv')\nprint(df_detailed.iloc[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bbc4d7d165df9d305a6c3feb7f3f6b1c13188674"},"cell_type":"code","source":"patientId = df_detailed['patientId'][0]\ndraw(parsed[patientId])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d9e2e82a1f4b01283c4da53c4b6306bc314c239"},"cell_type":"code","source":"summary = {}\nfor n, row in df_detailed.iterrows():\n    if row['class'] not in summary:\n        summary[row['class']] = 0\n    summary[row['class']] += 1\n    \nprint(summary)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80c69737add6e4deef6dbac97a99aa54ab4ae479"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}