{"cells":[{"metadata":{"_uuid":"4b952ca03320b50851b24591302a9b4f0a090c85"},"cell_type":"markdown","source":"# Overview\nThis notebook aims to explore the dataset and implement a basic (dumb) detection benchmark.\nMuch of the code here comes from these excellent kernels:\n- [Exploratory Data Analysis](https://www.kaggle.com/peterchang77/exploratory-data-analysis)\n- [Lung Opacity Overview](https://www.kaggle.com/kmader/lung-opacity-overview/notebook)\n\n## Combine all data into a single table"},{"metadata":{"trusted":true,"_uuid":"bffcfbd60ae2a765fa59d12bd8446e505c9db900","scrolled":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pydicom\nimport pylab\nimport pandas as pd\nfrom glob import glob\nimport os.path as op\n\nclass_info_path = '../input/stage_1_detailed_class_info.csv'\ntrain_labels_path = '../input/stage_1_train_labels.csv'\nimages_dir = '../input/stage_1_train_images/'\n\n# data frames\nclass_info_df = pd.read_csv(class_info_path)\ntrain_labels_df = pd.read_csv(train_labels_path)\nimages_df = pd.DataFrame({'path': glob(op.join(images_dir, '*.dcm'))})\nimages_df['patientId'] = images_df['path'].map(lambda x: op.splitext(op.basename(x))[0])\n# parse DICOM header into dataframe\nDICOM_TAGS = ['PatientAge', 'ViewPosition', 'PatientSex']\ndef get_tags(image_path):\n    tag_data = pydicom.read_file(image_path, stop_before_pixels = True)\n    tag_dict = {tag: getattr(tag_data, tag, '') for tag in DICOM_TAGS}\n    tag_dict['path'] = image_path\n    return pd.Series(tag_dict)\nmeta_df = images_df.apply(lambda x: get_tags(x['path']), 1)\nmeta_df['PatientAge'] = meta_df['PatientAge'].map(int)\nmeta_df.drop('path', 1).describe(exclude=np.number)\n\n# concatenate the data frames\ninfo_df = pd.concat([class_info_df, train_labels_df.drop('patientId', 1)], 1)\nimage_with_meta_df = pd.merge(images_df, meta_df, on='path')\nbbox_with_info_df = pd.merge(info_df, image_with_meta_df, on='patientId', how='left')\n\nbbox_with_info_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a17436e7e9caaa212327b4429b78848df7d62f6"},"cell_type":"markdown","source":"## Parse data into a dictionary by patient ID\nKeeping the data in the dataframe format makes it easy to compute all sorts of gross statistics, but it is also useful to have a simple way for accessing all the data belonging to a single patient."},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"5976eb556d6eca5c00deee0b402116892eb5ced5"},"cell_type":"code","source":"def parse_patient_data(df):\n    \"\"\"\n    Parse pandas dataframe into the following dictionary:\n      data = {\n        patientID: {\n          'dicom': path/to/dicom/file,\n          'target': 0 if normal, 1 if pneumonia,\n          'boxes': list of box(es), each box is an array of number [x y width height],\n          'class': one of the three values 'Lung Opacity', 'No Lung Opacity / Not Norma', 'Normal',\n          'age': age of the patient,\n          'view': either 'AP' - anteriorposterior, or 'PA' - posterioranterior,\n          'sex': either 'Male' or 'Female'\n        },\n        ...\n      }\n    \"\"\"\n    \n    extract_box = lambda row: [row['x'], row['y'], row['width'], row['height']]\n    \n    data = {}\n    for n, row in df.iterrows():\n        pid = row['patientId']\n        if pid not in data:\n            data[pid] = {\n                'dicom': '%s/%s.dcm' % (images_dir, pid),\n                'target': row['Target'],\n                'class': row['class'],\n                'age': row['PatientAge'],\n                'view': row['ViewPosition'],\n                'sex': row['PatientSex'],\n                'boxes': []}\n            \n        if data[pid]['target'] == 1:\n            data[pid]['boxes'].append(extract_box(row))\n    return data\n\npatients_data = parse_patient_data(bbox_with_info_df)\npatient_ids = list(patients_data.keys())\nprint(patients_data[np.random.choice(patient_ids)])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dd3409c57ab43dc06b2e47b54755f2420aa3ef89"},"cell_type":"markdown","source":"## Visualize patient data\nOverlay color boxes of opacities on the original grayscacle DICOM image of a patient."},{"metadata":{"trusted":true,"_uuid":"0ec9a2483a80b8da615b8a42ad39b0649e22307b"},"cell_type":"code","source":"def draw_single_patient(data, mean_image = None):\n    \"\"\"\n      Draw a single patient with overalying bounding boxes of opacities (if present)\n    \"\"\"\n    d = pydicom.read_file(data['dicom'])\n    im = d.pixel_array\n    \n    im = np.stack([im] * 3, axis = 2)\n    if (mean_image is np.ndarray):\n        im = np.subtract(im, mean_image)\n    \n    for box in data['boxes']:\n        im = overlay_box(im = im, box = box, stroke = 6)\n    pylab.imshow(im, cmap = pylab.cm.gist_gray)\n\ndef single_patient_info(patient_data, pid):\n    patient = patients_data[pid]\n    return ('id: %s\\ntarget: %s\\nclass: %s\\nage: %s\\nview: %s\\nsex: %s' % (pid, patient['target'], patient['class'], patient['age'], patient['view'], patient['sex']))\n\ndef overlay_box(im, box, color = [256, 200, 200], stroke = 1):\n    \"\"\"\n      Overlay a single box on the image\n    \"\"\"\n    \n    box = [int(d) for d in box]\n    [x1, y1, width, height] = box\n    x2 = x1 + width\n    y2 = y1 + height\n    \n    im[y1:y1 + stroke, x1:x2] = color\n    im[y2:y2 + stroke, x1:x2] = color\n    im[y1:y2, x1:x1 + stroke] = color\n    im[y1:y2, x2:x2 + stroke] = color\n    \n    return im\n\nplt.figure(num=None, figsize=(8, 6), dpi=80, facecolor='w', edgecolor='k')\n\nsample_pid = '00436515-870c-4b36-a041-de91049b9ab4'\nplt.title(single_patient_info(patients_data, sample_pid))\ndraw_single_patient(patients_data[sample_pid])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24f61ec1ad26f7b8718e67509c113a88ac649cdc"},"cell_type":"markdown","source":"## Data analysis\n### Mean image"},{"metadata":{"trusted":true,"_uuid":"4fe2cb32742653cc2b635819a06bf0a5ca5bdc02"},"cell_type":"code","source":"def mean_image(patients_data, pids):\n    mean_im = np.zeros((1024, 1024), dtype='uint8')\n    \n    for pid in pids:\n        d = pydicom.read_file(patients_data[pid]['dicom'])\n        im = d.pixel_array\n        mean_im = np.add(mean_im, im)\n    mean_im = np.round(np.multiply(np.divide(mean_im, np.amax(mean_im)), 255)).astype(int)\n    mean_im = np.stack([mean_im] * 3, axis = 2)\n    return mean_im\n\nglobal_mean = mean_image(patients_data, patient_ids)\npylab.imshow(global_mean, cmap = pylab.cm.gist_gray)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a82e46dbd6634537bff5151ff3fbdd81448b04ce"},"cell_type":"markdown","source":"### AP vs PA"},{"metadata":{"trusted":true,"_uuid":"4d0410450ee13ad4d6831797e3b39bdf21b81741"},"cell_type":"code","source":"ap_patient_ids = bbox_with_info_df.loc[bbox_with_info_df['ViewPosition'] == 'AP', 'patientId'].values\npa_patient_ids = bbox_with_info_df.loc[bbox_with_info_df['ViewPosition'] == 'PA', 'patientId'].values\n\nfig, ax = plt.subplots()\nplt.bar(np.arange(1, 3), list(map(lambda x: x.shape[0], [ap_patient_ids, pa_patient_ids])))\nax.set_ylabel('# patients')\nax.set_xticks(np.arange(1,3))\nax.set_xticklabels(['AP', 'PA'])\n\nmean_ap_img = mean_image(patients_data, ap_patient_ids)\nmean_pa_img = mean_image(patients_data, pa_patient_ids)\n\nplt.figure(num=None, figsize=(16, 8), dpi=80, facecolor='w', edgecolor='k')\nplt.subplot(221)\nsample_ap_pid = np.random.choice(ap_patient_ids)\nplt.title(single_patient_info(patients_data, sample_ap_pid))\ndraw_single_patient(patients_data[sample_ap_pid])\nplt.subplot(223)\ndraw_single_patient(patients_data[sample_ap_pid], mean_ap_img)\n\nplt.subplot(222)\nsample_pa_pid = np.random.choice(pa_patient_ids)\nplt.title(single_patient_info(patients_data, sample_pa_pid))\ndraw_single_patient(patients_data[sample_pa_pid])\nplt.subplot(224)\ndraw_single_patient(patients_data[sample_pa_pid], mean_pa_img)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}