{"cells":[{"metadata":{"trusted":true,"_uuid":"660dbbbcf7c2fd8b1127afa7f8a97693e8633a0d"},"cell_type":"markdown","source":"## Sections\n- [Parsing Tabular Data](#Parsing-Tabular-Data)\n- [Parsing Metadata from DICOM Object](#Parsing-Metadata-from-DICOM-Object)\n- [Cleaning DICOM Metadata and Merging to Tabular Data](#Cleaning-DICOM-Metadata-and-Merging-to-Tabular-Data)\n- [EDA on Metadata and Initial Tabular Data](#EDA-on-Metadata-and-Initial-Tabular-Data)\n    - [ViewPosition](#ViewPosition)\n    - [PatientAge](#PatientAge)\n    - [PatientSex](#PatientSex)\n    - [PixelSpacing](#PixelSpacing)\n    - [Initial Tabular Data](#Initial-Tabular-Data)\n    - [Bounding Box](#Bounding-Box)\n        - [Bounding Box Data Manipulation](#Bounding-Box-Data-Manipulation)\n        - [Bounding Box Plots](#Bounding-Box-Plots)\n    - [Image Intensity Distribution](#Image-Intensity-Distribution)\n- [Summary (TL;DR)](#Summary-%28TL%3BDR%29)\n- [Simple LGBM Binary Classifier](#Simple-LGBM-Binary-Classifier)"},{"metadata":{"_uuid":"275a16df7ba4eff5c63707b14a23eea7b480da34"},"cell_type":"markdown","source":"## Summary (TL;DR)\n- ViewPosition, PatientAge, PatientSex, PixelSpacing likely the only fields in metadata with possible value [[relevant section](#Initial-takeaways)]\n- ViewPosition seems to be a vital feature [[relevant section](#ViewPosition)]\n- Bounding boxes are more prevalent on the right lung [[relevant section](#Bounding-Box-Plots)]"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from functools import partial\nfrom collections import defaultdict\nimport pydicom\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\nfrom joblib import Parallel, delayed\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport seaborn as sns\nsns.set_style('whitegrid')\n%matplotlib inline\n\nnp.warnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"07d6186c26ac1715838fe1010cbb5c8a64ff1d22"},"cell_type":"markdown","source":"## Parsing Tabular Data"},{"metadata":{"trusted":true,"_uuid":"a1541704567e653eadf051a18b64ce6a20a3784f"},"cell_type":"code","source":"labels = pd.read_csv('../input/stage_1_train_labels.csv')\ndetails = pd.read_csv('../input/stage_1_detailed_class_info.csv')\n# duplicates in details just have the same class so can be safely dropped\ndetails = details.drop_duplicates('patientId').reset_index(drop=True)\nlabels_w_class = labels.merge(details, how='inner', on='patientId')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6dd362c52ed402b6e8f5831192fad76661af0810"},"cell_type":"markdown","source":"## Parsing Metadata from DICOM Object"},{"metadata":{"trusted":true,"_uuid":"5cc571798a0721065f3148a8be0ac646d7869972"},"cell_type":"code","source":"# get lists of all train/test dicom filepaths\ntrain_dcm_fps = glob.glob('../input/stage_1_train_images/*.dcm')\ntest_dcm_fps = glob.glob('../input/stage_1_test_images/*.dcm')\n\n# read each file into a list (using stop_before_pixels to avoid reading the image for speed and memory savings)\ntrain_dcms = [pydicom.read_file(x, stop_before_pixels=True) for x in train_dcm_fps]\ntest_dcms = [pydicom.read_file(x, stop_before_pixels=True) for x in test_dcm_fps]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0b69d9b09971f9f18c5290920f494c78611b72a","_kg_hide-input":false,"_kg_hide-output":false},"cell_type":"code","source":"def parse_dcm_metadata(dcm):\n    unpacked_data = {}\n    group_elem_to_keywords = {}\n    # iterating here to force conversion from lazy RawDataElement to DataElement\n    for d in dcm:\n        pass\n    # keys are pydicom.tag.BaseTag, values are pydicom.dataelem.DataElement\n    for tag, elem in dcm.items():\n        tag_group = tag.group\n        tag_elem = tag.elem\n        keyword = elem.keyword\n        group_elem_to_keywords[(tag_group, tag_elem)] = keyword\n        value = elem.value\n        unpacked_data[keyword] = value\n    return unpacked_data, group_elem_to_keywords\n\ntrain_meta_dicts, tag_to_keyword_train = zip(*[parse_dcm_metadata(x) for x in train_dcms])\ntest_meta_dicts, tag_to_keyword_test = zip(*[parse_dcm_metadata(x) for x in test_dcms])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"25badb83a5f6b4ea14389819dfd92ededce4c982"},"cell_type":"markdown","source":"### Using the easily interpretable keyword instead of the DICOM tag (group, element) for column names. However, the DICOM tag might be useful for searching online for more info / detailed explanation (e.g. DICOMLookup)."},{"metadata":{"trusted":true,"_uuid":"3f9997fcef6c15f8c0cdf029ebcbc05d05b1df4c"},"cell_type":"code","source":"# join all the dicts\nunified_tag_to_key_train = {k:v for dict_ in tag_to_keyword_train for k,v in dict_.items()}\nunified_tag_to_key_test = {k:v for dict_ in tag_to_keyword_test for k,v in dict_.items()}\n\n# quick check to make sure there are no different keys between test/train\nassert len(set(unified_tag_to_key_test.keys()).symmetric_difference(set(unified_tag_to_key_train.keys()))) == 0\n\ntag_to_key = {**unified_tag_to_key_test, **unified_tag_to_key_train}\ntag_to_key","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36e6984521246a5e773e5d8d76be9bfb7a9c0108"},"cell_type":"code","source":"# using from_records here since some values in the dicts will be iterables and some are constants\ntrain_df = pd.DataFrame.from_records(data=train_meta_dicts)\ntest_df = pd.DataFrame.from_records(data=test_meta_dicts)\ntrain_df['dataset'] = 'train'\ntest_df['dataset'] = 'test'\ndf = pd.concat([train_df, test_df])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e47196d88384b92fb9c1d61c43c6e4f7600d5489"},"cell_type":"code","source":"df.head(1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2674536f774df4e0c38dbfc8abd91d98c68809cf"},"cell_type":"markdown","source":"## Cleaning DICOM Metadata and Merging to Tabular Data"},{"metadata":{"trusted":true,"_uuid":"1e286fd35e98ee975c36c9cab4ef9b074efab893","scrolled":true},"cell_type":"code","source":"# separating PixelSpacing list to single values\ndf['PixelSpacing_x'] = df['PixelSpacing'].apply(lambda x: x[0])\ndf['PixelSpacing_y'] = df['PixelSpacing'].apply(lambda x: x[1])\ndf = df.drop(['PixelSpacing'], axis='columns')\n\n# x and y are always the same\nassert sum(df['PixelSpacing_x'] != df['PixelSpacing_y']) == 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9d481c575edb05065619388ef3db2528be39c5b"},"cell_type":"code","source":"# ReferringPhysicianName appears to just be empty strings\nassert sum(df['ReferringPhysicianName'] != '') == 0\n\n# SeriesDescription appears to be 'view: {}'.format(ViewPosition)\nset(df['SeriesDescription'].unique())\n\n# so these two columns don't have any useful info and can be safely dropped","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"edc3dcaf681557842bad67eba026e08747221249"},"cell_type":"markdown","source":"### Initial takeaways\n- Many of the fields are identical throughout all the samples (probably little to gain from looking into these)\n- Looks like PatientAge, PatientSex, PixelSpacing, and ViewPosition are the only metadata items with possible value\n- Wide range of ages\n- Small number of different Pixel Spacings (maybe relevant in terms of resolution and specific to certain imaging machines or setups?)\n- 2 different view positions"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"a05cfbb3b930730a9b6eed194678784fcda2e792"},"cell_type":"code","source":"nunique_all = df.aggregate('nunique')\nnunique_all","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f0828983800c42a77313eb18fd0e332fdd12480"},"cell_type":"code","source":"# drop constant cols and other two from above\ndf = df.drop(nunique_all[nunique_all == 1].index.tolist() + ['ReferringPhysicianName', 'SeriesDescription'], axis='columns')\n\n# now that we have a clean metadata dataframe we can merge back to our initial tabular data with target and class info\ndf = df.merge(labels_w_class, how='left', left_on='PatientID', right_on='patientId')\n\ndf['PatientAge'] = df['PatientAge'].astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f805e219926a7d2fc6a701c1df8cbb29835f096"},"cell_type":"code","source":"# df now has multiple rows for some patients (those with multiple bounding boxes in label_w_class)\n# so creating one with no duplicates for patients\ndf_deduped = df.drop_duplicates('PatientID', keep='first')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9484e897fa4af52b6f34ec2b74ac4a5e9b0310b4"},"cell_type":"markdown","source":"## EDA on Metadata and Initial Tabular Data"},{"metadata":{"_uuid":"74f87e14b4341931202f77da399465f6a2215828"},"cell_type":"markdown","source":"## ViewPosition\n\n### View Position is likely important since it describes the positioning of the patient when the radiograph is taken. PA = posterior-anterior (ray enters back first, so back would be at the 'top' of the image) and AP = anterior-posterior (ray enters chest first). My understanding is that PA would be the desired position for detecting lung opacity, however it frequently can't be done in that way as the patient is to ill to stand (which is required for PA). I think there are several important elements here: the pose of the patient is different (standing vs laying), the image has a different 'view' (back-first vs chest-first) and as such different parts of the body may be occluded or obscured to an extent, image quality itself is probably better in PA.\n\nNote: I am not a medical professional so this may be an incorrect or incomplete understanding.\n\n### Takeaways from below plots:\n- Test/Train seems relatively well balanced AP vs PA\n- AP vs PA have a big shift in the 'Normal' and 'Lung Opacity' classes\n- Target 1/0 is well split for AP, but for PA there are many more 0s\n- Gender seems relatively well balanced AP vs PA\n- Age distributions seem similar AP vs PA\n- Different pixel spacing AP vs PA (probably due to different location of patient relative to imaging machine)"},{"metadata":{"trusted":true,"_uuid":"8dd03bf77f46e0b65483d9a8a5e27ff3dd173a24"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='ViewPosition', hue='dataset', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='ViewPosition', hue='dataset', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d09b29ccf0a631292d84f8798bf0007c980d2bd2"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='ViewPosition', hue='class', data=df[df['dataset']=='train'], ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='ViewPosition', hue='class', data=df_deduped[df_deduped['dataset']=='train'], ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c94b75a3539d1d4178477edd9bc533f14739d84f"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='ViewPosition', hue='Target', data=df[df['dataset']=='train'], ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='ViewPosition', hue='Target', data=df_deduped[df_deduped['dataset']=='train'], ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ccb619d5132fb588929bde468a16e00c12f4841"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='ViewPosition', hue='PatientSex', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='ViewPosition', hue='PatientSex', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee5a76318d7b65fccfc25d992cf88eabeb684b32"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.distplot(df[df['ViewPosition']=='AP']['PatientAge'], hist=True, kde=False, color='red', label='AP', ax=axes[0])\np = sns.distplot(df[df['ViewPosition']=='PA']['PatientAge'], hist=True, kde=False, color='gray', label='PA', ax=axes[0])\n_ = p.set_ylabel('Count')\n_ = p.legend()\n_ = p.set_title('With multiple')\n\np = sns.distplot(df_deduped[df_deduped['ViewPosition']=='AP']['PatientAge'], hist=True, kde=False, color='red', label='AP', ax=axes[1])\np = sns.distplot(df_deduped[df_deduped['ViewPosition']=='PA']['PatientAge'], hist=True, kde=False, color='gray', label='PA', ax=axes[1])\n_ = p.set_ylabel('Count')\n_ = p.legend()\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3fc78c457c62a265ff06c848c81e7882dfc64d42"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.countplot(df['PixelSpacing_x'], hue='ViewPosition', data=df, ax=axes[0])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('With multiple')\n\np = sns.countplot(df_deduped['PixelSpacing_x'], hue='ViewPosition', data=df_deduped, ax=axes[1])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3ffbd385282437d54d6b48b98380b07b98d908f"},"cell_type":"markdown","source":"### Viewing the actual images for AP vs PA. The PA images seem much clearer to my untrained eye."},{"metadata":{"trusted":true,"_uuid":"6535d0d0aaf99511eb6bab2ba99557fceba7a27a"},"cell_type":"code","source":"def read_img(patient_id):\n    train_fp = '../input/stage_1_train_images/{}.dcm'.format(patient_id)\n    if os.path.exists(train_fp):\n        dcm = pydicom.read_file(train_fp)\n    else:\n        test_fp = '../input/stage_1_test_images/{}.dcm'.format(patient_id)\n        dcm = pydicom.read_file(test_fp)\n    return dcm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c2730cc2571ea8c1a12565ea9f7bd94d7da1f07"},"cell_type":"code","source":"def plot_grid(df, pid_sample_list, nrows=3, ncols=3, draw_bbox=True, ax_off=True):\n    fig = plt.figure(figsize=(16, 12))\n    for i in range(nrows * ncols):\n        patient_id = pid_sample_list[i]\n        img = read_img(patient_id).pixel_array\n        ax = fig.add_subplot(nrows, ncols, i + 1)\n        plt.imshow(img, cmap='gray')\n        ax.set_title(patient_id)\n        if ax_off: \n            ax.set_axis_off()\n        if draw_bbox:\n            bbox_rows = df[df['PatientID'] == patient_id]\n            for _, row in bbox_rows.iterrows():\n                x, y = row['x'], row['y']\n                width, height = row['width'], row['height']\n                bbox = patches.Rectangle((x, y), width, height, linewidth=.5, edgecolor='r', facecolor='none')\n                ax.add_patch(bbox)\n    plt.tight_layout()\n    plt.subplots_adjust(wspace=.01, hspace=.01)\n    return fig","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d630d55683e195308a3a45c069b299d02358e85"},"cell_type":"code","source":"pa_ids = df[df['ViewPosition']=='PA']['PatientID'].sample(20).tolist()\n_ = plot_grid(df, pa_ids, nrows=2, ncols=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebf39e494f55c032085e0fdabac6fdd095e37d05"},"cell_type":"code","source":"ap_ids = df[df['ViewPosition']=='AP']['PatientID'].sample(20).tolist()\n_ = plot_grid(df, ap_ids, nrows=2, ncols=3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"327c5961b50512e883862ea96ff996dcae72aba1"},"cell_type":"markdown","source":"## PatientAge\n\n### Not a whole lot to say about age other than there are 5 outliers with ages greater than 140 (likely human entry error). Otherwise it seems like the distribution of age is fairly consistent when split by target and class."},{"metadata":{"trusted":true,"_uuid":"9b619333ba5fa9943e73646381b750605dc301b8"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.distplot(df[df['Target']==1]['PatientAge'], hist=True, kde=False, color='red', label='Target 1', ax=axes[0])\np = sns.distplot(df[df['Target']==0]['PatientAge'], hist=True, kde=False, color='gray', label='Target 0', ax=axes[0])\n_ = p.set_ylabel('Count')\n_ = p.set_title('With multiple')\n\np = sns.distplot(df_deduped[df_deduped['Target']==1]['PatientAge'], hist=True, kde=False, color='red', label='Target 1', ax=axes[1])\np = sns.distplot(df_deduped[df_deduped['Target']==0]['PatientAge'], hist=True, kde=False, color='gray', label='Target 0', ax=axes[1])\n_ = p.set_ylabel('Count')\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad101d7da1631ebf2d90d4f4652fc1af631c604e"},"cell_type":"code","source":"fig, axes = plt.subplots(3, 2, figsize=(14, 9), sharex=True)\nfor i, _class in enumerate(df['class'].dropna().unique()):\n    p = sns.distplot(df[df['class']==_class]['PatientAge'], hist=True, kde=False, ax=axes[i, 0])\n    _ = p.set_ylabel('Count')\n    _ = p.set_xlabel(f'PatientAge - {_class}')\n    if i == 0: p.set_title('With multiple')\n    \n    p = sns.distplot(df_deduped[df_deduped['class']==_class]['PatientAge'], hist=True, kde=False, ax=axes[i, 1])\n    _ = p.set_ylabel('Count')\n    _ = p.set_xlabel(f'PatientAge - {_class}')\n    if i == 0: p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"32bc4824e6235397730f6c7d8a411d5abba0b40e"},"cell_type":"markdown","source":"## PatientSex"},{"metadata":{"trusted":true,"_uuid":"ed4b17719aaa2e0da2cb585367e4a1480aaea146"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='PatientSex', hue='Target', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='PatientSex', hue='Target', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4099ae0f5df2ea64428c2814e0016677e6b93382"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.countplot(x='PatientSex', hue='class', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.countplot(x='PatientSex', hue='class', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b088e6b1f3209981878acb21116561c2ee5d442"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='Target', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='Target', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e68c062a1b0609fba37013159988889828c7d438"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='class', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='class', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e74de4e2c58ec6a55f8c77458ffee0134afb2d55"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='ViewPosition', data=df, ax=axes[0])\n_ = p.set_title('With multiple')\np = sns.boxplot(x='PatientSex', y='PatientAge', hue='ViewPosition', data=df_deduped, ax=axes[1])\n_ = p.set_title('Single row per patient')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c0f251e2d6125ef5dce30fe49b999b86479ea40d"},"cell_type":"markdown","source":"## PixelSpacing"},{"metadata":{"trusted":true,"_uuid":"d86b6a010f02653f4e70705eb9e6b083723afcb4"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.countplot(df['PixelSpacing_x'], hue='Target', data=df, ax=axes[0])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('With multiple')\n_ = p.legend(loc='upper right')\n\np = sns.countplot(df_deduped['PixelSpacing_x'], hue='Target', data=df_deduped, ax=axes[1])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('Single row per patient')\n_ = p.legend(loc='upper right')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd3d655073145253081953e627c9ffb0d6c4deec"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.countplot(df['PixelSpacing_x'], hue='class', data=df, ax=axes[0])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('With multiple')\n_ = p.legend(loc='upper right')\n\np = sns.countplot(df_deduped['PixelSpacing_x'], hue='class', data=df_deduped, ax=axes[1])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('Single row per patient')\n_ = p.legend(loc='upper right')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc1d1b9909f137873e3021ba6dada0634df6f2a0"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n\np = sns.countplot(df['PixelSpacing_x'], hue='PatientSex', data=df, ax=axes[0])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('With multiple')\n_ = p.legend(loc='upper right')\n\np = sns.countplot(df_deduped['PixelSpacing_x'], hue='PatientSex', data=df_deduped, ax=axes[1])\n_ = p.set_xticklabels([x for x in p.get_xticklabels()],rotation=90)\n_ = p.set_title('Single row per patient')\n_ = p.legend(loc='upper right')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ddecc2fdc0380581c21aef0481cebcad9d6475d"},"cell_type":"markdown","source":"## Initial Tabular Data"},{"metadata":{"trusted":true,"_uuid":"f038d0be8c78ba9d5c79f00a033a6d4fe712e149"},"cell_type":"code","source":"p = sns.countplot(x='Target', hue='class', data=labels_w_class)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c47ffdd5b60f970cb5b1864b5ff8d09644391e2a"},"cell_type":"code","source":"# check every row with Target==1 has a bounding box\nassert sum(labels_w_class['Target']==1) == sum(~labels_w_class['x'].isnull())\n\nbbox_counts = labels_w_class.groupby('patientId')['Target'].sum()\nlabels_w_class.index = labels_w_class.patientId\nlabels_w_class['bbox_counts'] = bbox_counts\nlabels_w_class = labels_w_class.reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"9b7802cdd44cad5f435bd9e8b21e5398d6af3fad"},"cell_type":"code","source":"p = sns.countplot(x='bbox_counts', hue='class', data=labels_w_class)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f39c869f17a2d3c66348573836fce49d6f8ad854"},"cell_type":"code","source":"p = sns.countplot(x='bbox_counts', hue='Target', data=labels_w_class)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9347feae9e95afa5ba332fc9ee19d7e7b40d5b50"},"cell_type":"code","source":"labels_w_class[['x', 'y', 'width', 'height']].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db898c099299fe46ec9fa983c681222bdc757d25"},"cell_type":"markdown","source":"## Bounding Box\n\n### Bounding Box Data Manipulation"},{"metadata":{"trusted":true,"_uuid":"be8ffeeb148b51f5e190de73c3416939d4bb07f1"},"cell_type":"code","source":"def build_bbox_arrays_by_id(df):\n    zeros_array_constructor = partial(np.zeros, shape=(1024,1024), dtype=np.uint8)\n    arrays = defaultdict(zeros_array_constructor)\n    for idx, row in df.iterrows():\n        patient_id = row['patientId']\n        x, y = int(row['x']), int(row['y'])\n        width, height = int(row['width']), int(row['height'])\n        array = arrays[patient_id]\n        array[y: y + height, x: x + width] += 1\n    return arrays","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7dfe35e366e616dd5fdb53753edc601e424238e0"},"cell_type":"code","source":"bbox_arrays = build_bbox_arrays_by_id(df[df['Target']==1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1874400ff9a5f6ece7f44899d6a229ec0f2672a6"},"cell_type":"code","source":"groups_to_ids = {\n    'pa': set(df['patientId'][df['ViewPosition']=='PA'].dropna().unique()),\n    'ap': set(df['patientId'][df['ViewPosition']=='AP'].dropna().unique()),\n    \n    'bbox_4': set(labels_w_class['patientId'][labels_w_class['bbox_counts']==4].dropna().unique()),\n    'bbox_3': set(labels_w_class['patientId'][labels_w_class['bbox_counts']==3].dropna().unique()),\n    'bbox_2': set(labels_w_class['patientId'][labels_w_class['bbox_counts']==2].dropna().unique()),\n    'bbox_1': set(labels_w_class['patientId'][labels_w_class['bbox_counts']==1].dropna().unique()),\n    \n    'f': set(df['patientId'][df['PatientSex']=='F'].dropna().unique()),\n    'm': set(df['patientId'][df['PatientSex']=='M'].dropna().unique()),\n    \n    'age_above_60': set(df['patientId'][df['PatientAge'] > 60].dropna().unique()),\n    'age_40_to_60': set(df['patientId'][(df['PatientAge'] <= 60) & (df['PatientAge'] >= 40)].dropna().unique()),\n    'age_below_40': set(df['patientId'][df['PatientAge'] < 40].dropna().unique()),\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1cf506d657c886776e8133f79e7feb69460f3977"},"cell_type":"code","source":"# construct arrays representing 'density' of bounding boxes by summing the arrays\nzeros_array_constructor = partial(np.zeros, shape=(1024,1024), dtype=np.uint32)\ngroups_to_bbox_sums = defaultdict(zeros_array_constructor)\ngroups_to_bbox_sums['all'] = np.zeros(shape=(1024,1024), dtype=np.uint32)\n\nfor patient_id, bbox_array in bbox_arrays.items():\n    # add to all group\n    groups_to_bbox_sums['all'] += bbox_array\n\n    # add to each other group where id is in that group's id set\n    for group, id_set in groups_to_ids.items():\n        if patient_id in id_set:\n            groups_to_bbox_sums[group] += bbox_array","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"759c4516de9ecea9b13e348c7e521c36623f4345"},"cell_type":"code","source":"def plot_density(array, ax, title, n_countour_levels=3):\n    contour_set = ax.contour(\n        np.arange(0, 1024, 1), \n        np.arange(1024, 0, -1),\n        array, \n        n_countour_levels, \n        linewidths=.5,\n        colors='black'\n    )\n    plt.clabel(contour_set, inline=True, fontsize=10, fmt='%.0f')\n\n    im = ax.imshow(\n        array, \n        extent=[0, 1024, 0, 1024], \n        origin='upper', \n        cmap='viridis', \n        alpha=.8\n    )\n    plt.colorbar(im, ax=ax)\n    ax.set_title(title)\n    return im","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39df02af06110828f9c692539fb3ae9f37fe7ab7"},"cell_type":"markdown","source":"## Bounding Box Plots\n\n### Takeaways from below plots\n* Right lung tends to have more in whole group\n* Right lung tendency seems more true in Age > 60 than the other age groups\n* Right lung tendency seems more true in PA than in AP (lungs close to balanced)\n* Right lung tendency seems more true in patients with 1 bbox vs 2/3/4"},{"metadata":{"trusted":true,"_uuid":"cb129c31f3c74f19c3fe5fcfd8522dc9b8ca14d9","scrolled":true},"cell_type":"code","source":"fig, axes = plt.subplots(1, 1, figsize=(14, 6), sharex=True)\n_ = plot_density(groups_to_bbox_sums['all'], axes, 'All Targets - Bounding Box Density')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1f12a2f969b713178ea0081e1a4fdbe4515a1f9"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 6), sharex=True)\n\n_ = plot_density(groups_to_bbox_sums['pa'], axes[0], 'PA - Bounding Box Density')\n_ = plot_density(groups_to_bbox_sums['ap'], axes[1], 'AP - Bounding Box Density')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40a10b536407772b5a7466c62f2f46041793f564"},"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(12, 12))\n\n_ = plot_density(groups_to_bbox_sums['bbox_1'], axes[0, 0], '1 Box - Bounding Box Density', n_countour_levels=3)\n_ = plot_density(groups_to_bbox_sums['bbox_2'], axes[0, 1], '2 Boxes - Bounding Box Density', n_countour_levels=3)\n_ = plot_density(groups_to_bbox_sums['bbox_3'], axes[1, 0], '3 Boxes - Bounding Box Density', n_countour_levels=1)\n_ = plot_density(groups_to_bbox_sums['bbox_4'], axes[1, 1], '4 Boxes - Bounding Box Density', n_countour_levels=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eba34329cb092f2ddec40dca5f2c81e5f4ecb05c"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n_ = plot_density(groups_to_bbox_sums['f'], axes[0], 'PatientSex - F', n_countour_levels=3)\n_ = plot_density(groups_to_bbox_sums['m'], axes[1], 'PatientSex - M', n_countour_levels=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e401f00ba18b2cfe9e9377aea2f4502351293560"},"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\n_ = plot_density(groups_to_bbox_sums['age_above_60'], axes[0], 'PatientAge > 60', n_countour_levels=3)\n_ = plot_density(groups_to_bbox_sums['age_40_to_60'], axes[1], '40 <= PatientAge <= 60', n_countour_levels=3)\n_ = plot_density(groups_to_bbox_sums['age_below_40'], axes[2], 'PatientAge < 40', n_countour_levels=3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a4c44fc825d88130ff992ac3da434d724d918e5e"},"cell_type":"markdown","source":"### Image Intensity Distribution"},{"metadata":{"trusted":true,"_uuid":"8b8d6cf71ce6a7079697971b76b280817154ed4c"},"cell_type":"code","source":"def mean_intensity(patientId):\n    img = read_img(patientId)\n    return np.mean(img.pixel_array)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5544a5130fd816060d9528bc5426126737a047db"},"cell_type":"code","source":"test_patients = df['PatientID'][df['class'].isnull()].tolist()\ntrain_patients = df['PatientID'][df['class'].isnull()].tolist()\n\ntest_mean_intensity = Parallel(n_jobs=4)(delayed(mean_intensity)(patientId) for patientId in test_patients)\ntrain_mean_intensity = Parallel(n_jobs=4)(delayed(mean_intensity)(patientId) for patientId in train_patients)\nall_mean_intensity = test_mean_intensity + train_mean_intensity\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\np = sns.distplot(all_mean_intensity, ax=axes[0], kde=False)\n_ = p.set_title('All - Image Mean Intensity')\n_ = p.set_xlabel('Mean Intensity')\n_ = p.set_ylabel('Count')\n\np = sns.distplot(train_mean_intensity, ax=axes[1], kde=False)\n_ = p.set_title('Train - Image Mean Intensity')\n_ = p.set_xlabel('Mean Intensity')\n_ = p.set_ylabel('Count')\n\np = sns.distplot(test_mean_intensity, ax=axes[2], kde=False)\n_ = p.set_title('Test - Image Mean Intensity')\n_ = p.set_xlabel('Mean Intensity')\n_ = p.set_ylabel('Count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b66f63d7bf8521617bf9be5c8ec50e5df3b78e3a"},"cell_type":"code","source":"pa_mean_intensity = Parallel(n_jobs=4)(delayed(mean_intensity)(patientId) for patientId in groups_to_ids['pa'])\nap_mean_intensity = Parallel(n_jobs=4)(delayed(mean_intensity)(patientId) for patientId in groups_to_ids['ap'])\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\np = sns.distplot(pa_mean_intensity, ax=axes[0], kde=False)\n_ = p.set_title('PA - Image Mean Intensity')\n_ = p.set_xlabel('Mean Intensity')\n_ = p.set_ylabel('Count')\np = sns.distplot(ap_mean_intensity, ax=axes[1], kde=False)\n_ = p.set_title('AP - Image Mean Intensity')\n_ = p.set_xlabel('Mean Intensity')\n_ = p.set_ylabel('Count')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac66312452e982d2d3aa874d072d1f30fd423f23"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"954cdf0b4931e05472fb6acb3c020429020699ef"},"cell_type":"markdown","source":"## Simple LGBM Binary Classifier"},{"metadata":{"trusted":true,"_uuid":"a2bd6d184a080b921ec80a58caab0f2d6aa97cd6"},"cell_type":"code","source":"df_ = df_deduped[~df_deduped['Target'].isnull()].reset_index(drop=True)\ndf_['patient_sex'] = df_['PatientSex'].replace({'F': 0, 'M': 1}).astype(np.int8)\ndf_['view_position'] = df_['ViewPosition'].replace({'AP': 0, 'PA': 1}).astype(np.int8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"4b86bff1283ea339f203432fcbd3e483ad583cb6"},"cell_type":"code","source":"cv = KFold(n_splits=10, random_state=2018, shuffle=True)\noof = np.zeros(len(df_))\nfeats = ['patient_sex', 'PatientAge', 'view_position', 'PixelSpacing_x']\n\nfor n_fold, (train_idx, val_idx) in enumerate(cv.split(df_[feats], df_['Target'])):\n    train_x, train_y = df_[feats].loc[train_idx], df_['Target'].loc[train_idx]\n    val_x, val_y = df_[feats].loc[val_idx], df_['Target'].loc[val_idx]\n\n    clf = LGBMClassifier(\n        nthread=0,\n        n_estimators=10000,\n        learning_rate=.001,\n        num_leaves=16,\n        colsample_bytree=.75,\n        subsample=.75,\n        max_depth=5,\n        reg_alpha=3,\n        reg_lambda=3,\n        min_child_weight=50,\n        silent=-1,\n        verbose=-1,\n    )\n\n    clf.fit(\n        train_x, \n        train_y, \n        eval_set=[(train_x, train_y), (val_x, val_y)], \n        eval_metric='auc', \n        verbose=500, \n        early_stopping_rounds=200,\n    )\n\n    oof[val_idx] = clf.predict_proba(val_x, num_iteration=clf.best_iteration_)[:, 1]\n    score = roc_auc_score(val_y, oof[val_idx])\n    print(f'\\nFold {n_fold + 1} AUC : {score}\\n')\n\nscore = roc_auc_score(df_['Target'], oof)\nprint(f'\\nAUC on All Folds : {score}\\n')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7911ca75b5accc387674af7f1e852551c3b0e0a5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}