{"cells":[{"metadata":{"_uuid":"9590411d1805a0280f23f5ed53bba98777b73694"},"cell_type":"markdown","source":"# Matching Overview\nSince we know the images came from the NIH dataset, let's try to match them back and maybe even use some of the predictions as another baseline submission.\n\n## Process\n- Get metadata from all the dicoms in the RSNA competition\n- Get the metadata from the NIH images\n- Match using age, gender, scan type, and even pixel data if needed."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pydicom\nimport pandas as pd\nfrom glob import glob\nimport os\nfrom matplotlib.patches import Rectangle","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"54213ec6c6d1350c534289ffe9e846c7bf90ff95"},"cell_type":"markdown","source":"# Read Metadata from RSNA Pneumonia Images\nHere we read and load all the metadata from the RSNA images"},{"metadata":{"trusted":true,"_uuid":"7988cda88cb07382c0b9a82f32146fa09b9a9813"},"cell_type":"code","source":"det_class_path = '../input/rsna-pneumonia-detection-challenge/stage_1_detailed_class_info.csv'\nbbox_path = '../input/rsna-pneumonia-detection-challenge/stage_1_train_labels.csv'\ndicom_dir = '../input/rsna-pneumonia-detection-challenge/stage_1_train_images/'\ntest_dicom_dir = '../input/rsna-pneumonia-detection-challenge/stage_1_test_images/'","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"image_df = pd.DataFrame({'path': glob(os.path.join(dicom_dir, '*.dcm'))+\n                         glob(os.path.join(test_dicom_dir, '*.dcm'))\n                        })\nimage_df['patientId'] = image_df['path'].map(lambda x: os.path.splitext(os.path.basename(x))[0])\nimage_df['data_split'] = image_df['path'].map(lambda x: x.split('/')[-2].split('_')[-2])\nprint(image_df.shape[0], 'images found')\nimage_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a811f1acf18dbf51251e5d1da57952c1a801b8e"},"cell_type":"code","source":"DCM_TAG_LIST = ['PatientAge', 'BodyPartExamined', 'ViewPosition', 'PatientSex']\ndef get_tags(in_path):\n    c_dicom = pydicom.read_file(in_path, stop_before_pixels=True)\n    tag_dict = {c_tag: getattr(c_dicom, c_tag, '') \n         for c_tag in DCM_TAG_LIST}\n    tag_dict['path'] = in_path\n    return pd.Series(tag_dict)\nimage_meta_df = image_df.apply(lambda x: get_tags(x['path']), 1)\n# show the summary\nimage_meta_df['PatientAge'] = image_meta_df['PatientAge'].map(int)\nimage_meta_df['PatientAge'].hist()\nimage_meta_df.drop('path',1).describe(exclude=np.number)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6803ec83ab7b437c9a0231ccbbcc5f7b4a770999"},"cell_type":"code","source":"rsna_df = pd.merge(image_meta_df, image_df, on='path')\nprint('Overlapping meta and name data', rsna_df.shape[0])\nrsna_df.drop('path',1).describe(exclude=np.number)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"874fa8fbf79210b60cf0aa43158449c1d61170fe"},"cell_type":"markdown","source":"# Read NIH Image Data\nThe NIH data has already been extracted and so should be much easier to get to"},{"metadata":{"trusted":true,"_uuid":"1cd15b48dfa0715adc09498c97884ebaabd27cab"},"cell_type":"code","source":"nih_data_df = pd.read_csv('../input/data/Data_Entry_2017.csv')\nnih_bbox_df = pd.read_csv('../input/data/BBox_List_2017.csv')\nnih_path_map = {os.path.basename(x): x for x in glob('../input/data/*/images/*.png')}\nprint('Number of images', len(nih_path_map))\nprint('Number of cases', nih_data_df.shape[0])\nprint('Number of bounding boxes', nih_bbox_df.shape[0])\nnih_data_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d981d23a21085528e173d5793eab2bb5be2c5fe"},"cell_type":"code","source":"simple_nih_df = nih_data_df.rename(columns={'Patient Age': 'PatientAge', \n                         'View Position': 'ViewPosition',\n                         'Patient Gender': 'PatientSex'})[['Image Index', 'PatientAge', 'PatientSex', 'ViewPosition', 'Finding Labels']]\nsimple_nih_df['path'] = simple_nih_df['Image Index'].map(nih_path_map.get)\nsimple_nih_df.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"33a828e6b565fcbe5a2a106607c1b1d141f6765d"},"cell_type":"markdown","source":"# Joining RSNA and NIH data\nWe join the data based on the similar columns and then compare the images to find an exact match"},{"metadata":{"trusted":true,"_uuid":"15a857fcb1c256484914e29583645d520cc5a157"},"cell_type":"code","source":"join_nih_fcn = lambda x: pd.merge(x, simple_nih_df, \n                         on=['PatientAge', 'PatientSex', 'ViewPosition'], \n                         suffixes=['_rsna', '_nih'])\ntest_merge_df = join_nih_fcn(rsna_df.sample(1))\nprint(test_merge_df.shape[0], 'matches for a single entry in the RSNA dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f746fb7f26806a59196788c838ef965a26ec8721"},"cell_type":"code","source":"from skimage.io import imread\nfrom functools import lru_cache\ndef read_png(in_path):\n    return imread(in_path, as_gray=True)\ndef read_dicom(in_path):\n    return pydicom.read_file(in_path).pixel_array\nread_dicom_cached = lru_cache(1)(read_dicom) # we have lots of double counting\ntest_merge_df['rsna_data'] = test_merge_df['path_rsna'].map(read_dicom_cached)\ntest_merge_df['nih_data'] = test_merge_df['path_nih'].map(read_png)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a070ed9043d323a64842ebf94683ca563b5d2f60"},"cell_type":"markdown","source":"## Scaling\nThe images have clearly been scaled to different values and so we need to correct for the scaling before subtracting"},{"metadata":{"trusted":true,"_uuid":"1610026b1f85548541bb8f4992c1febc2eea9528"},"cell_type":"code","source":"mean_rsna = test_merge_df['rsna_data'].map(np.mean)\nmean_nih = test_merge_df['nih_data'].map(np.mean)\nfig, (ax1) = plt.subplots(1, 1, figsize = (8, 4))\nax1.hist(mean_rsna, np.linspace(0, 255, 30), label='RSNA Data')\nax1.hist(mean_nih, np.linspace(0, 255, 30), label='NIH Data', alpha = 0.5)\nax1.legend();","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"36884b6694567599cee294dfcda9921cbc349c1e"},"cell_type":"markdown","source":"## Normalization\n"},{"metadata":{"trusted":true,"_uuid":"171244b5cafa2e0ed09d12cbcf9fefa4c1dd361c"},"cell_type":"code","source":"norm_func = lambda x: (x-1.0*np.mean(x)).astype(np.float32)/np.std(x)\nmean_rsna = test_merge_df['rsna_data'].map(lambda x: np.max(norm_func(x)))\nmean_nih = test_merge_df['nih_data'].map(lambda x: np.max(norm_func(x)))\nfig, (ax1) = plt.subplots(1, 1, figsize = (8, 4))\nax1.hist(mean_rsna, np.linspace(0, 3, 30), label='RSNA Data')\nax1.hist(mean_nih, np.linspace(0, 3, 30), label='NIH Data', alpha = 0.5)\nax1.legend();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e00c8c9a05e02c9567007011c53e3df2a495df70","scrolled":true},"cell_type":"code","source":"test_merge_df['img_diff'] = test_merge_df.apply(lambda c_row: np.mean(np.abs(norm_func(c_row['rsna_data'])-norm_func(c_row['nih_data']))),1)\ntest_merge_df['img_corr'] = test_merge_df.apply(lambda c_row: np.corrcoef(c_row['rsna_data'].ravel(), \n                                                                          c_row['nih_data'].ravel())[0, 1],1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a630063d75cd54e92df6968f83af796f9c802378"},"cell_type":"code","source":"fig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (21, 7))\nax1.hist(test_merge_df['img_diff'])\nax1.set_title('MAE')\nax2.hist(test_merge_df['img_corr'])\nax2.set_title('Correlation')\nax3.plot(test_merge_df['img_diff'], test_merge_df['img_corr'], '.')\nax3.set_title('MAE vs Correlation')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dfe6f9cf5d9a841452055136e1c8e5d24c836689"},"cell_type":"markdown","source":"## Show the top 3 matches"},{"metadata":{"trusted":true,"_uuid":"f267cf8c4b74fd412d55981b87f89879b4c563d5"},"cell_type":"code","source":"fig, m_axs = plt.subplots(2, 3, figsize = (25, 10))\n[c_ax.axis('off') for c_ax in m_axs.flatten()]\nfor (i, (ax1, ax2)), (_, c_row) in zip(enumerate(m_axs.T, 1),\n                                  test_merge_df.sort_values('img_corr', ascending=False).iterrows()\n                                 ):\n    ax1.imshow(c_row['rsna_data'], cmap='gray')\n    ax1.set_title('RSNA Data')\n    ax2.imshow(c_row['nih_data'], cmap='gray')\n    ax2.set_title('NIH Data Rank:{}\\nCorrelation: {:2.1f}%, MAE: {:2.2f}'.format(i, 100*c_row['img_corr'], c_row['img_diff']))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b566834e44e33b63006ac80f496fe4bfd558f187"},"cell_type":"markdown","source":"### Seems to be a reasonable match\n0.01 might be a good cut-off for matching datasets well"},{"metadata":{"trusted":true,"_uuid":"98cddfee17c8de2a8e7a0db0c191c72f92bc708b"},"cell_type":"code","source":"full_merge_df = join_nih_fcn(rsna_df)\nprint(full_merge_df.shape[0], 'total number of overlapping entires')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c05c75a06eb88803ad49884be0b0a592746b4c4c"},"cell_type":"markdown","source":"## One Group at a time\nWe see there are too many (13million load operations is a lot) to try a brute force match and so we can load the images one group at a time and see which ones match best"},{"metadata":{"trusted":true,"_uuid":"0a7e97b0e3d985fc2bbda052ade29d18fb280452"},"cell_type":"code","source":"matched_groups = rsna_df.groupby(['PatientAge', 'PatientSex', 'ViewPosition'])\nprint('Number of unique patient age, sex and view combinations:', len(matched_groups))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12afe1418f9fb1dd71435fe2d29bc6566ac13833"},"cell_type":"code","source":"DS_FACTOR = 4\n@lru_cache(maxsize=None)\ndef read_png_cached(in_path):\n    return imread(in_path, as_gray=True)[::DS_FACTOR, ::DS_FACTOR]\n@lru_cache(maxsize=None)\ndef read_dicom_cached(in_path):\n    return pydicom.read_file(in_path).pixel_array[::DS_FACTOR, ::DS_FACTOR]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"22e86b17d2bc24790f5ed71322b4c8fd4cf46ec0"},"cell_type":"code","source":"out_matches_list = []\nfrom tqdm import tqdm_notebook\nimport gc\ngc.enable()\ngc.collect()\ndef calc_match_dist(in_row):\n    nih_img = read_png_cached(in_row['path_nih'])\n    rsna_img = read_dicom_cached(in_row['path_rsna'])\n    return np.corrcoef(nih_img.ravel(), rsna_img.ravel())[0, 1]\n\nfor _, rsna_grp_rows_df in tqdm_notebook(matched_groups):\n    # reset the caches\n    gc.collect()\n    gr_combo_df = join_nih_fcn(rsna_grp_rows_df)\n    gr_combo_df['img_corr'] = gr_combo_df.apply(calc_match_dist,1)\n    # add the top match for each group to the list\n    out_matches_list += [gr_combo_df.groupby('patientId').\\\n                             apply(lambda x: x.sort_values('img_corr', ascending=False).head(1)).\\\n                             reset_index(drop=True)]\n    read_dicom_cached.cache_clear()\n    read_png_cached.cache_clear()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9160e85dffee7735fed9c42d9b656697ddd5c840"},"cell_type":"code","source":"matches_df = pd.concat(out_matches_list)\nmatches_df.to_csv('matched_images.csv', index=False)\nmatches_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30be53429520b4bcf724b8f6b7af2efa887772f6"},"cell_type":"code","source":"matches_df['img_corr'].plot.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8d8854602c7b0f367048fc1a4b55e4665a1db69"},"cell_type":"code","source":"matches_df['Finding Labels'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"719dbb7a86fcee77b16ac14e355deec8c6e3ecd0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}