{"cells":[{"metadata":{"_uuid":"8c993b4a066ee7b20443912340433f5f3c88321c"},"cell_type":"markdown","source":"In this notebook, we focus on enhancing and combining bounding box information in order to glean boundaries within which to focus classification/segmentation. "},{"metadata":{"trusted":true,"_uuid":"83c74cdf44b27e6b206556b8979bf54eda588e30","collapsed":true},"cell_type":"code","source":"# basic imports\nimport os, random\nimport pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcab09c98e7e02740e00ac5bd2e533658316484d","collapsed":true},"cell_type":"code","source":"# what do we have here?\nprint (os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12173594123e32289906a8d18507a069337e1631","collapsed":true},"cell_type":"code","source":"TRAIN_LABELS_CSV_FILE=\"../input/stage_1_train_labels.csv\"\n# pedantic nit: we are changing 'Target' to 'label' on the way in\nTRAIN_LABELS_CSV_COLUMN_NAMES=['patientId', 'x1', 'y1', 'width', 'height', 'label']\n\n# we will pre-process bounding boxes into the following format\n# we will add x2=x1+width and y2=x2+height\n# NaN rows (non 'Lung Opacity' rows) which do not have bounding boxes will be discarded\nTRAIN_BOUNDINGBOX_CSV_FILE=\"stage_1_train_boundingboxes.csv\"\nTRAIN_BOUNDINGBOX_CSV_COLUMN_NAMES=['patientId', 'x1', 'y1', 'width', 'height', 'x2', 'y2']\n\n# we will compute 'superset' bounding boxes for each patientId\nTRAIN_COMBINED_BOUNDINGBOX_CSV_FILE=\"stage_1_train_combinedboxes.csv\"\nTRAIN_COMBINED_BOUNDINGBOX_CSV_COLUMN_NAMES=['patientId', 'x1min', 'y1min', 'maxwidth', 'maxheight', 'x2max', 'y2max']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c98ce0e223f497a3c18951e1b0fa5e8863675d88","collapsed":true},"cell_type":"code","source":"# read TRAIN_LABELS_CSV_FILE into a pandas dataframe\nlabelsbboxdf = pd.read_csv(TRAIN_LABELS_CSV_FILE,\n                           names=TRAIN_LABELS_CSV_COLUMN_NAMES,\n                           # skip the header line\n                           header=0,\n                           # index the dataframe on patientId\n                           index_col='patientId')\nprint (labelsbboxdf.shape)\n#print (labelsbboxdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7e024b147701f90fe8f14587a1814d3ad47feba","collapsed":true},"cell_type":"code","source":"# grab labels by unique patienId\nlabelsdf=pd.DataFrame(labelsbboxdf.pop('label'), columns=['label'])\n# remove duplicates\nlabelsdf=pd.DataFrame(labelsdf.groupby(['patientId'])['label'].first(), columns=['label'])\nprint (labelsdf.shape)\n#print (labelsdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b65e0eb0753e9eb5997539e7a1c107edc59dde0","collapsed":true},"cell_type":"code","source":"# after 'label' is popped off, x1,y1,width,height are left in labelsbboxdf\n# drop missing values (all rows except ones labeled as having  'Lung Opacity' will be dropped)\nbboxesdf=labelsbboxdf.dropna()\nprint (bboxesdf.shape)\n#print(bboxesdf.head(n=10))\n\n# create coordinates for right hand bottom corner for all bounding boxes\nbboxesdf['x2']=bboxesdf['x1']+bboxesdf['width']\nbboxesdf['y2']=bboxesdf['y1']+bboxesdf['height']\nprint (bboxesdf.shape)\n#print(bboxesdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b24c858afa934f6426e681f1a46a47ffb8e961f","collapsed":true},"cell_type":"code","source":"# let's view the bounding box information with the new fields\nbboxesdf.head(n=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ce712105f34d32dc7a403c78c662f78a7572482","collapsed":true},"cell_type":"code","source":"# what can we glean from the raw bounding boxes?\nbboxesdf.describe(percentiles=[0.05, 0.95])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d6a025eb36b98e966766da8538a34970f8c32f8","collapsed":true},"cell_type":"code","source":"# the bounding boxes are situated across the entire image dimensions\n# from lowest values of x1,y1=2.0,2.0 to largest values of x2,y2=1024,1024\n\n# we can focus our classification and segmentation efforts inside the top-left\n# and bottom-right corners created at 5th and 95th percentile boundaries by\n# (x1, y1) and (x2, y2) respectively\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fae374d44f63f3654df25923c59a3a7405720d0a","collapsed":true},"cell_type":"code","source":"# create unique bounding boxes by patientId that 'subsume' smallest and largest endpoints\nx1y1df=bboxesdf.groupby(['patientId'])['x1', 'y1'].apply(np.min, axis=0)\nx2y2df=bboxesdf.groupby(['patientId'])['x2', 'y2'].apply(np.max, axis=0)\ncombinedbboxdf=pd.concat([x1y1df, x2y2df], axis=1)\n# modify the column names to provide context\ncombinedbboxdf.rename(columns={'x1':'x1min', 'y1':'y1min', 'x2':'x2max', 'y2':'y2max'}, inplace=True)\n# recompute width and height for combined bounding boxes\ncombinedbboxdf['maxwidth']=combinedbboxdf['x2max']-combinedbboxdf['x1min']\ncombinedbboxdf['maxheight']=combinedbboxdf['y2max']-combinedbboxdf['y1min']\n# make the column order consistent\ncombinedbboxdf=combinedbboxdf[['x1min', 'y1min', 'maxwidth', 'maxheight', 'x2max', 'y2max']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59bd91efe7ea83d64d018c4ddfaf5de4c195c79b","collapsed":true},"cell_type":"code","source":"# let's view the superset bounding boxes\ncombinedbboxdf.head(n=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b676fdceb167480e400d0a527a6b1268d411df1","collapsed":true},"cell_type":"code","source":"# what can we glean from the superset bounding boxes?\ncombinedbboxdf.describe(percentiles=[0.05, 0.95])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"353527b99cec8cf7b3514737dde3bd95ff58a5d3","collapsed":true},"cell_type":"code","source":"# again, the bounding boxes are situated across the entire image dimensions,\n# but we can focus inside the 5th and 95th percentile boundaries as above","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d16fa30fae12fac642120333d8ffddb221a20cb","collapsed":true},"cell_type":"code","source":"bboxesdf.to_csv(TRAIN_BOUNDINGBOX_CSV_FILE)\ncombinedbboxdf.to_csv(TRAIN_COMBINED_BOUNDINGBOX_CSV_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a65958bca8a61d2c26c130bf838472580d59768e","collapsed":true},"cell_type":"code","source":"!head -10 stage_1_train_boundingboxes.csv\n!head -10 stage_1_train_combinedboxes.csv","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}