{"cells":[{"metadata":{"_uuid":"ce5a0b0eb73fd9e7bb458bccc52894895c3a988c"},"cell_type":"markdown","source":"# Box Guess\nHere we take the predictions made in the [transfer learning](https://www.kaggle.com/kmader/lung-opacity-classification-transfer-learning/notebook) kernel and refine them into reasonable bounding boxes in order to make a submission.\n- We use the original input bounding boxes as a starting point\n- We divide into left and right lung since they appear to be broken up that way\n- We find the 95th percentile (`PERCENTILE_TO_KEEP`) for each of the parameters $x, y, width, height$\n- We submit two boxes (left and right) for each case above a certain threshold (`THRESHOLD_FOR_PREDICTION`)\n- We make define the two parameters below to make hyperparameter optimization easier"},{"metadata":{"trusted":true,"_uuid":"d82da601e4cc351b059948b31f9a64af5c4a6d10"},"cell_type":"code","source":"THRESHOLD_FOR_PREDICTION = 0.6\nPERCENTILE_TO_KEEP = 95","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nsub_df = pd.read_csv('../input/lung-opacity-classification-transfer-learning/submission.csv')\nsub_df['score'] = sub_df['PredictionString'].map(lambda x: float(x[:4]) if isinstance(x, str) else 0)\nsub_df.drop(['PredictionString'], axis=1, inplace=True)\nsub_df['score'].plot.hist()\nsub_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"all_bbox_df = pd.read_csv('../input/lung-opacity-overview/image_bbox_full.csv')\nall_bbox_df.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c58b52d775c2a208e3d8f32cc142276e50c3403c"},"cell_type":"code","source":"mini_df = all_bbox_df.\\\n             query('Target==1')[['x', 'y', 'width', 'height', 'boxes']]\nsns.pairplot(mini_df,\n            hue='boxes', \n             plot_kws={'alpha': 0.1})","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"all_bbox_df['x'].plot.hist()\nright_box = all_bbox_df.query('x>450')\nleft_box = all_bbox_df.query('y<450')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af8d44a54836e815353a68b761d75fa0c3563d7e"},"cell_type":"code","source":"def percentile_box(in_df, pct=95):\n    return (\n        np.percentile(in_df['x'], 100-pct),\n        np.percentile(in_df['y'], 100-pct),\n        np.percentile(in_df['width'], pct),\n        np.percentile(in_df['height'], pct)\n    )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dcffdfb80d46b64c1fb535fe1dc4528700e97e09"},"cell_type":"markdown","source":"## Show the boxes\nHere we show the boxes that we predict"},{"metadata":{"trusted":true,"_uuid":"173675edb2cb24e8cca8bf337ddfd38a0a6651a9"},"cell_type":"code","source":"right_bbox = percentile_box(right_box, PERCENTILE_TO_KEEP)\nleft_bbox = percentile_box(left_box, PERCENTILE_TO_KEEP)\nprint(right_bbox)\nprint(left_bbox)\nfig, c_ax = plt.subplots(1, 1, figsize = (10, 10))\nc_ax.set_xlim(0, 1024)\nc_ax.set_ylim(0, 1024)\nfor i, (x, y, width, height) in enumerate([right_bbox, left_bbox]):\n    c_ax.add_patch(Rectangle(xy=(x, y),\n                                    width=width,\n                                    height=height, \n                                     alpha = 0.5+0.25*i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ffbcd6dd429dfd5b7c236ad5729ae88a17bdeab"},"cell_type":"code","source":"def proc_score(in_score):\n    out_str = []\n    if in_score>THRESHOLD_FOR_PREDICTION:\n        for n_box in [left_bbox, right_bbox]:\n            out_str+=['%2.2f %f %f %f %f' % (in_score, *n_box)]\n    if len(out_str)==0:\n        return ''\n    else:\n        return ' '.join(out_str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1185e56f95369089bb923cf50d223537bb01cad"},"cell_type":"code","source":"sub_df['PredictionString'] = sub_df['score'].map(proc_score)\nsub_df.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d57598a50a4f61f8392748ad28535ccf1bb47b94"},"cell_type":"code","source":"sub_df[['patientId','PredictionString']].to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3be668df2a5fbf9ee5119fd12018d29efe125619"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}