{"cells":[{"metadata":{"_uuid":"8466faddaa492e417e80cea07989dceb72017246"},"cell_type":"markdown","source":"# Objective: Obtain out of fold predictions on the entire training set using cross validation and then using a mean average precision IoU metric, that closely resembles the competition metric, to improve validation"},{"metadata":{"trusted":true,"_uuid":"eec3fbc59b4528da7fa867dd148c52d6391edb5b"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"97a0f040fd76bb2970fb721615570defb6efb949"},"cell_type":"markdown","source":"# Prepare out of fold training predictions for implementation of MAP IoU matching competition evaluation description"},{"metadata":{"_uuid":"351227cc50e0711d25350d1f1bdc1b54bbf3d505"},"cell_type":"markdown","source":"Load oof predictions from CNN segmentation CV kernel https://www.kaggle.com/cchadha/cnn-segmentation-cv-with-oof-preds-on-train-set/notebook"},{"metadata":{"trusted":true,"_uuid":"b141987cdeb3559470e514de28492ae28a770733"},"cell_type":"code","source":"oof_preds0 = pd.read_csv('../input/cnn-segmentation-cv-with-oof-preds-on-train-set/oof_preds0.csv')\noof_preds1 = pd.read_csv('../input/cnn-segmentation-cv-with-oof-preds-on-train-set/oof_preds1.csv')\noof_preds2 = pd.read_csv('../input/cnn-segmentation-cv-with-oof-preds-on-train-set/oof_preds2.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0ef41d62accd5165bf381de599419c83cab1a42"},"cell_type":"code","source":"oof_preds0.head()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"e4f9fc9593ded7314463ddbb437fd0ea95f3cd1c"},"cell_type":"code","source":"oof_preds1.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b15e2aaf7ceb8da8d8c2c522f0a2aa1ff4c7125c"},"cell_type":"code","source":"oof_preds2.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"debe3e4e95b80f6db124c7e731b4cf529073bc78"},"cell_type":"markdown","source":"Read in training labels"},{"metadata":{"trusted":true,"_uuid":"ac96897c3549427881c993eab140ff3bedc2d071"},"cell_type":"code","source":"df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_1_train_labels.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48da1d7e7db105ae2671a802e2332308576f6a4b"},"cell_type":"code","source":"df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b92a23ed6f5e57c1bacc1df9cc5ed011d35658d5"},"cell_type":"markdown","source":"Parse bounding box labels into correct format for Mean Average Precision IoU metric"},{"metadata":{"trusted":true,"_uuid":"1e4af8e36bb8b661b536772673e7adae4ae0e2c6"},"cell_type":"code","source":"df['bbox_target'] = (df['x'].astype(str) +\n                    ' ' + \n                    df['y'].astype(str) +\n                    ' ' +\n                    df['width'].astype(str) +\n                    ' ' +\n                    df['height'].astype(str))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"3d54c292a627d731491e2bee2cba24a3ecfb48fc"},"cell_type":"code","source":"df.loc[:, 'bbox_target'] = df.loc[:, 'bbox_target'].map(lambda x: x.split(' '))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"fb06c0dfd67c66ce23372dac97d4706b782d6751"},"cell_type":"code","source":"df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e55119e423db8ea20c29f61d740b23591a094a59"},"cell_type":"code","source":"df = df.groupby(['patientId'], as_index = False)['bbox_target'].agg('sum')    ","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"954d73d69538cd74bf49e3ce2603539458fa5389"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7cf7b27a2edbf7b5684126c45e7ea7c75b5e7a92"},"cell_type":"markdown","source":"Merge labels and oof preds"},{"metadata":{"trusted":true,"_uuid":"8a0c0153ebf06ed2dd227782630d660106fb519b"},"cell_type":"code","source":"df = df.merge(oof_preds0, on = 'patientId', how = 'left')\ndf = df.merge(oof_preds1, on = 'patientId', how = 'left')\ndf = df.merge(oof_preds2, on = 'patientId', how = 'left')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"365436c7300e1890f4ab1109c1cf2a4b3002188b"},"cell_type":"code","source":"df = df.fillna('')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"14b6f7ebaceceb6727908562ac5a88daca87fe4a"},"cell_type":"markdown","source":"Parse oof preds for MAP IoU"},{"metadata":{"trusted":true,"_uuid":"6681d853459f11192220fa14792f2807c5af518b"},"cell_type":"code","source":"df.loc[:, 'bbox_pred'] = (df.loc[:, 'PredictionString'] +\n                         ' ' +\n                         df.loc[:, 'PredictionString_x'] +\n                         ' ' +\n                         df.loc[:, 'PredictionString_y'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32e395811a42bd2d71dd51c1cb9c82881f61bcd0"},"cell_type":"code","source":"df = df.drop(['PredictionString','PredictionString_x', 'PredictionString_y'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"4ab2cf9a14c340641889834313eccf5be711ef3b"},"cell_type":"code","source":"df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1b0b788e5e0e63b17b0a0c5b6a0248d988985621"},"cell_type":"markdown","source":"Stripping whitespace from PredictionString column"},{"metadata":{"trusted":true,"_uuid":"af078767395bf03bce22f5372cea04b82a9f772b"},"cell_type":"code","source":"df.loc[:, 'bbox_pred'] = df.loc[:, 'bbox_pred'].str.strip()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"a049a5edb4bd5b86a56c0a207a07c5b55aff9283"},"cell_type":"code","source":"df.loc[:, 'bbox_pred'] = df.loc[:, 'bbox_pred'].map(lambda x: x.split(' '))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"542e3a63b81ae76249b2d12602031b736ea5cd64"},"cell_type":"code","source":"df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6311de8cc3169bddfc5745801fe2bbbfd3fdc440"},"cell_type":"code","source":"def parse_scores(x):\n    if len(x)!=1:\n        scores = [x[k] for k in range(0,len(x),5)]\n        for score in range(len(scores)):\n            scores[score] = float(scores[score])\n        return np.asarray(scores)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"745c94eeec245e2e6684ed872ea6758a23d4ac54"},"cell_type":"code","source":"df.loc[:, 'bbox_scores'] = df.loc[:, 'bbox_pred'].map(parse_scores)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"83d733c2ba4c9b1beaffce5dce69e7b0cae0a0e5"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e233df9c64144f027138d47efd5f3b4674a25ccd"},"cell_type":"code","source":"def parse_bbox(x):\n    if len(x)!=1:\n        bbox = [int(x[k]) for k in range(0,len(x)) if k%5 != 0]\n        return np.asarray(bbox).reshape(int(len(bbox)/4),4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"595da96de44909439638e259161744589df5276e"},"cell_type":"code","source":"df.loc[:, 'bbox_preds'] = df.loc[:, 'bbox_pred'].map(parse_bbox)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"545f6e8f67a9676d6d9451f46341feeeee77d524"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d099b5720fd97b12eb87ad6995d6ea27369fcf8"},"cell_type":"code","source":"df = df.drop(['bbox_pred'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"4540e86f658293242835827c314802ca821a2c4c"},"cell_type":"code","source":"df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08a4df1090663cc4523ed85ebddc76f36b900d76"},"cell_type":"markdown","source":"Edit NaN or None values to empty numpy arrays to fit MAP IoU metric implementation"},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"c66ad0761803b12e377a62904f192534546df443"},"cell_type":"code","source":"df.loc[df['bbox_scores'].isnull(),['bbox_scores']] = df.loc[df['bbox_scores'].isnull(),'bbox_scores'].apply(lambda x: np.asarray([]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec0aaf94c955093a98664f8ecebb728fba058695"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"d4a5a592d671322648a3cfff47e9fccb791e0aac"},"cell_type":"code","source":"df.loc[df['bbox_preds'].isnull(),['bbox_preds']] = df.loc[df['bbox_preds'].isnull(),'bbox_preds'].apply(lambda x: np.asarray([]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bff01ed1d7af24de3f655063386f5a33db36b5a2"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb9e3dc75a8c3ec9d2424e0496e0124f9661af3f"},"cell_type":"code","source":"def parse_target_str(x):\n    if x[0] != 'nan':\n        bbox = np.asarray([int(float(x[k])) for k in range(0,len(x))])\n        return bbox.reshape(int(len(bbox)/4),4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"360f0605fcb1c3d1bd6466adbd561fd33688b279"},"cell_type":"code","source":"df.loc[:,'bbox_target'] = df.loc[:,'bbox_target'].map(parse_target_str)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"0f56fbb9bccd180f743f7551c31408b6fdaac46b"},"cell_type":"code","source":"df.loc[df['bbox_target'].isnull(),['bbox_target']] = df.loc[df['bbox_target'].isnull(),'bbox_target'].apply(lambda x: np.asarray([]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f95413c0eccef734f9d19e2f134b4c7a691c1ac9"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc462c04aa4d250e862280e207d25444b18ae7bc"},"cell_type":"markdown","source":"# Find mean average precision IoU using implementation by chenyc15 https://www.kaggle.com/chenyc15/mean-average-precision-metric and edited herein"},{"metadata":{"trusted":true,"_uuid":"5194314ded5199dad9ea8725ff428f2102d8ac5f"},"cell_type":"code","source":"# helper function to calculate IoU\ndef iou(box1, box2):\n    x11, y11, w1, h1 = box1\n    x21, y21, w2, h2 = box2\n    assert w1 * h1 > 0\n    assert w2 * h2 > 0\n    x12, y12 = x11 + w1, y11 + h1\n    x22, y22 = x21 + w2, y21 + h2\n\n    area1, area2 = w1 * h1, w2 * h2\n    xi1, yi1, xi2, yi2 = max([x11, x21]), max([y11, y21]), min([x12, x22]), min([y12, y22])\n    \n    if xi2 <= xi1 or yi2 <= yi1:\n        return 0\n    else:\n        intersect = (xi2-xi1) * (yi2-yi1)\n        union = area1 + area2 - intersect\n        return intersect / union","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0589be0aee106c1369a4dce22d750958b34a2d57"},"cell_type":"code","source":"def map_iou(boxes_true, boxes_pred, scores, thresholds = [0.4, 0.45, 0.5, 0.55, 0.6, 0.65, 0.7, 0.75]):\n    \"\"\"\n    Mean average precision at differnet intersection over union (IoU) threshold\n    \n    input:\n        boxes_true: Mx4 numpy array of ground true bounding boxes of one image. \n                    bbox format: (x1, y1, w, h)\n        boxes_pred: Nx4 numpy array of predicted bounding boxes of one image. \n                    bbox format: (x1, y1, w, h)\n        scores:     length N numpy array of scores associated with predicted bboxes\n        thresholds: IoU shresholds to evaluate mean average precision on\n    output: \n        map: mean average precision of the image\n    \"\"\"\n    \n    # According to the introduction, images with no ground truth bboxes will not be \n    # included in the map score unless there is a false positive detection (?)\n        \n    # return None if both are empty, don't count the image in final evaluation (?)\n    if len(boxes_true) == 0 and len(boxes_pred) == 0:\n        return None\n    elif len(boxes_true) == 0 and len(boxes_pred) > 0:\n        return 0\n    elif len(boxes_true) > 0 and len(boxes_pred) == 0:\n        return 0\n    elif len(boxes_true) > 0 and len(boxes_pred) > 0:\n        assert boxes_true.shape[1] == 4 or boxes_pred.shape[1] == 4, \"boxes should be 2D arrays with shape[1]=4\"\n        if len(boxes_pred):\n            assert len(scores) == len(boxes_pred), \"boxes_pred and scores should be same length\"\n            # sort boxes_pred by scores in decreasing order\n            boxes_pred = boxes_pred[np.argsort(scores)[::-1], :]\n\n        map_total = 0\n\n        # loop over thresholds\n        for t in thresholds:\n            matched_bt = set()\n            tp, fn = 0, 0\n            for i, bt in enumerate(boxes_true):\n                matched = False\n                for j, bp in enumerate(boxes_pred):\n                    miou = iou(bt, bp)\n                    if miou >= t and not matched and j not in matched_bt:\n                        matched = True\n                        tp += 1 # bt is matched for the first time, count as TP\n                        matched_bt.add(j)\n                if not matched:\n                    fn += 1 # bt has no match, count as FN\n\n            fp = len(boxes_pred) - len(matched_bt) # FP is the bp that not matched to any bt\n            m = tp / (tp + fn + fp)\n            map_total += m\n    \n    return map_total / len(thresholds)","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":true,"_uuid":"87d7996c6211191766a4b9dc8c6880d15ddb30c3"},"cell_type":"code","source":"df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2213344b1af6ef259fe790da786e01c644525fdb"},"cell_type":"code","source":"for row in range(10):\n    print(map_iou(df['bbox_target'][row], df['bbox_preds'][row], df['bbox_scores'][row]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ef669b34be8b2e66b91cad868575b94f84b810c"},"cell_type":"code","source":"map_scores = [x for x in [map_iou(df['bbox_target'][row], df['bbox_preds'][row], df['bbox_scores'][row]) for row in range(len(df))] if x is not None]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02407979a332f12667eea6bfab4fdde7c4a2ea12"},"cell_type":"code","source":"np.mean(map_scores)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b47765ae333aa3dcae70c2130a0376b99ad64af6"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}