{"cells":[{"metadata":{"_uuid":"fb3af82636323181ddd6a53af0996c1f922f6163"},"cell_type":"markdown","source":"# Attempt of metric's implementation"},{"metadata":{"_uuid":"d38c96abc1455d2c7de9034169bceb5170377b1c"},"cell_type":"markdown","source":"This is an attempt to step-by-step implement the code of the metric described by the organizers of the competition. \n\nResult is very similar on LB score!\n\n__Request for help:__ how can I rewrite the code in terms of tensors? I need to realize a metric function such as:\n\n`def mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis = [1, 2, 3])\n    union = tf.reduce_sum(y_true, axis = [1, 2, 3]) + tf.reduce_sum(y_pred, axis = [1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))`"},{"metadata":{"trusted":true,"_uuid":"119d71ba9a87863b9d1de7986f1dc4aa78f221ed"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3be3ffb4e0c3a1a20a3ffd7e77631acb2e5eba39"},"cell_type":"code","source":"prediction_strings = pd.read_csv('../input/rsna-pred-train/model_1_train.csv')\nprediction_strings.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9375bf1d4a12fcbfa7cb246386b0325a5b8a89f"},"cell_type":"code","source":"ground_truth = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_1_train_labels.csv')\nground_truth.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee8dad92204383647b190982086c048d24dd7322"},"cell_type":"code","source":"prediction = pd.DataFrame(index = ['patientId', 'confidence', 'x', 'y', 'width', 'height'])\n\nfor i, row in prediction_strings.iterrows():\n    if len(str(row[1])) > 3:\n        row_array = row[1].split(' ')\n        for i in range(int(len(row_array) / 5)):\n            prediction[prediction.shape[1]] = [row[0], float(row_array[i * 5])] + \\\n                                              [int(b) for b in row_array[i * 5 + 1 : i * 5 + 5]]\n            \nprediction = prediction.T\nprediction.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21e5c4bb51bcaeded80d48b4b94f594d9db18f12"},"cell_type":"code","source":"def iou(x1, y1, width1, height1, x2, y2, width2, height2):\n    x1, y1, width1, height1, x2, y2, width2, height2 = [int(v) for v in [x1, y1, width1, height1, x2, y2, width2, height2]]\n\n    mask_1 = np.zeros((max(y1 + height1, y2 + height2), max(x1 + width1, x2 + width2)))\n    mask_2 = np.zeros((max(y1 + height1, y2 + height2), max(x1 + width1, x2 + width2)))\n    \n    mask_1[y1 : y1 + height1, x1 : x1 + width1] = 1\n    mask_2[y2 : y2 + height2, x2 : x2 + width2] = 1\n\n    mask_i = mask_1 * mask_2\n    mask_u = mask_1 + mask_2 - mask_i\n    \n    return float(sum(sum(mask_i))) / sum(sum(mask_u))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b71f6a698bc73ac4a082b386b24993d8aa592d3d"},"cell_type":"code","source":"threshold = .4","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":true,"_uuid":"b1edcb7ccd21fcd07eae980b1d4c253cf6bc6bc3"},"cell_type":"code","source":"tp_candidates = pd.DataFrame(index = ['patientId', 'index_prediction', 'index_gtruth', 'iou'])\n\nfor patient in list(prediction['patientId'].unique()):\n    pred = prediction[prediction['patientId'] == patient].sort_values('confidence', ascending = False)\n    g_tr = ground_truth[ground_truth['patientId'] == patient]\n    \n    if g_tr['Target'].sum() > 0:\n        for ig, g in g_tr.iterrows():\n            for ip, p in pred.iterrows():\n                iou_score = iou(p['x'], p['y'], p['width'], p['height'], \n                                g['x'], g['y'], g['width'], g['height'])\n                if iou_score > threshold:\n                    tp_candidates[tp_candidates.shape[1]] = [patient, ip, ig, iou_score]\n                    break\n                    \ntp_candidates = tp_candidates.T\ntp_candidates.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6144579d484dcd24311bcd7f0ae0abd533d36b6e"},"cell_type":"code","source":"rates = pd.DataFrame(index = ['TP', 'FP', 'FN', 'score'])\n\nfor threshold in [.4, .45, .5, .55, .6, .65, .7, .75]:\n    tp = sum(tp_candidates['iou'] > threshold)\n    fp = prediction.shape[0] - tp\n    fn = ground_truth.dropna().shape[0] - \\\n         ground_truth.loc[tp_candidates[tp_candidates['iou'] > threshold]['index_gtruth']].dropna().shape[0]\n    rates[threshold] = [tp, fp, fn, float(tp) / (tp + fp + fn)]\n    \nrates.T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"112d9f06e28b732d3a21c0c07fb4704936f1b473"},"cell_type":"code","source":"rates.T['score'].mean()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8279a5914b426a32ad72f0bf08f0fd2a8ffecf39"},"cell_type":"markdown","source":"__LB score: 0.014__"},{"metadata":{"trusted":true,"_uuid":"45cb0f706d31eea32623a45323beba7cfb829379"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}