{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Order of instances in metric / submission\n\nThis notebook demonstrates that the LB score might depend on order of instance row in the submission.\n\nThe code was taken from [here](https://www.kaggle.com/theoviel/competition-metric-map-iou) and modified for the demo.\n\n[Here](https://www.kaggle.com/c/sartorius-cell-instance-segmentation/discussion/288762) is discussion thread for your comments and feedback.","metadata":{}},{"cell_type":"code","source":"import os\nimport skimage\nimport numpy as np\nimport pandas as pd\nimport skimage.segmentation\nimport matplotlib.pyplot as plt\n\nfrom random import shuffle","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-18T10:10:21.323089Z","iopub.execute_input":"2021-11-18T10:10:21.3235Z","iopub.status.idle":"2021-11-18T10:10:21.329199Z","shell.execute_reply.started":"2021-11-18T10:10:21.32346Z","shell.execute_reply":"2021-11-18T10:10:21.328141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Example","metadata":{}},{"cell_type":"code","source":"def rles_to_mask(encs, shape):\n    \"\"\"\n    Decodes a rle.\n\n    Args:\n        encs (list of str): Rles for each class.\n        shape (tuple [2]): Mask size.\n\n    Returns:\n        np array [shape]: Mask.\n    \"\"\"\n    img = np.zeros(shape[0] * shape[1], dtype=np.uint)\n    for m, enc in enumerate(encs):\n        if isinstance(enc, np.float) and np.isnan(enc):\n            continue\n        enc_split = enc.split()\n        for i in range(len(enc_split) // 2):\n            start = int(enc_split[2 * i]) - 1\n            length = int(enc_split[2 * i + 1])\n            img[start: start + length] = 1 + m\n    return img.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:04:44.84807Z","iopub.execute_input":"2021-11-18T10:04:44.848491Z","iopub.status.idle":"2021-11-18T10:04:44.857188Z","shell.execute_reply.started":"2021-11-18T10:04:44.848443Z","shell.execute_reply":"2021-11-18T10:04:44.856294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/sartorius-cell-instance-segmentation/train.csv')\ndf = df.groupby('id').agg(list).reset_index()\n\nfor col in df.columns[2:]:\n    df[col] = df[col].apply(\n        lambda x: np.unique(x)[0] if len(np.unique(x)) == 1 else np.unique(x)\n    )","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:05:10.850443Z","iopub.execute_input":"2021-11-18T10:05:10.850748Z","iopub.status.idle":"2021-11-18T10:05:12.10409Z","shell.execute_reply.started":"2021-11-18T10:05:10.850715Z","shell.execute_reply":"2021-11-18T10:05:12.10339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ground truth","metadata":{}},{"cell_type":"code","source":"i = 0  # feel free to change that\n\nshape = df[['height', 'width']].values[i]\n\nrles = df['annotation'][i]\nmasks = rles_to_mask(rles, shape).astype(np.uint16)\n\nrles_shuffled = rles.copy()\nshuffle(rles_shuffled)\nmasks_shuffled = rles_to_mask(rles_shuffled, shape).astype(np.uint16)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:12:20.791206Z","iopub.execute_input":"2021-11-18T10:12:20.791845Z","iopub.status.idle":"2021-11-18T10:12:20.83778Z","shell.execute_reply.started":"2021-11-18T10:12:20.79179Z","shell.execute_reply":"2021-11-18T10:12:20.837133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f,ax = plt.subplots(1, 2, figsize=(15, 10))\nax[0].axis(False)\nax[0].imshow(masks)\nax[1].axis(False)\nax[1].imshow(masks_shuffled)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:20:49.692083Z","iopub.execute_input":"2021-11-18T10:20:49.692403Z","iopub.status.idle":"2021-11-18T10:20:50.009997Z","shell.execute_reply.started":"2021-11-18T10:20:49.692357Z","shell.execute_reply":"2021-11-18T10:20:50.009323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric","metadata":{}},{"cell_type":"markdown","source":"## IoU","metadata":{}},{"cell_type":"code","source":"def compute_iou(labels, y_pred):\n    \"\"\"\n    Computes the IoU for instance labels and predictions.\n\n    Args:\n        labels (np array): Labels.\n        y_pred (np array): predictions\n\n    Returns:\n        np array: IoU matrix, of size true_objects x pred_objects.\n    \"\"\"\n\n    true_objects = len(np.unique(labels))\n    pred_objects = len(np.unique(y_pred))\n\n    # Compute intersection between all objects\n    intersection = np.histogram2d(\n        labels.flatten(), y_pred.flatten(), bins=(true_objects, pred_objects)\n    )[0]\n\n    # Compute areas (needed for finding the union between all objects)\n    area_true = np.histogram(labels, bins=true_objects)[0]\n    area_pred = np.histogram(y_pred, bins=pred_objects)[0]\n    area_true = np.expand_dims(area_true, -1)\n    area_pred = np.expand_dims(area_pred, 0)\n\n    # Compute union\n    union = area_true + area_pred - intersection\n    iou = intersection / union\n    \n    return iou[1:, 1:]  # exclude background","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:21:03.79515Z","iopub.execute_input":"2021-11-18T10:21:03.796038Z","iopub.status.idle":"2021-11-18T10:21:03.804709Z","shell.execute_reply.started":"2021-11-18T10:21:03.795977Z","shell.execute_reply":"2021-11-18T10:21:03.803788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Precision","metadata":{}},{"cell_type":"code","source":"def precision_at(threshold, iou):\n    \"\"\"\n    Computes the precision at a given threshold.\n\n    Args:\n        threshold (float): Threshold.\n        iou (np array [n_truths x n_preds]): IoU matrix.\n\n    Returns:\n        int: Number of true positives,\n        int: Number of false positives,\n        int: Number of false negatives.\n    \"\"\"\n    matches = iou > threshold\n    true_positives = np.sum(matches, axis=1) >= 1  # Correct objects\n    false_negatives = np.sum(matches, axis=1) == 0  # Missed objects\n    false_positives = np.sum(matches, axis=0) == 0  # Extra objects\n    tp, fp, fn = (\n        np.sum(true_positives),\n        np.sum(false_positives),\n        np.sum(false_negatives),\n    )\n    return tp, fp, fn","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:21:08.252943Z","iopub.execute_input":"2021-11-18T10:21:08.253379Z","iopub.status.idle":"2021-11-18T10:21:08.259318Z","shell.execute_reply.started":"2021-11-18T10:21:08.253344Z","shell.execute_reply":"2021-11-18T10:21:08.258643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overall Metric","metadata":{}},{"cell_type":"markdown","source":"### IoU","metadata":{}},{"cell_type":"code","source":"def iou_map(truths, preds, verbose=0):\n    \"\"\"\n    Computes the metric for the competition.\n    Masks contain the segmented pixels where each object has one value associated,\n    and 0 is the background.\n\n    Args:\n        truths (list of masks): Ground truths.\n        preds (list of masks): Predictions.\n        verbose (int, optional): Whether to print infos. Defaults to 0.\n\n    Returns:\n        float: mAP.\n    \"\"\"\n    ious = [compute_iou(truth, pred) for truth, pred in zip(truths, preds)]\n    \n    print(ious[0].shape)\n\n    if verbose:\n        print(\"Thresh\\tTP\\tFP\\tFN\\tPrec.\")\n\n    prec = []\n    for t in np.arange(0.5, 1.0, 0.05):\n        tps, fps, fns = 0, 0, 0\n        for iou in ious:\n            tp, fp, fn = precision_at(t, iou)\n            tps += tp\n            fps += fp\n            fns += fn\n\n        p = tps / (tps + fps + fns)\n        prec.append(p)\n\n        if verbose:\n            print(\"{:1.3f}\\t{}\\t{}\\t{}\\t{:1.3f}\".format(t, tps, fps, fns, p))\n\n    if verbose:\n        print(\"AP\\t-\\t-\\t-\\t{:1.3f}\".format(np.mean(prec)))\n\n    return np.mean(prec)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:21:11.175334Z","iopub.execute_input":"2021-11-18T10:21:11.175782Z","iopub.status.idle":"2021-11-18T10:21:11.185516Z","shell.execute_reply.started":"2021-11-18T10:21:11.175728Z","shell.execute_reply":"2021-11-18T10:21:11.184817Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compute","metadata":{}},{"cell_type":"code","source":"iou_map([masks], [masks], verbose=1)  # This should score 1","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:21:43.523917Z","iopub.execute_input":"2021-11-18T10:21:43.524238Z","iopub.status.idle":"2021-11-18T10:21:43.583196Z","shell.execute_reply.started":"2021-11-18T10:21:43.524203Z","shell.execute_reply":"2021-11-18T10:21:43.582312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iou_map([masks], [masks_shuffled], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-18T10:21:59.937354Z","iopub.execute_input":"2021-11-18T10:21:59.937663Z","iopub.status.idle":"2021-11-18T10:21:59.996246Z","shell.execute_reply.started":"2021-11-18T10:21:59.93763Z","shell.execute_reply":"2021-11-18T10:21:59.995268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see that shuffling RLEs do change the score.","metadata":{}}]}