{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Credits: https://www.kaggle.com/pestipeti/competition-metric-map-0-4","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:51:21.31434Z","iopub.execute_input":"2021-06-05T19:51:21.31478Z","iopub.status.idle":"2021-06-05T19:51:21.319542Z","shell.execute_reply.started":"2021-06-05T19:51:21.314741Z","shell.execute_reply":"2021-06-05T19:51:21.317743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competiton metric calculator\n\n> The challenge uses the standard [PASCAL VOC 2010 mean Average Precision (mAP)](http://host.robots.ox.ac.uk/pascal/VOC/voc2010/devkit_doc_08-May-2010.pdf) at IoU > 0.5.","metadata":{}},{"cell_type":"code","source":"!pip install pycocotools","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:22:34.832786Z","iopub.execute_input":"2021-06-05T19:22:34.833474Z","iopub.status.idle":"2021-06-05T19:22:52.370133Z","shell.execute_reply.started":"2021-06-05T19:22:34.833411Z","shell.execute_reply":"2021-06-05T19:22:52.369097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport plotly.express as px\nimport plotly.graph_objects as go\n\nfrom pycocotools.coco import COCO\nfrom pycocotools.cocoeval import COCOeval","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-06-05T20:01:33.874287Z","iopub.execute_input":"2021-06-05T20:01:33.874718Z","iopub.status.idle":"2021-06-05T20:01:33.880781Z","shell.execute_reply.started":"2021-06-05T20:01:33.874683Z","shell.execute_reply":"2021-06-05T20:01:33.879811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:31:05.72754Z","iopub.execute_input":"2021-06-05T19:31:05.728119Z","iopub.status.idle":"2021-06-05T19:31:05.74612Z","shell.execute_reply.started":"2021-06-05T19:31:05.728084Z","shell.execute_reply":"2021-06-05T19:31:05.744833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:31:05.980957Z","iopub.execute_input":"2021-06-05T19:31:05.981381Z","iopub.status.idle":"2021-06-05T19:31:05.995123Z","shell.execute_reply.started":"2021-06-05T19:31:05.981332Z","shell.execute_reply":"2021-06-05T19:31:05.993644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = np.array(['negative', 'typical', 'indeterminate', 'atypical'])\nstudy_level_df['class_id'] = np.where(study_level_df.iloc[:,1:])[1]\nstudy_level_df['class_name'] = [names[i] for i in study_level_df['class_id'].values]","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:31:37.965709Z","iopub.execute_input":"2021-06-05T19:31:37.966138Z","iopub.status.idle":"2021-06-05T19:31:37.98348Z","shell.execute_reply.started":"2021-06-05T19:31:37.966099Z","shell.execute_reply":"2021-06-05T19:31:37.982001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level_df[['x_min','y_min', 'x_max', 'y_max']] = np.array([0.,0.,1.,1.])","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:33:21.384432Z","iopub.execute_input":"2021-06-05T19:33:21.384794Z","iopub.status.idle":"2021-06-05T19:33:21.394793Z","shell.execute_reply.started":"2021-06-05T19:33:21.384753Z","shell.execute_reply":"2021-06-05T19:33:21.393191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level_df['image_id'] = study_level_df['id']","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:36:45.093782Z","iopub.execute_input":"2021-06-05T19:36:45.094166Z","iopub.status.idle":"2021-06-05T19:36:45.101468Z","shell.execute_reply.started":"2021-06-05T19:36:45.094135Z","shell.execute_reply":"2021-06-05T19:36:45.099982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_level_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:36:45.571304Z","iopub.execute_input":"2021-06-05T19:36:45.571698Z","iopub.status.idle":"2021-06-05T19:36:45.590549Z","shell.execute_reply.started":"2021-06-05T19:36:45.571665Z","shell.execute_reply":"2021-06-05T19:36:45.589506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class VinBigDataEval:\n    \"\"\"Helper class for calculating the competition metric.\n    \n    You should remove the duplicated annoatations from the `true_df` dataframe\n    before using this script. Otherwise it may give incorrect results.\n\n        >>> vineval = VinBigDataEval(valid_df)\n        >>> cocoEvalResults = vineval.evaluate(pred_df)\n\n    Arguments:\n        true_df: pd.DataFrame Clean (no duplication) Training/Validating dataframe.\n\n    Authors:\n        Peter (https://kaggle.com/pestipeti)\n\n    See:\n        https://www.kaggle.com/pestipeti/competition-metric-map-0-4\n\n    Returns: None\n    \n    \"\"\"\n    def __init__(self, true_df):\n        \n        self.true_df = true_df\n\n        self.image_ids = true_df[\"image_id\"].unique()\n        self.annotations = {\n            \"type\": \"instances\",\n            \"images\": self.__gen_images(self.image_ids),\n            \"categories\": self.__gen_categories(self.true_df),\n            \"annotations\": self.__gen_annotations(self.true_df, self.image_ids)\n        }\n        \n        self.predictions = {\n            \"images\": self.annotations[\"images\"].copy(),\n            \"categories\": self.annotations[\"categories\"].copy(),\n            \"annotations\": None\n        }\n\n        \n    def __gen_images(self, image_ids):\n        print(\"Generating image data...\")\n        results = []\n\n        for idx, image_id in enumerate(image_ids):\n\n            # Add image identification.\n            results.append({\n                \"id\": idx,\n            })\n            \n        return results\n    \n    \n    def __gen_categories(self, df):\n        print(\"Generating category data...\")\n        \n        if \"class_name\" not in df.columns:\n            df[\"class_name\"] = df[\"class_id\"]\n        \n        cats = df[[\"class_name\", \"class_id\"]]\n        cats = cats.drop_duplicates().sort_values(by='class_id').values\n        \n        results = []\n        \n        for cat in cats:\n            results.append({\n                \"id\": cat[1],\n                \"name\": cat[0],\n                \"supercategory\": \"none\",\n            })\n            \n        return results\n\n    \n    def __gen_annotations(self, df, image_ids):\n        print(\"Generating annotation data...\")\n        k = 0\n        results = []\n        \n        for idx, image_id in enumerate(image_ids):\n\n            # Add image annotations\n            for i, row in df[df[\"image_id\"] == image_id].iterrows():\n\n                results.append({\n                    \"id\": k,\n                    \"image_id\": idx,\n                    \"category_id\": row[\"class_id\"],\n                    \"bbox\": np.array([\n                        row[\"x_min\"],\n                        row[\"y_min\"],\n                        row[\"x_max\"],\n                        row[\"y_max\"]]\n                    ),\n                    \"segmentation\": [],\n                    \"ignore\": 0,\n                    \"area\":(row[\"x_max\"] - row[\"x_min\"]) * (row[\"y_max\"] - row[\"y_min\"]),\n                    \"iscrowd\": 0,\n                })\n\n                k += 1\n                \n        return results\n\n    def __decode_prediction_string(self, pred_str):\n        data = list(map(float, pred_str.split(\" \")))\n        data = np.array(data)\n\n        return data.reshape(-1, 6)    \n    \n    def __gen_predictions(self, df, image_ids):\n        print(\"Generating prediction data...\")\n        k = 0\n        results = []\n        \n        for i, row in df.iterrows():\n            \n            image_id = row[\"image_id\"]\n            preds = self.__decode_prediction_string(row[\"PredictionString\"])\n\n            for j, pred in enumerate(preds):\n\n                results.append({\n                    \"id\": k,\n                    \"image_id\": int(np.where(image_ids == image_id)[0]),\n                    \"category_id\": int(pred[0]),\n                    \"bbox\": np.array([\n                        pred[2], pred[3], pred[4], pred[5]\n                    ]),\n                    \"segmentation\": [],\n                    \"ignore\": 0,\n                    \"area\": (pred[4] - pred[2]) * (pred[5] - pred[3]),\n                    \"iscrowd\": 0,\n                    \"score\": pred[1]\n                })\n\n                k += 1\n                \n        return results\n                \n    def evaluate(self, pred_df, n_imgs = -1):\n        \"\"\"Evaluating your results\n        \n        Arguments:\n            pred_df: pd.DataFrame your predicted results in the\n                     competition output format.\n\n            n_imgs:  int Number of images use for calculating the\n                     result.All of the images if `n_imgs` <= 0\n                     \n        Returns:\n            COCOEval object\n        \"\"\"\n        \n        if pred_df is not None:\n            self.predictions[\"annotations\"] = self.__gen_predictions(pred_df, self.image_ids)\n\n        coco_ds = COCO()\n        coco_ds.dataset = self.annotations\n        coco_ds.createIndex()\n        \n        coco_dt = COCO()\n        coco_dt.dataset = self.predictions\n        coco_dt.createIndex()\n        \n        imgIds=sorted(coco_ds.getImgIds())\n        \n        if n_imgs > 0:\n            imgIds = np.random.choice(imgIds, n_imgs)\n\n        cocoEval = COCOeval(coco_ds, coco_dt, 'bbox')\n        cocoEval.params.imgIds  = imgIds\n        cocoEval.params.useCats = True\n        cocoEval.params.iouType = \"bbox\"\n        cocoEval.params.iouThrs = np.array([0.5])\n\n        cocoEval.evaluate()\n        cocoEval.accumulate()\n        cocoEval.summarize()\n        \n        return cocoEval","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:37:47.045633Z","iopub.execute_input":"2021-06-05T19:37:47.045984Z","iopub.status.idle":"2021-06-05T19:37:47.084257Z","shell.execute_reply.started":"2021-06-05T19:37:47.045954Z","shell.execute_reply":"2021-06-05T19:37:47.083073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Usage","metadata":{}},{"cell_type":"code","source":"# df = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\n# df.fillna(0, inplace=True)\n# df.loc[df[\"class_id\"] == 14, ['x_max', 'y_max']] = 1.0\n\n# df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:37:10.373148Z","iopub.execute_input":"2021-06-05T19:37:10.373679Z","iopub.status.idle":"2021-06-05T19:37:10.379441Z","shell.execute_reply.started":"2021-06-05T19:37:10.373634Z","shell.execute_reply":"2021-06-05T19:37:10.378178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Removing duplications! DO NOT USE THIS in your training!!!\n# df = df.groupby(by=['image_id', 'class_id']).first().reset_index()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:37:11.940381Z","iopub.execute_input":"2021-06-05T19:37:11.940779Z","iopub.status.idle":"2021-06-05T19:37:11.945645Z","shell.execute_reply.started":"2021-06-05T19:37:11.940745Z","shell.execute_reply":"2021-06-05T19:37:11.944184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You only need to run this once.\nvineval = VinBigDataEval(study_level_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:37:49.198185Z","iopub.execute_input":"2021-06-05T19:37:49.198589Z","iopub.status.idle":"2021-06-05T19:37:59.210421Z","shell.execute_reply.started":"2021-06-05T19:37:49.198556Z","shell.execute_reply":"2021-06-05T19:37:59.209169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict single class for each study","metadata":{}},{"cell_type":"code","source":"# Predicting with 1 class\n# {0: 'negative', 1: 'typical', 2: 'indeterminate', 3: 'atypical'}\npred_df = study_level_df[[\"image_id\"]]\npred_df = pred_df.drop_duplicates()\nclass_id = 0\npred_df[\"PredictionString\"] = f\"{class_id} 1.0 0 0 1 1\"\npred_df.reset_index(drop=True, inplace=True)\n\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:57:19.231857Z","iopub.execute_input":"2021-06-05T19:57:19.232272Z","iopub.status.idle":"2021-06-05T19:57:19.252669Z","shell.execute_reply.started":"2021-06-05T19:57:19.232237Z","shell.execute_reply":"2021-06-05T19:57:19.251303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You should evaluate after every n epochs.\ncocoEvalRes = vineval.evaluate(pred_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:57:20.527588Z","iopub.execute_input":"2021-06-05T19:57:20.52814Z","iopub.status.idle":"2021-06-05T19:57:27.775597Z","shell.execute_reply.started":"2021-06-05T19:57:20.528091Z","shell.execute_reply":"2021-06-05T19:57:27.774285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We get same results as LB probing for negative which is 0.050. We need to multiply mAP for study by 4/6 to get contributions of 4 study classes to final LB score. You can check this [discussion](https://www.kaggle.com/c/siim-covid19-detection/discussion/244066) for LB probing to study level predictions.","metadata":{}},{"cell_type":"code","source":"cocoEvalRes.stats[1]*2/3 ","metadata":{"execution":{"iopub.status.busy":"2021-06-05T20:01:58.238734Z","iopub.execute_input":"2021-06-05T20:01:58.239305Z","iopub.status.idle":"2021-06-05T20:01:58.246047Z","shell.execute_reply.started":"2021-06-05T20:01:58.239254Z","shell.execute_reply":"2021-06-05T20:01:58.244929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict all classes for each study","metadata":{}},{"cell_type":"code","source":"class_probas = pd.value_counts(study_level_df['class_id'], normalize=True); class_probas","metadata":{"execution":{"iopub.status.busy":"2021-06-05T20:22:02.282144Z","iopub.execute_input":"2021-06-05T20:22:02.282530Z","iopub.status.idle":"2021-06-05T20:22:02.293759Z","shell.execute_reply.started":"2021-06-05T20:22:02.282498Z","shell.execute_reply":"2021-06-05T20:22:02.292108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Probability doesn't matter when predicting all classes since all IOUs = 1.0 and 1 box is TP and remaining 3 all always FP.","metadata":{}},{"cell_type":"code","source":"# Predicting with all classes\ndfs = []\nfor class_id in range(4):\n    pred_df = study_level_df[[\"image_id\"]]\n    pred_df = pred_df.drop_duplicates()\n    proba = class_probas[class_id]\n    pred_df[\"PredictionString\"] = f\"{class_id} {proba} 0 0 1 1\"\n    pred_df.reset_index(drop=True, inplace=True)\n    dfs.append(pred_df)\npred_df = pd.concat(dfs)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T20:21:55.830160Z","iopub.execute_input":"2021-06-05T20:21:55.830523Z","iopub.status.idle":"2021-06-05T20:21:55.866610Z","shell.execute_reply.started":"2021-06-05T20:21:55.830491Z","shell.execute_reply":"2021-06-05T20:21:55.865262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You should evaluate after every n epochs.\ncocoEvalRes = vineval.evaluate(pred_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-05T20:22:05.892951Z","iopub.execute_input":"2021-06-05T20:22:05.893388Z","iopub.status.idle":"2021-06-05T20:22:26.701430Z","shell.execute_reply.started":"2021-06-05T20:22:05.893335Z","shell.execute_reply":"2021-06-05T20:22:26.700478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cocoEvalRes.stats[1]*2/3 ","metadata":{"execution":{"iopub.status.busy":"2021-06-05T20:22:26.703423Z","iopub.execute_input":"2021-06-05T20:22:26.703749Z","iopub.status.idle":"2021-06-05T20:22:26.710456Z","shell.execute_reply.started":"2021-06-05T20:22:26.703713Z","shell.execute_reply":"2021-06-05T20:22:26.709637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Recalculating with random samples","metadata":{}},{"cell_type":"code","source":"%%capture\nstats = []\n\n# Recalculate the validation score using randomly selected images\nfor i in range(100):\n    cocoEvalRes = vineval.evaluate(pred_df = None, n_imgs = 300)\n    stats.append(cocoEvalRes.stats[0])\n    \navg = np.array(stats).mean()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:48:41.90747Z","iopub.execute_input":"2021-06-05T19:48:41.907853Z","iopub.status.idle":"2021-06-05T19:49:05.969191Z","shell.execute_reply.started":"2021-06-05T19:48:41.907822Z","shell.execute_reply":"2021-06-05T19:49:05.968354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\nfig.add_trace(go.Scatter(x=[x for x in range(len(stats))], y=stats, mode=\"markers\", name=\"Stats\"))\nfig.add_trace(go.Scatter(x=[0, 100], y=[avg, avg], mode=\"lines\", name=\"Mean\"))\nfig.add_trace(go.Scatter(x=[0, 100], y=[0.052, 0.052], mode=\"lines\", name=\"Public Baseline\"))\n\nfig.update_yaxes(\n    range=[0.03, 0.07]\n)\n\nfig.update_layout(title='Results of mAP@0.4 (randomly selected 300 images)',\n                  yaxis_title='Score',\n                  xaxis_title='')\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-05T19:46:59.996773Z","iopub.execute_input":"2021-06-05T19:46:59.997116Z","iopub.status.idle":"2021-06-05T19:47:00.199314Z","shell.execute_reply.started":"2021-06-05T19:46:59.997087Z","shell.execute_reply":"2021-06-05T19:47:00.198165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}