{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Version history\n* V1 - Basic training with default settings\n* V2 - Added a custom evaluator to track the competition metric during training","metadata":{}},{"cell_type":"markdown","source":"## v1 294\nv2 0.0005","metadata":{}},{"cell_type":"markdown","source":"## Training\n\nAll the heavy lifting here is done by the [detectron](https://github.com/facebookresearch/detectron2) library. What's needed from us is pointing it to the annotation files of our dataset (see [part one](https://www.kaggle.com/slawekbiel/positive-score-with-detectron-1-3-input-data/) for details), setting some hyperparameters and calling `trainer.train()`\n\nMost of the code here is just for displaying things to make sure everything is set up correctly and the training worked.","metadata":{}},{"cell_type":"code","source":"!pip install 'git+https://github.com/facebookresearch/detectron2.git'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-11-09T01:06:05.338584Z","iopub.execute_input":"2021-11-09T01:06:05.339317Z","iopub.status.idle":"2021-11-09T01:09:07.040361Z","shell.execute_reply.started":"2021-11-09T01:06:05.339227Z","shell.execute_reply":"2021-11-09T01:09:07.039543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import detectron2\nfrom pathlib import Path\nimport random, cv2, os\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pycocotools.mask as mask_util\n# import some common detectron2 utilities\nfrom detectron2 import model_zoo\nfrom detectron2.engine import DefaultPredictor, DefaultTrainer\nfrom detectron2.config import get_cfg\nfrom detectron2.utils.visualizer import Visualizer, ColorMode\nfrom detectron2.data import MetadataCatalog, DatasetCatalog\nfrom detectron2.data.datasets import register_coco_instances\nfrom detectron2.utils.logger import setup_logger\nfrom detectron2.evaluation.evaluator import DatasetEvaluator\nsetup_logger()","metadata":{"execution":{"iopub.status.busy":"2021-11-09T01:09:07.042577Z","iopub.execute_input":"2021-11-09T01:09:07.042859Z","iopub.status.idle":"2021-11-09T01:09:08.192426Z","shell.execute_reply.started":"2021-11-09T01:09:07.042824Z","shell.execute_reply":"2021-11-09T01:09:08.191737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load the competition data\nThis is very simple once we have our data in the COCO format. See the [part one notebook](https://www.kaggle.com/slawekbiel/positive-score-with-detectron-1-3-input-data/) for details.","metadata":{}},{"cell_type":"code","source":"dataDir=Path('../input/sartorius-cell-instance-segmentation/')\ncfg = get_cfg()\ncfg.INPUT.MASK_FORMAT='bitmask'\nregister_coco_instances('sartorius_train',{}, '../input/positive-score-with-detectron-1-3-input-data/annotations_train.json', dataDir)\nregister_coco_instances('sartorius_val',{},'../input/positive-score-with-detectron-1-3-input-data/annotations_val.json', dataDir)\nmetadata = MetadataCatalog.get('sartorius_train')\ntrain_ds = DatasetCatalog.get('sartorius_train')","metadata":{"execution":{"iopub.status.busy":"2021-11-09T01:09:08.193923Z","iopub.execute_input":"2021-11-09T01:09:08.194357Z","iopub.status.idle":"2021-11-09T01:09:13.052859Z","shell.execute_reply.started":"2021-11-09T01:09:08.194317Z","shell.execute_reply":"2021-11-09T01:09:13.051974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Display a sample file to check the data is loaded correctly","metadata":{"execution":{"iopub.status.busy":"2021-10-20T14:54:34.693045Z","iopub.execute_input":"2021-10-20T14:54:34.693326Z","iopub.status.idle":"2021-10-20T14:54:34.814645Z","shell.execute_reply.started":"2021-10-20T14:54:34.69329Z","shell.execute_reply":"2021-10-20T14:54:34.813452Z"}}},{"cell_type":"code","source":"d = train_ds[42]\nimg = cv2.imread(d[\"file_name\"])\nvisualizer = Visualizer(img[:, :, ::-1], metadata=metadata)\nout = visualizer.draw_dataset_dict(d)\nplt.figure(figsize = (20,15))\nplt.imshow(out.get_image()[:, :, ::-1])","metadata":{"execution":{"iopub.status.busy":"2021-11-09T01:09:13.055192Z","iopub.execute_input":"2021-11-09T01:09:13.055467Z","iopub.status.idle":"2021-11-09T01:09:14.393016Z","shell.execute_reply.started":"2021-11-09T01:09:13.055430Z","shell.execute_reply":"2021-11-09T01:09:14.391360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define evaluator \nGenerates lines like this in the training output:\n`[10/27 18:31:26 d2.evaluation.testing]: copypaste: MaP IoU=0.2192638391201311` \n\nSee here for definition: https://www.kaggle.com/c/sartorius-cell-instance-segmentation/overview/evaluation","metadata":{}},{"cell_type":"code","source":"# Taken from https://www.kaggle.com/theoviel/competition-metric-map-iou\ndef precision_at(threshold, iou):\n    matches = iou > threshold\n    true_positives = np.sum(matches, axis=1) == 1  # Correct objects\n    false_positives = np.sum(matches, axis=0) == 0  # Missed objects\n    false_negatives = np.sum(matches, axis=1) == 0  # Extra objects\n    return np.sum(true_positives), np.sum(false_positives), np.sum(false_negatives)\n\ndef score(pred, targ):\n    pred_masks = pred['instances'].pred_masks.cpu().numpy()\n    enc_preds = [mask_util.encode(np.asarray(p, order='F')) for p in pred_masks]\n    enc_targs = list(map(lambda x:x['segmentation'], targ))\n    ious = mask_util.iou(enc_preds, enc_targs, [0]*len(enc_targs))\n    prec = []\n    for t in np.arange(0.5, 1.0, 0.05):\n        tp, fp, fn = precision_at(t, ious)\n        p = tp / (tp + fp + fn)\n        prec.append(p)\n    return np.mean(prec)\n\nclass MAPIOUEvaluator(DatasetEvaluator):\n    def __init__(self, dataset_name):\n        dataset_dicts = DatasetCatalog.get(dataset_name)\n        self.annotations_cache = {item['image_id']:item['annotations'] for item in dataset_dicts}\n            \n    def reset(self):\n        self.scores = []\n\n    def process(self, inputs, outputs):\n        for inp, out in zip(inputs, outputs):\n            if len(out['instances']) == 0:\n                self.scores.append(0)    \n            else:\n                targ = self.annotations_cache[inp['image_id']]\n                self.scores.append(score(out, targ))\n\n    def evaluate(self):\n        return {\"MaP IoU\": np.mean(self.scores)}\n\nclass Trainer(DefaultTrainer):\n    @classmethod\n    def build_evaluator(cls, cfg, dataset_name, output_folder=None):\n        return MAPIOUEvaluator(dataset_name)\n    ","metadata":{"execution":{"iopub.status.busy":"2021-11-09T01:09:14.394046Z","iopub.execute_input":"2021-11-09T01:09:14.394274Z","iopub.status.idle":"2021-11-09T01:09:14.411849Z","shell.execute_reply.started":"2021-11-09T01:09:14.394244Z","shell.execute_reply":"2021-11-09T01:09:14.411029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train\nI haven't done any hyperparameter optimization yet, this is mostly taken as is from the Detectron tutorial. \n\nTraining for 1000 iterations here for demonstration. For a high scoring model you will need to train it longer, closer to 10000 with these settings","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.merge_from_file(model_zoo.get_config_file(\"COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml\"))\ncfg.DATASETS.TRAIN = (\"sartorius_train\",)\ncfg.DATASETS.TEST = (\"sartorius_val\",)\ncfg.DATALOADER.NUM_WORKERS = 2\ncfg.MODEL.WEIGHTS = model_zoo.get_checkpoint_url(\"COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml\")  # Let training initialize from model zoo\ncfg.SOLVER.IMS_PER_BATCH = 2\ncfg.SOLVER.BASE_LR = 0.000025 \ncfg.SOLVER.MAX_ITER = 10000    \ncfg.SOLVER.CHECKPOINT_PERIOD = len(DatasetCatalog.get('sartorius_train')) // cfg.SOLVER.IMS_PER_BATCH \ncfg.SOLVER.STEPS = []        \ncfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 512   \ncfg.MODEL.ROI_HEADS.NUM_CLASSES = 3  \ncfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = .5\ncfg.TEST.EVAL_PERIOD = len(DatasetCatalog.get('sartorius_train')) // cfg.SOLVER.IMS_PER_BATCH  # Once per epoch\n\nos.makedirs(cfg.OUTPUT_DIR, exist_ok=True)\ntrainer = Trainer(cfg) \ntrainer.resume_or_load(resume=False)\ntrainer.train()","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2021-11-09T01:09:14.413604Z","iopub.execute_input":"2021-11-09T01:09:14.414157Z","iopub.status.idle":"2021-11-09T01:09:37.205972Z","shell.execute_reply.started":"2021-11-09T01:09:14.414089Z","shell.execute_reply":"2021-11-09T01:09:37.204610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lets look at some of the validation files to check if things look reasonable\nWe show predictions on the left and ground truth on the right","metadata":{}},{"cell_type":"code","source":"cfg.MODEL.WEIGHTS = os.path.join(cfg.OUTPUT_DIR, \"model_final.pth\")  # path to the model we just trained\ncfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.5   # set a custom testing threshold\npredictor = DefaultPredictor(cfg)\ndataset_dicts = DatasetCatalog.get('sartorius_val')\nouts = []\nfor d in random.sample(dataset_dicts, 3):    \n    im = cv2.imread(d[\"file_name\"])\n    outputs = predictor(im)  # format is documented at https://detectron2.readthedocs.io/tutorials/models.html#model-output-format\n    v = Visualizer(im[:, :, ::-1],\n                   metadata = MetadataCatalog.get('sartorius_train'), \n                    \n                   instance_mode=ColorMode.IMAGE_BW   # remove the colors of unsegmented pixels. This option is only available for segmentation models\n    )\n    out_pred = v.draw_instance_predictions(outputs[\"instances\"].to(\"cpu\"))\n    visualizer = Visualizer(im[:, :, ::-1], metadata=MetadataCatalog.get('sartorius_train'))\n    out_target = visualizer.draw_dataset_dict(d)\n    outs.append(out_pred)\n    outs.append(out_target)\n_,axs = plt.subplots(len(outs)//2,2,figsize=(40,45))\nfor ax, out in zip(axs.reshape(-1), outs):\n    ax.imshow(out.get_image()[:, :, ::-1])","metadata":{"execution":{"iopub.status.busy":"2021-11-08T09:43:21.489774Z","iopub.status.idle":"2021-11-08T09:43:21.490351Z","shell.execute_reply.started":"2021-11-08T09:43:21.490089Z","shell.execute_reply":"2021-11-08T09:43:21.490115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We can see that while it is not perfect, it did learn something\nWe can now take our model file and use it to generate submission in the [final notebook](https://www.kaggle.com/slawekbiel/positive-score-with-detectron-3-3-inference)","metadata":{}},{"cell_type":"code","source":"!ls ./output/model_final.pth","metadata":{"execution":{"iopub.status.busy":"2021-11-08T09:43:21.491523Z","iopub.status.idle":"2021-11-08T09:43:21.492067Z","shell.execute_reply.started":"2021-11-08T09:43:21.491834Z","shell.execute_reply":"2021-11-08T09:43:21.49186Z"},"trusted":true},"execution_count":null,"outputs":[]}]}