{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"detectron2 using cocolike structure training mask rcnn for 5000 iterations.","metadata":{}},{"cell_type":"code","source":"!pip install openmim\n!mim install mmdet\n!git clone https://github.com/open-mmlab/mmdetection.git\n%cd mmdetection\n!pip install -q -e .","metadata":{"execution":{"iopub.status.busy":"2021-09-06T15:07:18.449545Z","iopub.execute_input":"2021-09-06T15:07:18.44992Z","iopub.status.idle":"2021-09-06T15:08:35.217998Z","shell.execute_reply.started":"2021-09-06T15:07:18.449886Z","shell.execute_reply":"2021-09-06T15:08:35.216992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile configs/mask_rcnn/custom_mask_rcnn_r50_fpn.py\n\n# dataset settings\ndataset_type = 'CocoDataset'\nclasses = ('ship',) # Added\ndata_root = '/kaggle/input/' \nimg_norm_cfg = dict(  # Image normalization config to normalize the input images\n    mean=[123.675, 116.28, 103.53],  # Mean values used to pre-training the pre-trained backbone models\n    std=[58.395, 57.12, 57.375],  # Standard variance used to pre-training the pre-trained backbone models\n    to_rgb=True\n)  # The channel orders of image used to pre-training the pre-trained backbone models\n\n# Augmentations and pipeline. \ntrain_pipeline = [  # Training pipeline\n    dict(type='LoadImageFromFile'),  # First pipeline to load images from file path\n    dict(\n        type='LoadAnnotations',  # Second pipeline to load annotations for current image\n        with_bbox=True,  # Whether to use bounding box, True for detection\n        with_mask=True,  # Whether to use instance mask, True for instance segmentation\n        poly2mask=False),  # Whether to convert the polygon mask to instance mask, set False for acceleration and to save memory\n    dict(\n        type='Resize',  # Augmentation pipeline that resize the images and their annotations\n        img_scale=(1333, 800),  # The largest scale of image\n        keep_ratio=True\n    ),  # whether to keep the ratio between height and width.\n    dict(\n        type='RandomFlip',  # Augmentation pipeline that flip the images and their annotations\n        flip_ratio=0.5),  # The ratio or probability to flip\n    dict(\n        type='Normalize',  # Augmentation pipeline that normalize the input images\n        mean=[123.675, 116.28, 103.53],  # These keys are the same of img_norm_cfg since the\n        std=[58.395, 57.12, 57.375],  # keys of img_norm_cfg are used here as arguments\n        to_rgb=True),\n    dict(\n        type='Pad',  # Padding config\n        size_divisor=32),  # The number the padded images should be divisible\n    dict(type='DefaultFormatBundle'),  # Default format bundle to gather data in the pipeline\n    dict(\n        type='Collect',  # Pipeline that decides which keys in the data should be passed to the detector\n        keys=['img', 'gt_bboxes', 'gt_labels', 'gt_masks'])\n]\ntest_pipeline = [\n    dict(type='LoadImageFromFile'),  # First pipeline to load images from file path\n    dict(\n        type='MultiScaleFlipAug',  # An encapsulation that encapsulates the testing augmentations\n        img_scale=(1333, 800),  # Decides the largest scale for testing, used for the Resize pipeline\n        flip=False,  # Whether to flip images during testing\n        transforms=[\n            dict(type='Resize',  # Use resize augmentation\n                 keep_ratio=True),  # Whether to keep the ratio between height and width, the img_scale set here will be suppressed by the img_scale set above.\n            dict(type='RandomFlip'),  # Thought RandomFlip is added in pipeline, it is not used because flip=False\n            dict(\n                type='Normalize',  # Normalization config, the values are from img_norm_cfg\n                mean=[123.675, 116.28, 103.53],\n                std=[58.395, 57.12, 57.375],\n                to_rgb=True),\n            dict(\n                type='Pad',  # Padding config to pad images divisable by 32.\n                size_divisor=32),\n            dict(\n                type='ImageToTensor',  # convert image to tensor\n                keys=['img']),\n            dict(\n                type='Collect',  # Collect pipeline that collect necessary keys for testing.\n                keys=['img'])\n        ])\n]\n\n\ndata = dict(\n    samples_per_gpu=4,  # Batch size of a single GPU\n    workers_per_gpu=4,  # Worker to pre-fetch data for each single GPU\n    train=dict(  # Train dataset config\n        type='CocoDataset',  # Type of dataset, refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/datasets/coco.py#L19 for details.\n               ann_file=data_root + 'ut-airbus-preprocess/train_annotations.json', # Modified\n        img_prefix=data_root + 'airbus-ship-detection/train_v2/', # Modified\n        classes=classes,\n        pipeline=[  # pipeline, this is passed by the train_pipeline created before.\n            dict(type='LoadImageFromFile'),\n            dict(\n                type='LoadAnnotations',\n                with_bbox=True,\n                with_mask=True,\n                poly2mask=False),\n            dict(type='Resize', img_scale=(1333, 800), keep_ratio=True),\n            dict(type='RandomFlip', flip_ratio=0.5),\n            dict(\n                type='Normalize',\n                mean=[123.675, 116.28, 103.53],\n                std=[58.395, 57.12, 57.375],\n                to_rgb=True),\n            dict(type='Pad', size_divisor=32),\n            dict(type='DefaultFormatBundle'),\n            dict(\n                type='Collect',\n                keys=['img', 'gt_bboxes', 'gt_labels', 'gt_masks'])\n        ]),\n    val=dict(  # Validation dataset config\n        type='CocoDataset',\n        ann_file=data_root + 'ut-airbus-preprocess/test_annotations.json', # Modified\n        img_prefix=data_root + 'airbus-ship-detection/test_v2/', # Modified\n        classes=classes,\n        pipeline=[  # Pipeline is passed by test_pipeline created before\n            dict(type='LoadImageFromFile'),\n            dict(\n                type='MultiScaleFlipAug',\n                img_scale=(1333, 800),\n                flip=False,\n                transforms=[\n                    dict(type='Resize', keep_ratio=True),\n                    dict(type='RandomFlip'),\n                    dict(\n                        type='Normalize',\n                        mean=[123.675, 116.28, 103.53],\n                        std=[58.395, 57.12, 57.375],\n                        to_rgb=True),\n                    dict(type='Pad', size_divisor=32),\n                    dict(type='ImageToTensor', keys=['img']),\n                    dict(type='Collect', keys=['img'])\n                ])\n        ]),\n    test=dict(  # Test dataset config, modify the ann_file for test-dev/test submission\n        type='CocoDataset',\n        ann_file=data_root + 'ut-airbus-preprocess/test_annotations.json', # Modified\n        img_prefix=data_root + 'airbus-ship-detection/test_v2/', # Modified\n        classes=classes,\n        pipeline=[  # Pipeline is passed by test_pipeline created before\n            dict(type='LoadImageFromFile'),\n            dict(\n                type='MultiScaleFlipAug',\n                img_scale=(1333, 800),\n                flip=False,\n                transforms=[\n                    dict(type='Resize', keep_ratio=True),\n                    dict(type='RandomFlip'),\n                    dict(\n                        type='Normalize',\n                        mean=[123.675, 116.28, 103.53],\n                        std=[58.395, 57.12, 57.375],\n                        to_rgb=True),\n                    dict(type='Pad', size_divisor=32),\n                    dict(type='ImageToTensor', keys=['img']),\n                    dict(type='Collect', keys=['img'])\n                ])\n        ],\n        samples_per_gpu=4  # Batch size of a single GPU used in testing\n        ))\n\n# model settings\nmodel = dict(\n    type='MaskRCNN',  # The name of detector\n    backbone=dict(  # The config of backbone\n        type='ResNet',  # The type of the backbone, refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/backbones/resnet.py#L308 for more details.\n        depth=50,  # The depth of backbone, usually it is 50 or 101 for ResNet and ResNext backbones.\n        num_stages=4,  # Number of stages of the backbone.\n        out_indices=(0, 1, 2, 3),  # The index of output feature maps produced in each stages\n        frozen_stages=1,  # The weights in the first 1 stage are fronzen\n        norm_cfg=dict(  # The config of normalization layers.\n            type='BN',  # Type of norm layer, usually it is BN or GN\n            requires_grad=True),  # Whether to train the gamma and beta in BN\n        norm_eval=True,  # Whether to freeze the statistics in BN\n        style='pytorch', # The style of backbone, 'pytorch' means that stride 2 layers are in 3x3 conv, 'caffe' means stride 2 layers are in 1x1 convs.\n        init_cfg=dict(type='Pretrained', checkpoint='torchvision://resnet50')),  # The ImageNet pretrained backbone to be loaded\n    neck=dict(\n        type='FPN',  # The neck of detector is FPN. We also support 'NASFPN', 'PAFPN', etc. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/necks/fpn.py#L10 for more details.\n        in_channels=[256, 512, 1024, 2048],  # The input channels, this is consistent with the output channels of backbone\n        out_channels=256,  # The output channels of each level of the pyramid feature map\n        num_outs=5),  # The number of output scales\n    rpn_head=dict(\n        type='RPNHead',  # The type of RPN head is 'RPNHead', we also support 'GARPNHead', etc. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/dense_heads/rpn_head.py#L12 for more details.\n        in_channels=256,  # The input channels of each input feature map, this is consistent with the output channels of neck\n        feat_channels=256,  # Feature channels of convolutional layers in the head.\n        anchor_generator=dict(  # The config of anchor generator\n            type='AnchorGenerator',  # Most of methods use AnchorGenerator, SSD Detectors uses `SSDAnchorGenerator`. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/anchor/anchor_generator.py#L10 for more details\n            scales=[8],  # Basic scale of the anchor, the area of the anchor in one position of a feature map will be scale * base_sizes\n            ratios=[0.5, 1.0, 2.0],  # The ratio between height and width.\n            strides=[4, 8, 16, 32, 64]),  # The strides of the anchor generator. This is consistent with the FPN feature strides. The strides will be taken as base_sizes if base_sizes is not set.\n        bbox_coder=dict(  # Config of box coder to encode and decode the boxes during training and testing\n            type='DeltaXYWHBBoxCoder',  # Type of box coder. 'DeltaXYWHBBoxCoder' is applied for most of methods. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/bbox/coder/delta_xywh_bbox_coder.py#L9 for more details.\n            target_means=[0.0, 0.0, 0.0, 0.0],  # The target means used to encode and decode boxes\n            target_stds=[1.0, 1.0, 1.0, 1.0]),  # The standard variance used to encode and decode boxes\n        loss_cls=dict(  # Config of loss function for the classification branch\n            type='CrossEntropyLoss',  # Type of loss for classification branch, we also support FocalLoss etc.\n            use_sigmoid=True,  # RPN usually perform two-class classification, so it usually uses sigmoid function.\n            loss_weight=1.0),  # Loss weight of the classification branch.\n        loss_bbox=dict(  # Config of loss function for the regression branch.\n            type='L1Loss',  # Type of loss, we also support many IoU Losses and smooth L1-loss, etc. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/losses/smooth_l1_loss.py#L56 for implementation.\n            loss_weight=1.0)),  # Loss weight of the regression branch.\n    roi_head=dict(  # RoIHead encapsulates the second stage of two-stage/cascade detectors.\n        type='StandardRoIHead',  # Type of the RoI head. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/roi_heads/standard_roi_head.py#L10 for implementation.\n        bbox_roi_extractor=dict(  # RoI feature extractor for bbox regression.\n            type='SingleRoIExtractor',  # Type of the RoI feature extractor, most of methods uses SingleRoIExtractor. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/roi_heads/roi_extractors/single_level.py#L10 for details.\n            roi_layer=dict(  # Config of RoI Layer\n                type='RoIAlign',  # Type of RoI Layer, DeformRoIPoolingPack and ModulatedDeformRoIPoolingPack are also supported. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/ops/roi_align/roi_align.py#L79 for details.\n                output_size=7,  # The output size of feature maps.\n                sampling_ratio=0),  # Sampling ratio when extracting the RoI features. 0 means adaptive ratio.\n            out_channels=256,  # output channels of the extracted feature.\n            featmap_strides=[4, 8, 16, 32]),  # Strides of multi-scale feature maps. It should be consistent to the architecture of the backbone.\n        bbox_head=dict(  # Config of box head in the RoIHead.\n            type='Shared2FCBBoxHead',  # Type of the bbox head, Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/roi_heads/bbox_heads/convfc_bbox_head.py#L177 for implementation details.\n            in_channels=256,  # Input channels for bbox head. This is consistent with the out_channels in roi_extractor\n            fc_out_channels=1024,  # Output feature channels of FC layers.\n            roi_feat_size=7,  # Size of RoI features\n            num_classes=80,  # Number of classes for classification\n            bbox_coder=dict(  # Box coder used in the second stage.\n                type='DeltaXYWHBBoxCoder',  # Type of box coder. 'DeltaXYWHBBoxCoder' is applied for most of methods.\n                target_means=[0.0, 0.0, 0.0, 0.0],  # Means used to encode and decode box\n                target_stds=[0.1, 0.1, 0.2, 0.2]),  # Standard variance for encoding and decoding. It is smaller since the boxes are more accurate. [0.1, 0.1, 0.2, 0.2] is a conventional setting.\n            reg_class_agnostic=False,  # Whether the regression is class agnostic.\n            loss_cls=dict(  # Config of loss function for the classification branch\n                type='CrossEntropyLoss',  # Type of loss for classification branch, we also support FocalLoss etc.\n                use_sigmoid=False,  # Whether to use sigmoid.\n                loss_weight=1.0),  # Loss weight of the classification branch.\n            loss_bbox=dict(  # Config of loss function for the regression branch.\n                type='L1Loss',  # Type of loss, we also support many IoU Losses and smooth L1-loss, etc.\n                loss_weight=1.0)),  # Loss weight of the regression branch.\n        mask_roi_extractor=dict(  # RoI feature extractor for mask generation.\n            type='SingleRoIExtractor',  # Type of the RoI feature extractor, most of methods uses SingleRoIExtractor.\n            roi_layer=dict(  # Config of RoI Layer that extracts features for instance segmentation\n                type='RoIAlign',  # Type of RoI Layer, DeformRoIPoolingPack and ModulatedDeformRoIPoolingPack are also supported\n                output_size=14,  # The output size of feature maps.\n                sampling_ratio=0),  # Sampling ratio when extracting the RoI features.\n            out_channels=256,  # Output channels of the extracted feature.\n            featmap_strides=[4, 8, 16, 32]),  # Strides of multi-scale feature maps.\n        mask_head=dict(  # Mask prediction head\n            type='FCNMaskHead',  # Type of mask head, refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/models/roi_heads/mask_heads/fcn_mask_head.py#L21 for implementation details.\n            num_convs=4,  # Number of convolutional layers in mask head.\n            in_channels=256,  # Input channels, should be consistent with the output channels of mask roi extractor.\n            conv_out_channels=256,  # Output channels of the convolutional layer.\n            num_classes=80,  # Number of class to be segmented.\n            loss_mask=dict(  # Config of loss function for the mask branch.\n                type='CrossEntropyLoss',  # Type of loss used for segmentation\n                use_mask=True,  # Whether to only train the mask in the correct class.\n                loss_weight=1.0))),\n    train_cfg = dict(  # Config of training hyperparameters for rpn and rcnn\n        rpn=dict(  # Training config of rpn\n            assigner=dict(  # Config of assigner\n                type='MaxIoUAssigner',  # Type of assigner, MaxIoUAssigner is used for many common detectors. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/bbox/assigners/max_iou_assigner.py#L10 for more details.\n                pos_iou_thr=0.7,  # IoU >= threshold 0.7 will be taken as positive samples\n                neg_iou_thr=0.3,  # IoU < threshold 0.3 will be taken as negative samples\n                min_pos_iou=0.3,  # The minimal IoU threshold to take boxes as positive samples\n                match_low_quality=True,  # Whether to match the boxes under low quality (see API doc for more details).\n                ignore_iof_thr=-1),  # IoF threshold for ignoring bboxes\n            sampler=dict(  # Config of positive/negative sampler\n                type='RandomSampler',  # Type of sampler, PseudoSampler and other samplers are also supported. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/bbox/samplers/random_sampler.py#L8 for implementation details.\n                num=256,  # Number of samples\n                pos_fraction=0.5,  # The ratio of positive samples in the total samples.\n                neg_pos_ub=-1,  # The upper bound of negative samples based on the number of positive samples.\n                add_gt_as_proposals=False),  # Whether add GT as proposals after sampling.\n            allowed_border=-1,  # The border allowed after padding for valid anchors.\n            pos_weight=-1,  # The weight of positive samples during training.\n            debug=False),  # Whether to set the debug mode\n        rpn_proposal=dict(  # The config to generate proposals during training\n            nms_across_levels=False,  # Whether to do NMS for boxes across levels. Only work in `GARPNHead`, naive rpn does not support do nms cross levels.\n            nms_pre=2000,  # The number of boxes before NMS\n            nms_post=1000,  # The number of boxes to be kept by NMS, Only work in `GARPNHead`.\n            max_per_img=1000,  # The number of boxes to be kept after NMS.\n            nms=dict( # Config of NMS\n                type='nms',  # Type of NMS\n                iou_threshold=0.7 # NMS threshold\n                ),\n            min_bbox_size=0),  # The allowed minimal box size\n        rcnn=dict(  # The config for the roi heads.\n            assigner=dict(  # Config of assigner for second stage, this is different for that in rpn\n                type='MaxIoUAssigner',  # Type of assigner, MaxIoUAssigner is used for all roi_heads for now. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/bbox/assigners/max_iou_assigner.py#L10 for more details.\n                pos_iou_thr=0.5,  # IoU >= threshold 0.5 will be taken as positive samples\n                neg_iou_thr=0.5,  # IoU < threshold 0.5 will be taken as negative samples\n                min_pos_iou=0.5,  # The minimal IoU threshold to take boxes as positive samples\n                match_low_quality=False,  # Whether to match the boxes under low quality (see API doc for more details).\n                ignore_iof_thr=-1),  # IoF threshold for ignoring bboxes\n            sampler=dict(\n                type='RandomSampler',  # Type of sampler, PseudoSampler and other samplers are also supported. Refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/bbox/samplers/random_sampler.py#L8 for implementation details.\n                num=512,  # Number of samples\n                pos_fraction=0.25,  # The ratio of positive samples in the total samples.\n                neg_pos_ub=-1,  # The upper bound of negative samples based on the number of positive samples.\n                add_gt_as_proposals=True\n            ),  # Whether add GT as proposals after sampling.\n            mask_size=28,  # Size of mask\n            pos_weight=-1,  # The weight of positive samples during training.\n            debug=False)),\n    test_cfg = dict(  # Config for testing hyperparameters for rpn and rcnn\n        rpn=dict(  # The config to generate proposals during testing\n            nms_across_levels=False,  # Whether to do NMS for boxes across levels. Only work in `GARPNHead`, naive rpn does not support do nms cross levels.\n            nms_pre=1000,  # The number of boxes before NMS\n            nms_post=1000,  # The number of boxes to be kept by NMS, Only work in `GARPNHead`.\n            max_per_img=1000,  # The number of boxes to be kept after NMS.\n            nms=dict( # Config of NMS\n                type='nms',  #Type of NMS\n                iou_threshold=0.7 # NMS threshold\n                ),\n            min_bbox_size=0),  # The allowed minimal box size\n        rcnn=dict(  # The config for the roi heads.\n            score_thr=0.05,  # Threshold to filter out boxes\n            nms=dict(  # Config of NMS in the second stage\n                type='nms',  # Type of NMS\n                iou_thr=0.5),  # NMS threshold\n            max_per_img=100,  # Max number of detections of each image\n            mask_thr_binary=0.5)))  # Threshold of mask prediction\n    \n# optimizer\nevaluation = dict(  # The config to build the evaluation hook, refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/evaluation/eval_hooks.py#L7 for more details.\n    interval=1,  # Evaluation interval\n    metric=['bbox', 'segm'])  # Metrics used during evaluation\noptimizer = dict(  # Config used to build optimizer, support all the optimizers in PyTorch whose arguments are also the same as those in PyTorch\n    type='SGD',  # Type of optimizers, refer to https://github.com/open-mmlab/mmdetection/blob/master/mmdet/core/optimizer/default_constructor.py#L13 for more details\n    lr=0.02,  # Learning rate of optimizers, see detail usages of the parameters in the documentaion of PyTorch\n    momentum=0.9,  # Momentum\n    weight_decay=0.0001)  # Weight decay of SGD\noptimizer_config = dict(  # Config used to build the optimizer hook, refer to https://github.com/open-mmlab/mmcv/blob/master/mmcv/runner/hooks/optimizer.py#L8 for implementation details.\n    grad_clip=None)  # Most of the methods do not use gradient clip\nlr_config = dict(  # Learning rate scheduler config used to register LrUpdater hook\n    policy='step',  # The policy of scheduler, also support CosineAnnealing, Cyclic, etc. Refer to details of supported LrUpdater from https://github.com/open-mmlab/mmcv/blob/master/mmcv/runner/hooks/lr_updater.py#L9.\n    warmup='linear',  # The warmup policy, also support `exp` and `constant`.\n    warmup_iters=500,  # The number of iterations for warmup\n    warmup_ratio=\n    0.001,  # The ratio of the starting learning rate used for warmup\n    step=[8, 11])  # Steps to decay the learning rate\nrunner = dict(\n    type='EpochBasedRunner', # Type of runner to use (i.e. IterBasedRunner or EpochBasedRunner)\n    max_epochs=1) # Runner that runs the workflow in total max_epochs. For IterBasedRunner use `max_iters`\ncheckpoint_config = dict(  # Config to set the checkpoint hook, Refer to https://github.com/open-mmlab/mmcv/blob/master/mmcv/runner/hooks/checkpoint.py for implementation.\n    interval=1)  # The save interval is 1\nlog_config = dict(  # config to register logger hook\n    interval=50,  # Interval to print the log\n    hooks=[\n        # dict(type='TensorboardLoggerHook')  # The Tensorboard logger is also supported\n        dict(type='TextLoggerHook')\n    ])  # The logger used to record the training process.\ndist_params = dict(backend='nccl')  # Parameters to setup distributed training, the port can also be set.\nlog_level = 'INFO'  # The level of logging.\n\nresume_from = None  # Resume checkpoints from a given path, the training will be resumed from the epoch when the checkpoint's is saved.\nworkflow = [('train', 1)]  # Workflow for runner. [('train', 1)] means there is only one workflow and the workflow named 'train' is executed once. The workflow trains the model by 12 epochs according to the total_epochs.\nwork_dir = 'work_dir'  # Directory to save the model checkpoints and logs for the current experiments.\n# load_from = None  # load models as a pre-trained model from a given path. This will not resume training.\nload_from = 'https://download.openmmlab.com/mmdetection/v2.0/mask_rcnn/mask_rcnn_r50_caffe_fpn_mstrain-poly_3x_coco/mask_rcnn_r50_caffe_fpn_mstrain-poly_3x_coco_bbox_mAP-0.408__segm_mAP-0.37_20200504_163245-42aa3d00.pth'  # Modified","metadata":{"execution":{"iopub.status.busy":"2021-09-06T15:52:45.989682Z","iopub.execute_input":"2021-09-06T15:52:45.990043Z","iopub.status.idle":"2021-09-06T15:52:46.007255Z","shell.execute_reply.started":"2021-09-06T15:52:45.989997Z","shell.execute_reply":"2021-09-06T15:52:46.006358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python tools/train.py configs/mask_rcnn/custom_mask_rcnn_r50_fpn.py","metadata":{"execution":{"iopub.status.busy":"2021-09-06T15:52:47.987507Z","iopub.execute_input":"2021-09-06T15:52:47.987834Z","iopub.status.idle":"2021-09-06T16:01:59.004987Z","shell.execute_reply.started":"2021-09-06T15:52:47.987801Z","shell.execute_reply":"2021-09-06T16:01:59.004167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import copy\n# import json\n# import pycocotools\n# import random \n\n# import matplotlib.pyplot as plt\n# from tqdm import tqdm\n# from pathlib import Path\n# from collections import defaultdict\n\n# import detectron2\n# import detectron2.data.transforms as T\n# import detectron2.utils.comm as comm\n\n# from detectron2.utils.logger import setup_logger\n# setup_logger()\n\n# from detectron2 import model_zoo\n# from detectron2.engine import DefaultTrainer, DefaultPredictor\n# from detectron2.config import get_cfg\n\n# from detectron2.data import MetadataCatalog,DatasetMapper,build_detection_train_loader,build_detection_test_loader\n# from detectron2.data import detection_utils as utils\n# from detectron2.data.catalog import DatasetCatalog\n# from detectron2.data.datasets import register_coco_instances \n\n# from detectron2.evaluation import COCOEvaluator, inference_on_dataset\n\n# from detectron2.projects.deeplab import add_deeplab_config, build_lr_scheduler\n\n# from PIL import ImageFile\n# ImageFile.LOAD_TRUNCATED_IMAGES = True #Truncated image -> https://github.com/keras-team/keras/issues/5475","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# register_coco_instances(\"my_dataset_train\",{},\"../input/ut-airbus-preprocess/test_annotations.json\",\"../input/airbus-ship-detection/train_v2/\") #TRAIN annotations got mixed up -.-\n# register_coco_instances(\"my_dataset_val\",{},\"../input/ut-airbus-preprocess/train_annotations.json\",\"../input/airbus-ship-detection/train_v2/\")\n\n# my_dataset_train_metadata = MetadataCatalog.get(\"my_dataset_train\")\n# dataset_dicts = DatasetCatalog.get(\"my_dataset_train\")\n\n# for d in random.sample(dataset_dicts, 6): #Random 6 pictures from the dataset, can be only sea or ship too... \n#     img = cv2.imread(d['file_name'])\n#     visualizer = Visualizer(img[:, :, ::-1], metadata=my_dataset_train_metadata, scale=0.5)\n#     vis = visualizer.draw_dataset_dict(d)\n#     plt.figure(figsize = (10,10))\n#     plt.imshow(vis.get_image()[:, :, ::-1])","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cfg = get_cfg()\n# cfg.merge_from_file(model_zoo.get_config_file(\"COCO-InstanceSegmentation/mask_rcnn_X_101_32x8d_FPN_3x.yaml\"))\n# cfg.DATASETS.TRAIN = (\"my_dataset_train\",)\n# cfg.DATASETS.TEST = (\"my_dataset_val\",)\n# #cfg.TEST.EVAL_PERIOD = 500\n# cfg.DATALOADER.NUM_WORKERS = 4\n# cfg.SOLVER.IMS_PER_BATCH = 4\n# cfg.SOLVER.BASE_LR = 0.00025  # pick a good LR\n# cfg.SOLVER.MAX_ITER = 5000\n# cfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 512   \n# cfg.MODEL.ROI_HEADS.NUM_CLASSES = 1  # only has one class (ship)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from detectron2.engine import DefaultTrainer\n# from detectron2.evaluation import COCOEvaluator\n# class CocoTrainer(DefaultTrainer):\n#     @classmethod\n#     def build_evaluator(cls, cfg, dataset_name, output_folder=None):\n#         if output_folder is None:\n#             os.makedirs(\"coco_eval\", exist_ok=True)\n#             output_folder = \"coco_eval\"\n#         return COCOEvaluator(dataset_name, cfg, False, output_folder)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext tensorboard\n%tensorboard --logdir logs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# os.makedirs(cfg.OUTPUT_DIR, exist_ok=True)\n# trainer = DefaultTrainer(cfg) \n# trainer.resume_or_load(resume=False)\n# trainer.train() #Trainer will throw out non-annotated pictures. ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cfg.MODEL.WEIGHTS = os.path.join(cfg.OUTPUT_DIR, \"model_final.pth\")  # path to the model we just trained\n# cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.7   # set a custom testing threshold\n# predictor = DefaultPredictor(cfg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Should build valuator inside the training. So it could be evaluated from there. ","metadata":{}},{"cell_type":"code","source":"# from detectron2.evaluation import COCOEvaluator, inference_on_dataset\n# from detectron2.data import build_detection_test_loader\n# evaluator = COCOEvaluator(\"my_dataset_val\", cfg, False, output_dir=\"../output/\") #Should bne val but ffs.  \n# val_loader = build_detection_test_loader(cfg, \"my_dataset_val\")\n# print(inference_on_dataset(trainer.model, val_loader, evaluator)) #https://medium.com/@yanfengliux/the-confusing-metrics-of-ap-and-map-for-object-detection-3113ba0386ef","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from detectron2.checkpoint import DetectionCheckpointer\n\n# checkpointer = DetectionCheckpointer(trainer.model, save_dir=\"./\")\n# checkpointer.save(\"model_mask_resnet101_rcnn\")  # save to save_dir","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #https://www.kaggle.com/ptaneja/detectron2-notebook\n# import glob\n\n# def create_test_datatset():    \n#     img_dir = '../input/airbus-ship-detection/test_v2/'\n#     dataset_dicts = []\n    \n#     for img_path in glob.glob(img_dir + '*.jpg'):\n#         record = {}\n#         file_path = img_path\n#         image_id = img_path.split('/')[-1].split('.')[0]\n#         record['file_name'] = file_path\n#         record['image_id'] = image_id\n#         dataset_dicts.append(record)\n#     return dataset_dicts\n\n# test_dataset = create_test_datatset()\n# img_ids = []\n# pred_string = []\n\n# DatasetCatalog.register(\"submit_test1\", create_test_datatset)\n# od_dataset = MetadataCatalog.get(\"submit_test1\")\n\n# cfg.MODEL.WEIGHTS = os.path.join(cfg.OUTPUT_DIR, \"model_final.pth\")\n# cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.7   # set the testing threshold for this model\n# cfg.DATASETS.TEST = (\"submit_test1\", )\n# predictor = DefaultPredictor(cfg)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for test_data in test_dataset[0:30]:\n#     image_id = test_data['file_name'].split('/')[-1].split('.')[0]\n#     img = plt.imread(test_data['file_name'])\n#     outputs = predictor(img)\n#     v = Visualizer(img[:, :, ::-1], metadata=od_dataset, scale=0.3,)\n#     v = v.draw_instance_predictions(outputs[\"instances\"].to(\"cpu\") )\n#     plt.figure(figsize=(25, 15))\n#     plt.imshow(cv2.cvtColor(v.get_image()[:, :, ::-1], cv2.COLOR_BGR2RGB))\n#     plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def PixelsToRLenc(pixels ,order='F',format=True):\n#     \"\"\"\n#     Based off code by https://www.kaggle.com/alexlzzz\n#     pixels is a list of absolute pixel values which need to be converted. (1-243600)\n#     order is down-then-right, i.e. Fortran\n#     format determines if the order needs to be preformatted (according to submission rules) or not\n    \n#     returns run length as an array or string (if format is True)\n#     \"\"\"\n    \n#     # Initialse empty array\n#     bytes = []\n#     for _ in range(0, 243600):\n#         bytes.append(0)\n    \n#     # Place values from input list into the array\n#     for x in pixels:\n#         p = x - 1\n#         bytes[p] = 1\n    \n#     runs = [] ## list of run lengths\n#     r = 0     ## the current run length\n#     pos = 1   ## count starts from 1 per WK\n#     for c in bytes:\n#         if ( c == 0 ):\n#             if r != 0:\n#                 runs.append((pos, r))\n#                 pos+=r\n#                 r=0\n#             pos+=1\n#         else:\n#             r+=1\n\n#     #if last run is unsaved (i.e. data ends with 1)\n#     if r != 0:\n#         runs.append((pos, r))\n#         pos += r\n#         r = 0\n\n#     if format:\n#         z = ''\n    \n#         for rr in runs:\n#             z+='{} {} '.format(rr[0],rr[1])\n#         return z[:-1]\n#     else:\n#         return runs\n\n\n# img_ids = []\n# pred_string = []\n# i = 0\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub={\"ImageId\":img_ids, \"EncodedPixels\":PixelsToRLenc(pred_string)}\n#sub=pd.DataFrame(sub)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub.to_csv('/kaggle/working/submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!kaggle competitions submit -c airbus-ship-detection -f submission.csv -m \"Test1\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}