{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **SIIM COVID-19 Detectron2 Training**","metadata":{"papermill":{"duration":0.019223,"end_time":"2021-06-08T06:00:48.919199","exception":false,"start_time":"2021-06-08T06:00:48.899976","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"papermill":{"duration":0.698188,"end_time":"2021-06-08T06:00:49.778916","exception":false,"start_time":"2021-06-08T06:00:49.080728","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:07:50.32264Z","iopub.execute_input":"2021-08-16T07:07:50.323054Z","iopub.status.idle":"2021-08-16T07:07:51.01112Z","shell.execute_reply.started":"2021-08-16T07:07:50.322947Z","shell.execute_reply":"2021-08-16T07:07:51.010163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvcc --version","metadata":{"papermill":{"duration":0.68008,"end_time":"2021-06-08T06:00:50.477984","exception":false,"start_time":"2021-06-08T06:00:49.797904","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:08:03.703391Z","iopub.execute_input":"2021-08-16T07:08:03.703726Z","iopub.status.idle":"2021-08-16T07:08:04.338058Z","shell.execute_reply.started":"2021-08-16T07:08:03.703687Z","shell.execute_reply":"2021-08-16T07:08:04.337102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch, torchvision\nprint(torch.__version__, torch.cuda.is_available())","metadata":{"papermill":{"duration":1.370763,"end_time":"2021-06-08T06:00:51.867488","exception":false,"start_time":"2021-06-08T06:00:50.496725","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:08:58.334956Z","iopub.execute_input":"2021-08-16T07:08:58.335314Z","iopub.status.idle":"2021-08-16T07:08:59.695317Z","shell.execute_reply.started":"2021-08-16T07:08:58.335281Z","shell.execute_reply":"2021-08-16T07:08:59.694429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* It seems CUDA=11.0 and torch==1.7.0 is used in this kaggle docker image.\n* See installation for details. https://detectron2.readthedocs.io/en/latest/tutorials/install.html","metadata":{"papermill":{"duration":0.040369,"end_time":"2021-06-08T06:00:51.936836","exception":false,"start_time":"2021-06-08T06:00:51.896467","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Installation of Pre-Built Detectron2","metadata":{"papermill":{"duration":0.029513,"end_time":"2021-06-08T06:00:52.001233","exception":false,"start_time":"2021-06-08T06:00:51.97172","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install detectron2 -f \\\n  https://dl.fbaipublicfiles.com/detectron2/wheels/cu110/torch1.7/index.html","metadata":{"_kg_hide-output":true,"papermill":{"duration":19.398785,"end_time":"2021-06-08T06:01:11.430609","exception":false,"start_time":"2021-06-08T06:00:52.031824","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:09:19.854297Z","iopub.execute_input":"2021-08-16T07:09:19.854625Z","iopub.status.idle":"2021-08-16T07:09:51.870217Z","shell.execute_reply.started":"2021-08-16T07:09:19.854595Z","shell.execute_reply":"2021-08-16T07:09:51.869266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{"papermill":{"duration":0.031195,"end_time":"2021-06-08T06:01:11.493668","exception":false,"start_time":"2021-06-08T06:01:11.462473","status":"completed"},"tags":[]}},{"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd \nfrom datetime import datetime\nimport time\nfrom tqdm import tqdm_notebook as tqdm # progress bar\nimport matplotlib.pyplot as plt\n\nimport os, json, cv2, random\nimport skimage.io as io\nimport copy\nimport pickle\nfrom pathlib import Path\nfrom typing import Optional\nfrom tqdm import tqdm\n\n# torch\nimport torch\n\n# Albumenatations\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\n#from pycocotools.coco import COCO\nfrom sklearn.model_selection import StratifiedKFold\n\n# glob\nfrom glob import glob\n\n# numba\nimport numba\nfrom numba import jit\n\nimport warnings\nwarnings.filterwarnings('ignore') #Ignore \"future\" warnings and Data-Frame-Slicing warnings.\n\n\n# detectron2\nfrom detectron2.structures import BoxMode\nfrom detectron2 import model_zoo\nfrom detectron2.config import get_cfg\nfrom detectron2.data import DatasetCatalog, MetadataCatalog\nfrom detectron2.engine import DefaultPredictor, DefaultTrainer, launch\nfrom detectron2.evaluation import COCOEvaluator\nfrom detectron2.structures import BoxMode\nfrom detectron2.utils.visualizer import ColorMode\nfrom detectron2.utils.logger import setup_logger\nfrom detectron2.utils.visualizer import Visualizer\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\n\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\nimport detectron2.data.transforms as T\nfrom detectron2.evaluation import COCOEvaluator, inference_on_dataset\n\nsetup_logger()","metadata":{"papermill":{"duration":3.159985,"end_time":"2021-06-08T06:01:14.684825","exception":false,"start_time":"2021-06-08T06:01:11.52484","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:11:05.682155Z","iopub.execute_input":"2021-08-16T07:11:05.682486Z","iopub.status.idle":"2021-08-16T07:11:05.696277Z","shell.execute_reply.started":"2021-08-16T07:11:05.682456Z","shell.execute_reply":"2021-08-16T07:11:05.695353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{"papermill":{"duration":0.031384,"end_time":"2021-06-08T06:01:14.747481","exception":false,"start_time":"2021-06-08T06:01:14.716097","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# --- Read data ---\nimgdir = \"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640\"\n# Read in the data CSV files\ntrain_df = pd.read_csv(\"../input/siimcovid19-detection-training-label/train_image_df.csv\")\nlen(train_df)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.123579,"end_time":"2021-06-08T06:01:14.902175","exception":false,"start_time":"2021-06-08T06:01:14.778596","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:11:15.76888Z","iopub.execute_input":"2021-08-16T07:11:15.769218Z","iopub.status.idle":"2021-08-16T07:11:15.89891Z","shell.execute_reply.started":"2021-08-16T07:11:15.769188Z","shell.execute_reply":"2021-08-16T07:11:15.897976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# configs","metadata":{"papermill":{"duration":0.031184,"end_time":"2021-06-08T06:01:14.965675","exception":false,"start_time":"2021-06-08T06:01:14.934491","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# --- configs ---\nthing_classes = [\n    \"atypical\",\n    \"indeterminate\",\n    \"negative\",\n    \"typical\"\n]\n\ndebug=False\nsplit_mode=\"valid20\" # Or  valid20 all_train\n\n\ncategory_name_to_id = {class_name: index for index, class_name in enumerate(thing_classes)}\ncategory_name_to_id","metadata":{"papermill":{"duration":0.040647,"end_time":"2021-06-08T06:01:15.038449","exception":false,"start_time":"2021-06-08T06:01:14.997802","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T07:30:22.922983Z","iopub.execute_input":"2021-08-16T07:30:22.923349Z","iopub.status.idle":"2021-08-16T07:30:22.930516Z","shell.execute_reply.started":"2021-08-16T07:30:22.923318Z","shell.execute_reply":"2021-08-16T07:30:22.929368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation\n* `detectron2` provides high-level API for training custom dataset.\n\nTo define custom dataset, we need to create **list of dict** (`dataset_dicts`) where each dict contains following:\n\n - file_name: file name of the image.\n - image_id: id of the image, index is used here.\n - height: height of the image.\n - width: width of the image.\n - annotation: This is the ground truth annotation data for object detection, which contains following\n     - bbox: bounding box pixel location with shape (n_boxes, 4)\n     - bbox_mode: `BoxMode.XYXY_ABS` is used here, meaning that absolute value of (x_min, y_min, x_max, y_max) annotation is used in the `bbox`.\n     - category_id: class label id for each bounding box, with shape (n_boxes,)\n\n`get_COVID19_data_dicts` is for train dataset preparation and `get_COVID19_data_dicts_test` is for test dataset preparation.","metadata":{"papermill":{"duration":0.032091,"end_time":"2021-06-08T06:01:15.102236","exception":false,"start_time":"2021-06-08T06:01:15.070145","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from glob import glob\n\ndef get_COVID19_data_dicts(\n    imgdir: Path,\n    train_df: pd.DataFrame,\n    use_cache: bool = True,\n    target_indices: Optional[np.ndarray] = None,\n    debug: bool = False,\n    data_type:str=\"train\"\n   \n):\n\n    cache_path = Path(\".\") / f\"dataset_dicts_cache_{data_type}.pkl\"\n    if not use_cache or not cache_path.exists():\n        print(\"Creating data...\")\n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/meta_sz_640.csv\")\n        train_meta=df_meta[df_meta.split==\"train\"]\n        if debug:\n            train_meta = train_meta.iloc[:100]  # For debug....\n\n        # Load 1 image to get image size.\n        image_id = train_meta.iloc[0,0]\n        #image_path = str(imgdir / \"train\" / f\"{image_id}.jpg\")\n        image_path = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/train/{image_id}.jpg')\n        image = cv2.imread(image_path)\n        resized_height, resized_width, ch = image.shape\n        print(f\"image shape: {image.shape}\")\n\n        dataset_dicts = []\n        for index, train_meta_row in tqdm(train_meta.iterrows(), total=len(train_meta)):\n            record = {}\n            image_id, height, width,s = train_meta_row.values\n            #filename = str(imgdir / \"train\" / f\"{image_id}.jpg\")\n            filename = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/train/{image_id}.jpg')\n            record[\"file_name\"] = filename\n            record[\"image_id\"] = image_id\n            record[\"height\"] = resized_height\n            record[\"width\"] = resized_width\n            objs = []\n            for index2, row in train_df.query(\"id == @image_id\").iterrows():\n                # print(row)\n                # print(row[\"class_name\"])\n                # class_name = row[\"class_name\"]\n                class_id = row[\"integer_label\"]\n                if class_id == 2: # NO class\n                    # It is \"No finding\"\n \n                    # Use this No finding class with the bbox covering all image area.\n                    #bbox_resized = [0, 0, resized_width, resized_height]\n                    bbox_resized = [50, 50, 200, 200]\n                    obj = {\n                        \"bbox\": bbox_resized,\n                        \"bbox_mode\": BoxMode.XYXY_ABS,\n                        \"category_id\": class_id,\n                    }\n                    #objs.append(obj)\n\n                else:\n                    # bbox_original = [int(row[\"x_min\"]), int(row[\"y_min\"]), int(row[\"x_max\"]), int(row[\"y_max\"])]\n                    h_ratio = resized_height / height\n                    w_ratio = resized_width / width\n                    bbox_resized = [\n                        float(row[\"x_min\"]) * w_ratio,\n                        float(row[\"y_min\"]) * h_ratio,\n                        float(row[\"x_max\"]) * w_ratio,\n                        float(row[\"y_max\"]) * h_ratio,\n                    ]\n                    obj = {\n                        \"bbox\": bbox_resized,\n                        \"bbox_mode\": BoxMode.XYXY_ABS,\n                        \"category_id\": class_id,\n                    }\n                    objs.append(obj)\n            record[\"annotations\"] = objs\n            dataset_dicts.append(record)\n        with open(cache_path, mode=\"wb\") as f:\n            pickle.dump(dataset_dicts, f)\n\n    print(f\"Load from cache {cache_path}\")\n    with open(cache_path, mode=\"rb\") as f:\n        dataset_dicts = pickle.load(f)\n    if target_indices is not None:\n        dataset_dicts = [dataset_dicts[i] for i in target_indices]\n    return dataset_dicts\n\n\ndef get_COVID19_data_dicts_test(\n    imgdir: Path, test_meta: pd.DataFrame, use_cache: bool = True, debug: bool = False,\n):\n    debug_str = f\"_debug{int(debug)}\"\n    cache_path = Path(\".\") / f\"dataset_dicts_cache_test.pkl\"\n    if not use_cache or not cache_path.exists():\n        print(\"Creating data...\")\n        # test_meta = pd.read_csv(imgdir / \"test_meta.csv\")\n        df_meta = pd.read_csv(\"../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/meta_sz_640.csv\")\n        test_meta=df_meta[df_meta.split==\"test\"]\n        if debug:\n            test_meta = test_meta.iloc[:100]  # For debug....\n        # Load 1 image to get image size.\n        image_id = test_meta.iloc[0,0]\n        #image_path = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n        image_path = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/test/{image_id}.jpg')\n        image = cv2.imread(image_path)\n        resized_height, resized_width, ch = image.shape\n        #print(f\"image shape: {image.shape}\")\n\n        dataset_dicts = []\n        for index, test_meta_row in tqdm(test_meta.iterrows(), total=len(test_meta)):\n            record = {}\n\n            image_id, height, width,s = test_meta_row.values\n            #filename = str(imgdir / \"test\" / f\"{image_id}.jpg\")\n            filename = str(f'../input/siim-covid19-resized-384512-and-640px/SIIM-COVID19-Resized/img_sz_640/test/{image_id}.jpg')\n            record[\"file_name\"] = filename\n            # record[\"image_id\"] = index\n            record[\"image_id\"] = image_id\n            record[\"height\"] = resized_height\n            record[\"width\"] = resized_width\n            # objs = []\n            # record[\"annotations\"] = objs\n            dataset_dicts.append(record)\n        with open(cache_path, mode=\"wb\") as f:\n            pickle.dump(dataset_dicts, f)\n\n    #print(f\"Load from cache {cache_path}\")\n    with open(cache_path, mode=\"rb\") as f:\n        dataset_dicts = pickle.load(f)\n    return dataset_dicts","metadata":{"papermill":{"duration":0.054862,"end_time":"2021-06-08T06:01:15.188776","exception":false,"start_time":"2021-06-08T06:01:15.133914","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-08-16T08:01:43.547632Z","iopub.execute_input":"2021-08-16T08:01:43.547977Z","iopub.status.idle":"2021-08-16T08:01:43.567949Z","shell.execute_reply.started":"2021-08-16T08:01:43.547946Z","shell.execute_reply":"2021-08-16T08:01:43.566778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if split_mode == \"all_train\":\n    DatasetCatalog.register(\n        \"COVID19_data_train\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            debug=debug,\n            data_type=\"train\"\n        ),\n    )\n    MetadataCatalog.get(\"COVID19_data_train\").set(thing_classes=thing_classes)\n    \n    \n    dataset_dicts_train = DatasetCatalog.get(\"COVID19_data_train\")\n    metadata_dicts_train = MetadataCatalog.get(\"COVID19_data_train\")\n    \n    \nelif split_mode == \"valid20\":\n\n    n_dataset = len(\n        get_COVID19_data_dicts(\n            imgdir, train_df, debug=debug,data_type=\"All\"\n        )\n    )\n    n_train = int(n_dataset * 0.90)\n    print(\"n_dataset\", n_dataset, \"n_train\", n_train)\n    rs = np.random.RandomState(42)\n    inds = rs.permutation(n_dataset)\n    train_inds, valid_inds = inds[:n_train], inds[n_train:]\n\n    DatasetCatalog.register(\n        \"COVID19_data_train\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            target_indices=train_inds,\n            debug=debug,\n            data_type=\"train\"\n        ),\n    )\n    MetadataCatalog.get(\"COVID19_data_train\").set(thing_classes=thing_classes)\n    \n\n    DatasetCatalog.register(\n        \"COVID19_data_valid\",\n        lambda: get_COVID19_data_dicts(\n            imgdir,\n            train_df,\n            target_indices=valid_inds,\n            debug=debug,\n            data_type=\"val\"\n            ),\n        )\n    MetadataCatalog.get(\"COVID19_data_valid\").set(thing_classes=thing_classes)\n    \n    dataset_dicts_train = DatasetCatalog.get(\"COVID19_data_train\")\n    metadata_dicts_train = MetadataCatalog.get(\"COVID19_data_train\")\n\n    dataset_dicts_valid = DatasetCatalog.get(\"COVID19_data_valid\")\n    metadata_dicts_valid = MetadataCatalog.get(\"COVID19_data_valid\")\n    \nelse:\n    raise ValueError(f\"[ERROR] Unexpected value split_mode={split_mode}\")","metadata":{"papermill":{"duration":18.053762,"end_time":"2021-06-08T06:01:33.274668","exception":false,"start_time":"2021-06-08T06:01:15.220906","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:07:57.671932Z","iopub.execute_input":"2021-07-31T13:07:57.672274Z","iopub.status.idle":"2021-07-31T13:08:53.613653Z","shell.execute_reply.started":"2021-07-31T13:07:57.672239Z","shell.execute_reply":"2021-07-31T13:08:53.612715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"data_vis\"></a>\n# Data Visualization\n\nIt's also very easy to visualize prepared training dataset with `detectron2`.<br/>\nIt provides `Visualizer` class, we can use it to draw an image with bounding box as following.","metadata":{"papermill":{"duration":0.075599,"end_time":"2021-06-08T06:01:33.427665","exception":false,"start_time":"2021-06-08T06:01:33.352066","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 4, figsize =(20,10))\nindices=[ax[0][0],ax[1][0],ax[0][1],ax[1][1],ax[0][2],ax[1][2],ax[0][3],ax[1][3]]\ni=-1\nfor d in random.sample(dataset_dicts_train, 8):\n    i=i+1    \n    img = cv2.imread(d[\"file_name\"])\n    v = Visualizer(img[:, :, ::-1],\n                   metadata=metadata_dicts_train, \n                   scale=0.3, \n                   instance_mode=ColorMode.IMAGE_BW   # remove the colors of unsegmented pixels. This option is only available for segmentation models\n    )\n    out = v.draw_dataset_dict(d)\n    indices[i].grid(False)\n    indices[i].axis('off')\n    indices[i].imshow(out.get_image()[:, :, ::-1])","metadata":{"_kg_hide-input":true,"papermill":{"duration":1.279878,"end_time":"2021-06-08T06:01:34.78284","exception":false,"start_time":"2021-06-08T06:01:33.502962","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:08:53.620427Z","iopub.execute_input":"2021-07-31T13:08:53.620929Z","iopub.status.idle":"2021-07-31T13:08:54.710746Z","shell.execute_reply.started":"2021-07-31T13:08:53.620889Z","shell.execute_reply":"2021-07-31T13:08:54.709797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = get_cfg()\nconfig_name = \"COCO-Detection/faster_rcnn_R_50_FPN_3x.yaml\" \n#config_name = \"COCO-Detection/faster_rcnn_X_101_32x8d_FPN_3x.yaml\"\n#config_name = \"COCO-Detection/faster_rcnn_R_101_C4_3x.yaml\"\n\ncfg.merge_from_file(model_zoo.get_config_file(config_name))\n\ncfg.DATASETS.TRAIN = (\"COVID19_data_train\",)\n\nif split_mode == \"all_train\":\n    cfg.DATASETS.TEST = ()\nelse:\n    cfg.DATASETS.TEST = (\"COVID19_data_valid\",)\n    cfg.TEST.EVAL_PERIOD = 1000\n\ncfg.DATALOADER.NUM_WORKERS = 0\n#cfg.MODEL.WEIGHTS = model_zoo.get_checkpoint_url(config_name)\ncfg.MODEL.WEIGHTS=\"../input/1siim-covid19-detectron2-weights/output/model_final.pth\"\n\n\ncfg.SOLVER.IMS_PER_BATCH = 2\ncfg.SOLVER.BASE_LR = 0.02\n\ncfg.SOLVER.WARMUP_ITERS = 1000\ncfg.SOLVER.MAX_ITER = 5000 #adjust up if val mAP is still rising, adjust down if overfit\n#cfg.SOLVER.STEPS = (100, 500) # must be less than  MAX_ITER \n#cfg.SOLVER.GAMMA = 0.05\n\n\ncfg.SOLVER.CHECKPOINT_PERIOD = 100000  # Small value=Frequent save need a lot of storage.\ncfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 128\ncfg.MODEL.ROI_HEADS.NUM_CLASSES = 4\n\n\nos.makedirs(cfg.OUTPUT_DIR, exist_ok=True)\n\n\n#Training using custom trainer defined above\ntrainer = AugTrainer(cfg) \n#trainer = DefaultTrainer(cfg) \ntrainer.resume_or_load(resume=False)\ntrainer.train()","metadata":{"_kg_hide-output":true,"papermill":{"duration":22570.609248,"end_time":"2021-06-08T12:17:46.007782","exception":false,"start_time":"2021-06-08T06:01:35.398534","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:26:00.271941Z","iopub.execute_input":"2021-07-31T13:26:00.272266Z","iopub.status.idle":"2021-07-31T13:26:40.643294Z","shell.execute_reply.started":"2021-07-31T13:26:00.272238Z","shell.execute_reply":"2021-07-31T13:26:40.642101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluator","metadata":{"papermill":{"duration":0.727722,"end_time":"2021-06-08T12:17:47.705251","exception":false,"start_time":"2021-06-08T12:17:46.977529","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"* Famouns dataset's evaluator is already implemented in detectron2.\n* For example, many kinds of AP (Average Precision) is calculted in COCOEvaluator.\n* COCOEvaluator only calculates AP with IoU from 0.50 to 0.95","metadata":{"papermill":{"duration":0.717068,"end_time":"2021-06-08T12:17:49.422976","exception":false,"start_time":"2021-06-08T12:17:48.705908","status":"completed"},"tags":[]}},{"cell_type":"code","source":"evaluator = COCOEvaluator(\"COVID19_data_valid\", cfg, False, output_dir=\"./output/\")\n#cfg.MODEL.WEIGHTS=\"./output/model_final.pth\"\n#cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.001   # set a custom testing threshold\nval_loader = build_detection_test_loader(cfg, \"COVID19_data_valid\")\ninference_on_dataset(trainer.model, val_loader, evaluator)","metadata":{"_kg_hide-input":true,"papermill":{"duration":432.076382,"end_time":"2021-06-08T12:25:02.220013","exception":false,"start_time":"2021-06-08T12:17:50.143631","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:38:29.420851Z","iopub.execute_input":"2021-07-31T13:38:29.42118Z","iopub.status.idle":"2021-07-31T13:40:18.532061Z","shell.execute_reply.started":"2021-07-31T13:38:29.42115Z","shell.execute_reply":"2021-07-31T13:40:18.531106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nmetrics_df = pd.read_json(\"./output/metrics.json\", orient=\"records\", lines=True)\nmdf = metrics_df.sort_values(\"iteration\")\nmdf.head(10).T","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.950982,"end_time":"2021-06-08T12:25:03.919134","exception":false,"start_time":"2021-06-08T12:25:02.968152","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:11:45.546516Z","iopub.execute_input":"2021-07-31T13:11:45.546888Z","iopub.status.idle":"2021-07-31T13:11:45.708195Z","shell.execute_reply.started":"2021-07-31T13:11:45.546848Z","shell.execute_reply":"2021-07-31T13:11:45.70715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Loss curve\nfig, ax = plt.subplots()\n\nmdf1 = mdf[~mdf[\"total_loss\"].isna()]\nax.plot(mdf1[\"iteration\"], mdf1[\"total_loss\"], c=\"C0\", label=\"train\")\nif \"validation_loss\" in mdf.columns:\n    mdf2 = mdf[~mdf[\"validation_loss\"].isna()]\n    ax.plot(mdf2[\"iteration\"], mdf2[\"validation_loss\"], c=\"C1\", label=\"validation\")\n\n# ax.set_ylim([0, 0.5])\nax.legend()\nax.set_title(\"Loss curve\")\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.893972,"end_time":"2021-06-08T12:25:05.563472","exception":false,"start_time":"2021-06-08T12:25:04.6695","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:11:45.709427Z","iopub.execute_input":"2021-07-31T13:11:45.709964Z","iopub.status.idle":"2021-07-31T13:11:45.845283Z","shell.execute_reply.started":"2021-07-31T13:11:45.709923Z","shell.execute_reply":"2021-07-31T13:11:45.844386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Loss curve\nfig, ax = plt.subplots()\n\nmdf1 = mdf[~mdf[\"fast_rcnn/cls_accuracy\"].isna()]\nax.plot(mdf1[\"iteration\"], mdf1[\"fast_rcnn/cls_accuracy\"], c=\"C0\", label=\"train\")\n# ax.set_ylim([0, 0.5])\nax.legend()\nax.set_title(\"Accuracy curve\")\nplt.show()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.892482,"end_time":"2021-06-08T12:25:07.210321","exception":false,"start_time":"2021-06-08T12:25:06.317839","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-31T13:11:45.846574Z","iopub.execute_input":"2021-07-31T13:11:45.846921Z","iopub.status.idle":"2021-07-31T13:11:46.004151Z","shell.execute_reply.started":"2021-07-31T13:11:45.846886Z","shell.execute_reply":"2021-07-31T13:11:45.99902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References\n1. https://www.kaggle.com/ammarnassanalhajali/training-detectron2-for-blood-cells-detection\n1. https://www.kaggle.com/corochann/vinbigdata-detectron2-train\n","metadata":{"papermill":{"duration":0.889466,"end_time":"2021-06-08T12:25:08.924018","exception":false,"start_time":"2021-06-08T12:25:08.034552","status":"completed"},"tags":[]}}]}