{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm.notebook import tqdm\nimport ast\nfrom pathlib import Path\nfrom sklearn.model_selection import GroupKFold\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport shutil\nimport json\nimport datetime as dt\ntqdm.pandas()\n\n# read csv\ntrain_csv_path = r\"../input/tensorflow-great-barrier-reef/train.csv\"\ntrain_img_path = Path(r\"../input/tensorflow-great-barrier-reef/train_images\")\ndf = pd.read_csv(train_csv_path)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:31:29.359082Z","iopub.execute_input":"2022-02-11T20:31:29.359629Z","iopub.status.idle":"2022-02-11T20:31:30.430741Z","shell.execute_reply.started":"2022-02-11T20:31:29.359504Z","shell.execute_reply":"2022-02-11T20:31:30.429909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Dataset for MMDetection","metadata":{}},{"cell_type":"code","source":"# prepare data for coco format\ndef get_img_path(row):\n    \"\"\"\n    get image path\n    \"\"\"\n    return os.fspath(train_img_path / f\"video_{row.video_id}\" / f\"{row.video_frame}.jpg\")\n\ndef get_bbox(annotations):\n    \"\"\"\n    convert bounding box to list\n    \"\"\"\n    bboxes = [list(annot.values()) for annot in annotations]\n    return bboxes\n\ndef bbox_checking(row):\n    \"\"\"\n    Fix wrong BB and remove some BB if cant fix it\n    \"\"\"\n    width = row[\"width\"]\n    height = row[\"height\"]\n    \n    annotations = row[\"annotations\"]\n    new_annotation = []\n    for anno in annotations:\n        x0, y0, w, h = anno.values()\n        x1, y1 = x0+w, y0+h\n        \n        # acceptable case\n        # start point case\n        if x0 < 0:\n            print(anno)\n            print(f\"Fix x0: {x0} = 0\")\n            x0 = 0\n            w = x1 - x0\n        if y0 < 0:\n            print(f\"Fix y0: {y0} = 0\")\n            y0 = 0\n            h = y1 - y0\n            \n        # end point case\n        if x1 > width:\n            print(anno)\n            print(f\"Fix w: {w} = {width - x0}\")\n            w = width - x0\n           \n            \n        if y1 > height:\n            print(anno)\n            print(f\"Fix h: {h} = {height - y0}\")\n            h = height - y0\n        \n        # cant acceptable case\n        if w <= 0 or h <= 0 or x0 >= width or y0 >= height:\n            print(anno, \"Cant ACCEPT\")\n            continue\n    \n        anno_dict = {\"x\": x0,\n                     \"y\": y0,\n                     \"width\": w,\n                     \"height\": h}\n        new_annotation.append(anno_dict)\n    return new_annotation\n\ndef prepare_coco_csv(df, df_type=0):\n    tqdm.pandas(desc=\"Convert string annotation to list and dict\")\n    df[\"annotations\"] = df[\"annotations\"].progress_apply(lambda anno: ast.literal_eval(anno))\n    \n    df[\"width\"] = 1280\n    df[\"height\"] = 720\n    \n    tqdm.pandas(desc=\"Bouding box fixing\")\n    df[\"annotations\"] = df.progress_apply(lambda row: bbox_checking(row), axis=1)\n    \n    tqdm.pandas(desc=\"Count bounding box of each frame\")\n    df[\"num_bbox\"] = df[\"annotations\"].progress_apply(lambda anno: len(anno))\n    \n    tqdm.pandas(desc=\"Get bounding box of each frame\")\n    df[\"bboxes\"] = df[\"annotations\"].progress_apply(lambda bboxes: get_bbox(bboxes))\n    \n    tqdm.pandas(desc=\"Get image path of each frame\")\n    df[\"image_path\"] = df.progress_apply(lambda row: get_img_path(row), axis=1)\n    \n\n    if df_type == 0:\n        return df\n    else:\n        return df[df[\"num_bbox\"] != 0]\n\n# K-Fold\ndef sequence2fold(sequence_frame):\n    curr_k = 0\n    results_k = []\n    for idx in range(len(sequence_frame)):\n        if idx == 0:\n            results_k.append(curr_k)\n            continue\n        diff_frame = sequence_frame[idx] - sequence_frame[idx-1]\n        curr_k = curr_k if diff_frame == 1 else curr_k+1\n        results_k.append(curr_k)\n    return results_k\n\ndef kfold_split(df, n_slits=5, inplace=False):\n    df = df.copy()\n    sequence_fold = sequence2fold(df[\"sequence_frame\"].tolist())\n    \n    df[\"sequence_fold\"] = sequence_fold\n    df = df[df[\"num_bbox\"] > 0]\n    sequence_fold = df[\"sequence_fold\"]\n    df.drop(columns=\"sequence_fold\", inplace=True)\n    df = df.reset_index(drop=True)\n    \n    group_kfold  = GroupKFold(n_splits=5)\n    df[\"fold\"] = -1\n    train_val_idx_tqdm = tqdm(list(group_kfold.split(df, sequence_fold, groups=df[\"sequence\"])))\n    train_val_idx_tqdm.set_description(\"Fold splitting\")\n    fold = 0\n    for train_idx, val_idx in train_val_idx_tqdm:\n        df.loc[val_idx, 'fold'] = fold\n        fold += 1\n    if inplace:\n        return df\n    return df[\"fold\"].tolist()\n\n\n\ndef starfish2coco(df):\n    pass","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:31:30.432656Z","iopub.execute_input":"2022-02-11T20:31:30.432932Z","iopub.status.idle":"2022-02-11T20:31:30.456036Z","shell.execute_reply.started":"2022-02-11T20:31:30.432897Z","shell.execute_reply":"2022-02-11T20:31:30.455381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = prepare_coco_csv(df)\ndf = kfold_split(df, inplace=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:31:30.457517Z","iopub.execute_input":"2022-02-11T20:31:30.457870Z","iopub.status.idle":"2022-02-11T20:31:32.942892Z","shell.execute_reply.started":"2022-02-11T20:31:30.457820Z","shell.execute_reply":"2022-02-11T20:31:32.942221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SELECTED_FOLD = 4\ntrain_df = df[df[\"fold\"] != SELECTED_FOLD]\nvalid_df = df[df[\"fold\"] == SELECTED_FOLD]\ntrain_df.shape, valid_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:31:32.944534Z","iopub.execute_input":"2022-02-11T20:31:32.945194Z","iopub.status.idle":"2022-02-11T20:31:32.954856Z","shell.execute_reply.started":"2022-02-11T20:31:32.945155Z","shell.execute_reply":"2022-02-11T20:31:32.954016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def copy_file(input_path, destination_path):\n    if not destination_path.parent.exists():\n        destination_path.parent.mkdir(parents=True, exist_ok=True)\n    shutil.copy(input_path, destination_path)\n    if destination_path.exists():\n        return True\n    return False\n\ndataset_dir = Path(\"./dataset/images\")\ntrain_img_path = dataset_dir / \"train_dataset\"\nvalid_img_path = dataset_dir / \"valid_dataset\"\n\nfor idx in tqdm(range(len(df)), desc=\"Copy file\"):\n    row = df.iloc[idx, :]\n    fold = row.fold\n    input_path = Path(row[\"image_path\"])\n    img_name = f\"video_{row.video_id}___frame_{row.video_frame}___fold_{row.fold}___numBB_{row.num_bbox}.jpg\"  # f\"{row['image_id']}.jpg\"\n    destination_dir = train_img_path if fold != SELECTED_FOLD else valid_img_path\n    destination_path = destination_dir / img_name\n    res = copy_file(input_path, destination_path)\n    if not res:\n        print(img_name)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:31:53.148886Z","iopub.execute_input":"2022-02-11T20:31:53.149417Z","iopub.status.idle":"2022-02-11T20:32:52.011183Z","shell.execute_reply.started":"2022-02-11T20:31:53.149381Z","shell.execute_reply":"2022-02-11T20:32:52.010435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Training dataset: {len(list(train_img_path.iterdir()))}\")\nprint(f\"Validation dataset: {len(list(valid_img_path.iterdir()))}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:32:52.012939Z","iopub.execute_input":"2022-02-11T20:32:52.013387Z","iopub.status.idle":"2022-02-11T20:32:52.031622Z","shell.execute_reply.started":"2022-02-11T20:32:52.013343Z","shell.execute_reply":"2022-02-11T20:32:52.030942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert dataset to coco format\n\ndef save_annot_json(json_annotation, filename):\n    with open(filename, 'w') as f:\n        output_json = json.dumps(json_annotation)\n        f.write(output_json)\n        \ndef convert_starfish_to_coco(df):\n    images = []\n    annotations = []\n    obj_count = 0\n    \n    for idx in tqdm(range(len(df))):\n        row = df.iloc[idx, :]\n        img_path = row.image_path\n        width, height = int(row.width), int(row.height)\n        filename = f\"video_{row.video_id}___frame_{row.video_frame}___fold_{row.fold}___numBB_{row.num_bbox}.jpg\"\n        image_data = dict(\n            id=idx,\n            file_name=filename,\n            height=height,\n            width=width)\n        images.append(image_data)\n        \n        row_annotations = row.annotations\n        for annot in row_annotations:\n            x, y, w, h = annot.values()\n            data_anno = dict(\n                image_id=idx,\n                id=obj_count,\n                category_id=0,\n                bbox=[x, y, w, h],\n                area=w*h,\n                segmentation=[],\n                iscrowd=0)\n            annotations.append(data_anno)\n            \n            obj_count += 1\n    \n    info = dict(\n        year=str(dt.datetime.now().year),\n        version=\"1\",\n        description=\"starfish COT dataset - COCO format\",\n        contributor=\"\",\n        url=\"kaggle\",\n        date_created=dt.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\"))\n    \n    licenses = dict(\n        id=1,\n        url=\"\",\n        name=\"Unknown\")\n    \n    categories = [{'id':0, 'name': 'starfish'}]\n    \n    coco_format_json = dict(\n        info=info,\n        licenses=licenses,\n        categories=categories,\n        images=images,\n        annotations=annotations)\n    print(f\"Convert starfish dataset to COCO format | Total bouding boxes {obj_count}\")\n    return coco_format_json","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:32:52.033037Z","iopub.execute_input":"2022-02-11T20:32:52.033476Z","iopub.status.idle":"2022-02-11T20:32:52.045448Z","shell.execute_reply.started":"2022-02-11T20:32:52.033442Z","shell.execute_reply":"2022-02-11T20:32:52.044719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert COTS dataset to JSON COCO\n\ntrain_annot_json = convert_starfish_to_coco(train_df)\nval_annot_json = convert_starfish_to_coco(valid_df)\n\n# Save converted annotations\nannotation_path = dataset_dir / \"annotations\"\nannotation_path.mkdir(parents=True, exist_ok=True)\n\ntrain_json_path = annotation_path / \"train.json\"\nvalid_json_path = annotation_path / \"valid.json\"\nsave_annot_json(train_annot_json, train_json_path)\nsave_annot_json(val_annot_json, valid_json_path)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:32:55.355444Z","iopub.execute_input":"2022-02-11T20:32:55.355869Z","iopub.status.idle":"2022-02-11T20:32:56.954820Z","shell.execute_reply.started":"2022-02-11T20:32:55.355826Z","shell.execute_reply":"2022-02-11T20:32:56.953673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"markdown","source":"### Setup MMDetection","metadata":{}},{"cell_type":"code","source":"%%time\n\n# just following official install instruction.\n# https://github.com/open-mmlab/mmdetection/blob/master/docs/get_started.md\n!pip install openmim\n!mim install mmdet\n!git clone https://github.com/open-mmlab/mmdetection.git\n%cd mmdetection\n!pip install -q -e .","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:34:20.755783Z","iopub.execute_input":"2022-02-11T20:34:20.756074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Setup Pipeline","metadata":{}},{"cell_type":"code","source":"# def cascade_rcnn_r50_fpn_py():\n#     # model settings\n#     model = dict(\n#         type='CascadeRCNN',\n#         backbone=dict(\n#             type='ResNet',\n#             depth=50,\n#             num_stages=4,\n#             out_indices=(0, 1, 2, 3),\n#             frozen_stages=1,\n#             norm_cfg=dict(type='BN', requires_grad=True),\n#             norm_eval=True,\n#             style='pytorch',\n#             init_cfg=dict(type='Pretrained', checkpoint='torchvision://resnet50')),\n#         neck=dict(\n#             type='FPN',\n#             in_channels=[256, 512, 1024, 2048],\n#             out_channels=256,\n#             num_outs=5),\n#         rpn_head=dict(\n#             type='RPNHead',\n#             in_channels=256,\n#             feat_channels=256,\n#             anchor_generator=dict(\n#                 type='AnchorGenerator',\n#                 scales=[8],\n#                 ratios=[0.5, 1.0, 2.0],\n#                 strides=[4, 8, 16, 32, 64]),\n#             bbox_coder=dict(\n#                 type='DeltaXYWHBBoxCoder',\n#                 target_means=[.0, .0, .0, .0],\n#                 target_stds=[1.0, 1.0, 1.0, 1.0]),\n#             loss_cls=dict(\n#                 type='CrossEntropyLoss', use_sigmoid=True, loss_weight=1.0),\n#             loss_bbox=dict(type='SmoothL1Loss', beta=1.0 / 9.0, loss_weight=1.0)),\n#         roi_head=dict(\n#             type='CascadeRoIHead',\n#             num_stages=3,\n#             stage_loss_weights=[1, 0.5, 0.25],\n#             bbox_roi_extractor=dict(\n#                 type='SingleRoIExtractor',\n#                 roi_layer=dict(type='RoIAlign', output_size=7, sampling_ratio=0),\n#                 out_channels=256,\n#                 featmap_strides=[4, 8, 16, 32]),\n#             bbox_head=[\n#                 dict(\n#                     type='Shared2FCBBoxHead',\n#                     in_channels=256,\n#                     fc_out_channels=1024,\n#                     roi_feat_size=7,\n#                     num_classes=1,\n#                     bbox_coder=dict(\n#                         type='DeltaXYWHBBoxCoder',\n#                         target_means=[0., 0., 0., 0.],\n#                         target_stds=[0.1, 0.1, 0.2, 0.2]),\n#                     reg_class_agnostic=True,\n#                     loss_cls=dict(\n#                         type='CrossEntropyLoss',\n#                         use_sigmoid=False,\n#                         loss_weight=1.0),\n#                     loss_bbox=dict(type='SmoothL1Loss', beta=1.0,\n#                                    loss_weight=1.0)),\n#                 dict(\n#                     type='Shared2FCBBoxHead',\n#                     in_channels=256,\n#                     fc_out_channels=1024,\n#                     roi_feat_size=7,\n#                     num_classes=1,\n#                     bbox_coder=dict(\n#                         type='DeltaXYWHBBoxCoder',\n#                         target_means=[0., 0., 0., 0.],\n#                         target_stds=[0.05, 0.05, 0.1, 0.1]),\n#                     reg_class_agnostic=True,\n#                     loss_cls=dict(\n#                         type='CrossEntropyLoss',\n#                         use_sigmoid=False,\n#                         loss_weight=1.0),\n#                     loss_bbox=dict(type='SmoothL1Loss', beta=1.0,\n#                                    loss_weight=1.0)),\n#                 dict(\n#                     type='Shared2FCBBoxHead',\n#                     in_channels=256,\n#                     fc_out_channels=1024,\n#                     roi_feat_size=7,\n#                     num_classes=1,\n#                     bbox_coder=dict(\n#                         type='DeltaXYWHBBoxCoder',\n#                         target_means=[0., 0., 0., 0.],\n#                         target_stds=[0.033, 0.033, 0.067, 0.067]),\n#                     reg_class_agnostic=True,\n#                     loss_cls=dict(\n#                         type='CrossEntropyLoss',\n#                         use_sigmoid=False,\n#                         loss_weight=1.0),\n#                     loss_bbox=dict(type='SmoothL1Loss', beta=1.0, loss_weight=1.0))\n#             ]),\n#         # model training and testing settings\n#         train_cfg=dict(\n#             rpn=dict(\n#                 assigner=dict(\n#                     type='MaxIoUAssigner',\n#                     pos_iou_thr=0.7,\n#                     neg_iou_thr=0.3,\n#                     min_pos_iou=0.3,\n#                     match_low_quality=True,\n#                     ignore_iof_thr=-1),\n#                 sampler=dict(\n#                     type='RandomSampler',\n#                     num=256,\n#                     pos_fraction=0.5,\n#                     neg_pos_ub=-1,\n#                     add_gt_as_proposals=False),\n#                 allowed_border=0,\n#                 pos_weight=-1,\n#                 debug=False),\n#             rpn_proposal=dict(\n#                 nms_pre=2000,\n#                 max_per_img=2000,\n#                 nms=dict(type='nms', iou_threshold=0.7),\n#                 min_bbox_size=0),\n#             rcnn=[\n#                 dict(\n#                     assigner=dict(\n#                         type='MaxIoUAssigner',\n#                         pos_iou_thr=0.5,\n#                         neg_iou_thr=0.5,\n#                         min_pos_iou=0.5,\n#                         match_low_quality=False,\n#                         ignore_iof_thr=-1),\n#                     sampler=dict(\n#                         type='RandomSampler',\n#                         num=512,\n#                         pos_fraction=0.25,\n#                         neg_pos_ub=-1,\n#                         add_gt_as_proposals=True),\n#                     pos_weight=-1,\n#                     debug=False),\n#                 dict(\n#                     assigner=dict(\n#                         type='MaxIoUAssigner',\n#                         pos_iou_thr=0.6,\n#                         neg_iou_thr=0.6,\n#                         min_pos_iou=0.6,\n#                         match_low_quality=False,\n#                         ignore_iof_thr=-1),\n#                     sampler=dict(\n#                         type='RandomSampler',\n#                         num=512,\n#                         pos_fraction=0.25,\n#                         neg_pos_ub=-1,\n#                         add_gt_as_proposals=True),\n#                     pos_weight=-1,\n#                     debug=False),\n#                 dict(\n#                     assigner=dict(\n#                         type='MaxIoUAssigner',\n#                         pos_iou_thr=0.7,\n#                         neg_iou_thr=0.7,\n#                         min_pos_iou=0.7,\n#                         match_low_quality=False,\n#                         ignore_iof_thr=-1),\n#                     sampler=dict(\n#                         type='RandomSampler',\n#                         num=512,\n#                         pos_fraction=0.25,\n#                         neg_pos_ub=-1,\n#                         add_gt_as_proposals=True),\n#                     pos_weight=-1,\n#                     debug=False)\n#             ]),\n#         test_cfg=dict(\n#             rpn=dict(\n#                 nms_pre=1000,\n#                 max_per_img=1000,\n#                 nms=dict(type='nms', iou_threshold=0.7),\n#                 min_bbox_size=0),\n#             rcnn=dict(\n#                 score_thr=0.05,\n#                 nms=dict(type='nms', iou_threshold=0.5),\n#                 max_per_img=100)))\n#     return model\n    \n\n# def coco_detection_py(train_img_path, valid_img_path, train_json_path, valid_json_path):\n#     train_img_path = os.fspath(\"/kaggle/working\" / train_img_path) + \"/\"\n#     valid_img_path = os.fspath(\"/kaggle/working\" / valid_img_path) + \"/\"\n#     train_json_path = os.fspath(\"/kaggle/working\" / train_json_path)\n#     valid_json_path = os.fspath(\"/kaggle/working\" / valid_json_path)\n    \n#     dataset_type = 'CocoDataset'\n#     classes = ('starfish',)\n#     img_norm_cfg = dict(\n#         mean=[123.675, 116.28, 103.53], std=[58.395, 57.12, 57.375], to_rgb=True)\n#     train_pipeline = [\n#         dict(type='LoadImageFromFile'),\n#         dict(type='LoadAnnotations', with_bbox=True),\n#         dict(type='Resize', img_scale=(1280, 720), keep_ratio=True),\n#         dict(type='RandomFlip', flip_ratio=0.5),\n#         dict(type='Normalize', **img_norm_cfg),\n#         dict(type='Pad', size_divisor=32),\n#         dict(type='DefaultFormatBundle'),\n#         dict(type='Collect', keys=['img', 'gt_bboxes', 'gt_labels']),\n#     ]\n#     test_pipeline = [\n#         dict(type='LoadImageFromFile'),\n#         dict(\n#             type='MultiScaleFlipAug',\n#             img_scale=(1280, 720),\n#             flip=False,\n#             transforms=[\n#                 dict(type='Resize', keep_ratio=True),\n#                 dict(type='RandomFlip'),\n#                 dict(type='Normalize', **img_norm_cfg),\n#                 dict(type='Pad', size_divisor=32),\n#                 dict(type='ImageToTensor', keys=['img']),\n#                 dict(type='Collect', keys=['img']),\n#             ])\n#     ]\n#     data = dict(\n#         samples_per_gpu=2,\n#         workers_per_gpu=2,\n#         train=dict(\n#             type=dataset_type,\n#             ann_file=train_json_path,\n#             img_prefix=train_img_path,\n#             classes=classes,\n#             pipeline=train_pipeline),\n#         val=dict(\n#             type=dataset_type,\n#             ann_file=valid_json_path,\n#             img_prefix=valid_img_path,\n#             classes=classes,\n#             pipeline=test_pipeline),\n#         test=dict(\n#             type=dataset_type,\n#             ann_file=valid_json_path,\n#             img_prefix=valid_img_path,\n#             classes=classes,\n#             pipeline=test_pipeline))\n#     evaluation = dict(interval=1, metric='bbox')\n    \n#     coco_detection_data = dict(dataset_type=dataset_type, \n#                                img_norm_cfg=img_norm_cfg, \n#                                train_pipeline=train_pipeline, \n#                                test_pipeline=test_pipeline,\n#                                data=data,\n#                                evaluation=evaluation)\n#     print(\"Created coco_detection_py\")\n#     return coco_detection_data\n\n# def schedules_1x_py():\n#     # optimizer\n#     optimizer = dict(type='SGD', lr=0.02, momentum=0.9, weight_decay=0.0001)\n#     optimizer_config = dict(grad_clip=None)\n#     # learning policy\n#     lr_config = dict(\n#         policy='step',\n#         warmup='linear',\n#         warmup_iters=500,\n#         warmup_ratio=0.001,\n#         step=[8, 11])\n#     runner = dict(type='EpochBasedRunner', max_epochs=12)\n    \n#     schedules_1x_data = dict(optimizer=optimizer,\n#                             optimizer_config=optimizer_config,\n#                             lr_config=lr_config,\n#                             runner=runner)\n#     print(\"Created schedules_1x.py\")\n#     return schedules_1x_data\n\n# def default_runtime_py():\n#     checkpoint_config = dict(interval=1)\n#     # yapf:disable\n#     log_config = dict(\n#         interval=50,\n#         hooks=[\n#             dict(type='TextLoggerHook'),\n# #             dict(type='TensorboardLoggerHook')\n#         ])\n#     # yapf:enable\n#     custom_hooks = [dict(type='NumClassCheckHook')]\n\n#     dist_params = dict(backend='nccl')\n#     log_level = 'INFO'\n#     load_from = 'https://download.openmmlab.com/mmdetection/v2.0/cascade_rcnn/cascade_rcnn_r50_fpn_20e_coco/cascade_rcnn_r50_fpn_20e_coco_bbox_mAP-0.41_20200504_175131-e9872a90.pth'\n#     resume_from = None\n#     workflow = [('train', 1), ('val', 1)]\n\n#     # disable opencv multithreading to avoid system being overloaded\n#     opencv_num_threads = 1\n#     # set multi-process start method as `fork` to speed up the training\n#     mp_start_method = 'fork'\n    \n#     default_runtime_data = dict(checkpoint_config=checkpoint_config,\n#                                 log_config=log_config,\n#                                 custom_hooks=custom_hooks,\n#                                 dist_params=dist_params,\n#                                 log_level=log_level,\n#                                 load_from=load_from,\n#                                 resume_from=resume_from,\n#                                 workflow=workflow,\n#                                 opencv_num_threads=opencv_num_threads,\n#                                 mp_start_method=mp_start_method)\n#     print(\"Created default_runtime.py\")\n#     return default_runtime_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %cd ..\n# !rm -rf mmdetection\n# !git clone https://github.com/open-mmlab/mmdetection.git\n# %cd mmdetection","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def read_json(path):\n#     file_obj = open(path, mode=\"r\")\n#     return json.load(file_obj)\n\n# #cascade_rcnn_r50_fpn.py\n# cascade_rcnn_r50_fpn_data = cascade_rcnn_r50_fpn_py()\n# cascade_rcnn_r50_fpn_path = r\"/kaggle/working/mmdetection/configs/_base_/models/cascade_rcnn_r50_fpn.py\"\n# save_annot_json(cascade_rcnn_r50_fpn_data, cascade_rcnn_r50_fpn_path)\n# display(read_json(cascade_rcnn_r50_fpn_path))\n\n# # coco_detection.py\n# coco_detection_data = coco_detection_py(train_img_path, valid_img_path, train_json_path, valid_json_path)\n# coco_detection_path = r\"/kaggle/working/mmdetection/configs/_base_/datasets/coco_detection.py\"\n# save_annot_json(coco_detection_data, coco_detection_path)\n# display(read_json(coco_detection_path))\n\n# # schedules_1x.py\n# schedules_1x_data = schedules_1x_py()\n# schedules_1x_path = r\"/kaggle/working/mmdetection/configs/_base_/schedules/schedule_1x.py\"\n# save_annot_json(schedules_1x_data, schedules_1x_path)\n# display(read_json(schedules_1x_path))\n\n# # default_runtime.py\n# default_runtime_data = default_runtime_py()\n# default_runtime_path = r\"/kaggle/working/mmdetection/configs/_base_/default_runtime.py\"\n# save_annot_json(default_runtime_data, default_runtime_path)\n# display(read_json(default_runtime_path))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile 'configs/cascade_rcnn/cascade_rcnn_r50_fpn_20e_coco.py'\n\n# model settings\nmodel = dict(\n    type='CascadeRCNN',\n    backbone=dict(\n        type='ResNet',\n        depth=50,\n        num_stages=4,\n        out_indices=(0, 1, 2, 3),\n        frozen_stages=1,\n        norm_cfg=dict(type='BN', requires_grad=True),\n        norm_eval=True,\n        style='pytorch',\n        init_cfg=dict(type='Pretrained', checkpoint='torchvision://resnet50')),\n    neck=dict(\n        type='FPN',\n        in_channels=[256, 512, 1024, 2048],\n        out_channels=256,\n        num_outs=5),\n    rpn_head=dict(\n        type='RPNHead',\n        in_channels=256,\n        feat_channels=256,\n        anchor_generator=dict(\n            type='AnchorGenerator',\n            scales=[8],\n            ratios=[0.5, 1.0, 2.0],\n            strides=[4, 8, 16, 32, 64]),\n        bbox_coder=dict(\n            type='DeltaXYWHBBoxCoder',\n            target_means=[.0, .0, .0, .0],\n            target_stds=[1.0, 1.0, 1.0, 1.0]),\n        loss_cls=dict(\n            type='CrossEntropyLoss', use_sigmoid=True, loss_weight=1.0),\n        loss_bbox=dict(type='SmoothL1Loss', beta=1.0 / 9.0, loss_weight=1.0)),\n    roi_head=dict(\n        type='CascadeRoIHead',\n        num_stages=3,\n        stage_loss_weights=[1, 0.5, 0.25],\n        bbox_roi_extractor=dict(\n            type='SingleRoIExtractor',\n            roi_layer=dict(type='RoIAlign', output_size=7, sampling_ratio=0),\n            out_channels=256,\n            featmap_strides=[4, 8, 16, 32]),\n        bbox_head=[\n            dict(\n                type='Shared2FCBBoxHead',\n                in_channels=256,\n                fc_out_channels=1024,\n                roi_feat_size=7,\n                num_classes=1,\n                bbox_coder=dict(\n                    type='DeltaXYWHBBoxCoder',\n                    target_means=[0., 0., 0., 0.],\n                    target_stds=[0.1, 0.1, 0.2, 0.2]),\n                reg_class_agnostic=True,\n                loss_cls=dict(\n                    type='CrossEntropyLoss',\n                    use_sigmoid=False,\n                    loss_weight=1.0),\n                loss_bbox=dict(type='SmoothL1Loss', beta=1.0,\n                               loss_weight=1.0)),\n            dict(\n                type='Shared2FCBBoxHead',\n                in_channels=256,\n                fc_out_channels=1024,\n                roi_feat_size=7,\n                num_classes=1,\n                bbox_coder=dict(\n                    type='DeltaXYWHBBoxCoder',\n                    target_means=[0., 0., 0., 0.],\n                    target_stds=[0.05, 0.05, 0.1, 0.1]),\n                reg_class_agnostic=True,\n                loss_cls=dict(\n                    type='CrossEntropyLoss',\n                    use_sigmoid=False,\n                    loss_weight=1.0),\n                loss_bbox=dict(type='SmoothL1Loss', beta=1.0,\n                               loss_weight=1.0)),\n            dict(\n                type='Shared2FCBBoxHead',\n                in_channels=256,\n                fc_out_channels=1024,\n                roi_feat_size=7,\n                num_classes=1,\n                bbox_coder=dict(\n                    type='DeltaXYWHBBoxCoder',\n                    target_means=[0., 0., 0., 0.],\n                    target_stds=[0.033, 0.033, 0.067, 0.067]),\n                reg_class_agnostic=True,\n                loss_cls=dict(\n                    type='CrossEntropyLoss',\n                    use_sigmoid=False,\n                    loss_weight=1.0),\n                loss_bbox=dict(type='SmoothL1Loss', beta=1.0, loss_weight=1.0))\n        ]),\n    # model training and testing settings\n    train_cfg=dict(\n        rpn=dict(\n            assigner=dict(\n                type='MaxIoUAssigner',\n                pos_iou_thr=0.7,\n                neg_iou_thr=0.3,\n                min_pos_iou=0.3,\n                match_low_quality=True,\n                ignore_iof_thr=-1),\n            sampler=dict(\n                type='RandomSampler',\n                num=256,\n                pos_fraction=0.5,\n                neg_pos_ub=-1,\n                add_gt_as_proposals=False),\n            allowed_border=0,\n            pos_weight=-1,\n            debug=False),\n        rpn_proposal=dict(\n            nms_pre=2000,\n            max_per_img=2000,\n            nms=dict(type='nms', iou_threshold=0.7),\n            min_bbox_size=0),\n        rcnn=[\n            dict(\n                assigner=dict(\n                    type='MaxIoUAssigner',\n                    pos_iou_thr=0.5,\n                    neg_iou_thr=0.5,\n                    min_pos_iou=0.5,\n                    match_low_quality=False,\n                    ignore_iof_thr=-1),\n                sampler=dict(\n                    type='RandomSampler',\n                    num=512,\n                    pos_fraction=0.25,\n                    neg_pos_ub=-1,\n                    add_gt_as_proposals=True),\n                pos_weight=-1,\n                debug=False),\n            dict(\n                assigner=dict(\n                    type='MaxIoUAssigner',\n                    pos_iou_thr=0.6,\n                    neg_iou_thr=0.6,\n                    min_pos_iou=0.6,\n                    match_low_quality=False,\n                    ignore_iof_thr=-1),\n                sampler=dict(\n                    type='RandomSampler',\n                    num=512,\n                    pos_fraction=0.25,\n                    neg_pos_ub=-1,\n                    add_gt_as_proposals=True),\n                pos_weight=-1,\n                debug=False),\n            dict(\n                assigner=dict(\n                    type='MaxIoUAssigner',\n                    pos_iou_thr=0.7,\n                    neg_iou_thr=0.7,\n                    min_pos_iou=0.7,\n                    match_low_quality=False,\n                    ignore_iof_thr=-1),\n                sampler=dict(\n                    type='RandomSampler',\n                    num=512,\n                    pos_fraction=0.25,\n                    neg_pos_ub=-1,\n                    add_gt_as_proposals=True),\n                pos_weight=-1,\n                debug=False)\n        ]),\n    test_cfg=dict(\n        rpn=dict(\n            nms_pre=1000,\n            max_per_img=1000,\n            nms=dict(type='nms', iou_threshold=0.7),\n            min_bbox_size=0),\n        rcnn=dict(\n            score_thr=0.05,\n            nms=dict(type='nms', iou_threshold=0.5),\n            max_per_img=100)))\n\n# dataset settings\ndataset_type = 'CocoDataset'  # Modify\ndata_root = '/kaggle/working/'\nclasses = ('starfish', )  # Modify\nimg_norm_cfg = dict(\n    mean=[123.675, 116.28, 103.53], std=[58.395, 57.12, 57.375], to_rgb=True)\ntrain_pipeline = [\n    dict(type='LoadImageFromFile'),\n    dict(type='LoadAnnotations', with_bbox=True),\n    dict(type='Resize', img_scale=(1280, 720), keep_ratio=True),\n    dict(type='RandomFlip', flip_ratio=0.5),\n    dict(type='Normalize', **img_norm_cfg),\n    dict(type='Pad', size_divisor=32),\n    dict(type='DefaultFormatBundle'),\n    dict(type='Collect', keys=['img', 'gt_bboxes', 'gt_labels']),\n]\ntest_pipeline = [\n    dict(type='LoadImageFromFile'),\n    dict(\n        type='MultiScaleFlipAug',\n        img_scale=(1280, 720),\n        flip=False,\n        transforms=[\n            dict(type='Resize', keep_ratio=True),\n            dict(type='RandomFlip'),\n            dict(type='Normalize', **img_norm_cfg),\n            dict(type='Pad', size_divisor=32),\n            dict(type='ImageToTensor', keys=['img']),\n            dict(type='Collect', keys=['img']),\n        ])\n]\ndata = dict(\n    samples_per_gpu=2,\n    workers_per_gpu=2,\n    train=dict(\n        type=dataset_type,\n        ann_file=data_root + 'dataset/images/annotations/train.json',  # Modify\n        img_prefix=data_root + 'dataset/images/train_dataset/',  # Modify\n        classes=classes,\n        pipeline=train_pipeline),\n    val=dict(\n        type=dataset_type,\n        ann_file=data_root + 'dataset/images/annotations/valid.json',  # Modify\n        img_prefix=data_root + 'dataset/images/valid_dataset/',  # Modify\n        classes=classes,\n        pipeline=test_pipeline),\n    test=dict(\n        type=dataset_type,\n        ann_file=data_root + 'dataset/images/annotations/valid.json',  # Modify\n        img_prefix=data_root + 'dataset/images/valid_dataset/',  # Modify\n        classes=classes,\n        pipeline=test_pipeline))\nevaluation = dict(interval=1, metric='bbox')\n\n# optimizer\noptimizer = dict(type='SGD', lr=0.002, momentum=0.9, weight_decay=0.0001)\noptimizer_config = dict(grad_clip=None)\n# learning policy\nlr_config = dict(\n    policy='step',\n    warmup='linear',\n    warmup_iters=500,\n    warmup_ratio=0.001,\n    step=[8, 11])\nrunner = dict(type='EpochBasedRunner', max_epochs=12)\n\ncheckpoint_config = dict(interval=1)\n# yapf:disable\nlog_config = dict(\n    interval=50,\n    hooks=[\n        dict(type='TextLoggerHook'),\n        dict(type='TensorboardLoggerHook')\n    ])\n# yapf:enable\ncustom_hooks = [dict(type='NumClassCheckHook')]\n\ndist_params = dict(backend='nccl')\nlog_level = 'INFO'\nload_from = None\nresume_from = 'https://download.openmmlab.com/mmdetection/v2.0/cascade_rcnn/cascade_rcnn_r50_fpn_20e_coco/cascade_rcnn_r50_fpn_20e_coco_bbox_mAP-0.41_20200504_175131-e9872a90.pth'  # Modify\nworkflow = [('train', 1)]\n\n# disable opencv multithreading to avoid system being overloaded\nopencv_num_threads = 1\n# set multi-process start method as `fork` to speed up the training\nmp_start_method = 'fork'","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:34:04.871545Z","iopub.execute_input":"2022-02-11T20:34:04.871845Z","iopub.status.idle":"2022-02-11T20:34:04.902860Z","shell.execute_reply.started":"2022-02-11T20:34:04.871808Z","shell.execute_reply":"2022-02-11T20:34:04.902237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training is just one line. \n# cascade_rcnn_cfg_path = '/kaggle/working/mmdetection/configs/cascade_rcnn/cascade_rcnn_r50_fpn_20e_coco.py'\n!python tools/train.py configs/cascade_rcnn/cascade_rcnn_r50_fpn_20e_coco.py\n\n# You can also test the model like this.\n# checkpoint_path = r\"/kaggle/working/models/custom_faster_rcnn_r50_fpn/latest.pth\"\n# !python tools/test.py configs/faster_rcnn/custom_faster_rcnn_r50_fpn.py {checkpoint_path} --eval bbox","metadata":{"execution":{"iopub.status.busy":"2022-02-11T20:34:04.906298Z","iopub.execute_input":"2022-02-11T20:34:04.907765Z","iopub.status.idle":"2022-02-11T20:34:05.793418Z","shell.execute_reply.started":"2022-02-11T20:34:04.907729Z","shell.execute_reply":"2022-02-11T20:34:05.792602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy model and cofig to certain directory\n%rm -rf /kaggle/working/mmdetection\n%rm -rf /kaggle/working/dataset\n!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}