{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch, torchvision\nprint(torch.__version__, torch.cuda.is_available())","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:11:31.366812Z","iopub.execute_input":"2022-01-04T03:11:31.36724Z","iopub.status.idle":"2022-01-04T03:11:33.200529Z","shell.execute_reply.started":"2022-01-04T03:11:31.367148Z","shell.execute_reply":"2022-01-04T03:11:33.199775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m pip install 'git+https://github.com/facebookresearch/detectron2.git'","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:11:35.227388Z","iopub.execute_input":"2022-01-04T03:11:35.227988Z","iopub.status.idle":"2022-01-04T03:14:35.981168Z","shell.execute_reply.started":"2022-01-04T03:11:35.227951Z","shell.execute_reply":"2022-01-04T03:14:35.98032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initialize Templates","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport pandas as pd\nimport numpy as np\nimport pandas as pd \n\nfrom tqdm import tqdm\nfrom tqdm import tqdm_notebook as tqdm # progress bar\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nfrom datetime import datetime\nimport time\nimport matplotlib.pyplot as plt\nfrom PIL import ImageDraw\nfrom PIL import Image\nimport os, json, cv2, random\nimport skimage.io as io\nimport copy\nfrom pathlib import Path\nfrom typing import Optional\nimport json\nimport matplotlib.pyplot as plt\n\nimport ast\n\n\nfrom tqdm import tqdm\nimport itertools\n\nimport torch\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\nfrom glob import glob\nimport numba\nfrom numba import jit\n\nfrom pycocotools.coco import COCO\n# detectron2\nfrom detectron2.structures import BoxMode\nfrom detectron2 import model_zoo\nfrom detectron2.config import get_cfg\nfrom detectron2.data import DatasetCatalog, MetadataCatalog\nfrom detectron2.engine import DefaultPredictor, DefaultTrainer, launch\nfrom detectron2.evaluation import COCOEvaluator\nfrom detectron2.structures import BoxMode\nfrom detectron2.utils.visualizer import ColorMode\nfrom detectron2.utils.logger import setup_logger\nfrom detectron2.utils.visualizer import Visualizer\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\n\n\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader\nfrom detectron2.data import detection_utils as utils\nimport detectron2.data.transforms as T\nfrom detectron2.evaluation import COCOEvaluator, inference_on_dataset\n\n\n\nfrom detectron2.evaluation.evaluator import DatasetEvaluator\nimport pycocotools.mask as mask_util\nfrom detectron2.engine import BestCheckpointer\nfrom detectron2.checkpoint import DetectionCheckpointer\n\nsetup_logger()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-04T03:31:46.838188Z","iopub.execute_input":"2022-01-04T03:31:46.838633Z","iopub.status.idle":"2022-01-04T03:31:46.860276Z","shell.execute_reply.started":"2022-01-04T03:31:46.838594Z","shell.execute_reply":"2022-01-04T03:31:46.859456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_templates():\n    coco_json_template = {\n        \"info\":{},\n        \"images\":[],\n        \"licenses\":[],\n        \"annotations\":[],\n        \"categories\":[]\n    }\n    image_template = {\n        'license': 0, \n        'file_name': '',\n        'coco_url': None,\n        'height': None,\n        'width': None,\n        'date_captured': None,\n        'flickr_url': None,\n        'id': None\n    }\n    annotation_template = {\n        'segmentation':[[]],\n        'area':None,\n        'iscrowd':0,\n        'image_id':None,\n        'bbox':[],\n        'category_id':None,\n        'id':None\n    }\n    category_template = {\n        'supercategory': 'Coral_creatures', \n        'id': None, \n        'name': ''\n    }\n    return coco_json_template,image_template, annotation_template, category_template","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:30:43.757004Z","iopub.execute_input":"2022-01-04T03:30:43.757726Z","iopub.status.idle":"2022-01-04T03:30:43.76444Z","shell.execute_reply.started":"2022-01-04T03:30:43.757691Z","shell.execute_reply":"2022-01-04T03:30:43.763624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bbox(annots):\n    bboxes = [list(annot.values()) for annot in annots]\n    return bboxes\ndef get_path(row):\n    row['image_path'] = f'{TRAIN_PATH}/train_images/video_{row.video_id}/{row.video_frame}.jpg'\n    return row\ndef load_image(image_path):\n    return cv2.cvtColor(cv2.imread(image_path), cv2.COLOR_BGR2RGB)\ndef to_json(df, base_dir, draw=False):\n    coco_json, image_template, annotation_template, category_template = get_templates()\n\n    category_ = copy.deepcopy(category_template)\n    category_[\"id\"] = 1\n    category_[\"name\"] =\"starfish\"\n    coco_json['categories'].append(category_)\n\n    annot_id = 1\n    image_id = 1\n    with tqdm(total=len(df)) as pbar:\n        for i, row in df.iterrows():\n            # get instances of annotations\n            image_prop = copy.deepcopy(image_template)\n\n            # get image properties\n            image_name = os.path.join(f\"video_{row['video_id']}\", f\"{row['video_frame']}.jpg\")\n\n            img_ = Image.open(os.path.join(base_dir, image_name))\n            img_width, img_height = img_.size\n\n            image_prop['file_name'] = image_name\n            image_prop['height'] = img_height\n            image_prop['width'] = img_width\n            image_prop['id'] = image_id\n            image_id+=1\n\n            # append the image\n            coco_json['images'].append(image_prop)\n\n            annotations = eval(row[\"annotations\"])\n            for annotation in annotations:\n                image_annot = copy.deepcopy(annotation_template)\n                bbox = [\n                    annotation[\"x\"],\n                    annotation[\"y\"],\n                    annotation[\"width\"],\n                    annotation[\"height\"]\n                ]\n\n                if draw:\n                    draw_handle = ImageDraw.Draw(img_)\n                    draw_handle.rectangle([(int(bbox[0]),int(bbox[1])),(int(bbox[2])+int(bbox[0]),int(bbox[3])+int(bbox[1]))],\n                                         width = 5)\n                    if not os.path.exists(\"output\"):\n                        os.mkdir(\"output\")\n\n                # populate the template\n                image_annot['segmentation'] = []\n                image_annot[\"area\"] = bbox[2]*bbox[3]\n                image_annot['image_id'] = image_id\n                image_annot['bbox'] = bbox\n                image_annot['category_id'] = 1\n                image_annot['id'] = annot_id\n                annot_id+=1\n\n                # append the annotations\n                coco_json['annotations'].append(image_annot)\n\n            pbar.update(1)\n            \n    return coco_json","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:27:11.257354Z","iopub.execute_input":"2022-01-04T03:27:11.25762Z","iopub.status.idle":"2022-01-04T03:27:11.272237Z","shell.execute_reply.started":"2022-01-04T03:27:11.25759Z","shell.execute_reply":"2022-01-04T03:27:11.271264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the data and convert to json","metadata":{}},{"cell_type":"code","source":"# --- Read data ---\nTRAIN_PATH = '/kaggle/input/tensorflow-great-barrier-reef'\n# Read in the data CSV files\ndf = pd.read_csv(\"/kaggle/input/tensorflow-great-barrier-reef/train.csv\")\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:00.10691Z","iopub.execute_input":"2022-01-04T03:34:00.107527Z","iopub.status.idle":"2022-01-04T03:34:00.14637Z","shell.execute_reply.started":"2022-01-04T03:34:00.107492Z","shell.execute_reply":"2022-01-04T03:34:00.145712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"NumBBox\"]=df['annotations'].apply(lambda x: str.count(x, 'x'))\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:06.215787Z","iopub.execute_input":"2022-01-04T03:34:06.216061Z","iopub.status.idle":"2022-01-04T03:34:06.245084Z","shell.execute_reply.started":"2022-01-04T03:34:06.216016Z","shell.execute_reply":"2022-01-04T03:34:06.24443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df[df[\"NumBBox\"]>0]\ndf_train.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:10.63691Z","iopub.execute_input":"2022-01-04T03:34:10.637608Z","iopub.status.idle":"2022-01-04T03:34:10.653389Z","shell.execute_reply.started":"2022-01-04T03:34:10.637572Z","shell.execute_reply":"2022-01-04T03:34:10.652607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.progress_apply(get_path, axis=1)\ndf_train.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:28.716606Z","iopub.execute_input":"2022-01-04T03:34:28.71687Z","iopub.status.idle":"2022-01-04T03:34:31.67362Z","shell.execute_reply.started":"2022-01-04T03:34:28.716842Z","shell.execute_reply":"2022-01-04T03:34:31.672858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_spl=5\nSelected_Fold=2 #0..2\n\nfrom sklearn.model_selection import GroupKFold\ngkf  = GroupKFold(n_splits = n_spl) # num_folds=3 as there are total 3 videos\ndf_train = df_train.reset_index(drop=True)\ndf_train['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df_train, y = df_train.video_id.tolist(), groups=df_train.sequence)):\n    df_train.loc[val_idx, 'fold'] = fold\ndisplay(df_train.fold.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:44.286432Z","iopub.execute_input":"2022-01-04T03:34:44.286674Z","iopub.status.idle":"2022-01-04T03:34:44.302588Z","shell.execute_reply.started":"2022-01-04T03:34:44.286647Z","shell.execute_reply":"2022-01-04T03:34:44.301903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df_train[df_train.fold != Selected_Fold]\nvalid_df = df_train[df_train.fold == Selected_Fold]\ntrain_df.sample(5)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:34:54.275795Z","iopub.execute_input":"2022-01-04T03:34:54.276059Z","iopub.status.idle":"2022-01-04T03:34:54.292666Z","shell.execute_reply.started":"2022-01-04T03:34:54.276015Z","shell.execute_reply":"2022-01-04T03:34:54.291972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_BASE_DIR = \"/kaggle/input/tensorflow-great-barrier-reef/train_images\"\n\ntrain_json = to_json(train_df, IMG_BASE_DIR)\nwith open(\"train.json\", \"w\") as file:\n    json.dump(train_json, file)\n    \ntest_json = to_json(valid_df, IMG_BASE_DIR)\nwith open(\"test.json\", \"w\") as file:\n    json.dump(test_json, file)","metadata":{"execution":{"iopub.status.busy":"2022-01-04T03:37:26.957896Z","iopub.execute_input":"2022-01-04T03:37:26.958305Z","iopub.status.idle":"2022-01-04T03:37:31.322291Z","shell.execute_reply.started":"2022-01-04T03:37:26.958274Z","shell.execute_reply":"2022-01-04T03:37:31.321578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# register datasets","metadata":{}},{"cell_type":"code","source":"from detectron2.data.datasets import register_coco_instances\nfrom detectron2.data import MetadataCatalog, DatasetCatalog\nfrom detectron2.data import detection_utils as utils\nfrom detectron2.utils.visualizer import Visualizer\nimport matplotlib.pyplot as plt\nimport random\n\ntry:\n    register_coco_instances(\"Coral_starfish_train\", {}, \"train.json\", IMG_BASE_DIR)\n    register_coco_instances(\"Coral_starfish_test\", {}, \"test.json\", IMG_BASE_DIR)\nexcept AssertionError:\n    print(\"dataset already created\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T14:15:19.289255Z","iopub.execute_input":"2021-12-31T14:15:19.289656Z","iopub.status.idle":"2021-12-31T14:15:19.29778Z","shell.execute_reply.started":"2021-12-31T14:15:19.289608Z","shell.execute_reply":"2021-12-31T14:15:19.296529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# visualize train Set","metadata":{}},{"cell_type":"code","source":"with open(IMG_BASE_DIR + \"/test.json\", \"r\") as file:\n    json.dump(test_json, file)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T04:16:07.925409Z","iopub.execute_input":"2021-12-31T04:16:07.925785Z","iopub.status.idle":"2021-12-31T04:16:07.951342Z","shell.execute_reply.started":"2021-12-31T04:16:07.92572Z","shell.execute_reply":"2021-12-31T04:16:07.950035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = 5\ndamage_metadata = MetadataCatalog.get(\"Coral_starfish_train\")\ndataset_dicts = DatasetCatalog.get(\"Coral_starfish_train\")\nimages_with_annot = [d for d in dataset_dicts if len(d[\"annotations\"])!=0]\nprint(f\"images with atleast one annotation in train Set :- {len(images_with_annot)}\")\nfor d in random.sample(images_with_annot, n):\n    print(d)\n    # Draw ground Truths\n    img = cv2.imread(d[\"file_name\"])\n    visualizer = Visualizer(img[:, :, ::-1], metadata=damage_metadata, scale=1)\n    vis = visualizer.draw_dataset_dict(d)\n    gt_image = vis.get_image()\n\n    plt.figure(figsize=(16,9))\n    plt.imshow(gt_image)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T14:15:32.169229Z","iopub.execute_input":"2021-12-31T14:15:32.169552Z","iopub.status.idle":"2021-12-31T14:15:36.289116Z","shell.execute_reply.started":"2021-12-31T14:15:32.169521Z","shell.execute_reply":"2021-12-31T14:15:36.288195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# visualize test Set","metadata":{}},{"cell_type":"code","source":"n = 5\ndamage_metadata = MetadataCatalog.get(\"Coral_starfish_test\")\ndataset_dicts = DatasetCatalog.get(\"Coral_starfish_test\")\nimages_with_annot = [d for d in dataset_dicts if len(d[\"annotations\"])!=0]\nprint(f\"images with atleast one annotation in test Set :- {len(images_with_annot)}\")\nfor d in random.sample(images_with_annot, n):\n    print(d)\n    # Draw ground Truths\n    img = cv2.imread(d[\"file_name\"])\n    visualizer = Visualizer(img[:, :, ::-1], metadata=damage_metadata, scale=1)\n    vis = visualizer.draw_dataset_dict(d)\n    gt_image = vis.get_image()\n\n    plt.figure(figsize=(16,9))\n    plt.imshow(gt_image)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T14:17:46.348158Z","iopub.execute_input":"2021-12-31T14:17:46.348445Z","iopub.status.idle":"2021-12-31T14:17:50.437528Z","shell.execute_reply.started":"2021-12-31T14:17:46.348415Z","shell.execute_reply":"2021-12-31T14:17:50.435895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training**","metadata":{}},{"cell_type":"markdown","source":"**Train**","metadata":{}}]}