{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!python -m pip install pyyaml==5.1\nimport sys, os, distutils.core\n# Note: This is a faster way to install detectron2 in Colab, but it does not include all functionalities (e.g. compiled operators).\n# See https://detectron2.readthedocs.io/tutorials/install.html for full installation instructions\n!git clone 'https://github.com/facebookresearch/detectron2'\ndist = distutils.core.run_setup(\"./detectron2/setup.py\")\n!python -m pip install {' '.join([f\"'{x}'\" for x in dist.install_requires])}\nsys.path.insert(0, os.path.abspath('./detectron2'))\n\n# Properly install detectron2. (Please do not install twice in both ways)\n# !python -m pip install 'git+https://github.com/facebookresearch/detectron2.git'\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-05T08:02:36.377288Z","iopub.execute_input":"2023-08-05T08:02:36.377825Z","iopub.status.idle":"2023-08-05T08:03:45.680321Z","shell.execute_reply.started":"2023-08-05T08:02:36.377789Z","shell.execute_reply":"2023-08-05T08:03:45.679026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch, detectron2\n!nvcc --version\nTORCH_VERSION = \".\".join(torch.__version__.split(\".\")[:2])\nCUDA_VERSION = torch.__version__.split(\"+\")[-1]\nprint(\"torch: \", TORCH_VERSION, \"; cuda: \", CUDA_VERSION)\nprint(\"detectron2:\", detectron2.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:18.976605Z","iopub.execute_input":"2023-08-05T15:12:18.977215Z","iopub.status.idle":"2023-08-05T15:12:20.288037Z","shell.execute_reply.started":"2023-08-05T15:12:18.977169Z","shell.execute_reply":"2023-08-05T15:12:20.286749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some basic setup:\n# Setup detectron2 logger\nimport detectron2\nfrom detectron2.utils.logger import setup_logger\nsetup_logger()\n\n# import some common libraries\nimport numpy as np\nimport os, json, cv2, random\n#from google.colab.patches import cv2_imshow\nimport cv2\nfrom matplotlib import pyplot as plt\n# import some common detectron2 utilities\nfrom detectron2 import model_zoo\nfrom detectron2.engine import DefaultPredictor\nfrom detectron2.config import get_cfg\nfrom detectron2.utils.visualizer import Visualizer\nfrom detectron2.data import MetadataCatalog, DatasetCatalog","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:21.724896Z","iopub.execute_input":"2023-08-05T15:12:21.726171Z","iopub.status.idle":"2023-08-05T15:12:21.734338Z","shell.execute_reply.started":"2023-08-05T15:12:21.726117Z","shell.execute_reply":"2023-08-05T15:12:21.732772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n# detectron2\nfrom detectron2.utils.memory import retry_if_cuda_oom\nfrom detectron2.utils.logger import setup_logger\nfrom detectron2.checkpoint import DetectionCheckpointer\nfrom detectron2.modeling import build_model\nfrom detectron2.evaluation import COCOEvaluator, inference_on_dataset\nimport detectron2.data.transforms as T\nfrom detectron2.data import detection_utils as utils\nfrom detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader, DatasetMapper\nfrom detectron2.utils.visualizer import Visualizer\nfrom detectron2.structures import BoxMode\nfrom detectron2.engine import DefaultPredictor, DefaultTrainer\nfrom detectron2.config import get_cfg\nfrom detectron2 import model_zoo\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm  # progress bar\nimport matplotlib.pyplot as plt\nimport json\nimport cv2\nimport copy\nfrom typing import Optional\nimport seaborn as sns\n\n!pip install -q pycocotools\nfrom pycocotools.coco import COCO\nfrom PIL import Image\nimport random\nfrom pathlib import Path\n%matplotlib inline\nsns.set_theme(style='darkgrid', palette='deep', font='sans-serif', font_scale=1)\n\nfrom IPython.display import FileLink\n# torch\nimport torch\nimport gc\nimport warnings\n# Ignore \"future\" warnings and Data-Frame-Slicing warnings.\nwarnings.filterwarnings('ignore')\n\nsetup_logger()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:22.427571Z","iopub.execute_input":"2023-08-05T15:12:22.429310Z","iopub.status.idle":"2023-08-05T15:12:36.211478Z","shell.execute_reply.started":"2023-08-05T15:12:22.429257Z","shell.execute_reply":"2023-08-05T15:12:36.210123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\nTRAIN_IMG_DIR = Path(\"/kaggle/input/dlsprint2/badlad/images/train\")\n\nTRAIN_COCO_PATH = Path(\"/kaggle/input/dlsprint2/badlad/labels/coco_format/train/badlad-train-coco.json\")\n\nTEST_IMG_DIR = Path(\"/kaggle/input/dlsprint2/badlad/images/test\")\n\nTEST_METADATA_PATH = Path(\"/kaggle/input/dlsprint2/badlad/badlad-test-metadata.json\")\n\n# Training output directory\nOUTPUT_DIR = Path(\"./output\")\nOUTPUT_MODEL = OUTPUT_DIR/\"model_final.pth\"\n\n\n# Path to your pretrained model weights \nPRETRAINED_PATH = Path(\"/kaggle/input/pretrainweight/model_final (7).pth\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:36.214301Z","iopub.execute_input":"2023-08-05T15:12:36.215076Z","iopub.status.idle":"2023-08-05T15:12:36.221839Z","shell.execute_reply.started":"2023-08-05T15:12:36.215033Z","shell.execute_reply":"2023-08-05T15:12:36.220532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with TRAIN_COCO_PATH.open() as f:\n    train_dict = json.load(f)\n\nwith TEST_METADATA_PATH.open() as f:\n    test_dict = json.load(f)\n    \ntrain_coco_labels=COCO(annotation_file=TRAIN_COCO_PATH)\n\nprint(\"#### LABELS AND METADATA LOADED ####\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:36.224076Z","iopub.execute_input":"2023-08-05T15:12:36.224598Z","iopub.status.idle":"2023-08-05T15:12:52.925740Z","shell.execute_reply.started":"2023-08-05T15:12:36.224554Z","shell.execute_reply":"2023-08-05T15:12:52.924433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\n# if False, model is set to `PRETRAINED_PATH` model\nis_train = True\n\n# if True, evaluate on validation dataset\nis_evaluate = True\n\n# if True, run inference on test dataset\nis_inference = False\n\n# if True and `is_train` == True, `PRETRAINED_PATH` model is trained further\nis_resume_training = True\n\n# Perform augmentation\nis_augment = False\n\nSEED = int(datetime.now().timestamp())\n\n# Model path based on Decisions\nMODEL_PATH = OUTPUT_MODEL if is_train else PRETRAINED_PATH","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:52.928947Z","iopub.execute_input":"2023-08-05T15:12:52.929447Z","iopub.status.idle":"2023-08-05T15:12:52.936945Z","shell.execute_reply.started":"2023-08-05T15:12:52.929405Z","shell.execute_reply":"2023-08-05T15:12:52.935589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are \" + str(len(train_dict['categories'])) + \" categories.\\n\")\nprint(\"There are \" + str(len(test_dict['images']) + len(train_dict['images'])) + \" images in the dataset.\")\nprint(\"There are \" + str(len(train_dict['images'])) + \" images in the train set.\")\nprint(\"There are \" + str(len(test_dict['images'])) + \" images in the test set.\\n\")\nprint(\"There are \" + str(len(train_dict['annotations'])) + \" annotations in the train set.\\n\")\n\nprint(\"We will focus on mainly categories, images and annotations.\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:52.938434Z","iopub.execute_input":"2023-08-05T15:12:52.940058Z","iopub.status.idle":"2023-08-05T15:12:52.951494Z","shell.execute_reply.started":"2023-08-05T15:12:52.940019Z","shell.execute_reply":"2023-08-05T15:12:52.950267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def organize_coco_data(data_dict: dict) -> tuple[list[str], list[dict], list[dict]]:\n    thing_classes: list[str] = []\n\n    # Map Category Names to IDs\n    for cat in data_dict['categories']:\n        thing_classes.append(cat['name'])\n\n    # Images\n    images_metadata: list[dict] = data_dict['images']\n\n    # Convert COCO annotations to detectron2 annotations format\n    data_annotations = []\n    for ann in data_dict['annotations']:\n        # coco format -> detectron2 format\n        annot_obj = {\n            # Annotation ID\n            \"id\": ann['id'],\n\n            # Segmentation Polygon (x, y) coordinnates\n            \"gt_masks\": ann['segmentation'],\n\n            # Image ID for this annotation (Which image does this annotation belong to?)\n            \"image_id\": ann['image_id'],\n\n            # Category Label (0: paragraph, 1: text box, 2: image, 3: table)\n            \"category_id\": ann['category_id'],\n\n            \"x_min\": ann['bbox'][0],  # left\n            \"y_min\": ann['bbox'][1],  # top\n            \"x_max\": ann['bbox'][0] + ann['bbox'][2],  # left+width\n            \"y_max\": ann['bbox'][1] + ann['bbox'][3]  # top+height\n        }\n        data_annotations.append(annot_obj)\n\n    return thing_classes, images_metadata, data_annotations","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:52.953057Z","iopub.execute_input":"2023-08-05T15:12:52.953515Z","iopub.status.idle":"2023-08-05T15:12:52.983980Z","shell.execute_reply.started":"2023-08-05T15:12:52.953477Z","shell.execute_reply":"2023-08-05T15:12:52.983029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thing_classes, images_metadata, data_annotations = organize_coco_data(train_dict)\n\nthing_classes_test, images_metadata_test, _ = organize_coco_data(test_dict)\n\nprint(thing_classes)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:52.985679Z","iopub.execute_input":"2023-08-05T15:12:52.986068Z","iopub.status.idle":"2023-08-05T15:12:54.033619Z","shell.execute_reply.started":"2023-08-05T15:12:52.986034Z","shell.execute_reply":"2023-08-05T15:12:54.032513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.DataFrame(images_metadata)\ntrain_metadata = train_metadata[['id', 'file_name', 'width', 'height']]\ntrain_metadata = train_metadata.rename(columns={\"id\": \"image_id\"})\nprint(\"train_metadata size=\", len(train_metadata))\ntrain_metadata.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:54.035199Z","iopub.execute_input":"2023-08-05T15:12:54.036328Z","iopub.status.idle":"2023-08-05T15:12:54.113579Z","shell.execute_reply.started":"2023-08-05T15:12:54.036289Z","shell.execute_reply":"2023-08-05T15:12:54.112513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_annot_df = pd.DataFrame(data_annotations)\nprint(\"train_annot_df size=\", len(train_annot_df))\ntrain_annot_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:54.115382Z","iopub.execute_input":"2023-08-05T15:12:54.116045Z","iopub.status.idle":"2023-08-05T15:12:55.580096Z","shell.execute_reply.started":"2023-08-05T15:12:54.116006Z","shell.execute_reply":"2023-08-05T15:12:55.578979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_metadata = pd.DataFrame(images_metadata_test)\ntest_metadata = test_metadata[['id', 'file_name', 'width', 'height']]\ntest_metadata = test_metadata.rename(columns={\"id\": \"image_id\"})\nprint(\"test_metadata size=\", len(test_metadata))\ntest_metadata.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:55.585218Z","iopub.execute_input":"2023-08-05T15:12:55.587227Z","iopub.status.idle":"2023-08-05T15:12:55.650808Z","shell.execute_reply.started":"2023-08-05T15:12:55.587188Z","shell.execute_reply":"2023-08-05T15:12:55.649759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef show_image(_image_id):\n    image_file = train_coco_labels.loadImgs([_image_id])[0]['file_name']\n    image = Image.open(TRAIN_IMG_DIR/image_file)\n    plt.imshow(np.asarray(image))\n    plt.axis('off')\n\ndef show_multiple_images(image_ids):\n    num_images = len(image_ids)\n    num_cols = 2\n    num_rows = num_images\n\n    fig = plt.figure(figsize=(15, 15 * num_rows))\n\n    for i, image_id in enumerate(image_ids, start=1):  # Use 1-based indexing\n        fig.add_subplot(num_rows, num_cols, i)\n        show_image(image_id)\n\n    plt.subplots_adjust(hspace=0.2)  # Adjust the vertical space between images\n    plt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:55.652512Z","iopub.execute_input":"2023-08-05T15:12:55.652885Z","iopub.status.idle":"2023-08-05T15:12:55.663495Z","shell.execute_reply.started":"2023-08-05T15:12:55.652849Z","shell.execute_reply":"2023-08-05T15:12:55.662416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef show_segmentations(_image_id):\n    show_image(_image_id)\n    annotation_ids = train_coco_labels.getAnnIds(imgIds=[_image_id])\n    annotations = train_coco_labels.loadAnns(annotation_ids)\n    train_coco_labels.showAnns(annotations)\n\ndef show_multiple_segmentations(image_ids):\n    num_images = len(image_ids)\n    num_cols = 2\n    num_rows = num_images\n\n    fig = plt.figure(figsize=(15, 15 * num_rows))\n\n    for i, image_id in enumerate(image_ids, start=1):  # Use 1-based indexing\n        fig.add_subplot(num_rows, num_cols, i)\n        show_segmentations(image_id)\n\n    plt.subplots_adjust(hspace=0.2)  # Adjust the vertical space between images\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:55.665216Z","iopub.execute_input":"2023-08-05T15:12:55.665675Z","iopub.status.idle":"2023-08-05T15:12:55.675767Z","shell.execute_reply.started":"2023-08-05T15:12:55.665636Z","shell.execute_reply":"2023-08-05T15:12:55.674340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage\nimage_ids = [60, 20, 20,80]  # Replace with the image IDs you want to display\nshow_multiple_images(image_ids)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:55.677775Z","iopub.execute_input":"2023-08-05T15:12:55.678204Z","iopub.status.idle":"2023-08-05T15:12:58.437100Z","shell.execute_reply.started":"2023-08-05T15:12:55.678166Z","shell.execute_reply":"2023-08-05T15:12:58.436220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage\nimage_ids = [60, 20, 20,80]  # Replace with the image IDs you want to display\nshow_multiple_segmentations(image_ids)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:12:58.438376Z","iopub.execute_input":"2023-08-05T15:12:58.439188Z","iopub.status.idle":"2023-08-05T15:13:01.361866Z","shell.execute_reply.started":"2023-08-05T15:12:58.439147Z","shell.execute_reply":"2023-08-05T15:13:01.360958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_annotations = pd.DataFrame(train_dict['annotations'])\n#train_annotations.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:01.363198Z","iopub.execute_input":"2023-08-05T15:13:01.363994Z","iopub.status.idle":"2023-08-05T15:13:02.874026Z","shell.execute_reply.started":"2023-08-05T15:13:01.363931Z","shell.execute_reply":"2023-08-05T15:13:02.872993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding bbox_area and bbox_aspect_ratio\nprint(\"\\nRenaming id to annotation_id.\\nAdding bbox_area, bbox_aspect_ratio.\\n\")\ntrain_annotations.rename(columns={\"id\":\"annotation_id\"}, inplace=True)\nbbox_area=[]\nbbox_aspect_ratio=[]\nfor idx in train_annotations.index:\n    bbox_area.append(train_annotations['bbox'][idx][3]*train_annotations['bbox'][idx][2])\n    bbox_aspect_ratio.append(train_annotations['bbox'][idx][3]/train_annotations['bbox'][idx][2])\ntrain_annotations['bbox_area']=bbox_area\ntrain_annotations['bbox_aspect_ratio']=bbox_aspect_ratio\nprint(\"train_annotations shape: \" + str(train_annotations.shape) + \"\\n\")\ntrain_annotations.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:02.875912Z","iopub.execute_input":"2023-08-05T15:13:02.876315Z","iopub.status.idle":"2023-08-05T15:13:19.922766Z","shell.execute_reply.started":"2023-08-05T15:13:02.876277Z","shell.execute_reply":"2023-08-05T15:13:19.921649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_annotations.describe()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:19.924796Z","iopub.execute_input":"2023-08-05T15:13:19.925901Z","iopub.status.idle":"2023-08-05T15:13:19.931192Z","shell.execute_reply.started":"2023-08-05T15:13:19.925857Z","shell.execute_reply":"2023-08-05T15:13:19.930084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_categories = pd.DataFrame(train_dict['categories'])\n\nprint(\"Dropping supercategory.\\nRenaming id to category_id.\\n\")\ntrain_categories.drop('supercategory', axis=1, inplace=True)\ntrain_categories.rename(columns={\"id\":\"category_id\"}, inplace=True)\ntrain_categories","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:19.932579Z","iopub.execute_input":"2023-08-05T15:13:19.933136Z","iopub.status.idle":"2023-08-05T15:13:19.953304Z","shell.execute_reply.started":"2023-08-05T15:13:19.933101Z","shell.execute_reply":"2023-08-05T15:13:19.952128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_cat_count = train_annotations[['image_id', 'category_id']].copy()\ncategory_names=[]\nfor idx in img_cat_count.index:\n    category_names.append(train_categories['name'][img_cat_count['category_id'][idx]])\nimg_cat_count['category_name']=category_names\nimg_cat_count[train_categories['name'][0]] = img_cat_count['category_id'] == train_categories['category_id'][0]\nimg_cat_count[train_categories['name'][1]] = img_cat_count['category_id'] == train_categories['category_id'][1]\nimg_cat_count[train_categories['name'][2]] = img_cat_count['category_id'] == train_categories['category_id'][2]\nimg_cat_count[train_categories['name'][3]] = img_cat_count['category_id'] == train_categories['category_id'][3]\nimg_cat_count.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:19.954793Z","iopub.execute_input":"2023-08-05T15:13:19.955843Z","iopub.status.idle":"2023-08-05T15:13:28.533406Z","shell.execute_reply.started":"2023-08-05T15:13:19.955799Z","shell.execute_reply":"2023-08-05T15:13:28.532179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the style for a better appearance\nsns.set_style(\"whitegrid\")\n\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\n\n# Create a figure and axes with customized dimensions and colors\nfig, ax = plt.subplots(figsize=(10, 4))\nscatter = ax.scatter(train_annotations['bbox_area'], train_annotations['bbox_aspect_ratio'], c=train_annotations['bbox_area'], cmap='viridis', alpha=0.7)\n\n# Customize the colorbar\ncbar = plt.colorbar(scatter)\ncbar.set_label('Bounding Box Area', rotation=270, labelpad=15)\n\n# Set labels and title with bold font weight\nax.set_xlabel('Bounding Box Area', fontweight='bold')\nax.set_ylabel('Bounding Box Aspect Ratio', fontweight='bold')\nax.set_title(' Bounding Box Area vs Aspect Ratio', fontweight='bold')\n\n# Customize grid and ticks\nax.grid(which='major', linestyle='--', linewidth=0.5, alpha=0.7)\nax.minorticks_on()\nax.grid(which='minor', linestyle=':', linewidth=0.5, alpha=0.5)\n\n# Adding legend with bold font weight\nlegend = ax.legend(*scatter.legend_elements(), title='Bounding Box Area', title_fontsize='14', prop={'weight': 'bold'})\n\n# Display the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:28.535044Z","iopub.execute_input":"2023-08-05T15:13:28.535474Z","iopub.status.idle":"2023-08-05T15:13:41.100809Z","shell.execute_reply.started":"2023-08-05T15:13:28.535438Z","shell.execute_reply":"2023-08-05T15:13:41.099895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nCategorywise Document Image Count.\\n\")\ncategorywise_image_count=img_cat_count.groupby('category_id', as_index=False)['image_id'].nunique()\ncategorywise_image_count['category_name']=train_categories['name']\ncategorywise_image_count.rename(columns={'image_id':'image_count'}, inplace=True)\ncategorywise_image_count = categorywise_image_count[['category_id', 'category_name', 'image_count']]\ncategorywise_image_count","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:41.102297Z","iopub.execute_input":"2023-08-05T15:13:41.103317Z","iopub.status.idle":"2023-08-05T15:13:41.148364Z","shell.execute_reply.started":"2023-08-05T15:13:41.103282Z","shell.execute_reply":"2023-08-05T15:13:41.147014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Categories vs Number of document images they appear\")\nsns.barplot(x=categorywise_image_count['category_name'], y=categorywise_image_count['image_count'])","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:41.150080Z","iopub.execute_input":"2023-08-05T15:13:41.150500Z","iopub.status.idle":"2023-08-05T15:13:41.529879Z","shell.execute_reply.started":"2023-08-05T15:13:41.150465Z","shell.execute_reply":"2023-08-05T15:13:41.528943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nImagewise Category Count.\\n\")\nimagewise_category_count=img_cat_count.groupby('image_id', as_index=False)[['paragraph', 'text_box', 'image', 'table']].sum()\nimagewise_category_count.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:41.531616Z","iopub.execute_input":"2023-08-05T15:13:41.531983Z","iopub.status.idle":"2023-08-05T15:13:41.579418Z","shell.execute_reply.started":"2023-08-05T15:13:41.531924Z","shell.execute_reply":"2023-08-05T15:13:41.578241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_area_count = train_annotations.groupby('bbox_area', as_index=False)['annotation_id'].count()\nbbox_area_count.rename(columns={'annotation_id':'annotation_count'}, inplace=True)\nprint(\"\\nTop 10 areas having maximum annotation count.\\n\")\nbbox_area_count.sort_values(by='annotation_count', ascending=False,inplace=True)\nbbox_area_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:41.581017Z","iopub.execute_input":"2023-08-05T15:13:41.581450Z","iopub.status.idle":"2023-08-05T15:13:41.635056Z","shell.execute_reply.started":"2023-08-05T15:13:41.581407Z","shell.execute_reply":"2023-08-05T15:13:41.634042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up a more visually appealing style\nsns.set_style(\"darkgrid\")\nsns.set_palette(\"viridis\")\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\n# Create a figure and axis with customized dimensions and colors\nplt.figure(figsize=(8, 4))\n\n# Plotting the top 10 areas with maximum annotation count\nax = sns.barplot(x='bbox_area', y='annotation_count', data=bbox_area_count.head(10))\n\n# Adding annotations on top of the bars\nfor p in ax.patches:\n    ax.annotate(f\"{int(p.get_height()):,}\",\n                (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center',\n                xytext=(0, 10), textcoords='offset points',\n                fontsize=10, fontweight='bold', color='black')\n\n# Set labels and title\nplt.xlabel('Bounding Box Area')\nplt.ylabel('Annotation Count')\nplt.title('Top 10 Bounding Box Areas with Maximum Annotation Count', fontweight='bold', fontsize=12)\n\n# Customize the x-axis tick labels to display as integers with comma separators\nplt.xticks()\nplt.gca().xaxis.set_major_formatter(plt.FuncFormatter(lambda x, _: f'{int(x):,}'))\n\n# Adding grid\nplt.grid(axis='y', alpha=0.7)\n\n\n# Display the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:41.636724Z","iopub.execute_input":"2023-08-05T15:13:41.637445Z","iopub.status.idle":"2023-08-05T15:13:42.222771Z","shell.execute_reply.started":"2023-08-05T15:13:41.637383Z","shell.execute_reply":"2023-08-05T15:13:42.221905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.ticker import FuncFormatter\n\n# Set the style for the plot\nsns.set_style(\"whitegrid\")\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\nplt.figure(figsize=(8, 6))  # Set the size of the figure\n\n# Create a bar plot using seaborn with a different palette\nax = sns.barplot(x=categorywise_image_count['category_name'], y=categorywise_image_count['image_count'], palette=\"Set2\")\n\n# Add labels and title with Cambria font style and bold\nfont_style = {'fontname': 'Cambria', 'fontweight': 'bold'}\n\nax.set_xlabel(\"Document Categories\", fontdict=font_style, fontsize=12, labelpad=15)\nax.set_ylabel(\"Number of Images\", fontdict=font_style, fontsize=12, labelpad=15)\nax.set_title(\"Distribution of Images Across Document Categories\", fontdict=font_style, fontsize=14, pad=20)\n\n# Rotate x-axis labels for better readability\nplt.xticks(rotation=45, ha='right', fontsize=12)\n\n# Add data labels on top of the bars with better formatting\ndef format_func(value, tick_number):\n    return f\"{int(value):,}\"\nax.yaxis.set_major_formatter(FuncFormatter(format_func))\n\nfor p in ax.patches:\n    ax.annotate(f'{int(p.get_height()):,}',\n                (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='bottom', fontsize=10,\n                xytext=(0, 9), textcoords='offset points')\n\n# Adjust layout to prevent cutting off labels\nplt.tight_layout()\n\n# Customize the background color of the plot\nax.set_facecolor('#f4f4f4')\n\n# Remove spines\nsns.despine()\n\n# Add grid lines\nax.grid(axis='y', linestyle='--', alpha=0.7)\n\nplt.show()  # Display the plot\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:42.224548Z","iopub.execute_input":"2023-08-05T15:13:42.225283Z","iopub.status.idle":"2023-08-05T15:13:42.674005Z","shell.execute_reply.started":"2023-08-05T15:13:42.225243Z","shell.execute_reply":"2023-08-05T15:13:42.672993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming you have loaded the data into train_metadata and train_annot_df\n\n# Select relevant columns from train_metadata\ntrainmetadata = train_metadata[['image_id', 'width', 'height']]\n\n# Set up a customized style using Seaborn\nsns.set(style='whitegrid')\nplt.figure(figsize=(10, 4))\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\n# Add a title for the entire plot\nplt.suptitle(\"Histogram Analysis for Training Document Images\", fontname='Cambria', fontsize=14, fontweight='bold')\n\n# Create a subplot with two histograms side by side\nplt.subplot(1, 2, 1)\nsns.histplot(trainmetadata['width'], bins=50, color='skyblue', kde=True)\nplt.title(\"Histogram of Image Widths\", fontname='Cambria', fontsize=10, fontweight='bold')\nplt.xlabel(\"Image Width\", fontname='Cambria', fontsize=12)\nplt.ylabel(\"Frequency\", fontname='Cambria', fontsize=12)\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nsns.histplot(trainmetadata['height'], bins=50, color='salmon', kde=True)\nplt.title(\"Histogram of Image Heights\", fontname='Cambria', fontsize=10, fontweight='bold')\nplt.xlabel(\"Image Height\", fontname='Cambria', fontsize=12)\nplt.ylabel(\"\", fontsize=12)  # Remove y-axis label for consistency\nplt.grid(True)\n\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])  # Adjust the top margin for the title\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:42.675635Z","iopub.execute_input":"2023-08-05T15:13:42.675961Z","iopub.status.idle":"2023-08-05T15:13:44.296990Z","shell.execute_reply.started":"2023-08-05T15:13:42.675919Z","shell.execute_reply":"2023-08-05T15:13:44.295928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming you have loaded the data into train_metadata and train_annot_df\n\n# Select relevant columns from train_metadata\ntestmetadata = test_metadata[['image_id', 'width', 'height']]\n\n# Set up a customized style using Seaborn\nsns.set(style='whitegrid')\nplt.figure(figsize=(10, 4))\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\n\n# Add a title for the entire plot\nplt.suptitle(\"Histogram Analysis for Testing Document Images\", fontname='Cambria', fontsize=14, fontweight='bold')\n\n\n# Create a subplot with two histograms side by side\nplt.subplot(1, 2, 1)\nsns.histplot(testmetadata['width'], bins=50, color='skyblue', kde=True)\nplt.title(\"Histogram of Image Widths\", fontname='Cambria', fontsize=10, fontweight='bold')\nplt.xlabel(\"Image Width\", fontname='Cambria', fontsize=12)\nplt.ylabel(\"Frequency\", fontname='Cambria', fontsize=12)\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nsns.histplot(testmetadata['height'], bins=50, color='salmon', kde=True)\nplt.title(\"Histogram of Image Heights\", fontname='Cambria', fontsize=10, fontweight='bold')\nplt.xlabel(\"Image Height\", fontname='Cambria', fontsize=12)\nplt.ylabel(\"\", fontsize=12)  # Remove y-axis label for consistency\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:44.303096Z","iopub.execute_input":"2023-08-05T15:13:44.303382Z","iopub.status.idle":"2023-08-05T15:13:45.617917Z","shell.execute_reply.started":"2023-08-05T15:13:44.303355Z","shell.execute_reply":"2023-08-05T15:13:45.616920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up a more visually appealing style\nsns.set_style(\"whitegrid\")\n\n# Set the font family to Cambria\nplt.rcParams['font.family'] = 'Cambria'\n\n# Define custom colors for the subplots\ncolors = ['#FFA07A', '#66CCCC', '#9ACD32', '#FFD700']\n\n# Create a figure and subplots in a 2x2 grid with customized dimensions and colors\nfig, axs = plt.subplots(2, 2, figsize=(10, 6), sharex=True)\n\n# Plotting the number of paragraphs per document image\nsns.lineplot(ax=axs[0, 0], x=imagewise_category_count['image_id'], y=imagewise_category_count['paragraph'], linewidth=2, color=colors[0])\naxs[0, 0].set_title(\"Number of Paragraphs per Document Image\", fontweight='bold', fontsize=10)\naxs[0, 0].set_ylabel(\"Count\", fontweight='bold', fontsize=10)\naxs[0, 0].tick_params(axis='both', which='major', labelsize=10)\n\n# Plotting the number of text boxes per document image\nsns.lineplot(ax=axs[0, 1], x=imagewise_category_count['image_id'], y=imagewise_category_count['text_box'], linewidth=2, color=colors[1])\naxs[0, 1].set_title(\"Number of Text Boxes per Document Image\", fontweight='bold', fontsize=10)\naxs[0, 1].set_ylabel(\"Count\", fontweight='bold', fontsize=10)\naxs[0, 1].tick_params(axis='both', which='major', labelsize=10)\n\n# Plotting the number of images per document image\nsns.lineplot(ax=axs[1, 0], x=imagewise_category_count['image_id'], y=imagewise_category_count['image'], linewidth=2, color=colors[2])\naxs[1, 0].set_title(\"Number of Images per Document Image\", fontweight='bold', fontsize=10)\naxs[1, 0].set_xlabel(\"Image ID\", fontweight='bold', fontsize=10)\naxs[1, 0].set_ylabel(\"Count\", fontweight='bold', fontsize=10)\naxs[1, 0].tick_params(axis='both', which='major', labelsize=10)\n\n# Plotting the number of tables per document image\nsns.lineplot(ax=axs[1, 1], x=imagewise_category_count['image_id'], y=imagewise_category_count['table'], linewidth=2, color=colors[3])\naxs[1, 1].set_title(\"Number of Tables per Document Image\", fontweight='bold', fontsize=10)\naxs[1, 1].set_xlabel(\"Image ID\", fontweight='bold', fontsize=10)\naxs[1, 1].set_ylabel(\"Count\", fontweight='bold', fontsize=10)\naxs[1, 1].tick_params(axis='both', which='major', labelsize=10)\n\n# Adding grid lines with lighter color\nfor row in axs:\n    for ax in row:\n        ax.grid(axis='y', alpha=0.5, linestyle='--')\n\n# Adjusting spacing between subplots\nplt.tight_layout()\n\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:45.621199Z","iopub.execute_input":"2023-08-05T15:13:45.621492Z","iopub.status.idle":"2023-08-05T15:13:47.733886Z","shell.execute_reply.started":"2023-08-05T15:13:45.621465Z","shell.execute_reply":"2023-08-05T15:13:47.732874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_SPLIT = 0.99","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:47.735395Z","iopub.execute_input":"2023-08-05T15:13:47.736025Z","iopub.status.idle":"2023-08-05T15:13:47.740835Z","shell.execute_reply.started":"2023-08-05T15:13:47.735987Z","shell.execute_reply":"2023-08-05T15:13:47.739812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_dataset = len(train_metadata)\nn_train = int(n_dataset * TRAIN_SPLIT)\nprint(\"n_dataset\", n_dataset, \"n_train\", n_train, \"n_val\", n_dataset-n_train)\n\nnp.random.seed(SEED)\n\ninds = np.random.permutation(n_dataset)\ntrain_inds, valid_inds = inds[:n_train], inds[n_train:]","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:47.742280Z","iopub.execute_input":"2023-08-05T15:13:47.742866Z","iopub.status.idle":"2023-08-05T15:13:47.753584Z","shell.execute_reply.started":"2023-08-05T15:13:47.742832Z","shell.execute_reply":"2023-08-05T15:13:47.752634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_coco_to_detectron2_format(\n    imgdir: Path,\n    metadata_df: pd.DataFrame,\n    annot_df: Optional[pd.DataFrame] = None,\n    target_indices: Optional[np.ndarray] = None,\n):\n\n    dataset_dicts = []\n    for _, train_meta_row in tqdm(metadata_df.iterrows(), total=len(metadata_df)):\n        # Iterate over each image\n        image_id, filename, width, height = train_meta_row.values\n\n        annotations = []\n\n        # If train/validation data, then there will be annotations\n        if annot_df is not None:\n            for _, ann in annot_df.query(\"image_id == @image_id\").iterrows():\n                # Get annotations of current iteration's image\n                class_id = ann[\"category_id\"]\n                gt_masks = ann[\"gt_masks\"]\n                bbox_resized = [\n                    float(ann[\"x_min\"]),\n                    float(ann[\"y_min\"]),\n                    float(ann[\"x_max\"]),\n                    float(ann[\"y_max\"]),\n                ]\n\n                annotation = {\n                    \"bbox\": bbox_resized,\n                    \"bbox_mode\": BoxMode.XYXY_ABS,\n                    \"segmentation\": gt_masks,\n                    \"category_id\": class_id,\n                }\n\n                annotations.append(annotation)\n\n        # coco format -> detectron2 format dict\n        record = {\n            \"file_name\": str(imgdir/filename),\n            \"image_id\": image_id,\n            \"width\": width,\n            \"height\": height,\n            \"annotations\": annotations\n        }\n\n        dataset_dicts.append(record)\n\n    if target_indices is not None:\n        dataset_dicts = [dataset_dicts[i] for i in target_indices]\n\n    return dataset_dicts","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:47.754917Z","iopub.execute_input":"2023-08-05T15:13:47.755341Z","iopub.status.idle":"2023-08-05T15:13:47.768292Z","shell.execute_reply.started":"2023-08-05T15:13:47.755307Z","shell.execute_reply":"2023-08-05T15:13:47.767160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_REGISTER_TRAINING = \"badlad_train\"\nDATA_REGISTER_VALID    = \"badlad_valid\"\nDATA_REGISTER_TEST     = \"badlad_test\"","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:47.770474Z","iopub.execute_input":"2023-08-05T15:13:47.771105Z","iopub.status.idle":"2023-08-05T15:13:47.784800Z","shell.execute_reply.started":"2023-08-05T15:13:47.771071Z","shell.execute_reply":"2023-08-05T15:13:47.783808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Register Training data\nif is_train:\n    DatasetCatalog.register(\n        DATA_REGISTER_TRAINING,\n        lambda: convert_coco_to_detectron2_format(\n            TRAIN_IMG_DIR,\n            train_metadata,\n            train_annot_df,\n            target_indices=train_inds,\n        ),\n    )\n\n    # Set Training data categories\n    MetadataCatalog.get(DATA_REGISTER_TRAINING).set(thing_classes=thing_classes)\n\n    dataset_dicts_train = DatasetCatalog.get(DATA_REGISTER_TRAINING)\n    metadata_dicts_train = MetadataCatalog.get(DATA_REGISTER_TRAINING)\n\n    print(\"dicts training size=\", len(dataset_dicts_train))\n    print(\"################\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:13:47.787473Z","iopub.execute_input":"2023-08-05T15:13:47.787909Z","iopub.status.idle":"2023-08-05T15:15:38.309790Z","shell.execute_reply.started":"2023-08-05T15:13:47.787831Z","shell.execute_reply":"2023-08-05T15:15:38.308678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Register Validation data\nif is_train or is_evaluate:\n    DatasetCatalog.register(\n        DATA_REGISTER_VALID,\n        lambda: convert_coco_to_detectron2_format(\n            TRAIN_IMG_DIR,\n            train_metadata,\n            train_annot_df,\n            target_indices=valid_inds,\n        ),\n    )\n\n    # Set Validation data categories\n    MetadataCatalog.get(DATA_REGISTER_VALID).set(thing_classes=thing_classes)\n\n    dataset_dicts_valid = DatasetCatalog.get(DATA_REGISTER_VALID)\n    metadata_dicts_valid = MetadataCatalog.get(DATA_REGISTER_VALID)\n\n    print(\"dicts valid size=\", len(dataset_dicts_valid))\n    print(\"################\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:15:38.311447Z","iopub.execute_input":"2023-08-05T15:15:38.312485Z","iopub.status.idle":"2023-08-05T15:17:30.708161Z","shell.execute_reply.started":"2023-08-05T15:15:38.312447Z","shell.execute_reply":"2023-08-05T15:17:30.707126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Register Test Inference data\nDatasetCatalog.register(\n    DATA_REGISTER_TEST,\n    lambda: convert_coco_to_detectron2_format(\n        TEST_IMG_DIR,\n        test_metadata,\n    )\n)\n\n# Set Test data categories\nMetadataCatalog.get(DATA_REGISTER_TEST).set(\n    thing_classes=thing_classes_test\n)\n\ndataset_dicts_test = DatasetCatalog.get(DATA_REGISTER_TEST)\nmetadata_dicts_test = MetadataCatalog.get(DATA_REGISTER_TEST)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:17:30.709973Z","iopub.execute_input":"2023-08-05T15:17:30.710713Z","iopub.status.idle":"2023-08-05T15:17:31.562220Z","shell.execute_reply.started":"2023-08-05T15:17:30.710672Z","shell.execute_reply":"2023-08-05T15:17:31.561239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"#### DATA REGISTERED ####\")","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:17:31.563786Z","iopub.execute_input":"2023-08-05T15:17:31.564620Z","iopub.status.idle":"2023-08-05T15:17:31.570767Z","shell.execute_reply.started":"2023-08-05T15:17:31.564581Z","shell.execute_reply":"2023-08-05T15:17:31.569460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_mapper(dataset_dict):\n    dataset_dict = copy.deepcopy(dataset_dict)\n    image = utils.read_image(dataset_dict[\"file_name\"], format=\"BGR\")\n\n    transform_list = [T.RandomBrightness(0.8, 1.2),\n                      T.RandomFlip(prob=0.5, horizontal=False, vertical=True),\n                      T.RandomFlip(prob=0.5, horizontal=True, vertical=False)\n                      ]\n    image, transforms = T.apply_transform_gens(transform_list, image)\n\n    dataset_dict[\"image\"] = torch.as_tensor(\n        image.transpose(2, 0, 1).astype(\"float32\"))\n\n    annos = [\n        utils.transform_instance_annotations(obj, transforms, image.shape[:2])\n        for obj in dataset_dict.pop(\"annotations\")\n        if obj.get(\"iscrowd\", 0) == 0\n    ]\n    instances = utils.annotations_to_instances(annos, image.shape[:2])\n\n    dataset_dict[\"instances\"] = utils.filter_empty_instances(instances)\n\n    return dataset_dict","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:50.827635Z","iopub.execute_input":"2023-08-05T15:20:50.828082Z","iopub.status.idle":"2023-08-05T15:20:50.837428Z","shell.execute_reply.started":"2023-08-05T15:20:50.828028Z","shell.execute_reply":"2023-08-05T15:20:50.836311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AugTrainer(DefaultTrainer):\n    @classmethod\n    def build_train_loader(cls, cfg):\n        return build_detection_train_loader(cfg, mapper=custom_mapper)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:51.174729Z","iopub.execute_input":"2023-08-05T15:20:51.175483Z","iopub.status.idle":"2023-08-05T15:20:51.181453Z","shell.execute_reply.started":"2023-08-05T15:20:51.175427Z","shell.execute_reply":"2023-08-05T15:20:51.180164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_train:\n    cfg = get_cfg()\n\n    # config_name = \"COCO-Detection/faster_rcnn_R_50_FPN_3x.yaml\"\n    config_name = \"COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml\"\n\n    cfg.merge_from_file(model_zoo.get_config_file(config_name))\n\n    cfg.DATASETS.TRAIN = (DATA_REGISTER_TRAINING,)\n    cfg.DATASETS.TEST = (DATA_REGISTER_VALID,)\n\n    # to evaluate during training, you have to implement `build_evaluator()` method of the trainer.\n    # https://github.com/facebookresearch/detectron2/blob/94113be6e12db36b8c7601e13747587f19ec92fe/detectron2/engine/defaults.py#L561\n    # cfg.TEST.EVAL_PERIOD = 500\n\n    cfg.DATALOADER.NUM_WORKERS = 0\n    #cfg.TEST.EVAL_PERIOD = 500\n    cfg.MODEL.WEIGHTS = model_zoo.get_checkpoint_url(config_name)\n    if (is_resume_training):\n        print(\"#### SETTING PRETRAINED WEIGHTS TO RESUME TRAINING ####\")\n        cfg.MODEL.WEIGHTS = str(PRETRAINED_PATH)\n    else:\n        print(\"#### TRAINING MODEL FROM SCRATCH ####\")\n\n    cfg.SOLVER.AMP.ENABLED = True\n    cfg.SOLVER.IMS_PER_BATCH = 16\n    cfg.SOLVER.BASE_LR = 0.001\n\n    cfg.SOLVER.WARMUP_ITERS = 5\n\n    # Maximum number of iterations\n    cfg.SOLVER.MAX_ITER = 6100\n\n    # cfg.SOLVER.STEPS = (500, 1000) # must be less than MAX_ITER\n\n    cfg.SOLVER.GAMMA = 0.05\n\n    # Small value == Frequent save need a lot of storage.\n    cfg.SOLVER.CHECKPOINT_PERIOD = 1000\n    cfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 128\n    cfg.MODEL.ROI_HEADS.NUM_CLASSES = 4\n\n    # Create Output Directory\n    cfg.OUTPUT_DIR = str(OUTPUT_DIR)\n    print(\"creating cfg.OUTPUT_DIR -> \", cfg.OUTPUT_DIR)\n    OUTPUT_DIR.mkdir(exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:51.528244Z","iopub.execute_input":"2023-08-05T15:20:51.528550Z","iopub.status.idle":"2023-08-05T15:20:51.560172Z","shell.execute_reply.started":"2023-08-05T15:20:51.528524Z","shell.execute_reply":"2023-08-05T15:20:51.559144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:52.294813Z","iopub.execute_input":"2023-08-05T15:20:52.296091Z","iopub.status.idle":"2023-08-05T15:20:52.374290Z","shell.execute_reply.started":"2023-08-05T15:20:52.295933Z","shell.execute_reply":"2023-08-05T15:20:52.373038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:52.980566Z","iopub.execute_input":"2023-08-05T15:20:52.981758Z","iopub.status.idle":"2023-08-05T15:20:54.255947Z","shell.execute_reply.started":"2023-08-05T15:20:52.981709Z","shell.execute_reply":"2023-08-05T15:20:54.254484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:54.260150Z","iopub.execute_input":"2023-08-05T15:20:54.260521Z","iopub.status.idle":"2023-08-05T15:20:54.266543Z","shell.execute_reply.started":"2023-08-05T15:20:54.260486Z","shell.execute_reply":"2023-08-05T15:20:54.265446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_train:\n    trainer = DefaultTrainer(cfg) if not is_augment else AugTrainer(cfg)\n        \n    trainer.resume_or_load(resume=is_resume_training)\n\n    trainer.train()\n    \n    print(\"#### TRAINING COMPLETE ####\")\n    _ = trainer.model.train(False)  # turn off training\n    \n    FileLink(str(OUTPUT_MODEL))","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:20:54.268486Z","iopub.execute_input":"2023-08-05T15:20:54.268835Z","iopub.status.idle":"2023-08-05T15:32:27.633766Z","shell.execute_reply.started":"2023-08-05T15:20:54.268801Z","shell.execute_reply":"2023-08-05T15:32:27.632661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if is_train:\n    # Load metrics\n    metrics_df = pd.read_json(\n        OUTPUT_DIR/\"metrics.json\", orient=\"records\", lines=True\n    )\n    mdf = metrics_df.sort_values(\"iteration\")\n    #.tail(3)\n    print(mdf.head(30).T)\n\n    # Plot loss\n    fig, ax = plt.subplots()\n\n    mdf1 = mdf[~mdf[\"total_loss\"].isna()]\n    ax.plot(mdf1[\"iteration\"], mdf1[\"total_loss\"], c=\"C0\", label=\"train\")\n\n    if \"validation_loss\" in mdf.columns:\n        mdf2 = mdf[~mdf[\"validation_loss\"].isna()]\n        ax.plot(mdf2[\"iteration\"], mdf2[\"validation_loss\"],\n                c=\"C1\", label=\"validation\")\n\n    ax.legend()\n    ax.set_title(\"Loss curve\")\n    plt.show()\n\n    # Plot Accuracy\n    fig, ax = plt.subplots()\n\n    mdf1 = mdf[~mdf[\"fast_rcnn/cls_accuracy\"].isna()]\n    ax.plot(mdf1[\"iteration\"], mdf1[\"fast_rcnn/cls_accuracy\"],\n            c=\"C0\", label=\"train\")\n\n    ax.legend()\n    ax.set_title(\"Accuracy curve\")\n    plt.show()\n\n    # Plot Bounding Box regressor loss\n    fig, ax = plt.subplots()\n\n    mdf1 = mdf[~mdf[\"loss_box_reg\"].isna()]\n    ax.plot(mdf1[\"iteration\"], mdf1[\"loss_box_reg\"], c=\"C0\", label=\"train\")\n\n    ax.legend()\n    ax.set_title(\"loss_box_reg\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-05T14:55:40.972217Z","iopub.execute_input":"2023-08-05T14:55:40.972873Z","iopub.status.idle":"2023-08-05T14:55:42.182886Z","shell.execute_reply.started":"2023-08-05T14:55:40.972834Z","shell.execute_reply":"2023-08-05T14:55:42.181982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if Path(\"/kaggle/working/output/model_final.pth\").exists:\n    display(FileLink(\"output/model_final.pth\"))","metadata":{"execution":{"iopub.status.busy":"2023-08-05T15:42:53.269439Z","iopub.execute_input":"2023-08-05T15:42:53.269887Z","iopub.status.idle":"2023-08-05T15:42:53.280850Z","shell.execute_reply.started":"2023-08-05T15:42:53.269849Z","shell.execute_reply":"2023-08-05T15:42:53.279594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install GPUtil\n\nfrom GPUtil import showUtilization as gpu_usage\ngpu_usage()                             \n","metadata":{"execution":{"iopub.status.busy":"2023-08-04T18:31:12.527725Z","iopub.status.idle":"2023-08-04T18:31:12.528208Z","shell.execute_reply.started":"2023-08-04T18:31:12.527953Z","shell.execute_reply":"2023-08-04T18:31:12.527975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inf_cfg = get_cfg()\n\nconfig_name = \"COCO-InstanceSegmentation/mask_rcnn_R_50_FPN_3x.yaml\"\n\ninf_cfg.merge_from_file(model_zoo.get_config_file(config_name))\ninf_cfg.MODEL.ROI_HEADS.BATCH_SIZE_PER_IMAGE = 128\ninf_cfg.MODEL.ROI_HEADS.NUM_CLASSES = 4\ninf_cfg.MODEL.ROI_HEADS.SCORE_THRESH_TEST = 0.5\ninf_cfg.MODEL.DEVICE = \"cuda\"\n\ninf_cfg.DATALOADER.NUM_WORKERS =  0 # lower this if CUDA overflow occurs\ninf_cfg.MODEL.WEIGHTS = str(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T21:09:39.934155Z","iopub.execute_input":"2023-08-04T21:09:39.934583Z","iopub.status.idle":"2023-08-04T21:09:39.961426Z","shell.execute_reply.started":"2023-08-04T21:09:39.934550Z","shell.execute_reply":"2023-08-04T21:09:39.960399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH = 10  # lower this if CUDA overflow occurs\ntest_loader = build_detection_test_loader(inf_cfg, DATA_REGISTER_TEST, batch_size=BATCH)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T21:09:55.490684Z","iopub.execute_input":"2023-08-04T21:09:55.491781Z","iopub.status.idle":"2023-08-04T21:09:56.715615Z","shell.execute_reply.started":"2023-08-04T21:09:55.491729Z","shell.execute_reply":"2023-08-04T21:09:56.714440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ACCEPTANCE_THRESHOLD = 0.6  # for all categories","metadata":{"execution":{"iopub.status.busy":"2023-08-04T21:09:58.105765Z","iopub.execute_input":"2023-08-04T21:09:58.106132Z","iopub.status.idle":"2023-08-04T21:09:58.111201Z","shell.execute_reply.started":"2023-08-04T21:09:58.106101Z","shell.execute_reply":"2023-08-04T21:09:58.110127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictor = DefaultPredictor(inf_cfg)\nprint(predictor)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T21:09:58.683084Z","iopub.execute_input":"2023-08-04T21:09:58.684476Z","iopub.status.idle":"2023-08-04T21:09:59.676523Z","shell.execute_reply.started":"2023-08-04T21:09:58.684432Z","shell.execute_reply":"2023-08-04T21:09:59.675511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize=(20, 20))\nindices = [ax[0][0], ax[1][0], ax[0][1], ax[1][1]]\n\n# Show some qualitative results by predicting on test set images\nNUM_TEST_SAMPLES = 4\nsamples = np.random.choice(dataset_dicts_test, NUM_TEST_SAMPLES)\n\nfor i, sample in enumerate(samples):\n    img = cv2.imread(sample[\"file_name\"])\n    outputs = predictor(img)\n    visualizer = Visualizer(img, metadata=metadata_dicts_test, scale=0.5,)\n    visualizer = visualizer.draw_instance_predictions(\n        outputs[\"instances\"].to(\"cpu\")\n    )\n    display_img = visualizer.get_image()[:, :, ::-1]\n    indices[i].grid(False)\n    indices[i].imshow(display_img)","metadata":{"execution":{"iopub.status.busy":"2023-08-04T21:10:01.429743Z","iopub.execute_input":"2023-08-04T21:10:01.430140Z","iopub.status.idle":"2023-08-04T21:10:08.266112Z","shell.execute_reply.started":"2023-08-04T21:10:01.430107Z","shell.execute_reply":"2023-08-04T21:10:08.265171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}