{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"!pip install -q pycocotools\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pycocotools.coco import COCO\nfrom PIL import Image\nimport json\nimport random\nfrom pathlib import Path\n%matplotlib inline\nsns.set_theme(style='darkgrid', palette='deep', font='sans-serif', font_scale=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:32:09.222522Z","iopub.execute_input":"2023-06-26T05:32:09.224491Z","iopub.status.idle":"2023-06-26T05:32:23.883023Z","shell.execute_reply.started":"2023-06-26T05:32:09.224410Z","shell.execute_reply":"2023-06-26T05:32:23.881307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load","metadata":{}},{"cell_type":"code","source":"train_images_directory = Path(\"/kaggle/input/dlsprint2/badlad/images/train\")\ntest_images_directory = Path(\"/kaggle/input/dlsprint2/badlad/images/test\")\n\ntrain_labels_path = Path(\"/kaggle/input/dlsprint2/badlad/labels/coco_format/train/badlad-train-coco.json\")\ntest_metadata_path = Path(\"/kaggle/input/dlsprint2/badlad/badlad-test-metadata.json\")","metadata":{"_uuid":"be8e6b11-f3ee-4b14-ae92-4ff7a5ccf547","_cell_guid":"b90db510-aa0d-499d-8ed9-fcbfd3b805e6","execution":{"iopub.status.busy":"2023-06-26T05:12:49.184891Z","iopub.execute_input":"2023-06-26T05:12:49.185382Z","iopub.status.idle":"2023-06-26T05:12:49.193457Z","shell.execute_reply.started":"2023-06-26T05:12:49.185344Z","shell.execute_reply":"2023-06-26T05:12:49.192102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with train_labels_path.open() as f:\n    train_dict = json.load(f)\n\nwith test_metadata_path.open() as f:\n    test_dict = json.load(f)\n\ntrain_coco_labels=COCO(annotation_file=train_labels_path)","metadata":{"_uuid":"299ffc3d-9eda-4561-a143-a4323c8e27fc","_cell_guid":"7c7dead0-414a-4951-9ec1-10d97bb4afbc","execution":{"iopub.status.busy":"2023-06-26T05:12:49.195334Z","iopub.execute_input":"2023-06-26T05:12:49.196935Z","iopub.status.idle":"2023-06-26T05:13:08.958284Z","shell.execute_reply.started":"2023-06-26T05:12:49.196858Z","shell.execute_reply":"2023-06-26T05:13:08.956789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# General Overview\n\n> **About The Dataset** \n<br>\n> The dataset for this competition comprises about 34,000 Bengali documents annotated using polygon boundaries and rectangular bounding boxes. These documents range from newspaper articles, official documents, notices, and government gadgets to novels, comics, magazines, and even liberation war records.","metadata":{}},{"cell_type":"code","source":"def print_columns(_columns, _description):\n    print(_description)\n    for column in _columns:\n        print(column)\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:08.961826Z","iopub.execute_input":"2023-06-26T05:13:08.962257Z","iopub.status.idle":"2023-06-26T05:13:08.967207Z","shell.execute_reply.started":"2023-06-26T05:13:08.962224Z","shell.execute_reply":"2023-06-26T05:13:08.966245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are \" + str(len(train_dict['categories'])) + \" categories.\\n\")\nprint(\"There are \" + str(len(test_dict['images']) + len(train_dict['images'])) + \" images in the dataset.\")\nprint(\"There are \" + str(len(train_dict['images'])) + \" images in the train set.\")\nprint(\"There are \" + str(len(test_dict['images'])) + \" images in the test set.\\n\")\nprint(\"There are \" + str(len(train_dict['annotations'])) + \" annotations in the train set.\\n\")\n\nprint_columns(train_dict.keys(), \"Fields in COCO format of train labels:\")\nprint(\"We will focus on mainly categories, images and annotations.\")","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:08.969257Z","iopub.execute_input":"2023-06-26T05:13:08.969740Z","iopub.status.idle":"2023-06-26T05:13:08.989332Z","shell.execute_reply.started":"2023-06-26T05:13:08.969663Z","shell.execute_reply":"2023-06-26T05:13:08.987842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring Documents\n> At first let's see some sample documents.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nfor i in range(6):\n    image_id = random.randint(2, 20364)\n    image_file = train_coco_labels.loadImgs([image_id])[0]['file_name']\n    image = Image.open(train_images_directory/image_file)\n    code = 231 + i;\n    plt.subplot(code)\n    plt.axis('off')\n    plt.imshow(np.asarray(image))","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:32:29.740512Z","iopub.execute_input":"2023-06-26T05:32:29.741074Z","iopub.status.idle":"2023-06-26T05:32:35.118526Z","shell.execute_reply.started":"2023-06-26T05:32:29.741032Z","shell.execute_reply":"2023-06-26T05:32:35.114204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_image(_image_id):\n    image_file = train_coco_labels.loadImgs([_image_id])[0]['file_name']\n    image = Image.open(train_images_directory/image_file)\n    plt.figure(figsize=(15, 15))\n    plt.axis('off')\n    plt.imshow(np.asarray(image))","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:11.648227Z","iopub.execute_input":"2023-06-26T05:13:11.648676Z","iopub.status.idle":"2023-06-26T05:13:11.660013Z","shell.execute_reply.started":"2023-06-26T05:13:11.648639Z","shell.execute_reply":"2023-06-26T05:13:11.658463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_segmentations(_image_id):\n    show_image(_image_id)\n    annotation_ids = train_coco_labels.getAnnIds(imgIds=[_image_id])\n    annotations = train_coco_labels.loadAnns(annotation_ids)\n    train_coco_labels.showAnns(annotations)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:11.661951Z","iopub.execute_input":"2023-06-26T05:13:11.662512Z","iopub.status.idle":"2023-06-26T05:13:11.681620Z","shell.execute_reply.started":"2023-06-26T05:13:11.662463Z","shell.execute_reply":"2023-06-26T05:13:11.680349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Now let's focus on a single document.","metadata":{}},{"cell_type":"code","source":"# you can change the image_id\nshow_image(14)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:11.683442Z","iopub.execute_input":"2023-06-26T05:13:11.684041Z","iopub.status.idle":"2023-06-26T05:13:12.432160Z","shell.execute_reply.started":"2023-06-26T05:13:11.683998Z","shell.execute_reply":"2023-06-26T05:13:12.430794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Let's see all the class's(paragraph, text box, image, table) segmentations in a single document.","metadata":{}},{"cell_type":"code","source":"# you can change the image_id\nshow_segmentations(14)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:12.439190Z","iopub.execute_input":"2023-06-26T05:13:12.440668Z","iopub.status.idle":"2023-06-26T05:13:13.237559Z","shell.execute_reply.started":"2023-06-26T05:13:12.440604Z","shell.execute_reply":"2023-06-26T05:13:13.236093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_binary_mask(_image_id, _category_id):\n    annotation_ids = train_coco_labels.getAnnIds(imgIds=[_image_id], catIds=[_category_id])\n    annotations = train_coco_labels.loadAnns(annotation_ids)\n    image_height = train_coco_labels.loadImgs([_image_id])[0]['height']\n    image_width = train_coco_labels.loadImgs([_image_id])[0]['width']\n    binary_mask=np.zeros((image_height, image_width))\n    for annotation in annotations:\n        current_mask=train_coco_labels.annToMask(annotation)\n        binary_mask += current_mask\n    plt.axis('off')\n    plt.imshow(binary_mask)\n\ndef show_categorywise_binary_mask(_image_id):\n    plt.figure(figsize=(15, 15))\n    \n    plt.subplot(221)\n    plt.title('Paragraph Binary Mask')\n    show_binary_mask(_image_id, 0)\n    \n    plt.subplot(222)\n    plt.title('Text Box Binary Mask')\n    show_binary_mask(_image_id, 1)\n    \n    plt.subplot(223)\n    plt.title('Image Binary Mask')\n    show_binary_mask(_image_id, 2)\n    \n    plt.subplot(224)\n    plt.title('Table Binary Mask')\n    show_binary_mask(_image_id, 3)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:13.239453Z","iopub.execute_input":"2023-06-26T05:13:13.240017Z","iopub.status.idle":"2023-06-26T05:13:13.252585Z","shell.execute_reply.started":"2023-06-26T05:13:13.239974Z","shell.execute_reply":"2023-06-26T05:13:13.251144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# What Should We Predict?\n\n> Your model should predict the masks for bounding polygons in each image. A mask is a 2D matrix corresponding to each input image pixel. The contents of each pixel position can be 0 or 1. If the pixel position is marked 0, then that position corresponds to the background category, no paragraph, text box, image, or table exists at that pixel. This mask is called a binary mask because of its content. Each prediction category (paragraph, text box, image, table) has a binary mask for each image. For example, image 846df66a-610e-4356-b369-6788885a0dc5.png has 4 binary masks, for each existing category - that predict where in the image, that specific category occurs.\n\nSo, let's generate the binary mask of a document for each class.","metadata":{}},{"cell_type":"code","source":"# you can change the image_id\nshow_categorywise_binary_mask(14)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:13.254810Z","iopub.execute_input":"2023-06-26T05:13:13.255370Z","iopub.status.idle":"2023-06-26T05:13:14.877475Z","shell.execute_reply.started":"2023-06-26T05:13:13.255325Z","shell.execute_reply":"2023-06-26T05:13:14.876100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring Categories\n> *`categories:` Maps instance categories to their assigned IDs.*\n\nThe annotations specify the regions in these documents that fall into 4 classes -\n\n1. Paragraph\n1. Text Box\n1. Image\n1. Table \n","metadata":{}},{"cell_type":"code","source":"train_categories = pd.DataFrame(train_dict['categories'])\ntrain_categories.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:14.879305Z","iopub.execute_input":"2023-06-26T05:13:14.879740Z","iopub.status.idle":"2023-06-26T05:13:14.894775Z","shell.execute_reply.started":"2023-06-26T05:13:14.879686Z","shell.execute_reply":"2023-06-26T05:13:14.893821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Dropping supercategory.\\nRenaming id to category_id.\\n\")\ntrain_categories.drop('supercategory', axis=1, inplace=True)\ntrain_categories.rename(columns={\"id\":\"category_id\"}, inplace=True)\ntrain_categories","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:14.896284Z","iopub.execute_input":"2023-06-26T05:13:14.896903Z","iopub.status.idle":"2023-06-26T05:13:14.919347Z","shell.execute_reply.started":"2023-06-26T05:13:14.896866Z","shell.execute_reply":"2023-06-26T05:13:14.917812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring Image Labels\n> *`images:` Maps image IDs (id) to their hashed name, height, and width.*\n","metadata":{}},{"cell_type":"code","source":"train_images = pd.DataFrame(train_dict[\"images\"])\ntrain_images.head()","metadata":{"_uuid":"3aff901c-ee7a-41ac-aed8-2e58b04a934a","_cell_guid":"4e40a7b2-7103-49ea-935e-1f329e04c60f","execution":{"iopub.status.busy":"2023-06-26T05:13:14.921596Z","iopub.execute_input":"2023-06-26T05:13:14.922841Z","iopub.status.idle":"2023-06-26T05:13:15.010666Z","shell.execute_reply.started":"2023-06-26T05:13:14.922788Z","shell.execute_reply":"2023-06-26T05:13:15.009140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nDropping license and date_captured.\\nRenaming id to image_id.\\n\")\ntrain_images.drop(['license', 'date_captured'], axis=1, inplace=True)\ntrain_images.rename(columns={\"id\":\"image_id\"}, inplace=True)\nprint(\"train_images shape: \" + str(train_images.shape) + \"\\n\")\ntrain_images.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.012528Z","iopub.execute_input":"2023-06-26T05:13:15.013072Z","iopub.status.idle":"2023-06-26T05:13:15.033764Z","shell.execute_reply.started":"2023-06-26T05:13:15.013026Z","shell.execute_reply":"2023-06-26T05:13:15.032461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding area and aspect_ratio\nprint(\"\\nAdding area and aspect_ratio of images.\\n\")\ntrain_images['area']=train_images['height']*train_images['width']\ntrain_images['aspect_ratio']=train_images['height']/train_images['width']\ntrain_images.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.035499Z","iopub.execute_input":"2023-06-26T05:13:15.036640Z","iopub.status.idle":"2023-06-26T05:13:15.059292Z","shell.execute_reply.started":"2023-06-26T05:13:15.036600Z","shell.execute_reply":"2023-06-26T05:13:15.057936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.060995Z","iopub.execute_input":"2023-06-26T05:13:15.061366Z","iopub.status.idle":"2023-06-26T05:13:15.101157Z","shell.execute_reply.started":"2023-06-26T05:13:15.061338Z","shell.execute_reply":"2023-06-26T05:13:15.099894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_aspect_ratio_count = train_images.groupby('aspect_ratio', as_index=False)['image_id'].nunique()\ntrain_images_aspect_ratio_count.rename(columns={'image_id':'image_count'}, inplace=True)\n\nplt.figure(figsize=(15, 6))\nplt.title(\"Aspect Ratio vs Image Count Line Plot\")\nsns.lineplot(x=train_images_aspect_ratio_count['aspect_ratio'], y = train_images_aspect_ratio_count['image_count'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.103017Z","iopub.execute_input":"2023-06-26T05:13:15.104301Z","iopub.status.idle":"2023-06-26T05:13:15.724381Z","shell.execute_reply.started":"2023-06-26T05:13:15.104252Z","shell.execute_reply":"2023-06-26T05:13:15.722970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nTop 10 aspect ratios having maximum image count.\\n\")\ntrain_images_aspect_ratio_count.sort_values(by='image_count', ascending=False,inplace=True)\ntrain_images_aspect_ratio_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.726218Z","iopub.execute_input":"2023-06-26T05:13:15.726595Z","iopub.status.idle":"2023-06-26T05:13:15.742752Z","shell.execute_reply.started":"2023-06-26T05:13:15.726564Z","shell.execute_reply":"2023-06-26T05:13:15.741784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_area_count = train_images.groupby('area', as_index=False)['image_id'].count()\ntrain_images_area_count.rename(columns={'image_id':'image_count'}, inplace=True)\n\nplt.figure(figsize=(15, 6))\nplt.title('Area vs Image Count Line Plot')\nsns.lineplot(x=train_images_area_count['area'], y = train_images_area_count['image_count'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:15.744688Z","iopub.execute_input":"2023-06-26T05:13:15.745164Z","iopub.status.idle":"2023-06-26T05:13:16.357583Z","shell.execute_reply.started":"2023-06-26T05:13:15.745128Z","shell.execute_reply":"2023-06-26T05:13:16.356475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nTop 10 areas having maximum image count.\\n\")\ntrain_images_area_count.sort_values(by='image_count', ascending=False,inplace=True)\ntrain_images_area_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:16.359030Z","iopub.execute_input":"2023-06-26T05:13:16.359928Z","iopub.status.idle":"2023-06-26T05:13:16.376010Z","shell.execute_reply.started":"2023-06-26T05:13:16.359889Z","shell.execute_reply":"2023-06-26T05:13:16.374485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring Annotations\n\n> *`annotations:` This is a list of the polygon bounding segments for each instance of the categories occurring in each image.*\n\n<br>\nThe contents of this category are the crux of the problem:\n\n* `segmentation:` Polygon (x, y) coordinate pairs of it's vertices.For example, the first annotation has its first vertex at (86.328, 179.007). The following pairs complete the polygon region of the annotation.\n* `image_id:` The image this annotation is for.\n* `category_id:` The category this annotation is bounding. It may be a paragraph, text box, image, or table.\n* `id:` Annotation ID, identifies each annotation.\n* `bbox:` The rectangular bounding box that best estimates the annotated region.\nThis list holds all the annotations for each instance of the categories for all images.","metadata":{}},{"cell_type":"code","source":"train_annotations = pd.DataFrame(train_dict['annotations'])\ntrain_annotations.head()","metadata":{"_uuid":"142e7a21-14f7-4047-9052-ee8c73d81578","_cell_guid":"9cac6f39-3274-4dee-8567-56f315e58181","execution":{"iopub.status.busy":"2023-06-26T05:13:16.377551Z","iopub.execute_input":"2023-06-26T05:13:16.377956Z","iopub.status.idle":"2023-06-26T05:13:18.203486Z","shell.execute_reply.started":"2023-06-26T05:13:16.377922Z","shell.execute_reply":"2023-06-26T05:13:18.202362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# adding bbox_area and bbox_aspect_ratio\nprint(\"\\nRenaming id to annotation_id.\\nAdding bbox_area, bbox_aspect_ratio.\\n\")\ntrain_annotations.rename(columns={\"id\":\"annotation_id\"}, inplace=True)\nbbox_area=[]\nbbox_aspect_ratio=[]\nfor idx in train_annotations.index:\n    bbox_area.append(train_annotations['bbox'][idx][3]*train_annotations['bbox'][idx][2])\n    bbox_aspect_ratio.append(train_annotations['bbox'][idx][3]/train_annotations['bbox'][idx][2])\ntrain_annotations['bbox_area']=bbox_area\ntrain_annotations['bbox_aspect_ratio']=bbox_aspect_ratio\nprint(\"train_annotations shape: \" + str(train_annotations.shape) + \"\\n\")\ntrain_annotations.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:18.204969Z","iopub.execute_input":"2023-06-26T05:13:18.205331Z","iopub.status.idle":"2023-06-26T05:13:37.575360Z","shell.execute_reply.started":"2023-06-26T05:13:18.205300Z","shell.execute_reply":"2023-06-26T05:13:37.573645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_annotations.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:37.577231Z","iopub.execute_input":"2023-06-26T05:13:37.577641Z","iopub.status.idle":"2023-06-26T05:13:37.780529Z","shell.execute_reply.started":"2023-06-26T05:13:37.577604Z","shell.execute_reply":"2023-06-26T05:13:37.778783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_aspect_ratio_count = train_annotations.groupby('bbox_aspect_ratio', as_index=False)['annotation_id'].count()\nbbox_aspect_ratio_count.rename(columns={'annotation_id':'annotation_count'}, inplace=True)\n\nplt.figure(figsize=(15, 6))\nplt.title('bbox aspect ratio vs Annotation Count Line Plot')\nsns.lineplot(x=bbox_aspect_ratio_count['bbox_aspect_ratio'], y = bbox_aspect_ratio_count['annotation_count'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:37.784376Z","iopub.execute_input":"2023-06-26T05:13:37.784985Z","iopub.status.idle":"2023-06-26T05:13:39.101425Z","shell.execute_reply.started":"2023-06-26T05:13:37.784934Z","shell.execute_reply":"2023-06-26T05:13:39.100172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nTop 10 aspect ratios having maximum annotation count.\\n\")\nbbox_aspect_ratio_count.sort_values(by='annotation_count', ascending=False,inplace=True)\nbbox_aspect_ratio_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:39.103056Z","iopub.execute_input":"2023-06-26T05:13:39.103443Z","iopub.status.idle":"2023-06-26T05:13:39.128880Z","shell.execute_reply.started":"2023-06-26T05:13:39.103412Z","shell.execute_reply":"2023-06-26T05:13:39.127533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bbox_area_count = train_annotations.groupby('bbox_area', as_index=False)['annotation_id'].count()\nbbox_area_count.rename(columns={'annotation_id':'annotation_count'}, inplace=True)\n\nplt.figure(figsize=(15, 6))\nplt.title('Area vs Annotation Count Line Plot')\nsns.lineplot(x=bbox_area_count['bbox_area'], y = bbox_area_count['annotation_count'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:39.136345Z","iopub.execute_input":"2023-06-26T05:13:39.136824Z","iopub.status.idle":"2023-06-26T05:13:40.250006Z","shell.execute_reply.started":"2023-06-26T05:13:39.136789Z","shell.execute_reply":"2023-06-26T05:13:40.249064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nTop 10 areas having maximum annotation count.\\n\")\nbbox_area_count.sort_values(by='annotation_count', ascending=False,inplace=True)\nbbox_area_count.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:40.251588Z","iopub.execute_input":"2023-06-26T05:13:40.252878Z","iopub.status.idle":"2023-06-26T05:13:40.276459Z","shell.execute_reply.started":"2023-06-26T05:13:40.252838Z","shell.execute_reply":"2023-06-26T05:13:40.275202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Categorical Classification","metadata":{}},{"cell_type":"code","source":"img_cat_count = train_annotations[['image_id', 'category_id']].copy()\ncategory_names=[]\nfor idx in img_cat_count.index:\n    category_names.append(train_categories['name'][img_cat_count['category_id'][idx]])\nimg_cat_count['category_name']=category_names\nimg_cat_count[train_categories['name'][0]] = img_cat_count['category_id'] == train_categories['category_id'][0]\nimg_cat_count[train_categories['name'][1]] = img_cat_count['category_id'] == train_categories['category_id'][1]\nimg_cat_count[train_categories['name'][2]] = img_cat_count['category_id'] == train_categories['category_id'][2]\nimg_cat_count[train_categories['name'][3]] = img_cat_count['category_id'] == train_categories['category_id'][3]\nimg_cat_count.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:40.278174Z","iopub.execute_input":"2023-06-26T05:13:40.278546Z","iopub.status.idle":"2023-06-26T05:13:49.573421Z","shell.execute_reply.started":"2023-06-26T05:13:40.278517Z","shell.execute_reply":"2023-06-26T05:13:49.572053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nCategorywise Document Image Count.\\n\")\ncategorywise_image_count=img_cat_count.groupby('category_id', as_index=False)['image_id'].nunique()\ncategorywise_image_count['category_name']=train_categories['name']\ncategorywise_image_count.rename(columns={'image_id':'image_count'}, inplace=True)\ncategorywise_image_count = categorywise_image_count[['category_id', 'category_name', 'image_count']]\ncategorywise_image_count","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:49.575051Z","iopub.execute_input":"2023-06-26T05:13:49.576002Z","iopub.status.idle":"2023-06-26T05:13:49.630748Z","shell.execute_reply.started":"2023-06-26T05:13:49.575966Z","shell.execute_reply":"2023-06-26T05:13:49.629761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insights**\n> * Text box and paragraph appears almost in all the documents.\n> * Image appears almost in one-fourth of the documents of trainset.\n> * Tables are not that much frequent like other 3 classes. 1 in 20 documents contains table.","metadata":{}},{"cell_type":"code","source":"plt.title(\"Categories vs Number of document images they appear\")\nsns.barplot(x=categorywise_image_count['category_name'], y=categorywise_image_count['image_count'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:49.632285Z","iopub.execute_input":"2023-06-26T05:13:49.633272Z","iopub.status.idle":"2023-06-26T05:13:50.071082Z","shell.execute_reply.started":"2023-06-26T05:13:49.633221Z","shell.execute_reply":"2023-06-26T05:13:50.069802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\nImagewise Category Count.\\n\")\nimagewise_category_count=img_cat_count.groupby('image_id', as_index=False)[['paragraph', 'text_box', 'image', 'table']].sum()\nimagewise_category_count.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:50.072645Z","iopub.execute_input":"2023-06-26T05:13:50.073158Z","iopub.status.idle":"2023-06-26T05:13:50.120505Z","shell.execute_reply.started":"2023-06-26T05:13:50.073113Z","shell.execute_reply":"2023-06-26T05:13:50.119264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imagewise_category_count.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:50.122309Z","iopub.execute_input":"2023-06-26T05:13:50.122801Z","iopub.status.idle":"2023-06-26T05:13:50.161195Z","shell.execute_reply.started":"2023-06-26T05:13:50.122765Z","shell.execute_reply":"2023-06-26T05:13:50.159822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 30))\n\nplt.subplot(411)\nplt.title(\"Number of paragraphs per document image\")\nsns.lineplot(x=imagewise_category_count['image_id'], y=imagewise_category_count['paragraph'])\n\nplt.subplot(412)\nplt.title(\"Number of text boxes per document image\")\nsns.lineplot(x=imagewise_category_count['image_id'], y=imagewise_category_count['text_box'])\n\nplt.subplot(413)\nplt.title(\"Number of images per document image\")\nsns.lineplot(x=imagewise_category_count['image_id'], y=imagewise_category_count['image'])\n\nplt.subplot(414)\nplt.title(\"Number of tables per document image\")\nsns.lineplot(x=imagewise_category_count['image_id'], y=imagewise_category_count['table'])","metadata":{"execution":{"iopub.status.busy":"2023-06-26T05:13:50.163049Z","iopub.execute_input":"2023-06-26T05:13:50.163461Z","iopub.status.idle":"2023-06-26T05:13:53.136140Z","shell.execute_reply.started":"2023-06-26T05:13:50.163428Z","shell.execute_reply":"2023-06-26T05:13:53.134657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Thank you if you reached this point. If you feel like it was useful, an upvote will be appreciated and help me keep going.**","metadata":{"_kg_hide-input":false}}]}