{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport ast\nimport json\nimport subprocess\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom pprint import pprint\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom IPython.display import Video","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.478737Z","iopub.execute_input":"2021-12-06T15:14:02.479012Z","iopub.status.idle":"2021-12-06T15:14:02.485356Z","shell.execute_reply.started":"2021-12-06T15:14:02.478984Z","shell.execute_reply":"2021-12-06T15:14:02.484382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport ast\nimport json\nimport subprocess\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom pprint import pprint\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom IPython.display import Video","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.487469Z","iopub.execute_input":"2021-12-06T15:14:02.488040Z","iopub.status.idle":"2021-12-06T15:14:02.505782Z","shell.execute_reply.started":"2021-12-06T15:14:02.487990Z","shell.execute_reply":"2021-12-06T15:14:02.504652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Root of input\nINPUT_PATH = '../input/tensorflow-great-barrier-reef'\nHEIGHT = 720 # image height\nWIDTH  = 1280 # image width","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.507715Z","iopub.execute_input":"2021-12-06T15:14:02.508507Z","iopub.status.idle":"2021-12-06T15:14:02.521309Z","shell.execute_reply.started":"2021-12-06T15:14:02.508457Z","shell.execute_reply":"2021-12-06T15:14:02.520300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(INPUT_PATH + '/train.csv')\ndisplay(df_train)\nprint(df_train.info())","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.523593Z","iopub.execute_input":"2021-12-06T15:14:02.525675Z","iopub.status.idle":"2021-12-06T15:14:02.608132Z","shell.execute_reply.started":"2021-12-06T15:14:02.525636Z","shell.execute_reply":"2021-12-06T15:14:02.607384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for video_id in df_train['video_id'].unique():\n    print(f'video_id: {video_id}')\n    print(f'w   annotations:  {sum(df_train[df_train[\"video_id\"]==video_id][\"annotations\"] == \"[]\")}')\n    print(f'w/o annotations:  {sum(df_train[df_train[\"video_id\"]==video_id][\"annotations\"] != \"[]\")}\\n')","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.609559Z","iopub.execute_input":"2021-12-06T15:14:02.610034Z","iopub.status.idle":"2021-12-06T15:14:02.641222Z","shell.execute_reply.started":"2021-12-06T15:14:02.610001Z","shell.execute_reply":"2021-12-06T15:14:02.640549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Change the type of 'annotations' from str to list\ndf_train['annotations'] = df_train['annotations'].apply(ast.literal_eval) # str -> list\n# Add columns of image path and number of bboxes\ndf_train['image_path'] = INPUT_PATH + '/train_images/video_' + df_train['video_id'].astype(str) + '/' + df_train['video_frame'].astype(str) + \".jpg\"\ndf_train['num_bboxes'] = df_train['annotations'].apply(lambda x: len(x))\ndisplay(df_train)","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:02.642487Z","iopub.execute_input":"2021-12-06T15:14:02.642912Z","iopub.status.idle":"2021-12-06T15:14:03.108482Z","shell.execute_reply.started":"2021-12-06T15:14:02.642881Z","shell.execute_reply":"2021-12-06T15:14:03.107544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_num_bboxes = max(df_train['num_bboxes'])\nindexes = df_train[df_train['num_bboxes']==max_num_bboxes].index.values\nprint(f'Maximum number of bboxes in an image: {max_num_bboxes}')\ndisplay(df_train.iloc[indexes])","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:03.110300Z","iopub.execute_input":"2021-12-06T15:14:03.110603Z","iopub.status.idle":"2021-12-06T15:14:03.160450Z","shell.execute_reply.started":"2021-12-06T15:14:03.110561Z","shell.execute_reply":"2021-12-06T15:14:03.159452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indexes = [indexes[0], indexes[2]]","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:03.161586Z","iopub.execute_input":"2021-12-06T15:14:03.161817Z","iopub.status.idle":"2021-12-06T15:14:03.166358Z","shell.execute_reply.started":"2021-12-06T15:14:03.161781Z","shell.execute_reply":"2021-12-06T15:14:03.165676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bboxes(annotations):\n    \"\"\"\n    annotations: list of annotations\n    return: bboxes as [x_min, y_min, x_max, y_max]\n    \"\"\"\n    if len(annotations)==0:\n        return []\n    boxes = pd.DataFrame(annotations, columns=['x', 'y', 'width', 'height']).astype(np.int32).values\n    # [x_min, y_min, w, h] -> [x_min, y_min, x_max, y_max]\n    boxes[:, 2] = boxes[:, 0] + boxes[:, 2]\n    boxes[:, 3] = boxes[:, 1] + boxes[:, 3]\n    return boxes   \n\ndef plot_img_and_bbox(img_path, anntations):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    fig, ax = plt.subplots(1, 1, figsize=(16,10))\n    if len(annotations)>0:\n        bboxes = get_bboxes(annotations)\n        for i, box in enumerate(bboxes):\n            # pur bbox on image\n            cv2.rectangle(img,\n                          (box[0], box[1]),\n                          (box[2], box[3]),\n                          color = (255, 0, 0),\n                          thickness = 2)\n            # numbering\n            ax.text(box[0], box[1]-5, i+1, color='red')\n\n    ax.set_axis_off()\n    ax.imshow(img)\n\n\ndef zoom_bbox(img_path, annotations):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    bboxes = get_bboxes(annotations)\n    \n    col = 6\n    row = np.ceil(len(bboxes)//6).astype(int)\n    fig, ax = plt.subplots(row, col, figsize=(16,9))\n    cnt = 0\n    for i in range(row):\n        if cnt >= len(bboxes):\n            break\n        for j in range(col):\n            bbox = bboxes[cnt]\n            sliced_img = img[bbox[1]:bbox[3], bbox[0]:bbox[2]]\n            ax[i,j].imshow(sliced_img)\n            ax[i,j].set_title(cnt+1, color='red')\n            ax[i,j].set_axis_off()\n            cnt += 1\n    plt.show() ","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:03.168451Z","iopub.execute_input":"2021-12-06T15:14:03.168956Z","iopub.status.idle":"2021-12-06T15:14:03.184533Z","shell.execute_reply.started":"2021-12-06T15:14:03.168924Z","shell.execute_reply":"2021-12-06T15:14:03.183858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = df_train.iloc[indexes].copy()\nfor idx, row in samples.iterrows():\n    img_path    = row['image_path']\n    annotations = row['annotations']\n    print('image_id:', row['image_id'])\n    # plot image with bboxes\n    plot_img_and_bbox(img_path, annotations)\n    # plot zoom of bboxes\n    zoom_bbox(img_path, annotations)","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:03.185880Z","iopub.execute_input":"2021-12-06T15:14:03.186153Z","iopub.status.idle":"2021-12-06T15:14:06.420203Z","shell.execute_reply.started":"2021-12-06T15:14:03.186122Z","shell.execute_reply":"2021-12-06T15:14:06.419161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:06.421621Z","iopub.execute_input":"2021-12-06T15:14:06.422062Z","iopub.status.idle":"2021-12-06T15:14:06.446309Z","shell.execute_reply.started":"2021-12-06T15:14:06.422027Z","shell.execute_reply":"2021-12-06T15:14:06.445150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(INPUT_PATH + '/train.csv').to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:26.854271Z","iopub.execute_input":"2021-12-06T15:14:26.855018Z","iopub.status.idle":"2021-12-06T15:14:26.994434Z","shell.execute_reply.started":"2021-12-06T15:14:26.854965Z","shell.execute_reply":"2021-12-06T15:14:26.993648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-06T15:14:06.449479Z","iopub.status.idle":"2021-12-06T15:14:06.450530Z","shell.execute_reply.started":"2021-12-06T15:14:06.449641Z","shell.execute_reply":"2021-12-06T15:14:06.449676Z"},"trusted":true},"execution_count":null,"outputs":[]}]}