{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TensorFlow Object Detection - Make compact videos\n\n**<span style=\"color:red\">If you liked this notebook, please don't forget to upvote it!</span>**\n\n* This notebook is based on https://www.kaggle.com/alexchwong/yolov5-is-all-you-need-make-compact-videos from alexchwong (version 3). Please upvote it as well!\n\nI used these notebooks to create this one:\n* https://www.kaggle.com/alexchwong/yolov5-is-all-you-need-make-compact-videos\n* https://www.kaggle.com/khanhlvg/cots-detection-w-tensorflow-object-detection-api\n* https://www.kaggle.com/bamps53/create-annotated-video\n\nAlso I used code from:\n* https://www.tensorflow.org/hub/tutorials/object_detection\n\nPlease upvote them as well!\n","metadata":{}},{"cell_type":"code","source":"# Path to TF model\n# We will use model from https://www.kaggle.com/khanhlvg/cots-detection-w-tensorflow-object-detection-api for demo purposes\nMODEL_DIR = '../input/cots-detection-w-tensorflow-object-detection-api/cots_efficientdet_d0'\n\n# Detection parameters\nDETECTION_THRESHOLD  = 0.3\nMAX_BOXES = 10","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:47:58.904272Z","iopub.execute_input":"2022-01-30T18:47:58.905278Z","iopub.status.idle":"2022-01-30T18:47:58.931990Z","shell.execute_reply.started":"2022-01-30T18:47:58.905134Z","shell.execute_reply":"2022-01-30T18:47:58.931302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport torch\nfrom tqdm import tqdm\nimport sys\nimport tensorflow as tf\n\nfrom PIL import Image\nfrom PIL import ImageColor\nfrom PIL import ImageDraw\nfrom PIL import ImageFont\nfrom PIL import ImageOps\n\nsys.path.append('../input/tensorflow-great-barrier-reef')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-30T18:47:58.933589Z","iopub.execute_input":"2022-01-30T18:47:58.934100Z","iopub.status.idle":"2022-01-30T18:48:04.150412Z","shell.execute_reply.started":"2022-01-30T18:47:58.934064Z","shell.execute_reply":"2022-01-30T18:48:04.149571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os\nimport importlib\nimport cv2 \n\nimport ast\nimport shutil\nimport sys\nimport time\n\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nfrom PIL import Image\nfrom IPython.display import display","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:04.151620Z","iopub.execute_input":"2022-01-30T18:48:04.151869Z","iopub.status.idle":"2022-01-30T18:48:04.340492Z","shell.execute_reply.started":"2022-01-30T18:48:04.151837Z","shell.execute_reply":"2022-01-30T18:48:04.339826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define the Model Here","metadata":{}},{"cell_type":"code","source":"start_time = time.time()\ntf.keras.backend.clear_session()\ndetect_fn_tf_odt = tf.saved_model.load(os.path.join(os.path.join(MODEL_DIR, 'output'), 'saved_model'))\nend_time = time.time()\nelapsed_time = end_time - start_time\nprint('Elapsed time: ' + str(elapsed_time) + 's')","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-01-30T18:48:04.342615Z","iopub.execute_input":"2022-01-30T18:48:04.342875Z","iopub.status.idle":"2022-01-30T18:48:36.421139Z","shell.execute_reply.started":"2022-01-30T18:48:04.342840Z","shell.execute_reply":"2022-01-30T18:48:36.420157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Videos","metadata":{}},{"cell_type":"markdown","source":"## Install ffmpeg for Kaggle","metadata":{}},{"cell_type":"code","source":"# Install ffmpeg for video compression\n%cd /kaggle/working\n\n! tar -xf ../input/ffmpeg-static-build/ffmpeg-git-amd64-static.tar.xz\n\nimport subprocess\n\nFFMPEG_BIN = \"/kaggle/working/ffmpeg-git-20191209-amd64-static/ffmpeg\"","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-01-30T18:48:36.422443Z","iopub.execute_input":"2022-01-30T18:48:36.423228Z","iopub.status.idle":"2022-01-30T18:48:41.040350Z","shell.execute_reply.started":"2022-01-30T18:48:36.423184Z","shell.execute_reply":"2022-01-30T18:48:41.039520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Utility Functions","metadata":{}},{"cell_type":"code","source":"# https://www.tensorflow.org/hub/tutorials/object_detection\ndef draw_bounding_box_on_image(image,\n                               ymin,\n                               xmin,\n                               ymax,\n                               xmax,\n                               color,\n                               font,\n                               thickness=4,\n                               display_str_list=()):\n    \"\"\"Adds a bounding box to an image.\"\"\"\n    draw = ImageDraw.Draw(image)\n    im_width, im_height = image.size\n    (left, right, top, bottom) = (xmin * im_width, xmax * im_width,\n                                ymin * im_height, ymax * im_height)\n    draw.line([(left, top), (left, bottom), (right, bottom), (right, top),\n             (left, top)],\n            width=thickness,\n            fill=color)\n\n    # If the total height of the display strings added to the top of the bounding\n    # box exceeds the top of the image, stack the strings below the bounding box\n    # instead of above.\n    display_str_heights = [font.getsize(ds)[1] for ds in display_str_list]\n    # Each display_str has a top and bottom margin of 0.05x.\n    total_display_str_height = (1 + 2 * 0.05) * sum(display_str_heights)\n\n    if top > total_display_str_height:\n        text_bottom = top\n    else:\n        text_bottom = top + total_display_str_height\n    # Reverse list and print from bottom to top.\n    for display_str in display_str_list[::-1]:\n        text_width, text_height = font.getsize(display_str)\n        margin = np.ceil(0.05 * text_height)\n        draw.rectangle([(left, text_bottom - text_height - 2 * margin),\n                        (left + text_width, text_bottom)],\n                       fill=color)\n        draw.text((left + margin, text_bottom - text_height - margin),\n                  display_str,\n                  fill=\"black\",\n                  font=font)\n        text_bottom -= text_height - 2 * margin\n\n\ndef draw_boxes(image, boxes, class_names, scores, color, max_boxes=MAX_BOXES, min_score=DETECTION_THRESHOLD):\n    \"\"\"Overlay labeled boxes on an image with formatted scores and label names.\"\"\"\n    colors = list(ImageColor.colormap.values())\n\n    font = ImageFont.load_default()\n    for i in range(min(boxes.shape[0], max_boxes)):\n        if scores[i] >= min_score and len(boxes[i]) > 1:\n            #print(boxes[i])\n            ymin, xmin, ymax, xmax = tuple(boxes[i])\n            display_str = \"{}: {}%\".format(class_names[i],\n                                         int(100 * scores[i]))\n            #color = colors[hash(class_names[i]) % len(colors)]\n            image_pil = Image.fromarray(np.uint8(image)).convert(\"RGB\")\n            draw_bounding_box_on_image(\n              image_pil,\n              ymin,\n              xmin,\n              ymax,\n              xmax,\n              color,\n              font,\n              display_str_list=[display_str])\n            np.copyto(image, np.array(image_pil))\n    return image","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:41.043750Z","iopub.execute_input":"2022-01-30T18:48:41.048158Z","iopub.status.idle":"2022-01-30T18:48:41.072544Z","shell.execute_reply.started":"2022-01-30T18:48:41.048111Z","shell.execute_reply":"2022-01-30T18:48:41.071852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing the Training Dataset and selecting videos\n\n* Note that it is important to also include videos with NO COTS, so you can understand where false positive detections may arise","metadata":{}},{"cell_type":"code","source":"# Modified from https://www.kaggle.com/remekkinas/yolox-inference-on-kaggle-for-cots-lb-0-507\n\n%cd /kaggle/working\n\nfrom sklearn.model_selection import GroupKFold\n\ndef get_bbox(annots):\n    bboxes = [list(annot.values()) for annot in annots]\n    return bboxes\n\ndef get_path(row):\n    row['image_path'] = f'{ROOT_DIR}/train_images/video_{row.video_id}/{row.video_frame}.jpg'\n    return row\n\nROOT_DIR  = '/kaggle/input/tensorflow-great-barrier-reef/'\n\ndf = pd.read_csv(\"/kaggle/input/tensorflow-great-barrier-reef/train.csv\")\n\n\ndf[\"num_bbox\"] = df['annotations'].apply(lambda x: str.count(x, 'x'))\ndf_train = df\n\n#Annotations \ndf_train['annotations'] = df_train['annotations'].progress_apply(lambda x: ast.literal_eval(x))\ndf_train['bboxes'] = df_train.annotations.progress_apply(get_bbox)\n\ndf_train = df_train.progress_apply(get_path, axis=1)\n\nkf = GroupKFold(n_splits = 5) \ndf_train = df_train.reset_index(drop=True)\ndf_train['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(kf.split(df_train, y = df_train.video_id.tolist(), groups=df_train.sequence)):\n    df_train.loc[val_idx, 'fold'] = fold\n\ndf_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:41.076732Z","iopub.execute_input":"2022-01-30T18:48:41.079549Z","iopub.status.idle":"2022-01-30T18:48:58.031527Z","shell.execute_reply.started":"2022-01-30T18:48:41.079494Z","shell.execute_reply":"2022-01-30T18:48:58.030653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Select the validation dataset\n\n* Take care not to shuffle the videos! Keep them in the original order by not sorting the data frame","metadata":{}},{"cell_type":"code","source":"df_test = df_train[df_train.fold == 4]","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:58.032742Z","iopub.execute_input":"2022-01-30T18:48:58.033102Z","iopub.status.idle":"2022-01-30T18:48:58.041011Z","shell.execute_reply.started":"2022-01-30T18:48:58.033064Z","shell.execute_reply":"2022-01-30T18:48:58.040087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define image paths and ground truth bounding boxes","metadata":{}},{"cell_type":"code","source":"image_paths = df_test.image_path.tolist()\n\ngt = []\nfor i, row in df_test.iterrows():\n    if len(row['bboxes']) > 1:\n        x0 = row['bboxes'][0][0] / 1280\n        y0 = row['bboxes'][0][1] / 720\n        x1 = x0 + row['bboxes'][0][2] / 1280\n        y1 = y0 + row['bboxes'][0][3] / 720\n        gt.append([y0, x0, y1, x1])\n    else:\n        gt.append([])","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:58.042364Z","iopub.execute_input":"2022-01-30T18:48:58.043994Z","iopub.status.idle":"2022-01-30T18:48:58.258284Z","shell.execute_reply.started":"2022-01-30T18:48:58.043891Z","shell.execute_reply":"2022-01-30T18:48:58.257652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference on Validation Videos and Recording the Video","metadata":{}},{"cell_type":"code","source":"def load_img(path):\n    img = tf.io.read_file(path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    return img\n\ndef detect(image_np):\n    input_tensor = np.expand_dims(image_np, 0)\n    start_time = time.time()\n    detections = detect_fn_tf_odt(input_tensor)\n    return detections\n\ndef run_detector(detector, path, color):\n    img = load_img(path)#load_image_into_numpy_array(path)\n\n    ##converted_img  = tf.image.convert_image_dtype(img, tf.float32)[tf.newaxis, ...]\n    start_time = time.time()\n    result = detect(img)\n    ##result = detector(converted_img)\n    end_time = time.time()\n\n    result = {key:value.numpy()[0] for key,value in result.items()}\n\n    #print(\"Found %d objects.\" % len(result[\"detection_scores\"]))\n    #print(\"Inference time: \", end_time-start_time)\n\n    detection_classes = ['cots']*len(result[\"detection_boxes\"])\n    image_with_boxes = draw_boxes(\n        img.numpy(), result[\"detection_boxes\"],\n        detection_classes, result[\"detection_scores\"],\n        color)\n\n    return image_with_boxes\n\n\ndef add_correct_box(img, ind, color):\n    detection_classes_t = ['cots']\n    detection_scores_t = [1]\n    image_with_boxes = draw_boxes(\n        img, np.array([gt[ind]]),\n        detection_classes_t, detection_scores_t,\n        color)\n    return image_with_boxes","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:58.260651Z","iopub.execute_input":"2022-01-30T18:48:58.260908Z","iopub.status.idle":"2022-01-30T18:48:58.271219Z","shell.execute_reply.started":"2022-01-30T18:48:58.260874Z","shell.execute_reply":"2022-01-30T18:48:58.270452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd /kaggle/working\n\nvideo_size = (1280, 720)\nCOCO_CLASSES = (\"starfish\")\n\nout1 = cv2.VideoWriter('Video.avi',cv2.VideoWriter_fourcc(*'DIVX'), 15, video_size)\n\nfor i in tqdm(range(1250, 2200)):\n    # Test a small video first. For the full video, substitute \"tqdm(range(start, finish))\" with \"tqdm(range(len(image_paths)))\"\n    TEST_IMAGE_PATH = image_paths[i]\n    img = cv2.imread(TEST_IMAGE_PATH)\n    out_image = run_detector(detect_fn_tf_odt, TEST_IMAGE_PATH, color='#ff0000')\n    out_image = add_correct_box(out_image, i, color='#00ff00')\n    out_image = cv2.cvtColor(out_image, cv2.COLOR_BGR2RGB)\n    out1.write(out_image)\n    \n# Finalize AVI\nout1.release()","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:48:58.272497Z","iopub.execute_input":"2022-01-30T18:48:58.272763Z","iopub.status.idle":"2022-01-30T18:49:08.111437Z","shell.execute_reply.started":"2022-01-30T18:48:58.272725Z","shell.execute_reply":"2022-01-30T18:49:08.110714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert AVI to compressed mp4 for more convenient downloading\n\n* The created AVI is a large file. Compress the video file so you can download it and watch it locally","metadata":{}},{"cell_type":"code","source":"AVI2MP4 = \"-ac 2 -b:v 2000k -c:a aac -c:v libx264 -b:a 160k -vprofile high -bf 0 -strict experimental -f mp4\"\n\ncommand = f\"{FFMPEG_BIN} -i Video.avi {AVI2MP4} Video.mp4\"\nsubprocess.call(command, shell=True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-01-30T18:49:08.112861Z","iopub.execute_input":"2022-01-30T18:49:08.113938Z","iopub.status.idle":"2022-01-30T18:49:08.722716Z","shell.execute_reply.started":"2022-01-30T18:49:08.113897Z","shell.execute_reply":"2022-01-30T18:49:08.721955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show off your video!\n\n* Green boxes are ground truth\n* Red boxes are model inference","metadata":{}},{"cell_type":"code","source":"from IPython.display import HTML\nfrom base64 import b64encode\n\ndef play(filename):\n    html = ''\n    video = open(filename,'rb').read()\n    src = 'data:video/mp4;base64,' + b64encode(video).decode()\n    html += '<video width=800 controls autoplay loop><source src=\"%s\" type=\"video/mp4\"></video>' % src \n    return HTML(html)\n\nplay('Video.mp4')","metadata":{"execution":{"iopub.status.busy":"2022-01-30T18:49:08.725230Z","iopub.execute_input":"2022-01-30T18:49:08.725507Z","iopub.status.idle":"2022-01-30T18:49:08.745153Z","shell.execute_reply.started":"2022-01-30T18:49:08.725478Z","shell.execute_reply":"2022-01-30T18:49:08.744020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleanup\n\n!rm *.avi\n!rm -r ffmpeg*","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-01-30T18:49:08.746156Z","iopub.execute_input":"2022-01-30T18:49:08.746470Z","iopub.status.idle":"2022-01-30T18:49:10.223825Z","shell.execute_reply.started":"2022-01-30T18:49:08.746354Z","shell.execute_reply":"2022-01-30T18:49:10.222885Z"},"trusted":true},"execution_count":null,"outputs":[]}]}