{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14315,"databundleVersionId":862230,"sourceType":"competition"},{"sourceId":9596805,"sourceType":"datasetVersion","datasetId":5854147},{"sourceId":11213840,"sourceType":"datasetVersion","datasetId":7002329},{"sourceId":11222089,"sourceType":"datasetVersion","datasetId":7008408}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics transformers sentencepiece\n!pip install ultralytics\n!pip install cvzone","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:39:39.770967Z","iopub.execute_input":"2025-03-31T04:39:39.771212Z","iopub.status.idle":"2025-03-31T04:40:11.416600Z","shell.execute_reply.started":"2025-03-31T04:39:39.771184Z","shell.execute_reply":"2025-03-31T04:40:11.415344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nimport matplotlib.pyplot as plt\nimport cv2\nimport cvzone\nimport math\nimport time\nimport os\n\n# Load YOLO model\nmodel = YOLO(\"../Yolo-Weights/yolov8n.pt\")\n\n# Video capture\ncap = cv2.VideoCapture(\"/kaggle/input/d/anandasau/object-detection-video-test/Huskey.mp4\")\n\n# Output folder for saving frames\noutput_folder = \"output_frames\"\nos.makedirs(output_folder, exist_ok=True)\n\nclassNames = [\"person\", \"bicycle\", \"car\", \"motorbike\", \"aeroplane\", \"bus\", \"train\", \"truck\", \"boat\",\n              \"traffic light\", \"fire hydrant\", \"stop sign\", \"parking meter\", \"bench\", \"bird\", \"cat\",\n              \"dog\", \"horse\", \"sheep\", \"cow\", \"elephant\", \"bear\", \"zebra\", \"giraffe\", \"backpack\", \"umbrella\",\n              \"handbag\", \"tie\", \"suitcase\", \"frisbee\", \"skis\", \"snowboard\", \"sports ball\", \"kite\", \"baseball bat\",\n              \"baseball glove\", \"skateboard\", \"surfboard\", \"tennis racket\", \"bottle\", \"wine glass\", \"cup\",\n              \"fork\", \"knife\", \"spoon\", \"bowl\", \"banana\", \"apple\", \"sandwich\", \"orange\", \"broccoli\",\n              \"carrot\", \"hot dog\", \"pizza\", \"donut\", \"cake\", \"chair\", \"sofa\", \"pottedplant\", \"bed\",\n              \"diningtable\", \"toilet\", \"tvmonitor\", \"laptop\", \"mouse\", \"remote\", \"keyboard\", \"cell phone\",\n              \"microwave\", \"oven\", \"toaster\", \"sink\", \"refrigerator\", \"book\", \"clock\", \"vase\", \"scissors\",\n              \"teddy bear\", \"hair drier\", \"toothbrush\"]\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:41:31.639963Z","iopub.execute_input":"2025-03-31T04:41:31.640598Z","iopub.status.idle":"2025-03-31T04:41:40.552847Z","shell.execute_reply.started":"2025-03-31T04:41:31.640559Z","shell.execute_reply":"2025-03-31T04:41:40.552130Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Video Detection","metadata":{}},{"cell_type":"code","source":"while True:\n    success, img = cap.read()\n    if not success:\n        break  # Break out of the loop if there are no more frames to read\n\n    # Perform object detection with YOLO\n    results = model(img, stream=True)\n    for r in results:\n        boxes = r.boxes\n        for box in boxes:\n            # Bounding Box\n            x1, y1, x2, y2 = box.xyxy[0]\n            x1, y1, x2, y2 = int(x1), int(y1), int(x2), int(y2)\n            w, h = x2 - x1, y2 - y1\n            cvzone.cornerRect(img, (x1, y1, w, h))\n            # Confidence\n            conf = math.ceil((box.conf[0] * 100)) / 100\n            # Class Name\n            cls = int(box.cls[0])\n            cvzone.putTextRect(img, f'{classNames[cls]} {conf}', (max(0, x1), max(35, y1)), scale=1, thickness=1)\n\n    # Save frame with YOLO detections applied\n    output_path = os.path.join(output_folder, f\"frame_{int(cap.get(cv2.CAP_PROP_POS_FRAMES)):06d}.jpg\")\n    cv2.imwrite(output_path, img)\n\n# Release video capture\ncap.release()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:45:48.468239Z","iopub.execute_input":"2025-03-31T04:45:48.469278Z","iopub.status.idle":"2025-03-31T04:45:59.358046Z","shell.execute_reply.started":"2025-03-31T04:45:48.469240Z","shell.execute_reply":"2025-03-31T04:45:59.357376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\n\noutput_folder = \"output_frames\"  # Change this to your folder path\noutput_video_path = \"output_video.mp4\"\n\nimg_array = []\nsize = None  # Initialize as None instead of 0\n\nfor filename in sorted(os.listdir(output_folder)):\n    if filename.endswith(\".jpg\"):\n        img_path = os.path.join(output_folder, filename)\n        img = cv2.imread(img_path)\n        if img is None:\n            continue  # Skip unreadable images\n\n        height, width, layers = img.shape\n        if size is None:  # Assign size once from the first valid image\n            size = (width, height)\n\n        img_array.append(img)\n\nif size is not None and img_array:  # Ensure we have valid images\n    out = cv2.VideoWriter(output_video_path, cv2.VideoWriter_fourcc(*'mp4v'), 30, size)\n    for img in img_array:\n        out.write(img)\n    out.release()\n    print(\"Video created successfully.\")\nelse:\n    print(\"No valid images found.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:47:59.534684Z","iopub.execute_input":"2025-03-31T04:47:59.535563Z","iopub.status.idle":"2025-03-31T04:48:02.168631Z","shell.execute_reply.started":"2025-03-31T04:47:59.535529Z","shell.execute_reply":"2025-03-31T04:48:02.167695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Object Detection with Generated Caption","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade ipywidgets jupyterlab-widgets\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:53:21.791548Z","iopub.execute_input":"2025-03-31T04:53:21.791910Z","iopub.status.idle":"2025-03-31T04:53:30.315102Z","shell.execute_reply.started":"2025-03-31T04:53:21.791856Z","shell.execute_reply":"2025-03-31T04:53:30.313934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import BlipProcessor, BlipForConditionalGeneration\nfrom PIL import Image\nimport shutil\n\n# Load the BLIP (Bootstrapping Language-Image Pre-training) model for captioning\nprocessor = BlipProcessor.from_pretrained(\"Salesforce/blip-image-captioning-base\")\nblip_model = BlipForConditionalGeneration.from_pretrained(\"Salesforce/blip-image-captioning-base\")\n\n# Path to the Open Images dataset\nimage_folder = \"/kaggle/input/open-images-2019-object-detection/test\"\n\n# Output folder to save images with predictions and bounding boxes\noutput_folder = \"predicted_images_with_captions\"\nos.makedirs(output_folder, exist_ok=True)\n\n# Function to perform object detection, draw bounding boxes, and generate image captions\ndef detect_draw_and_caption(image_path, model, output_folder, processor, blip_model):\n    # Read the image\n    img = cv2.imread(image_path)\n    \n    # Perform object detection with YOLO\n    results = model(img)\n    \n    # Loop through each result (there may be multiple objects detected)\n    for result in results:\n        boxes = result.boxes  # Get the bounding boxes\n        \n        for box in boxes:\n            # Get bounding box coordinates\n            x1, y1, x2, y2 = box.xyxy[0]  # These are the bounding box corner points\n            x1, y1, x2, y2 = int(x1), int(y1), int(x2), int(y2)\n            \n            # Get confidence and class id\n            conf = box.conf[0]\n            cls = int(box.cls[0])\n            \n            # Draw the bounding box\n            cv2.rectangle(img, (x1, y1), (x2, y2), (0, 255, 0), 2)\n            \n            # Get class name\n            class_name = model.names[cls]\n            \n            # Display class and confidence\n            label = f'{class_name} {conf:.2f}'\n            cv2.putText(img, label, (x1, y1 - 10), cv2.FONT_HERSHEY_SIMPLEX, 0.9, (255, 0, 0), 2)\n\n    # Generate caption using the BLIP model\n    pil_img = Image.open(image_path).convert(\"RGB\")\n    inputs = processor(pil_img, return_tensors=\"pt\")\n    caption = blip_model.generate(**inputs)\n    description = processor.decode(caption[0], skip_special_tokens=True)\n    \n    # Save the image with bounding boxes\n    output_path = os.path.join(output_folder, os.path.basename(image_path))\n    cv2.imwrite(output_path, img)\n    \n    # Display the image and its caption\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.title(f\"Caption: {description}\")\n    plt.show()\n\n# Example usage for a few images in the folder\nfor i, image_name in enumerate(os.listdir(image_folder)):\n    if image_name.endswith(\".jpg\") and i < 5:  # Change the number '5' to process more images\n        image_path = os.path.join(image_folder, image_name)\n        detect_draw_and_caption(image_path, model, output_folder, processor, blip_model)\n\n# Compress the output folder into a ZIP file for easy downloading\nshutil.make_archive(\"predicted_images_with_captions\", 'zip', output_folder)\n\nprint(\"Object detection and captioning completed, images saved and compressed for download.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T04:53:36.653526Z","iopub.execute_input":"2025-03-31T04:53:36.653977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = '/kaggle/input/video-detect/PetalsWebDesignerCushionChair_Beige_packof2.jpg'\ndetect_draw_and_caption(image_path, model, output_folder, processor, blip_model)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T05:02:48.603485Z","iopub.execute_input":"2025-03-31T05:02:48.604280Z","iopub.status.idle":"2025-03-31T05:02:52.564421Z","shell.execute_reply.started":"2025-03-31T05:02:48.604246Z","shell.execute_reply":"2025-03-31T05:02:52.563515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = '/kaggle/input/video-detect/107032274-1647540069295-gettyimages-1084167640-2018_10_13-n1_office_0312.jpeg'\ndetect_draw_and_caption(image_path, model, output_folder, processor, blip_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T05:03:33.601053Z","iopub.execute_input":"2025-03-31T05:03:33.601741Z","iopub.status.idle":"2025-03-31T05:03:37.687084Z","shell.execute_reply.started":"2025-03-31T05:03:33.601708Z","shell.execute_reply":"2025-03-31T05:03:37.686107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = '/kaggle/input/video-detect/img.jpeg'\ndetect_draw_and_caption(image_path, model, output_folder, processor, blip_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T05:05:31.528477Z","iopub.execute_input":"2025-03-31T05:05:31.529307Z","iopub.status.idle":"2025-03-31T05:05:35.351187Z","shell.execute_reply.started":"2025-03-31T05:05:31.529272Z","shell.execute_reply":"2025-03-31T05:05:35.350164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_path = '/kaggle/input/video-detect/images (2).jpeg'\ndetect_draw_and_caption(image_path, model, output_folder, processor, blip_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-31T05:06:05.218654Z","iopub.execute_input":"2025-03-31T05:06:05.219056Z","iopub.status.idle":"2025-03-31T05:06:08.157522Z","shell.execute_reply.started":"2025-03-31T05:06:05.219022Z","shell.execute_reply":"2025-03-31T05:06:08.156671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}