{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"},{"sourceId":1462296,"sourceType":"datasetVersion","datasetId":857191},{"sourceId":13260419,"sourceType":"datasetVersion","datasetId":8402892},{"sourceId":13285165,"sourceType":"datasetVersion","datasetId":7609993},{"sourceId":13291533,"sourceType":"datasetVersion","datasetId":8424096}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport shutil\nimport os\nfrom PIL import Image\nfrom pycocotools.coco import COCO","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-04T13:47:12.573562Z","iopub.execute_input":"2025-10-04T13:47:12.574324Z","iopub.status.idle":"2025-10-04T13:47:12.578126Z","shell.execute_reply.started":"2025-10-04T13:47:12.574294Z","shell.execute_reply":"2025-10-04T13:47:12.57724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-04T13:47:15.038651Z","iopub.execute_input":"2025-10-04T13:47:15.03892Z","iopub.status.idle":"2025-10-04T13:47:18.20586Z","shell.execute_reply.started":"2025-10-04T13:47:15.038897Z","shell.execute_reply":"2025-10-04T13:47:18.204862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport shutil\nimport os\nfrom PIL import Image\nfrom pycocotools.coco import COCO\nimport random\n\n\noutput_dir = '/kaggle/input/tp-finetunedatanew/finetune_ds'\n\n\n\nfrom ultralytics import YOLO\n\n# Load pre-trained model (có thể dùng yolo11n.pt, yolo11s.pt, yolo11m.pt)\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-04T13:48:04.132012Z","iopub.execute_input":"2025-10-04T13:48:04.132802Z","iopub.status.idle":"2025-10-04T13:48:07.447541Z","shell.execute_reply.started":"2025-10-04T13:48:04.132769Z","shell.execute_reply":"2025-10-04T13:48:07.446934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = YOLO(\"yolo11l.pt\")  # Dùng model lớn hơn để có độ chính xác cao hơn\n# Fine-tune với các tham số tối ưu\nresults = model.train(\n    data=os.path.join(output_dir, 'data.yaml'),\n    epochs=50,  # Tăng epochs\n    imgsz=640,\n    batch=16,\n    patience=10,\n    save=True,\n    device=0,  # GPU\n    workers=4,\n    lr0=0.001,  # Learning rate thấp hơn cho fine-tuning\n    lrf=0.01,\n    momentum=0.937,\n    weight_decay=0.0005,\n    warmup_epochs=3,\n    warmup_momentum=0.8,\n    box=0.05,  # Box loss gain\n    cls=0.5,   # Class loss gain\n    dfl=1.5,   # DFL loss gain\n    hsv_h=0.015,  # Image HSV-Hue augmentation\n    hsv_s=0.7,    # Image HSV-Saturation augmentation\n    hsv_v=0.4,    # Image HSV-Value augmentation\n    degrees=0.0,  # Image rotation\n    translate=0.1,  # Image translation\n    scale=0.5,    # Image scale\n    shear=0.0,    # Image shear\n    perspective=0.0,  # Image perspective\n    flipud=0.0,   # Image flip up-down\n    fliplr=0.5,   # Image flip left-right\n    mosaic=1.0,   # Mosaic augmentation\n    mixup=0.5,    # Mixup augmentation\n    copy_paste=0.1,  # Copy-paste augmentation\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-04T13:48:13.320667Z","iopub.execute_input":"2025-10-04T13:48:13.321327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validate model\nmetrics = model.val()\nprint(f\"\\nmAP50: {metrics.box.map50}\")\nprint(f\"mAP50-95: {metrics.box.map}\")\n\nprint(\"\\n=== Lưu mô hình và kết quả ===\")\n\n# Tạo thư mục lưu kết quả\nresults_dir = '/kaggle/working/model_results'\nos.makedirs(results_dir, exist_ok=True)\n\n# 1. Lưu model weights (file .pt)\nbest_model_path = '/kaggle/working/best_model.pt'\nmodel.save(best_model_path)\nprint(f\"Đã lưu best model tại: {best_model_path}\")\n\n# 2. Lưu model ở định dạng khác nhau\ntry:\n    # Export ONNX\n    onnx_path = '/kaggle/working/model.onnx'\n    model.export(format='onnx', dynamic=True)\n    print(f\"Đã export ONNX model\")\n    \n    # Export TorchScript\n    torchscript_path = '/kaggle/working/model.torchscript'\n    model.export(format='torchscript')\n    print(f\"Đã export TorchScript model\")\n    \nexcept Exception as e:\n    print(f\"Lỗi khi export model: {e}\")\n\n# 3. Lưu kết quả training\ntraining_results = {\n    'mAP50': float(metrics.box.map50) if metrics.box.map50 is not None else 0,\n    'mAP50-95': float(metrics.box.map) if metrics.box.map is not None else 0,\n    'epochs_trained': 50,\n    'classes': ['person', 'phone', 'reflex_camera', 'polaroid_camera'],\n    'class_counts': class_counts\n}\n\n# Lưu kết quả dưới dạng JSON\nimport json\nwith open('/kaggle/working/training_results.json', 'w') as f:\n    json.dump(training_results, f, indent=2)\nprint(\"Đã lưu kết quả training tại: /kaggle/working/training_results.json\")\n\n# 4. Lưu confusion matrix và các metrics chi tiết\ntry:\n    import matplotlib.pyplot as plt\n    \n    # Lưu confusion matrix\n    if hasattr(metrics, 'confusion_matrix') and metrics.confusion_matrix is not None:\n        plt.figure(figsize=(10, 8))\n        plt.imshow(metrics.confusion_matrix.matrix, cmap='Blues')\n        plt.title('Confusion Matrix')\n        plt.colorbar()\n        plt.savefig('/kaggle/working/confusion_matrix.png', dpi=300, bbox_inches='tight')\n        plt.close()\n        print(\"Đã lưu confusion matrix\")\n    \nexcept Exception as e:\n    print(f\"Lỗi khi lưu confusion matrix: {e}\")\n\n# 5. Copy training logs và charts từ runs/detect/train\nimport glob\ntry:\n    # Tìm thư mục runs mới nhất\n    train_dirs = glob.glob('/kaggle/working/runs/detect/train*')\n    if train_dirs:\n        latest_train_dir = max(train_dirs, key=os.path.getctime)\n        \n        # Copy các file quan trọng\n        important_files = ['results.png', 'confusion_matrix.png', 'results.csv', 'weights/best.pt', 'weights/last.pt']\n        \n        for file_pattern in important_files:\n            source_files = glob.glob(os.path.join(latest_train_dir, file_pattern))\n            for source_file in source_files:\n                if os.path.exists(source_file):\n                    filename = os.path.basename(source_file)\n                    if filename.endswith('.pt'):\n                        # Đổi tên weights để phân biệt\n                        if 'best' in filename:\n                            filename = 'yolo_best_weights.pt'\n                        elif 'last' in filename:\n                            filename = 'yolo_last_weights.pt'\n                    \n                    dest_file = os.path.join('/kaggle/working', filename)\n                    shutil.copy(source_file, dest_file)\n                    print(f\"Đã copy {filename}\")\n        \n        print(f\"Training results available in: {latest_train_dir}\")\n    \nexcept Exception as e:\n    print(f\"Lỗi khi copy training files: {e}\")\n\n# 6. Tạo file README với thông tin mô hình\nreadme_content = f\"\"\"# YOLO Fine-tuned Model\n\n## Model Information\n- Base Model: YOLOv11s\n- Classes: person, phone, reflex_camera, polaroid_camera\n- Training Epochs: 50\n- Image Size: 640x640\n\n## Performance Metrics\n- mAP50: {training_results['mAP50']:.4f}\n- mAP50-95: {training_results['mAP50-95']:.4f}\n\n## Class Distribution\n\"\"\"\n\nfor i, (class_id, count) in enumerate(class_counts.items()):\n    class_names = ['person', 'phone', 'reflex_camera', 'polaroid_camera']\n    readme_content += f\"- {class_names[i]}: {count} instances\\n\"\n\nreadme_content += f\"\"\"\n## Files Included\n- best_model.pt: Best model weights\n- model.onnx: ONNX format model\n- model.torchscript: TorchScript format model\n- training_results.json: Detailed training metrics\n- results.png: Training charts\n- confusion_matrix.png: Confusion matrix visualization\n\n## Usage\n```python\nfrom ultralytics import YOLO\nmodel = YOLO('best_model.pt')\nresults = model('image.jpg')\n```\n\"\"\"\n\nwith open('/kaggle/working/README.md', 'w') as f:\n    f.write(readme_content)\nprint(\"Đã tạo README.md\")\n\n# 7. Hiển thị danh sách tất cả files đã tạo\nprint(\"\\n=== Files đã tạo trong /kaggle/working/ ===\")\nworking_files = os.listdir('/kaggle/working/')\nfor file in sorted(working_files):\n    file_path = os.path.join('/kaggle/working/', file)\n    if os.path.isfile(file_path):\n        size = os.path.getsize(file_path) / (1024*1024)  # Size in MB\n        print(f\"{file} ({size:.2f} MB)\")\n\nprint(f\"\\nTổng cộng {len(working_files)} files/folders trong /kaggle/working/\")\nprint(\"Các file này sẽ có thể download được thông qua Kaggle Kernels Output!\")\n\n# Test model với 1 ảnh mẫu (nếu có)\ntry:\n    if len(all_images) > 0:\n        test_image_path = os.path.join(output_dir, 'images', all_images[0])\n        if os.path.exists(test_image_path):\n            print(f\"\\n=== Test model với ảnh mẫu ===\")\n            results = model(test_image_path)\n            print(f\"Detected {len(results[0].boxes)} objects in test image\")\n            \n            # Lưu kết quả detection\n            results[0].save('/kaggle/working/test_detection.jpg')\n            print(\"Đã lưu kết quả detection tại: /kaggle/working/test_detection.jpg\")\n            \nexcept Exception as e:\n    print(f\"Lỗi khi test model: {e}\")\n\nprint(\"\\n🎉 Hoàn thành! Model và tất cả files đã được lưu vào /kaggle/working/\")\nprint(\"Bạn có thể download chúng bằng lệnh: kaggle kernels output username/kernel-name -p ./\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport zipfile\nimport json\n\nprint(\"🚀 GIẢI PHÁP DOWNLOAD MODEL TỪ KAGGLE\\n\")\n\n# === GIẢI PHÁP 1: Copy và nén file nhỏ hơn ===\ndef solution_1_compress():\n    print(\"📦 GIẢI PHÁP 1: Nén file model\")\n    print(\"-\" * 50)\n    \n    # Tìm tất cả file .pt trong runs\n    model_files = []\n    for root, dirs, files in os.walk('/kaggle/working'):\n        for file in files:\n            if file.endswith('.pt'):\n                full_path = os.path.join(root, file)\n                size_mb = os.path.getsize(full_path) / (1024**2)\n                model_files.append((full_path, size_mb))\n                print(f\"Tìm thấy: {file} ({size_mb:.1f} MB)\")\n    \n    if not model_files:\n        print(\"❌ Không tìm thấy file .pt nào!\")\n        return\n    \n    # Copy file best.pt ra working directory\n    best_pt = None\n    for path, size in model_files:\n        if 'best.pt' in path:\n            best_pt = path\n            break\n    \n    if best_pt:\n        # Copy và nén\n        output_path = '/kaggle/working/best_model.pt'\n        shutil.copy(best_pt, output_path)\n        \n        # Tạo zip file\n        zip_path = '/kaggle/working/yolo_model.zip'\n        with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zf:\n            zf.write(output_path, 'best_model.pt')\n        \n        zip_size = os.path.getsize(zip_path) / (1024**2)\n        print(f\"\\n✅ Đã tạo: yolo_model.zip ({zip_size:.1f} MB)\")\n        print(f\"📥 Download file này từ Output tab\")\n\nsolution_1_compress()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài đặt thư viện Ultralytics\n!pip install -q ultralytics\n\nimport os\nimport random\nfrom PIL import Image, ImageDraw, ImageFont\nimport matplotlib.pyplot as plt\nfrom ultralytics import YOLO\nimport numpy as np\n\n# --- Load YOLOv8 model ---\n# Dùng mô hình YOLOv8 đã được huấn luyện trước với dữ liệu COCO\nmodel = YOLO('/kaggle/working/runs/detect/train/weights/best.pt')  # Sử dụng model YOLOv8 medium pre-trained\n\n# --- Load ảnh ---\n# Chọn một ảnh ngẫu nhiên từ bộ validation của COCO\nimg_path = '/kaggle/input/testimage/WIN_20250607_19_34_01_Pro.jpg'  # Cập nhật đúng đường dẫn ảnh nếu cần\nimg = Image.open(img_path).convert('RGB')\n\n# --- Chạy inference với YOLOv8 ---\nresults = model(img_path)[0]  # Sử dụng mô hình đã tải để chạy inference\nboxes = results.boxes.xyxy.cpu().numpy()  # Lấy toạ độ bounding boxes (x1, y1, x2, y2)\nclasses = results.boxes.cls.cpu().numpy().astype(int)  # Lấy lớp của các object\nconfidences = results.boxes.conf.cpu().numpy()  # Lấy độ tự tin của các prediction\n\n# --- Vẽ kết quả và đánh giá hành động ---\ndraw = ImageDraw.Draw(img)\nfont = ImageFont.load_default()\n\n# Tập hợp các box theo class\nby_cls = {}\nfor box, cls, conf in zip(boxes, classes, confidences):\n    by_cls.setdefault(cls, []).append((box, conf))\n\n# Kiểm tra hành động \"chụp ảnh màn hình\"\npersons = by_cls.get(0, [])  # Lớp person (ID=0)\nphones = by_cls.get(1, [])  # Lớp cell phone (ID=67)\n\ndetected = False\nfor p_box, _ in persons:\n    for ph_box, _ in phones:\n        # Kiểm tra xem điện thoại có gần màn hình (có thể bạn cần thêm \"screen\" nếu muốn)\n        cx, cy = (ph_box[0] + ph_box[2]) / 2, (ph_box[1] + ph_box[3]) / 2  # Tọa độ trung tâm của điện thoại\n        px, py = (p_box[0] + p_box[2]) / 2, (p_box[1] + p_box[3]) / 2  # Tọa độ trung tâm của người\n\n        # Kiểm tra xem điện thoại có nằm trong vùng tầm tay của người hay không\n        if p_box[0] < cx < p_box[2] and p_box[1] < cy < p_box[3]:\n            # Vẽ các bounding boxes\n            draw.rectangle(p_box.tolist(), outline='blue', width=2)\n            draw.rectangle(ph_box.tolist(), outline='green', width=2)\n            draw.text((ph_box[0], ph_box[1]-10), \"likely taking photo\", fill='green', font=font)\n            detected = True\n            break\n    if detected: break\n\nif detected:\n    print(\"Detected taking-photo action based on rule-based heuristic.\")\nelse:\n    print(\"No taking-photo action detected.\")\n\n# --- Hiển thị kết quả ---\nplt.figure(figsize=(8, 8))\nplt.imshow(img)\nplt.axis('off')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nAdvanced Screen Capture Detection Algorithm\nA Computer Vision Approach for Detecting Screen Photography Behavior\n\nThis algorithm implements a multi-stage detection pipeline combining:\n1. Object detection (YOLO)\n2. Spatial relationship analysis\n3. Pose estimation features\n4. Temporal consistency (for video)\n\"\"\"\n\nimport numpy as np\nimport cv2\nfrom scipy.spatial.distance import euclidean, cosine\nfrom scipy.stats import norm\nimport matplotlib.pyplot as plt\nfrom dataclasses import dataclass\nfrom typing import List, Tuple, Dict, Optional\nfrom collections import deque\nimport json\n\n@dataclass\nclass Detection:\n    \"\"\"Data structure for object detection results\"\"\"\n    bbox: List[float]  # [x1, y1, x2, y2]\n    confidence: float\n    class_id: int\n    class_name: str\n    center: Tuple[float, float]\n    area: float\n    aspect_ratio: float\n    \n    @classmethod\n    def from_yolo_box(cls, box, class_names):\n        \"\"\"Create Detection from YOLO box object\"\"\"\n        x1, y1, x2, y2 = box.xyxy[0].tolist()\n        class_id = int(box.cls)\n        \n        center = ((x1 + x2) / 2, (y1 + y2) / 2)\n        width = x2 - x1\n        height = y2 - y1\n        area = width * height\n        aspect_ratio = width / height if height > 0 else 0\n        \n        return cls(\n            bbox=[x1, y1, x2, y2],\n            confidence=float(box.conf),\n            class_id=class_id,\n            class_name=class_names[class_id],\n            center=center,\n            area=area,\n            aspect_ratio=aspect_ratio\n        )\n\nclass ScreenCaptureDetector:\n    \"\"\"\n    Advanced detector for screen capture behavior analysis\n    \n    Key innovations:\n    1. Multi-modal feature extraction\n    2. Probabilistic confidence scoring\n    3. Temporal consistency for video streams\n    4. Explainable AI outputs\n    \"\"\"\n    \n    def __init__(self, model_path: str, config: Optional[Dict] = None):\n        \"\"\"Initialize detector with YOLO model and configuration\"\"\"\n        from ultralytics import YOLO\n        \n        self.model = YOLO(model_path)\n        self.class_names = ['person', 'phone', 'reflex_camera', 'polaroid_camera']\n        \n        # Default configuration (can be tuned)\n        self.config = config or {\n            'spatial': {\n                'max_distance_ratio': 0.8,  # Max distance as ratio of person height\n                'optimal_distance_ratio': 0.4,  # Optimal holding distance\n                'min_overlap_ratio': 0.1,  # Minimum spatial overlap\n                'camera_height_range': (0.3, 0.8),  # Camera position relative to person\n            },\n            'geometric': {\n                'phone_tilt_range': (-30, 45),  # Degrees, negative = toward camera\n                'camera_angle_range': (15, 75),  # Viewing angle to phone\n                'min_phone_visibility': 0.3,  # Minimum visible phone area\n            },\n            'temporal': {\n                'window_size': 10,  # Frames for temporal analysis\n                'min_consistency': 0.7,  # Minimum temporal consistency\n            },\n            'weights': {\n                'spatial': 0.3,\n                'geometric': 0.25,\n                'pose': 0.25,\n                'context': 0.2\n            }\n        }\n        \n        # Temporal buffer for video analysis\n        self.temporal_buffer = deque(maxlen=self.config['temporal']['window_size'])\n        \n    def detect_screen_capture(self, image_path: str, \n                            visualize: bool = True,\n                            return_details: bool = True) -> Dict:\n        \"\"\"\n        Main detection pipeline\n        \n        Returns:\n            Dict containing:\n            - is_capturing: bool\n            - confidence: float (0-1)\n            - evidence: detailed analysis results\n            - visualization: annotated image (if requested)\n        \"\"\"\n        # Run YOLO detection\n        results = self.model(image_path, conf=0.2, verbose=False)\n        \n        if len(results[0].boxes) == 0:\n            return {\n                'is_capturing': False,\n                'confidence': 0.0,\n                'evidence': {'reason': 'No objects detected'},\n                'visualization': None\n            }\n        \n        # Parse detections\n        detections = [Detection.from_yolo_box(box, self.class_names) \n                     for box in results[0].boxes]\n        \n        # Group by class\n        detections_by_class = self._group_detections_by_class(detections)\n        \n        # Analyze screen capture behavior\n        analysis = self._analyze_capture_behavior(detections_by_class, image_path)\n        \n        # Visualize if requested\n        visualization = None\n        if visualize:\n            visualization = self._visualize_analysis(\n                image_path, detections, analysis\n            )\n        \n        # Compile results\n        result = {\n            'is_capturing': analysis['final_score'] > 0.5,\n            'confidence': analysis['final_score'],\n            'evidence': analysis if return_details else None,\n            'visualization': visualization\n        }\n        \n        # Update temporal buffer for video analysis\n        self.temporal_buffer.append(result)\n        \n        return result\n    \n    def _group_detections_by_class(self, detections: List[Detection]) -> Dict:\n        \"\"\"Group detections by class for easier analysis\"\"\"\n        groups = {name: [] for name in self.class_names}\n        for det in detections:\n            groups[det.class_name].append(det)\n        return groups\n    \n    def _analyze_capture_behavior(self, detections_by_class: Dict, \n                                 image_path: str) -> Dict:\n        \"\"\"\n        Core analysis algorithm combining multiple evidence sources\n        \"\"\"\n        analysis = {\n            'spatial_score': 0.0,\n            'geometric_score': 0.0,\n            'pose_score': 0.0,\n            'context_score': 0.0,\n            'final_score': 0.0,\n            'details': {}\n        }\n        \n        # Check prerequisites\n        persons = detections_by_class['person']\n        phones = detections_by_class['phone']\n        cameras = (detections_by_class['reflex_camera'] + \n                  detections_by_class['polaroid_camera'])\n        \n        if not persons or not phones:\n            analysis['details']['missing'] = 'Required objects not detected'\n            return analysis\n        \n        # Find best person-phone-camera combination\n        best_combo = None\n        best_score = 0.0\n        \n        for person in persons:\n            for phone in phones:\n                # Analyze with camera\n                if cameras:\n                    for camera in cameras:\n                        score, details = self._analyze_combination(\n                            person, phone, camera, with_camera=True\n                        )\n                        if score > best_score:\n                            best_score = score\n                            best_combo = (person, phone, camera)\n                            analysis['details'] = details\n                else:\n                    # Analyze without camera (phone self-camera)\n                    score, details = self._analyze_combination(\n                        person, phone, None, with_camera=False\n                    )\n                    if score > best_score:\n                        best_score = score\n                        best_combo = (person, phone, None)\n                        analysis['details'] = details\n        \n        # Calculate component scores\n        if best_combo:\n            analysis.update(self._calculate_component_scores(best_combo, analysis['details']))\n        \n        # Weighted final score\n        weights = self.config['weights']\n        analysis['final_score'] = (\n            weights['spatial'] * analysis['spatial_score'] +\n            weights['geometric'] * analysis['geometric_score'] +\n            weights['pose'] * analysis['pose_score'] +\n            weights['context'] * analysis['context_score']\n        )\n        \n        # Apply temporal consistency if available\n        if len(self.temporal_buffer) > 0:\n            analysis['final_score'] = self._apply_temporal_smoothing(\n                analysis['final_score']\n            )\n        \n        return analysis\n    \n    def _analyze_combination(self, person: Detection, phone: Detection, \n                           camera: Optional[Detection], with_camera: bool) -> Tuple[float, Dict]:\n        \"\"\"Analyze specific person-phone-camera combination\"\"\"\n        details = {\n            'with_camera': with_camera,\n            'spatial_relations': {},\n            'geometric_features': {},\n            'pose_indicators': {}\n        }\n        \n        # 1. Spatial relationship analysis\n        spatial_score = self._analyze_spatial_relations(\n            person, phone, camera, details['spatial_relations']\n        )\n        \n        # 2. Geometric configuration analysis\n        geometric_score = self._analyze_geometric_config(\n            person, phone, camera, details['geometric_features']\n        )\n        \n        # 3. Pose-based indicators\n        pose_score = self._analyze_pose_indicators(\n            person, phone, camera, details['pose_indicators']\n        )\n        \n        # Combined score\n        total_score = (spatial_score + geometric_score + pose_score) / 3\n        \n        return total_score, details\n    \n    def _analyze_spatial_relations(self, person: Detection, phone: Detection,\n                                 camera: Optional[Detection], details: Dict) -> float:\n        \"\"\"Analyze spatial relationships between objects\"\"\"\n        score = 0.0\n        \n        # Person-phone distance\n        person_height = person.bbox[3] - person.bbox[1]\n        phone_distance = euclidean(person.center, phone.center)\n        normalized_distance = phone_distance / person_height\n        \n        details['phone_distance_normalized'] = normalized_distance\n        \n        # Optimal distance scoring (Gaussian distribution)\n        optimal = self.config['spatial']['optimal_distance_ratio']\n        distance_score = norm.pdf(normalized_distance, optimal, 0.2) / norm.pdf(optimal, optimal, 0.2)\n        score += distance_score * 0.4\n        \n        # Phone position relative to person\n        phone_in_front = (\n            person.bbox[0] < phone.center[0] < person.bbox[2] and\n            phone.center[1] < person.center[1]  # Phone above center\n        )\n        details['phone_in_front'] = phone_in_front\n        if phone_in_front:\n            score += 0.3\n        \n        # Camera analysis if present\n        if camera:\n            # Camera-phone alignment\n            camera_to_phone = np.array(phone.center) - np.array(camera.center)\n            camera_to_person = np.array(person.center) - np.array(camera.center)\n            \n            # Check if camera points toward phone\n            cos_angle = np.dot(camera_to_phone, camera_to_person) / (\n                np.linalg.norm(camera_to_phone) * np.linalg.norm(camera_to_person)\n            )\n            alignment_score = max(0, cos_angle)\n            details['camera_alignment'] = alignment_score\n            score += alignment_score * 0.3\n        \n        return min(1.0, score)\n    \n    def _analyze_geometric_config(self, person: Detection, phone: Detection,\n                                 camera: Optional[Detection], details: Dict) -> float:\n        \"\"\"Analyze geometric configuration and angles\"\"\"\n        score = 0.0\n        \n        # Phone orientation (aspect ratio indicates orientation)\n        phone_vertical = phone.aspect_ratio < 0.7\n        details['phone_vertical'] = phone_vertical\n        if not phone_vertical:  # Horizontal phone more likely for viewing\n            score += 0.3\n        \n        # Estimate viewing angle\n        if camera:\n            # Vector from camera to phone\n            view_vector = np.array(phone.center) - np.array(camera.center)\n            view_angle = np.degrees(np.arctan2(view_vector[1], view_vector[0]))\n            \n            # Check if angle is suitable for screen capture\n            angle_range = self.config['geometric']['camera_angle_range']\n            if angle_range[0] <= abs(view_angle) <= angle_range[1]:\n                score += 0.4\n            details['viewing_angle'] = view_angle\n        \n        # Phone size relative to person (indicates distance)\n        person_area = (person.bbox[2] - person.bbox[0]) * (person.bbox[3] - person.bbox[1])\n        phone_relative_size = phone.area / person_area\n        details['phone_relative_size'] = phone_relative_size\n        \n        # Optimal relative size (not too close, not too far)\n        if 0.01 < phone_relative_size < 0.1:\n            score += 0.3\n        \n        return min(1.0, score)\n    \n    def _analyze_pose_indicators(self, person: Detection, phone: Detection,\n                                camera: Optional[Detection], details: Dict) -> float:\n        \"\"\"Analyze pose-based behavioral indicators\"\"\"\n        score = 0.0\n        \n        # Hand position estimation (simplified without keypoints)\n        # Assume hands are near phone if phone is in person's bounding box\n        phone_center_y = phone.center[1]\n        person_upper_body = person.bbox[1] + (person.bbox[3] - person.bbox[1]) * 0.4\n        \n        hands_raised = phone_center_y < person_upper_body\n        details['hands_raised'] = hands_raised\n        if hands_raised:\n            score += 0.4\n        \n        # Body orientation (simplified)\n        if camera:\n            # Check if person faces camera while phone faces away\n            person_to_camera = np.array(camera.center) - np.array(person.center)\n            person_to_phone = np.array(phone.center) - np.array(person.center)\n            \n            # Angle between vectors\n            cos_angle = np.dot(person_to_camera, person_to_phone) / (\n                np.linalg.norm(person_to_camera) * np.linalg.norm(person_to_phone) + 1e-6\n            )\n            \n            # Negative correlation expected (opposite directions)\n            if cos_angle < 0:\n                score += 0.3\n            details['body_phone_camera_angle'] = np.degrees(np.arccos(np.clip(cos_angle, -1, 1)))\n        \n        # Stability indicator (low confidence might indicate motion blur)\n        avg_confidence = (person.confidence + phone.confidence) / 2\n        if camera:\n            avg_confidence = (avg_confidence + camera.confidence) / 1.5\n        \n        details['detection_stability'] = avg_confidence\n        if avg_confidence > 0.7:\n            score += 0.3\n        \n        return min(1.0, score)\n    \n    def _calculate_component_scores(self, combo: Tuple, details: Dict) -> Dict:\n        \"\"\"Calculate individual component scores from analysis details\"\"\"\n        scores = {\n            'spatial_score': 0.0,\n            'geometric_score': 0.0,\n            'pose_score': 0.0,\n            'context_score': 0.0\n        }\n        \n        # Spatial score\n        if details.get('spatial_relations', {}).get('phone_in_front'):\n            scores['spatial_score'] += 0.5\n        if details.get('spatial_relations', {}).get('camera_alignment', 0) > 0.7:\n            scores['spatial_score'] += 0.5\n        \n        # Geometric score\n        if not details.get('geometric_features', {}).get('phone_vertical', True):\n            scores['geometric_score'] += 0.4\n        if 0.01 < details.get('geometric_features', {}).get('phone_relative_size', 0) < 0.1:\n            scores['geometric_score'] += 0.6\n        \n        # Pose score\n        if details.get('pose_indicators', {}).get('hands_raised'):\n            scores['pose_score'] += 0.5\n        if details.get('pose_indicators', {}).get('detection_stability', 0) > 0.7:\n            scores['pose_score'] += 0.5\n        \n        # Context score (additional evidence)\n        if combo[2] is not None:  # Has camera\n            scores['context_score'] += 0.6\n        if len(self.temporal_buffer) > 5:  # Consistent behavior over time\n            scores['context_score'] += 0.4\n        \n        return scores\n    \n    def _apply_temporal_smoothing(self, current_score: float) -> float:\n        \"\"\"Apply temporal smoothing for video sequences\"\"\"\n        if len(self.temporal_buffer) < 2:\n            return current_score\n        \n        # Get recent scores\n        recent_scores = [b['confidence'] for b in list(self.temporal_buffer)[-5:]]\n        recent_scores.append(current_score)\n        \n        # Weighted average with emphasis on recent frames\n        weights = np.exp(np.linspace(0, 1, len(recent_scores)))\n        weights /= weights.sum()\n        \n        smoothed = np.average(recent_scores, weights=weights)\n        return float(smoothed)\n    \n    def _visualize_analysis(self, image_path: str, detections: List[Detection],\n                          analysis: Dict) -> np.ndarray:\n        \"\"\"Create visualization with analysis results\"\"\"\n        img = cv2.imread(image_path)\n        img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Color scheme\n        colors = {\n            'person': (255, 0, 0),      # Red\n            'phone': (0, 255, 0),       # Green\n            'reflex_camera': (0, 0, 255), # Blue\n            'polaroid_camera': (255, 255, 0) # Yellow\n        }\n        \n        # Draw detections\n        for det in detections:\n            color = colors.get(det.class_name, (128, 128, 128))\n            x1, y1, x2, y2 = map(int, det.bbox)\n            \n            # Bounding box\n            cv2.rectangle(img_rgb, (x1, y1), (x2, y2), color, 2)\n            \n            # Label with confidence\n            label = f\"{det.class_name} {det.confidence:.2f}\"\n            label_size, _ = cv2.getTextSize(label, cv2.FONT_HERSHEY_SIMPLEX, 0.5, 2)\n            cv2.rectangle(img_rgb, (x1, y1-20), (x1+label_size[0], y1), color, -1)\n            cv2.putText(img_rgb, label, (x1, y1-5), cv2.FONT_HERSHEY_SIMPLEX, \n                       0.5, (255, 255, 255), 2)\n        \n        # Add analysis overlay\n        overlay = img_rgb.copy()\n        \n        # Draw connections and indicators\n        if analysis['details']:\n            # Add visual indicators based on analysis\n            # This is simplified - in practice, would add more sophisticated visualization\n            pass\n        \n        # Add analysis results panel\n        panel_height = 150\n        panel = np.ones((panel_height, img_rgb.shape[1], 3), dtype=np.uint8) * 255\n        \n        # Add text to panel\n        y_offset = 30\n        cv2.putText(panel, f\"Screen Capture Detection Results\", \n                   (10, y_offset), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (0, 0, 0), 2)\n        \n        y_offset += 30\n        status = \"DETECTED\" if analysis['final_score'] > 0.5 else \"NOT DETECTED\"\n        color = (0, 128, 0) if analysis['final_score'] > 0.5 else (128, 0, 0)\n        cv2.putText(panel, f\"Status: {status}\", \n                   (10, y_offset), cv2.FONT_HERSHEY_SIMPLEX, 0.6, color, 2)\n        \n        y_offset += 25\n        cv2.putText(panel, f\"Confidence: {analysis['final_score']:.1%}\", \n                   (10, y_offset), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 0, 0), 1)\n        \n        # Component scores\n        y_offset += 25\n        cv2.putText(panel, f\"Spatial: {analysis['spatial_score']:.1%} | \"\n                          f\"Geometric: {analysis['geometric_score']:.1%} | \"\n                          f\"Pose: {analysis['pose_score']:.1%} | \"\n                          f\"Context: {analysis['context_score']:.1%}\", \n                   (10, y_offset), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (0, 0, 0), 1)\n        \n        # Combine image and panel\n        result = np.vstack([img_rgb, panel])\n        \n        return result\n    \n    def evaluate_on_dataset(self, dataset_path: str, \n                          ground_truth_path: str) -> Dict:\n        \"\"\"\n        Evaluate algorithm performance on annotated dataset\n        For research paper evaluation\n        \"\"\"\n        from sklearn.metrics import precision_recall_fscore_support, roc_auc_score\n        import json\n        \n        # Load ground truth\n        with open(ground_truth_path, 'r') as f:\n            ground_truth = json.load(f)\n        \n        predictions = []\n        true_labels = []\n        confidence_scores = []\n        \n        # Process each image\n        for img_name, label in ground_truth.items():\n            img_path = f\"{dataset_path}/{img_name}\"\n            result = self.detect_screen_capture(img_path, visualize=False)\n            \n            predictions.append(result['is_capturing'])\n            true_labels.append(label)\n            confidence_scores.append(result['confidence'])\n        \n        # Calculate metrics\n        precision, recall, f1, _ = precision_recall_fscore_support(\n            true_labels, predictions, average='binary'\n        )\n        \n        auc = roc_auc_score(true_labels, confidence_scores)\n        \n        return {\n            'precision': precision,\n            'recall': recall,\n            'f1_score': f1,\n            'auc': auc,\n            'accuracy': np.mean(np.array(predictions) == np.array(true_labels))\n        }\n\n\n# === USAGE EXAMPLE ===\ndef main():\n    \"\"\"Example usage for research paper\"\"\"\n    \n    # Initialize detector\n    detector = ScreenCaptureDetector(\n        model_path='/kaggle/input/testimage/best_model.pt'\n    )\n    \n    # Single image analysis\n    result = detector.detect_screen_capture(\n        '/kaggle/input/testimage/WIN_20250607_19_34_01_Pro.jpg',\n        visualize=True,\n        return_details=True\n    )\n    \n    # Display results\n    print(\"=\" * 60)\n    print(\"SCREEN CAPTURE DETECTION RESULTS\")\n    print(\"=\" * 60)\n    print(f\"Detection: {'YES' if result['is_capturing'] else 'NO'}\")\n    print(f\"Confidence: {result['confidence']:.1%}\")\n    \n    if result['evidence']:\n        print(\"\\nDetailed Analysis:\")\n        print(f\"- Spatial Score: {result['evidence']['spatial_score']:.1%}\")\n        print(f\"- Geometric Score: {result['evidence']['geometric_score']:.1%}\")\n        print(f\"- Pose Score: {result['evidence']['pose_score']:.1%}\")\n        print(f\"- Context Score: {result['evidence']['context_score']:.1%}\")\n        \n        print(\"\\nKey Evidence:\")\n        details = result['evidence'].get('details', {})\n        for category, features in details.items():\n            if isinstance(features, dict):\n                print(f\"\\n{category.upper()}:\")\n                for key, value in features.items():\n                    print(f\"  - {key}: {value}\")\n    \n    # Visualize\n    if result['visualization'] is not None:\n        plt.figure(figsize=(15, 10))\n        plt.imshow(result['visualization'])\n        plt.axis('off')\n        plt.title('Screen Capture Detection Analysis')\n        plt.tight_layout()\n        plt.show()\n    \n    # For research evaluation\n    # metrics = detector.evaluate_on_dataset(\n    #     dataset_path='/path/to/test/images',\n    #     ground_truth_path='/path/to/annotations.json'\n    # )\n    # print(f\"\\nEvaluation Metrics:\")\n    # print(f\"- Precision: {metrics['precision']:.3f}\")\n    # print(f\"- Recall: {metrics['recall']:.3f}\")\n    # print(f\"- F1-Score: {metrics['f1_score']:.3f}\")\n    # print(f\"- AUC: {metrics['auc']:.3f}\")\n\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nSimplified Screen Capture Detection\nQuick implementation for practical use\n\"\"\"\n\nimport cv2\nimport numpy as np\nfrom ultralytics import YOLO\nimport matplotlib.pyplot as plt\n\nclass SimpleScreenCaptureDetector:\n    def __init__(self, model_path):\n        self.model = YOLO(model_path)\n        self.class_map = {\n            0: 'person',\n            1: 'phone', \n            2: 'reflex_camera',\n            3: 'polaroid_camera'\n        }\n    \n    def detect(self, image_path):\n        # Run detection\n        results = self.model(image_path, conf=0.25)\n        \n        if len(results[0].boxes) == 0:\n            return False, 0.0, \"No objects detected\"\n        \n        # Parse detections\n        detections = {'person': [], 'phone': [], 'camera': []}\n        \n        for box in results[0].boxes:\n            cls_id = int(box.cls)\n            cls_name = self.class_map[cls_id]\n            \n            if cls_name == 'person':\n                detections['person'].append(box)\n            elif cls_name == 'phone':\n                detections['phone'].append(box)\n            elif cls_name in ['reflex_camera', 'polaroid_camera']:\n                detections['camera'].append(box)\n        \n        # Check conditions\n        if not detections['person'] or not detections['phone']:\n            return False, 0.0, \"Missing person or phone\"\n        \n        # Simple rule-based detection\n        confidence = 0.0\n        reason = []\n        \n        # Check each person-phone pair\n        for person in detections['person']:\n            p_box = person.xyxy[0].cpu().numpy()\n            p_center = [(p_box[0] + p_box[2])/2, (p_box[1] + p_box[3])/2]\n            p_height = p_box[3] - p_box[1]\n            \n            for phone in detections['phone']:\n                ph_box = phone.xyxy[0].cpu().numpy()\n                ph_center = [(ph_box[0] + ph_box[2])/2, (ph_box[1] + ph_box[3])/2]\n                \n                # Distance check\n                distance = np.linalg.norm(np.array(p_center) - np.array(ph_center))\n                normalized_dist = distance / p_height\n                \n                # Phone in front and raised\n                phone_raised = ph_center[1] < p_center[1]\n                phone_in_range = 0.2 < normalized_dist < 0.8\n                \n                if phone_raised and phone_in_range:\n                    confidence = max(confidence, 0.6)\n                    reason.append(\"Phone held up\")\n                    \n                    # Bonus if camera present\n                    if detections['camera']:\n                        confidence = min(confidence + 0.3, 0.9)\n                        reason.append(\"Camera detected\")\n        \n        is_capturing = confidence > 0.5\n        reason_str = \", \".join(reason) if reason else \"No capture behavior detected\"\n        \n        return is_capturing, confidence, reason_str\n    \n    def visualize(self, image_path):\n        \"\"\"Quick visualization\"\"\"\n        results = self.model(image_path)\n        \n        # Get annotated image\n        annotated = results[0].plot()\n        \n        # Detect\n        is_capturing, confidence, reason = self.detect(image_path)\n        \n        # Add text overlay\n        status = \"SCREEN CAPTURE DETECTED\" if is_capturing else \"NO SCREEN CAPTURE\"\n        color = (0, 255, 0) if is_capturing else (0, 0, 255)\n        \n        cv2.putText(annotated, status, (10, 30), \n                   cv2.FONT_HERSHEY_SIMPLEX, 1, color, 2)\n        cv2.putText(annotated, f\"Confidence: {confidence:.0%}\", (10, 60),\n                   cv2.FONT_HERSHEY_SIMPLEX, 0.7, color, 2)\n        cv2.putText(annotated, reason, (10, 90),\n                   cv2.FONT_HERSHEY_SIMPLEX, 0.6, (255, 255, 255), 1)\n        \n        return annotated\n\n# Usage\ndetector = SimpleScreenCaptureDetector('/kaggle/input/testimage/best_model.pt')\n\n# Detect\nimage_path = '/kaggle/input/testimage/5e4c00bead0bcb001cd9294b.jpg'\nis_capturing, confidence, reason = detector.detect(image_path)\n\nprint(f\"Screen capture: {is_capturing}\")\nprint(f\"Confidence: {confidence:.0%}\")\nprint(f\"Reason: {reason}\")\n\n# Visualize\nannotated = detector.visualize(image_path)\nplt.figure(figsize=(12, 8))\nplt.imshow(cv2.cvtColor(annotated, cv2.COLOR_BGR2RGB))\nplt.axis('off')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đường dẫn dataset\nimagenet_dir = '/kaggle/input/imagenet-object-localization-challenge'\ncoco_dir = '/kaggle/input/coco-2017-dataset/coco2017'\noutput_dir = '/kaggle/working/dataset'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo thư mục output\nos.makedirs(os.path.join(output_dir, 'images'), exist_ok=True)\nos.makedirs(os.path.join(output_dir, 'labels'), exist_ok=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý ImageNet\nannotations_file = os.path.join(imagenet_dir, 'LOC_train_solution.csv')\nannotations = pd.read_csv(annotations_file)\ndesired_classes = ['n02992529', 'n04069434', 'n03976467']\nclass_map = {'n02992529': 1, 'n04069434': 2, 'n03976467': 3}  # phone, reflex_camera, polaroid_camera","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfiltered_annotations = annotations[annotations['PredictionString'].str.contains('|'.join(desired_classes))]\nfor idx, row in filtered_annotations.iterrows():\n    image_id = row['ImageId']\n    prediction = row['PredictionString'].split()\n    class_id = prediction[0]\n    if class_id not in desired_classes:\n        continue\n    xmin, ymin, xmax, ymax = map(float, prediction[1:5])\n    \n    # Đường dẫn hình ảnh\n    image_path = os.path.join(imagenet_dir, 'ILSVRC/Data/CLS-LOC/train', class_id, f'{image_id}.JPEG')\n    if not os.path.exists(image_path):\n        continue\n    \n    # Sao chép hình ảnh\n    output_image_path = os.path.join(output_dir, 'images', f'imagenet_{image_id}.jpg')\n    shutil.copy(image_path, output_image_path)\n    \n    # Lấy kích thước hình ảnh\n    with Image.open(image_path) as img:\n        image_width, image_height = img.size\n    \n    # Tạo label YOLO\n    x_center = (xmin + xmax) / 2 / image_width\n    y_center = (ymin + ymax) / 2 / image_height\n    width_norm = (xmax - xmin) / image_width\n    height_norm = (ymax - ymin) / image_height\n    our_class_id = class_map[class_id]\n    \n    label = f\"{our_class_id} {x_center:.6f} {y_center:.6f} {width_norm:.6f} {height_norm:.6f}\\n\"\n    \n    # Lưu label\n    with open(os.path.join(output_dir, 'labels', f'imagenet_{image_id}.txt'), 'w') as f:\n        f.write(label)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xử lý COCO\ncoco = COCO(os.path.join(coco_dir, 'annotations/instances_train2017.json'))\ndesired_category_ids = [1, 68]  # person, cell phone\ncategory_map = {1: 0, 68: 1}  # person: 0, cell phone: 1 (phone)\nimg_ids = coco.getImgIds(catIds=desired_category_ids)\n\nfor img_id in img_ids:\n    img_info = coco.loadImgs(img_id)[0]\n    ann_ids = coco.getAnnIds(imgIds=img_id, catIds=desired_category_ids)\n    anns = coco.loadAnns(ann_ids)\n    \n    # Sao chép hình ảnh\n    image_path = os.path.join(coco_dir, 'train2017', img_info['file_name'])\n    output_image_path = os.path.join(output_dir, 'images', f'coco_{img_info[\"file_name\"]}')\n    shutil.copy(image_path, output_image_path)\n    \n    # Tạo label YOLO\n    label_path = os.path.join(output_dir, 'labels', os.path.splitext(f'coco_{img_info[\"file_name\"]}')[0] + '.txt')\n    with open(label_path, 'w') as f:\n        for ann in anns:\n            category_id = ann['category_id']\n            if category_id in category_map:\n                our_class_id = category_map[category_id]\n                bbox = ann['bbox']  # [x, y, width, height]\n                x, y, w, h = bbox\n                img_width = img_info['width']\n                img_height = img_info['height']\n                x_center = (x + w / 2) / img_width\n                y_center = (y + h / 2) / img_height\n                width_norm = w / img_width\n                height_norm = h / img_height\n                f.write(f\"{our_class_id} {x_center:.6f} {y_center:.6f} {width_norm:.6f} {height_norm:.6f}\\n\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo train.txt\nimages = [os.path.join(output_dir, 'images', f) for f in os.listdir(os.path.join(output_dir, 'images')) if f.endswith('.jpg')]\nwith open(os.path.join(output_dir, 'train.txt'), 'w') as f:\n    for image in images:\n        f.write(image + '\\n')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo data.yaml\nyaml_content = f\"\"\"\ntrain: {os.path.join(output_dir, 'train.txt')}\nval: {os.path.join(output_dir, 'train.txt')}\nnc: 4\nnames: ['person', 'phone', 'reflex_camera', 'polaroid_camera']\n\"\"\"\nwith open(os.path.join(output_dir, 'data.yaml'), 'w') as f:\n    f.write(yaml_content)\n\n# Fine-tune YOLO11\nfrom ultralytics import YOLO\n\n# Load mô hình pre-trained\nmodel = YOLO(\"yolo11n.pt\")\n\n# Huấn luyện mô hình\nmodel.train(data=os.path.join(output_dir, 'data.yaml'), epochs=10, imgsz=640)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}