{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"},{"sourceId":1462296,"sourceType":"datasetVersion","datasetId":857191}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nImproved Kaggle Data Processor for YOLO Training\nProcesses ImageNet + COCO datasets and creates downloadable ZIP\n\"\"\"\n\nimport pandas as pd\nimport shutil\nimport os\nfrom PIL import Image\nfrom pycocotools.coco import COCO\nimport random\nimport zipfile\nfrom pathlib import Path\nimport json\nimport time\n\n# Đường dẫn dataset (Kaggle paths)\nIMAGENET_DIR = '/kaggle/input/imagenet-object-localization-challenge'\nCOCO_DIR = '/kaggle/input/coco-2017-dataset/coco2017'\nOUTPUT_DIR = '/kaggle/working/yolo_dataset'\n\n# Mapping classes cho YOLO\nIMAGENET_CLASSES = {\n    'n02992529': 1,  # mobile phone -> phone\n    'n04069434': 2,  # reflex camera\n    'n03976467': 3   # Polaroid camera\n}\n\nCOCO_CLASSES = {\n    1: 0,   # person (COCO ID: 1) -> YOLO ID: 0\n    77: 1   # cell phone (COCO ID: 77) -> YOLO ID: 1\n}\n\nCLASS_NAMES = ['person', 'phone', 'reflex_camera', 'polaroid_camera']\n\ndef setup_directories():\n    \"\"\"Tạo cấu trúc thư mục\"\"\"\n    print(\"📁 Setting up directories...\")\n    \n    dirs = [\n        Path(OUTPUT_DIR) / 'images' / 'train',\n        Path(OUTPUT_DIR) / 'images' / 'val',\n        Path(OUTPUT_DIR) / 'labels' / 'train', \n        Path(OUTPUT_DIR) / 'labels' / 'val',\n        Path(OUTPUT_DIR) / 'temp'\n    ]\n    \n    for dir_path in dirs:\n        dir_path.mkdir(parents=True, exist_ok=True)\n        print(f\"   Created: {dir_path}\")\n    \n    return True\n\ndef debug_imagenet_format():\n    \"\"\"Debug format của ImageNet annotations\"\"\"\n    print(\"🔍 Debugging ImageNet annotation format...\")\n    \n    try:\n        annotations_file = Path(IMAGENET_DIR) / 'LOC_train_solution.csv'\n        annotations = pd.read_csv(annotations_file)\n        \n        print(f\"📊 Total annotations: {len(annotations)}\")\n        print(f\"📋 Columns: {list(annotations.columns)}\")\n        \n        # Xem sample predictions\n        print(\"\\n📝 Sample PredictionStrings:\")\n        for i in range(min(10, len(annotations))):\n            pred_str = annotations.iloc[i]['PredictionString']\n            pred_parts = pred_str.split()\n            print(f\"   {i+1}: {pred_str}\")\n            print(f\"      Parts count: {len(pred_parts)}\")\n            print(f\"      Parts: {pred_parts[:8]}...\")  # First 8 parts\n        \n        # Tìm pattern với classes cần thiết\n        desired_classes = list(IMAGENET_CLASSES.keys())\n        for class_id in desired_classes:\n            matching = annotations[annotations['PredictionString'].str.contains(class_id, na=False)]\n            if len(matching) > 0:\n                sample = matching.iloc[0]['PredictionString']\n                parts = sample.split()\n                print(f\"\\n🎯 Sample for {class_id}:\")\n                print(f\"   Full string: {sample}\")\n                print(f\"   Parts: {parts}\")\n                \n                # Tìm vị trí của class_id\n                try:\n                    idx = parts.index(class_id)\n                    print(f\"   Class at index: {idx}\")\n                    if idx + 5 < len(parts):\n                        print(f\"   Next 5 values: {parts[idx+1:idx+6]}\")\n                except ValueError:\n                    print(f\"   Class not found in parts\")\n                break\n        \n        return True\n        \n    except Exception as e:\n        print(f\"❌ Debug failed: {e}\")\n        return False\n\ndef process_imagenet():\n    \"\"\"Xử lý ImageNet dataset với flexible parsing\"\"\"\n    print(\"\\n🖼️  Processing ImageNet dataset...\")\n    \n    class_counts = {0: 0, 1: 0, 2: 0, 3: 0}\n    processed_images = []\n    \n    try:\n        # Debug format trước\n        if not debug_imagenet_format():\n            return [], class_counts\n        \n        # Load annotations\n        annotations_file = Path(IMAGENET_DIR) / 'LOC_train_solution.csv'\n        annotations = pd.read_csv(annotations_file)\n        \n        # Lọc annotations \n        desired_classes = list(IMAGENET_CLASSES.keys())\n        pattern = '|'.join(desired_classes)\n        filtered_annotations = annotations[\n            annotations['PredictionString'].str.contains(pattern, na=False)\n        ]\n        \n        print(f\"📊 Found {len(filtered_annotations)} relevant annotations\")\n        \n        processed_count = 0\n        for idx, row in filtered_annotations.iterrows():\n            if processed_count % 100 == 0:\n                print(f\"   Processing: {processed_count}/{len(filtered_annotations)}\")\n            \n            image_id = row['ImageId']\n            predictions = row['PredictionString'].split()\n            \n            labels_for_image = []\n            image_copied = False\n            \n            # Parse predictions với flexible format\n            i = 0\n            while i < len(predictions):\n                if predictions[i] in desired_classes:\n                    try:\n                        class_id = predictions[i]\n                        \n                        # Determine format dựa trên số values còn lại\n                        remaining = len(predictions) - i - 1\n                        \n                        if remaining >= 5:\n                            # Format: class_id confidence xmin ymin xmax ymax\n                            confidence = float(predictions[i+1])\n                            coords = list(map(float, predictions[i+2:i+6]))\n                            i += 6\n                        elif remaining >= 4:\n                            # Format: class_id xmin ymin xmax ymax (no confidence)\n                            coords = list(map(float, predictions[i+1:i+5]))\n                            confidence = 0.9  # Default confidence\n                            i += 5\n                        elif remaining >= 4:\n                            # Format: class_id x_center y_center width height\n                            coords = list(map(float, predictions[i+1:i+5]))\n                            confidence = 0.9\n                            i += 5\n                        else:\n                            print(f\"   ⚠️  Not enough values for bbox: {remaining}\")\n                            i += 1\n                            continue\n                        \n                        # Handle different coordinate formats\n                        if len(coords) == 4:\n                            # Check if coords are normalized (0-1) or absolute\n                            if all(c <= 1.0 for c in coords):\n                                # Normalized format (x_center, y_center, width, height)\n                                x_center, y_center, width_norm, height_norm = coords\n                            else:\n                                # Absolute format (xmin, ymin, xmax, ymax)\n                                xmin, ymin, xmax, ymax = coords\n                                \n                                # Cần kích thước image để normalize\n                                image_path = Path(IMAGENET_DIR) / 'ILSVRC/Data/CLS-LOC/train' / class_id / f'{image_id}.JPEG'\n                                if not image_path.exists():\n                                    continue\n                                \n                                with Image.open(image_path) as img:\n                                    img_width, img_height = img.size\n                                \n                                x_center = (xmin + xmax) / 2 / img_width\n                                y_center = (ymin + ymax) / 2 / img_height\n                                width_norm = (xmax - xmin) / img_width\n                                height_norm = (ymax - ymin) / img_height\n                        else:\n                            print(f\"   ⚠️  Invalid coords length: {len(coords)}\")\n                            continue\n                        \n                        # Validate coordinates\n                        if (0 <= x_center <= 1 and 0 <= y_center <= 1 and \n                            0 < width_norm <= 1 and 0 < height_norm <= 1):\n                            \n                            # Copy image if not done yet\n                            if not image_copied:\n                                image_path = Path(IMAGENET_DIR) / 'ILSVRC/Data/CLS-LOC/train' / class_id / f'{image_id}.JPEG'\n                                if image_path.exists():\n                                    output_image_path = Path(OUTPUT_DIR) / 'temp' / f'imagenet_{image_id}.jpg'\n                                    shutil.copy(str(image_path), str(output_image_path))\n                                    image_copied = True\n                            \n                            yolo_class_id = IMAGENET_CLASSES[class_id]\n                            label = f\"{yolo_class_id} {x_center:.6f} {y_center:.6f} {width_norm:.6f} {height_norm:.6f}\"\n                            labels_for_image.append(label)\n                            class_counts[yolo_class_id] += 1\n                        \n                    except (ValueError, IndexError) as e:\n                        print(f\"   ⚠️  Error parsing prediction: {e}\")\n                        # Debug what we're trying to parse\n                        if i < len(predictions):\n                            debug_slice = predictions[i:min(i+8, len(predictions))]\n                            print(f\"      Trying to parse: {debug_slice}\")\n                        i += 1\n                else:\n                    i += 1\n            \n            # Save processed data\n            if labels_for_image and image_copied:\n                processed_images.append({\n                    'image_file': f'imagenet_{image_id}.jpg',\n                    'labels': labels_for_image,\n                    'source': 'imagenet'\n                })\n            \n            processed_count += 1\n                \n    except Exception as e:\n        print(f\"❌ Error processing ImageNet: {e}\")\n        import traceback\n        traceback.print_exc()\n    \n    print(f\"✅ ImageNet processed: {len(processed_images)} images\")\n    return processed_images, class_counts\n\ndef process_coco():\n    \"\"\"Xử lý COCO dataset\"\"\"\n    print(\"\\n🏷️  Processing COCO dataset...\")\n    \n    class_counts = {0: 0, 1: 0, 2: 0, 3: 0}\n    processed_images = []\n    \n    try:\n        # Load COCO annotations\n        ann_file = Path(COCO_DIR) / 'annotations/instances_train2017.json'\n        if not ann_file.exists():\n            print(f\"❌ COCO annotations not found: {ann_file}\")\n            return [], class_counts\n        \n        print(f\"📋 Loading COCO annotations...\")\n        coco = COCO(str(ann_file))\n        \n        # Lấy image IDs\n        person_img_ids = coco.getImgIds(catIds=[1])  # person\n        phone_img_ids = coco.getImgIds(catIds=[77])  # cell phone\n        \n        # Giới hạn số lượng để cân bằng\n        max_person_images = 3000\n        person_img_ids = random.sample(person_img_ids, min(len(person_img_ids), max_person_images))\n        \n        # Combine và remove duplicates\n        all_img_ids = list(set(person_img_ids + phone_img_ids))\n        random.shuffle(all_img_ids)\n        \n        print(f\"📊 Found {len(person_img_ids)} person images\")\n        print(f\"📊 Found {len(phone_img_ids)} phone images\") \n        print(f\"📊 Total unique images: {len(all_img_ids)}\")\n        \n        processed_count = 0\n        for img_id in all_img_ids:\n            if processed_count % 500 == 0:\n                print(f\"   Processing: {processed_count}/{len(all_img_ids)}\")\n            \n            try:\n                img_info = coco.loadImgs(img_id)[0]\n                \n                # Lấy annotations\n                ann_ids = coco.getAnnIds(imgIds=img_id, catIds=[1, 77])\n                anns = coco.loadAnns(ann_ids)\n                \n                if not anns:\n                    processed_count += 1\n                    continue\n                \n                # Kiểm tra hình ảnh tồn tại\n                image_path = Path(COCO_DIR) / 'train2017' / img_info['file_name']\n                if not image_path.exists():\n                    processed_count += 1\n                    continue\n                \n                # Sao chép hình ảnh\n                output_image_path = Path(OUTPUT_DIR) / 'temp' / f'coco_{img_info[\"file_name\"]}'\n                shutil.copy(str(image_path), str(output_image_path))\n                \n                # Tạo labels\n                labels_for_image = []\n                for ann in anns:\n                    if ann['category_id'] in COCO_CLASSES:\n                        bbox = ann['bbox']  # [x, y, width, height]\n                        x, y, w, h = bbox\n                        \n                        if w <= 0 or h <= 0:\n                            continue\n                        \n                        # YOLO format conversion\n                        x_center = (x + w/2) / img_info['width']\n                        y_center = (y + h/2) / img_info['height']\n                        width_norm = w / img_info['width']\n                        height_norm = h / img_info['height']\n                        \n                        # Clamp values\n                        x_center = max(0, min(1, x_center))\n                        y_center = max(0, min(1, y_center))\n                        width_norm = max(0, min(1, width_norm))\n                        height_norm = max(0, min(1, height_norm))\n                        \n                        yolo_class_id = COCO_CLASSES[ann['category_id']]\n                        label = f\"{yolo_class_id} {x_center:.6f} {y_center:.6f} {width_norm:.6f} {height_norm:.6f}\"\n                        labels_for_image.append(label)\n                        class_counts[yolo_class_id] += 1\n                \n                if labels_for_image:\n                    processed_images.append({\n                        'image_file': f'coco_{img_info[\"file_name\"]}',\n                        'labels': labels_for_image,\n                        'source': 'coco'\n                    })\n                \n            except Exception as e:\n                print(f\"   ⚠️  Error processing image {img_id}: {e}\")\n            \n            processed_count += 1\n                \n    except Exception as e:\n        print(f\"❌ Error processing COCO: {e}\")\n        import traceback\n        traceback.print_exc()\n    \n    print(f\"✅ COCO processed: {len(processed_images)} images\")\n    return processed_images, class_counts\n\ndef create_train_val_split(all_images, train_ratio=0.8):\n    \"\"\"Tạo train/val split\"\"\"\n    print(f\"\\n📂 Creating train/val split ({train_ratio:.0%} train)...\")\n    \n    random.shuffle(all_images)\n    split_idx = int(train_ratio * len(all_images))\n    \n    train_images = all_images[:split_idx]\n    val_images = all_images[split_idx:]\n    \n    print(f\"📊 Train: {len(train_images)} images\")\n    print(f\"📊 Val: {len(val_images)} images\")\n    \n    return train_images, val_images\n\ndef save_dataset(train_images, val_images):\n    \"\"\"Lưu dataset với cấu trúc YOLO\"\"\"\n    print(\"\\n💾 Saving dataset...\")\n    \n    # Tạo train set\n    for img_data in train_images:\n        # Copy image\n        src_path = Path(OUTPUT_DIR) / 'temp' / img_data['image_file']\n        dst_path = Path(OUTPUT_DIR) / 'images' / 'train' / img_data['image_file']\n        shutil.move(str(src_path), str(dst_path))\n        \n        # Save labels\n        label_file = dst_path.stem + '.txt'\n        label_path = Path(OUTPUT_DIR) / 'labels' / 'train' / label_file\n        \n        with open(label_path, 'w') as f:\n            f.write('\\n'.join(img_data['labels']) + '\\n')\n    \n    # Tạo val set\n    for img_data in val_images:\n        # Copy image\n        src_path = Path(OUTPUT_DIR) / 'temp' / img_data['image_file']\n        dst_path = Path(OUTPUT_DIR) / 'images' / 'val' / img_data['image_file']\n        shutil.move(str(src_path), str(dst_path))\n        \n        # Save labels\n        label_file = dst_path.stem + '.txt'\n        label_path = Path(OUTPUT_DIR) / 'labels' / 'val' / label_file\n        \n        with open(label_path, 'w') as f:\n            f.write('\\n'.join(img_data['labels']) + '\\n')\n    \n    # Clean temp directory\n    shutil.rmtree(Path(OUTPUT_DIR) / 'temp')\n    \n    print(\"✅ Dataset saved successfully\")\n\ndef create_config_files():\n    \"\"\"Tạo các file cấu hình cho YOLO\"\"\"\n    print(\"\\n⚙️  Creating configuration files...\")\n    \n    # Tạo data.yaml\n    yaml_content = f\"\"\"# YOLO Dataset Configuration\n# Generated from ImageNet + COCO\n\ntrain: images/train\nval: images/val\n\nnc: 4\nnames: {CLASS_NAMES}\n\n# Augmentation settings\nhsv_h: 0.015\nhsv_s: 0.7\nhsv_v: 0.4\ndegrees: 0.0\ntranslate: 0.1\nscale: 0.5\nshear: 0.0\nperspective: 0.0\nflipud: 0.0\nfliplr: 0.5\nmosaic: 1.0\nmixup: 0.0\n\"\"\"\n    \n    with open(Path(OUTPUT_DIR) / 'data.yaml', 'w') as f:\n        f.write(yaml_content)\n    \n    # Tạo README\n    readme_content = f\"\"\"# YOLO Dataset - ImageNet + COCO\n\n## Dataset Information\n- **Classes**: {len(CLASS_NAMES)}\n- **Class Names**: {', '.join(CLASS_NAMES)}\n- **Sources**: ImageNet Object Localization Challenge + COCO 2017\n- **Format**: YOLO (txt annotations)\n\n## Directory Structure\n```\nyolo_dataset/\n├── data.yaml              # Dataset configuration\n├── images/\n│   ├── train/            # Training images  \n│   └── val/              # Validation images\n├── labels/\n│   ├── train/            # Training labels (YOLO format)\n│   └── val/              # Validation labels (YOLO format)\n└── README.md             # This file\n```\n\n## Usage\n1. Extract this dataset\n2. Update paths in data.yaml if needed\n3. Train with: `yolo train data=data.yaml model=yolov8n.pt epochs=100`\n\n## Class Mapping\n- 0: person (from COCO)\n- 1: phone (from COCO cell_phone + ImageNet mobile_phone)  \n- 2: reflex_camera (from ImageNet)\n- 3: polaroid_camera (from ImageNet)\n\nGenerated on: {time.strftime('%Y-%m-%d %H:%M:%S')}\n\"\"\"\n    \n    with open(Path(OUTPUT_DIR) / 'README.md', 'w') as f:\n        f.write(readme_content)\n    \n    print(\"✅ Configuration files created\")\n\ndef create_zip_archive():\n    \"\"\"Tạo file ZIP để download\"\"\"\n    print(\"\\n📦 Creating ZIP archive...\")\n    \n    zip_path = '/kaggle/working/yolo_dataset.zip'\n    \n    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, dirs, files in os.walk(OUTPUT_DIR):\n            for file in files:\n                file_path = os.path.join(root, file)\n                arcname = os.path.relpath(file_path, OUTPUT_DIR)\n                zipf.write(file_path, arcname)\n                \n                if len(zipf.namelist()) % 1000 == 0:\n                    print(f\"   Added {len(zipf.namelist())} files to ZIP...\")\n    \n    # Thống kê ZIP\n    zip_size = os.path.getsize(zip_path) / (1024*1024)  # MB\n    print(f\"✅ ZIP created: {zip_path}\")\n    print(f\"📊 ZIP size: {zip_size:.1f} MB\")\n    \n    return zip_path\n\ndef print_final_stats(total_counts):\n    \"\"\"In thống kê cuối cùng\"\"\"\n    print(\"\\n\" + \"=\"*50)\n    print(\"📊 FINAL DATASET STATISTICS\")\n    print(\"=\"*50)\n    \n    total_instances = sum(total_counts.values())\n    \n    for class_id, count in total_counts.items():\n        percentage = count / total_instances * 100 if total_instances > 0 else 0\n        print(f\"{CLASS_NAMES[class_id]:>15}: {count:>6} instances ({percentage:.1f}%)\")\n    \n    print(f\"{'TOTAL':>15}: {total_instances:>6} instances\")\n    print(\"=\"*50)\n\ndef main():\n    \"\"\"Main processing function\"\"\"\n    print(\"🚀 KAGGLE YOLO DATASET PROCESSOR\")\n    print(\"=\"*50)\n    \n    start_time = time.time()\n    \n    try:\n        # Setup\n        setup_directories()\n        \n        # Process datasets\n        imagenet_images, imagenet_counts = process_imagenet()\n        coco_images, coco_counts = process_coco()\n        \n        # Combine results\n        all_images = imagenet_images + coco_images\n        total_counts = {k: imagenet_counts[k] + coco_counts[k] for k in imagenet_counts}\n        \n        if not all_images:\n            print(\"❌ No images processed! Check dataset paths and availability.\")\n            return\n        \n        print(f\"\\n🎯 Total processed: {len(all_images)} images\")\n        \n        # Create splits\n        train_images, val_images = create_train_val_split(all_images)\n        \n        # Save dataset\n        save_dataset(train_images, val_images)\n        \n        # Create config files  \n        create_config_files()\n        \n        # Create ZIP\n        zip_path = create_zip_archive()\n        \n        # Final statistics\n        print_final_stats(total_counts)\n        \n        elapsed_time = time.time() - start_time\n        print(f\"\\n⏱️  Total processing time: {elapsed_time:.1f} seconds\")\n        \n        print(f\"\\n🎉 SUCCESS! Dataset ready for download:\")\n        print(f\"📁 ZIP file: {zip_path}\")\n        print(f\"💾 Size: {os.path.getsize(zip_path)/(1024*1024):.1f} MB\")\n        \n        print(f\"\\n💡 Next steps:\")\n        print(\"1. Download the ZIP file from Kaggle\")\n        print(\"2. Extract and use with: yolo train data=data.yaml model=yolov8n.pt\")\n        \n    except Exception as e:\n        print(f\"\\n❌ Fatal error: {e}\")\n        import traceback\n        traceback.print_exc()\n\nif __name__ == \"__main__\":\n    main()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-07T12:05:02.487709Z","iopub.execute_input":"2025-08-07T12:05:02.4881Z","iopub.status.idle":"2025-08-07T12:08:25.681418Z","shell.execute_reply.started":"2025-08-07T12:05:02.488068Z","shell.execute_reply":"2025-08-07T12:08:25.680196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}