{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":128792,"databundleVersionId":15456921,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================\n# CELL 1: SETUP & CLEAN ENVIRONMENT\n# ============================================\n\n# ⚡ CRITICAL: Start with a clean slate\nimport os\nimport shutil\nimport subprocess\n\nWORK_DIR = \"/kaggle/working\"\n\n# Remove ALL existing files in working directory\nprint(\"🧹 Cleaning working directory...\")\nfor item in os.listdir(WORK_DIR):\n    item_path = os.path.join(WORK_DIR, item)\n    try:\n        if os.path.isdir(item_path):\n            shutil.rmtree(item_path)\n        else:\n            os.remove(item_path)\n    except:\n        pass\n\n# Clear cache directories\ncache_dirs = [\n    \"/root/.cache\",\n    \"/tmp/wandb\",\n    \"/root/.wandb\",\n]\nfor cache_dir in cache_dirs:\n    if os.path.exists(cache_dir):\n        shutil.rmtree(cache_dir, ignore_errors=True)\n\n# Check initial disk space\nprint(\"\\n📊 Initial Disk Space:\")\nsubprocess.run(['df', '-h', '/kaggle/working'])\n\n# Install required packages (minimal)\nprint(\"\\n📦 Installing packages...\")\n!pip install -q ultralytics --no-cache-dir\n\n# Disable wandb to save space\nos.environ['WANDB_DISABLED'] = 'true'\nos.environ['WANDB_MODE'] = 'disabled'\n\nprint(\"\\n✅ Setup complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:53:51.345253Z","iopub.execute_input":"2026-01-27T16:53:51.345441Z","iopub.status.idle":"2026-01-27T16:54:00.945368Z","shell.execute_reply.started":"2026-01-27T16:53:51.345422Z","shell.execute_reply":"2026-01-27T16:54:00.944245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 2: IMPORTS\n# ============================================\n\nimport os\nimport json\nimport cv2\nimport random\nimport shutil\nimport subprocess\nfrom pathlib import Path\nfrom collections import defaultdict\nfrom tqdm import tqdm\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom ultralytics import YOLO\n\n# Reproducibility\nrandom.seed(42)\nnp.random.seed(42)\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Constants\nKAGGLE_INPUT = \"/kaggle/input\"\nWORK_DIR = \"/kaggle/working\"\n\nprint(\"✅ All imports successful!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:54:00.945850Z","iopub.execute_input":"2026-01-27T16:54:00.946026Z","iopub.status.idle":"2026-01-27T16:54:14.392483Z","shell.execute_reply.started":"2026-01-27T16:54:00.946007Z","shell.execute_reply":"2026-01-27T16:54:14.391350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# ============================================\n# CELL 3: DISCOVER DATASET STRUCTURE\n# ============================================\n\ndef explore_dataset(root_path, max_depth=4):\n    \"\"\"Explore and print dataset structure\"\"\"\n    print(f\"📁 Exploring: {root_path}\\n\")\n    \n    def _explore(path, depth=0):\n        if depth >= max_depth:\n            return\n        \n        try:\n            items = sorted(os.listdir(path))\n        except PermissionError:\n            return\n        \n        indent = \"  \" * depth\n        \n        # Separate directories and files\n        dirs = [i for i in items if os.path.isdir(os.path.join(path, i))]\n        files = [i for i in items if os.path.isfile(os.path.join(path, i))]\n        \n        # Print directories\n        for d in dirs[:15]:\n            full_path = os.path.join(path, d)\n            n_items = len(os.listdir(full_path)) if os.path.isdir(full_path) else 0\n            print(f\"{indent}📂 {d}/ ({n_items} items)\")\n            _explore(full_path, depth + 1)\n        \n        if len(dirs) > 15:\n            print(f\"{indent}   ... and {len(dirs) - 15} more directories\")\n        \n        # Print files\n        for f in files[:10]:\n            size = os.path.getsize(os.path.join(path, f)) / 1024\n            unit = \"KB\"\n            if size > 1024:\n                size /= 1024\n                unit = \"MB\"\n            print(f\"{indent}📄 {f} ({size:.1f} {unit})\")\n        \n        if len(files) > 10:\n            print(f\"{indent}   ... and {len(files) - 10} more files\")\n    \n    _explore(root_path)\n\n# Find Vista dataset\nprint(\"🔍 Looking for Vista dataset...\\n\")\navailable_datasets = os.listdir(KAGGLE_INPUT)\nprint(f\"Available datasets: {available_datasets}\\n\")\n\nDATA_DIR = None\nfor ds in available_datasets:\n    if 'vista' in ds.lower():\n        DATA_DIR = os.path.join(KAGGLE_INPUT, ds)\n        break\n\nif DATA_DIR is None and available_datasets:\n    DATA_DIR = os.path.join(KAGGLE_INPUT, available_datasets[0])\n\nprint(f\"📂 Using: {DATA_DIR}\\n\")\nprint(\"=\" * 60)\nexplore_dataset(DATA_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:54:14.393570Z","iopub.execute_input":"2026-01-27T16:54:14.393887Z","iopub.status.idle":"2026-01-27T16:58:15.211105Z","shell.execute_reply.started":"2026-01-27T16:54:14.393867Z","shell.execute_reply":"2026-01-27T16:58:15.209924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 4: AUTO-DETECT & SET PATHS\n# ============================================\n\ndef find_all_paths(root_dir):\n    \"\"\"Automatically find all required paths\"\"\"\n    \n    paths = {\n        'train_images': None,\n        'test_images': None,\n        'val_images': None,\n        'train_ann': None,\n        'test_ann': None,\n        'val_ann': None,\n        'categories': None,\n    }\n    \n    json_files = []\n    image_dirs = []\n    \n    for root, dirs, files in os.walk(root_dir):\n        # Collect JSON files\n        for f in files:\n            if f.endswith('.json'):\n                json_files.append(os.path.join(root, f))\n        \n        # Check if directory contains images\n        img_count = sum(1 for f in files if f.lower().endswith(('.jpg', '.jpeg', '.png', '.bmp')))\n        if img_count > 0:\n            image_dirs.append((root, img_count))\n    \n    # Sort image directories by count (descending)\n    image_dirs.sort(key=lambda x: x[1], reverse=True)\n    \n    # Assign image directories\n    for img_dir, count in image_dirs:\n        dir_lower = img_dir.lower()\n        dir_name = os.path.basename(img_dir).lower()\n        parent_name = os.path.basename(os.path.dirname(img_dir)).lower()\n        \n        combined = f\"{parent_name}/{dir_name}\"\n        \n        if ('train' in dir_name or 'train' in parent_name) and paths['train_images'] is None:\n            paths['train_images'] = img_dir\n        elif ('test' in dir_name or 'test' in parent_name) and paths['test_images'] is None:\n            paths['test_images'] = img_dir\n        elif ('val' in dir_name or 'valid' in dir_name or \n              'val' in parent_name or 'valid' in parent_name) and paths['val_images'] is None:\n            paths['val_images'] = img_dir\n    \n    # Assign JSON files\n    for jf in json_files:\n        jf_lower = jf.lower()\n        fname = os.path.basename(jf_lower)\n        \n        if 'categor' in jf_lower:\n            paths['categories'] = jf\n        elif 'train' in jf_lower and 'ann' in jf_lower:\n            paths['train_ann'] = jf\n        elif 'test' in jf_lower and 'ann' in jf_lower:\n            paths['test_ann'] = jf\n        elif 'val' in jf_lower and 'ann' in jf_lower:\n            paths['val_ann'] = jf\n        elif 'train' in fname:\n            paths['train_ann'] = jf\n        elif 'test' in fname:\n            paths['test_ann'] = jf\n    \n    # Fallback: if no specific annotations found, look for instance files\n    if paths['train_ann'] is None:\n        for jf in json_files:\n            if 'instance' in jf.lower() and 'train' in jf.lower():\n                paths['train_ann'] = jf\n                break\n    \n    if paths['test_ann'] is None:\n        for jf in json_files:\n            if 'instance' in jf.lower() and 'test' in jf.lower():\n                paths['test_ann'] = jf\n                break\n    \n    return paths, json_files, image_dirs\n\n# Auto-detect paths\ndetected_paths, all_jsons, all_img_dirs = find_all_paths(DATA_DIR)\n\nprint(\"📋 AUTO-DETECTED PATHS:\")\nprint(\"=\" * 60)\nfor key, path in detected_paths.items():\n    if path:\n        if os.path.isdir(path):\n            count = len([f for f in os.listdir(path) if f.lower().endswith(('.jpg', '.jpeg', '.png'))])\n            print(f\"✅ {key}: {path} ({count} images)\")\n        else:\n            size = os.path.getsize(path) / (1024 * 1024)\n            print(f\"✅ {key}: {path} ({size:.2f} MB)\")\n    else:\n        print(f\"❌ {key}: Not found\")\n\nprint(\"\\n📄 All JSON files found:\")\nfor jf in all_jsons:\n    size = os.path.getsize(jf) / (1024 * 1024)\n    print(f\"   {jf} ({size:.2f} MB)\")\n\nprint(\"\\n📂 All image directories found:\")\nfor img_dir, count in all_img_dirs[:10]:\n    print(f\"   {img_dir} ({count} images)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:00:52.618453Z","iopub.execute_input":"2026-01-27T17:00:52.618785Z","iopub.status.idle":"2026-01-27T17:01:33.063095Z","shell.execute_reply.started":"2026-01-27T17:00:52.618765Z","shell.execute_reply":"2026-01-27T17:01:33.062119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 5: SET FINAL PATHS\n# ============================================\n\n# Use auto-detected paths\nTRAIN_IMG_DIR = detected_paths['train_images']\nTEST_IMG_DIR = detected_paths['test_images']\nVAL_IMG_DIR = detected_paths['val_images']\nTRAIN_ANN_FILE = detected_paths['train_ann']\nTEST_ANN_FILE = detected_paths['test_ann']\nCATEGORY_FILE = detected_paths['categories']\n\n# ============================================\n# ⚠️ MANUAL OVERRIDE SECTION ⚠️\n# If auto-detection failed, uncomment and set manually:\n# ============================================\n\n# TRAIN_IMG_DIR = \"/kaggle/input/vista26/train/images\"\n# TEST_IMG_DIR = \"/kaggle/input/vista26/test/images\"\n# VAL_IMG_DIR = \"/kaggle/input/vista26/validation/images\"\n# TRAIN_ANN_FILE = \"/kaggle/input/vista26/train/train_annotations.json\"\n# TEST_ANN_FILE = \"/kaggle/input/vista26/test/test_annotations.json\"\n# CATEGORY_FILE = \"/kaggle/input/vista26/category.json\"\n\n# If validation images are in a different location:\n# VAL_IMG_DIR = \"/kaggle/input/vista26/validation\"\n\n# ============================================\n\n# Verify final paths\nprint(\"🎯 FINAL PATHS:\")\nprint(\"=\" * 60)\npaths_to_verify = [\n    (\"Train Images\", TRAIN_IMG_DIR),\n    (\"Test Images\", TEST_IMG_DIR),\n    (\"Validation Images\", VAL_IMG_DIR),\n    (\"Train Annotations\", TRAIN_ANN_FILE),\n    (\"Test Annotations\", TEST_ANN_FILE),\n    (\"Categories\", CATEGORY_FILE),\n]\n\nall_valid = True\nfor name, path in paths_to_verify:\n    if path and os.path.exists(path):\n        if os.path.isdir(path):\n            count = len(os.listdir(path))\n            print(f\"✅ {name}: {path} ({count} items)\")\n        else:\n            print(f\"✅ {name}: {path}\")\n    else:\n        print(f\"❌ {name}: {path} - NOT FOUND!\")\n        all_valid = False\n\nif all_valid:\n    print(\"\\n✅ All paths verified successfully!\")\nelse:\n    print(\"\\n⚠️ Some paths are missing. Please set them manually in the override section above.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:01:45.908562Z","iopub.execute_input":"2026-01-27T17:01:45.908907Z","iopub.status.idle":"2026-01-27T17:01:45.938183Z","shell.execute_reply.started":"2026-01-27T17:01:45.908884Z","shell.execute_reply":"2026-01-27T17:01:45.937285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 6: LOAD & PARSE ANNOTATIONS\n# ============================================\n\ndef load_json_safe(filepath):\n    \"\"\"Safely load JSON file\"\"\"\n    if filepath and os.path.exists(filepath):\n        try:\n            with open(filepath, 'r') as f:\n                return json.load(f)\n        except Exception as e:\n            print(f\"⚠️ Error loading {filepath}: {e}\")\n    return None\n\ndef parse_categories(data):\n    \"\"\"Parse category data from various formats\"\"\"\n    if data is None:\n        return {}\n    \n    if isinstance(data, list):\n        # Format: [{\"id\": 1, \"name\": \"product\"}, ...]\n        return {item['id']: item['name'] for item in data if 'id' in item and 'name' in item}\n    \n    elif isinstance(data, dict):\n        if 'categories' in data:\n            # Format: {\"categories\": [...]}\n            return {item['id']: item['name'] for item in data['categories']}\n        else:\n            # Format: {\"1\": \"product\", ...} or {1: \"product\", ...}\n            return {int(k): v for k, v in data.items()}\n    \n    return {}\n\ndef parse_coco_annotations(data):\n    \"\"\"Parse COCO format annotations\"\"\"\n    if data is None:\n        return {}\n    \n    images = {}\n    \n    # Parse images\n    for img in data.get('images', []):\n        img_id = img['id']\n        images[img_id] = {\n            'id': img_id,\n            'file_name': img.get('file_name', f\"{img_id}.jpg\"),\n            'width': img.get('width', 0),\n            'height': img.get('height', 0),\n            'difficulty': img.get('difficulty', 'unknown'),\n            'annotations': []\n        }\n    \n    # Parse annotations\n    for ann in data.get('annotations', []):\n        img_id = ann.get('image_id')\n        if img_id in images:\n            images[img_id]['annotations'].append({\n                'id': ann.get('id', 0),\n                'category_id': ann['category_id'],\n                'bbox': ann['bbox'],  # [x, y, width, height]\n                'area': ann.get('area', 0),\n            })\n    \n    return images\n\n# Load categories\nprint(\"📦 Loading categories...\")\ncat_data = load_json_safe(CATEGORY_FILE)\ncategory_id_to_name = parse_categories(cat_data)\nprint(f\"   Found {len(category_id_to_name)} categories\")\n\nif category_id_to_name:\n    sample_cats = dict(list(category_id_to_name.items())[:5])\n    print(f\"   Sample: {sample_cats}\")\n\n# Load training annotations\nprint(\"\\n📊 Loading training annotations...\")\ntrain_data = load_json_safe(TRAIN_ANN_FILE)\ntrain_images = parse_coco_annotations(train_data)\nprint(f\"   Found {len(train_images)} training images\")\n\nif train_images:\n    sample = list(train_images.values())[0]\n    print(f\"   Sample image: {sample['file_name']} with {len(sample['annotations'])} objects\")\n\n# Load test annotations\nprint(\"\\n📊 Loading test annotations...\")\ntest_data = load_json_safe(TEST_ANN_FILE)\ntest_images = parse_coco_annotations(test_data)\nprint(f\"   Found {len(test_images)} test images\")\n\n# Count total objects\ntotal_train_objects = sum(len(img['annotations']) for img in train_images.values())\ntotal_test_objects = sum(len(img['annotations']) for img in test_images.values())\n\nprint(f\"\\n📈 Dataset Summary:\")\nprint(f\"   Training: {len(train_images)} images, {total_train_objects} objects\")\nprint(f\"   Test: {len(test_images)} images, {total_test_objects} objects\")\nprint(f\"   Categories: {len(category_id_to_name)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:01:50.906621Z","iopub.execute_input":"2026-01-27T17:01:50.906964Z","iopub.status.idle":"2026-01-27T17:01:53.134191Z","shell.execute_reply.started":"2026-01-27T17:01:50.906933Z","shell.execute_reply":"2026-01-27T17:01:53.132891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 7: CREATE YOLO DATASET WITH SYMLINKS\n# ⚡ THIS USES ALMOST ZERO DISK SPACE!\n# ============================================\n\nYOLO_DIR = os.path.join(WORK_DIR, \"yolo_dataset\")\n\ndef create_yolo_dataset_symlinks(images_dict, img_src_dir, split_name, get_dimensions=True):\n    \"\"\"\n    Create YOLO format dataset using SYMLINKS for images.\n    Only label files (.txt) are created - minimal disk usage!\n    \n    YOLO format: class_id x_center y_center width height (normalized 0-1)\n    \"\"\"\n    \n    img_dir = os.path.join(YOLO_DIR, \"images\", split_name)\n    lbl_dir = os.path.join(YOLO_DIR, \"labels\", split_name)\n    \n    os.makedirs(img_dir, exist_ok=True)\n    os.makedirs(lbl_dir, exist_ok=True)\n    \n    # Build category ID to YOLO class mapping (0-indexed)\n    all_category_ids = set()\n    for img_info in images_dict.values():\n        for ann in img_info['annotations']:\n            all_category_ids.add(ann['category_id'])\n    \n    cat_id_to_yolo = {cat_id: idx for idx, cat_id in enumerate(sorted(all_category_ids))}\n    \n    processed = 0\n    skipped = 0\n    \n    for img_id, img_info in tqdm(images_dict.items(), desc=f\"Creating {split_name}\"):\n        src_path = os.path.join(img_src_dir, img_info['file_name'])\n        \n        # Check if source exists\n        if not os.path.exists(src_path):\n            # Try without subdirectory\n            alt_path = os.path.join(img_src_dir, os.path.basename(img_info['file_name']))\n            if os.path.exists(alt_path):\n                src_path = alt_path\n            else:\n                skipped += 1\n                continue\n        \n        # Create SYMLINK for image (no disk space used!)\n        dst_img_path = os.path.join(img_dir, os.path.basename(img_info['file_name']))\n        if not os.path.exists(dst_img_path):\n            try:\n                os.symlink(src_path, dst_img_path)\n            except Exception as e:\n                skipped += 1\n                continue\n        \n        # Get image dimensions\n        img_w = img_info.get('width', 0)\n        img_h = img_info.get('height', 0)\n        \n        if (img_w == 0 or img_h == 0) and get_dimensions:\n            try:\n                img = cv2.imread(src_path)\n                if img is not None:\n                    img_h, img_w = img.shape[:2]\n            except:\n                skipped += 1\n                continue\n        \n        if img_w == 0 or img_h == 0:\n            skipped += 1\n            continue\n        \n        # Create label file (small text file)\n        label_filename = os.path.splitext(os.path.basename(img_info['file_name']))[0] + \".txt\"\n        label_path = os.path.join(lbl_dir, label_filename)\n        \n        with open(label_path, 'w') as f:\n            for ann in img_info['annotations']:\n                bbox = ann['bbox']  # [x_min, y_min, width, height]\n                x_min, y_min, bbox_w, bbox_h = bbox\n                \n                # Convert to YOLO format (center coordinates, normalized)\n                x_center = (x_min + bbox_w / 2) / img_w\n                y_center = (y_min + bbox_h / 2) / img_h\n                w_norm = bbox_w / img_w\n                h_norm = bbox_h / img_h\n                \n                # Clamp to valid range\n                x_center = max(0.001, min(0.999, x_center))\n                y_center = max(0.001, min(0.999, y_center))\n                w_norm = max(0.001, min(0.999, w_norm))\n                h_norm = max(0.001, min(0.999, h_norm))\n                \n                yolo_class = cat_id_to_yolo[ann['category_id']]\n                f.write(f\"{yolo_class} {x_center:.6f} {y_center:.6f} {w_norm:.6f} {h_norm:.6f}\\n\")\n        \n        processed += 1\n    \n    print(f\"   ✅ Processed: {processed}, Skipped: {skipped}\")\n    return cat_id_to_yolo\n\n# Create training set\nprint(\"\\n🔄 Creating YOLO training set (using symlinks)...\")\ncat_id_to_yolo = create_yolo_dataset_symlinks(train_images, TRAIN_IMG_DIR, \"train\")\n\n# Create validation set (from test)\nprint(\"\\n🔄 Creating YOLO validation set (using symlinks)...\")\n_ = create_yolo_dataset_symlinks(test_images, TEST_IMG_DIR, \"val\")\n\n# Reverse mapping for inference\nyolo_to_cat_id = {v: k for k, v in cat_id_to_yolo.items()}\n\n# Save mappings\nmapping_data = {\n    'cat_id_to_yolo': {str(k): v for k, v in cat_id_to_yolo.items()},\n    'yolo_to_cat_id': {str(k): v for k, v in yolo_to_cat_id.items()},\n    'category_names': {str(k): v for k, v in category_id_to_name.items()}\n}\n\nmapping_path = os.path.join(YOLO_DIR, \"mapping.json\")\nwith open(mapping_path, 'w') as f:\n    json.dump(mapping_data, f, indent=2)\n\nprint(f\"\\n✅ YOLO dataset created with {len(cat_id_to_yolo)} classes\")\nprint(f\"📁 Location: {YOLO_DIR}\")\n\n# Check disk usage\nprint(\"\\n📊 Disk usage after YOLO dataset creation:\")\nsubprocess.run(['df', '-h', '/kaggle/working'])\n\n# Show size of YOLO directory\nyolo_size = sum(os.path.getsize(os.path.join(dp, f)) \n                for dp, dn, fn in os.walk(YOLO_DIR) for f in fn)\nprint(f\"📦 YOLO dataset size: {yolo_size / (1024*1024):.2f} MB (mostly label files)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:02:03.397117Z","iopub.execute_input":"2026-01-27T17:02:03.397449Z","iopub.status.idle":"2026-01-27T17:06:10.022861Z","shell.execute_reply.started":"2026-01-27T17:02:03.397429Z","shell.execute_reply":"2026-01-27T17:06:10.021763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 8: CREATE YOLO CONFIG FILE\n# ============================================\n\nnum_classes = len(cat_id_to_yolo)\n\n# Create class names list (ordered by YOLO class ID)\nclass_names = []\nfor i in range(num_classes):\n    original_cat_id = yolo_to_cat_id[i]\n    name = category_id_to_name.get(original_cat_id, f\"class_{original_cat_id}\")\n    # Clean name for YAML\n    name = str(name).replace(\"'\", \"\").replace('\"', \"\").strip()\n    class_names.append(name)\n\n# Create YAML config\nyaml_content = f\"\"\"# Vista Dataset YOLO Configuration\n# Auto-generated for Vista'26 Competition\n\npath: {YOLO_DIR}\ntrain: images/train\nval: images/val\n\n# Number of classes\nnc: {num_classes}\n\n# Class names\nnames:\n\"\"\"\n\nfor i, name in enumerate(class_names):\n    yaml_content += f\"  {i}: {name}\\n\"\n\nyaml_path = os.path.join(YOLO_DIR, \"dataset.yaml\")\nwith open(yaml_path, 'w') as f:\n    f.write(yaml_content)\n\nprint(f\"✅ Created YOLO config: {yaml_path}\")\nprint(f\"📊 Number of classes: {num_classes}\")\nprint(f\"\\n📋 Config preview:\")\nprint(yaml_content[:500])\nif len(yaml_content) > 500:\n    print(\"...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:06:34.282421Z","iopub.execute_input":"2026-01-27T17:06:34.282748Z","iopub.status.idle":"2026-01-27T17:06:34.288595Z","shell.execute_reply.started":"2026-01-27T17:06:34.282726Z","shell.execute_reply":"2026-01-27T17:06:34.287891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 9: VISUALIZE SAMPLES (SANITY CHECK)\n# ============================================\n\ndef visualize_yolo_samples(yolo_dir, split=\"train\", n_samples=4):\n    \"\"\"Visualize YOLO format samples to verify correctness\"\"\"\n    \n    img_dir = os.path.join(yolo_dir, \"images\", split)\n    lbl_dir = os.path.join(yolo_dir, \"labels\", split)\n    \n    if not os.path.exists(img_dir):\n        print(f\"⚠️ Directory not found: {img_dir}\")\n        return\n    \n    # Get sample files\n    img_files = [f for f in os.listdir(img_dir) if f.lower().endswith(('.jpg', '.jpeg', '.png'))]\n    \n    if not img_files:\n        print(\"⚠️ No images found\")\n        return\n    \n    sample_files = random.sample(img_files, min(n_samples, len(img_files)))\n    \n    # Create plot\n    n_cols = 2\n    n_rows = (len(sample_files) + n_cols - 1) // n_cols\n    fig, axes = plt.subplots(n_rows, n_cols, figsize=(14, 7 * n_rows))\n    axes = axes.flatten() if n_rows > 1 else [axes] if n_samples == 1 else axes\n    \n    colors = plt.cm.tab20(np.linspace(0, 1, 20))\n    \n    for idx, img_file in enumerate(sample_files):\n        img_path = os.path.join(img_dir, img_file)\n        lbl_file = os.path.splitext(img_file)[0] + \".txt\"\n        lbl_path = os.path.join(lbl_dir, lbl_file)\n        \n        # Read image\n        img = cv2.imread(img_path)\n        if img is None:\n            continue\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        h, w = img.shape[:2]\n        \n        # Read labels\n        n_objects = 0\n        if os.path.exists(lbl_path):\n            with open(lbl_path, 'r') as f:\n                for line in f:\n                    parts = line.strip().split()\n                    if len(parts) >= 5:\n                        cls_id = int(parts[0])\n                        xc, yc, bw, bh = map(float, parts[1:5])\n                        \n                        # Convert back to pixel coordinates\n                        x1 = int((xc - bw/2) * w)\n                        y1 = int((yc - bh/2) * h)\n                        x2 = int((xc + bw/2) * w)\n                        y2 = int((yc + bh/2) * h)\n                        \n                        # Draw box\n                        color = tuple(int(c * 255) for c in colors[cls_id % 20][:3])\n                        cv2.rectangle(img, (x1, y1), (x2, y2), color, 2)\n                        cv2.putText(img, str(cls_id), (x1, y1 - 5),\n                                   cv2.FONT_HERSHEY_SIMPLEX, 0.6, color, 2)\n                        n_objects += 1\n        \n        ax = axes[idx]\n        ax.imshow(img)\n        ax.set_title(f\"{img_file}\\nObjects: {n_objects}\")\n        ax.axis('off')\n    \n    # Hide empty subplots\n    for idx in range(len(sample_files), len(axes)):\n        axes[idx].axis('off')\n    \n    plt.suptitle(f\"YOLO Dataset Verification - {split}\", fontsize=14, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig(os.path.join(WORK_DIR, f'verification_{split}.png'), dpi=100, bbox_inches='tight')\n    plt.show()\n    \n    print(f\"✅ Visualization saved!\")\n\n# Visualize training samples\nvisualize_yolo_samples(YOLO_DIR, \"train\", n_samples=4)\n\nprint(\"\\n⚠️ VERIFY: Bounding boxes should align correctly with objects!\")\nprint(\"   If boxes are wrong, check annotation parsing in Cell 6.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T17:06:40.550652Z","iopub.execute_input":"2026-01-27T17:06:40.550985Z","iopub.status.idle":"2026-01-27T17:06:43.874002Z","shell.execute_reply.started":"2026-01-27T17:06:40.550962Z","shell.execute_reply":"2026-01-27T17:06:43.872844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 10: CLEAN + QUIET YOLOv8 TRAINING\n# ============================================\n\nimport os, shutil, subprocess\nfrom ultralytics import YOLO\n\n# --- Hard cleanup (no mercy) ---\nshutil.rmtree(os.path.join(WORK_DIR, \"runs\"), ignore_errors=True)\nshutil.rmtree(\"/root/.cache/torch\", ignore_errors=True)\n\n# Optional but useful sanity check (no spam)\nsubprocess.run([\"df\", \"-h\", \"/kaggle/working\"], stdout=subprocess.DEVNULL)\n\n# --- Model ---\nmodel = YOLO(\"yolov8m.pt\")  # Correct choice for balance\n\n# --- Core config ---\nEPOCHS = 50\nIMG_SIZE = 640\nBATCH_SIZE = 16\n\n# --- Train ---\nmodel.train(\n    data=yaml_path,\n    epochs=EPOCHS,\n    imgsz=IMG_SIZE,\n    batch=BATCH_SIZE,\n\n    # 🔇 Silence & disk control\n    verbose=False,\n    plots=False,\n    save=True,              # Save best + last only\n    save_period=-1,         # No intermediate checkpoints\n    cache=False,\n\n    # 📁 Output control\n    project=WORK_DIR,\n    name=\"train\",\n    exist_ok=True,\n\n    # 🎯 Optimization\n    optimizer=\"AdamW\",\n    lr0=1e-3,\n    lrf=1e-2,\n    weight_decay=5e-4,\n    warmup_epochs=3,\n    patience=12,\n\n    # ⚡ Performance\n    amp=True,\n    workers=4,\n\n    # 🧠 Augmentation (kept realistic, not Kaggle-meme)\n    hsv_h=0.015,\n    hsv_s=0.7,\n    hsv_v=0.4,\n    degrees=10,\n    translate=0.1,\n    scale=0.5,\n    shear=2.0,\n    perspective=0.0001,\n    fliplr=0.5,\n    mosaic=1.0,\n    mixup=0.1\n)\n\n# Final disk check (quiet)\nsubprocess.run([\"df\", \"-h\", \"/kaggle/working\"], stdout=subprocess.DEVNULL)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 11: SAVE BEST MODEL & CLEANUP\n# ============================================\n\n# Find the best model\ntrain_dir = os.path.join(WORK_DIR, \"yolo_train\")\nbest_model_src = os.path.join(train_dir, \"weights\", \"best.pt\")\nlast_model_src = os.path.join(train_dir, \"weights\", \"last.pt\")\n\n# Determine which model to use\nif os.path.exists(best_model_src):\n    model_src = best_model_src\n    print(f\"✅ Found best.pt\")\nelif os.path.exists(last_model_src):\n    model_src = last_model_src\n    print(f\"⚠️ best.pt not found, using last.pt\")\nelse:\n    print(\"❌ No trained model found!\")\n    model_src = None\n\n# Save model to working directory root\nif model_src:\n    final_model_path = os.path.join(WORK_DIR, \"best_model.pt\")\n    shutil.copy2(model_src, final_model_path)\n    model_size = os.path.getsize(final_model_path) / (1024 * 1024)\n    print(f\"✅ Saved model: {final_model_path} ({model_size:.1f} MB)\")\n\n# Also save the mapping file\nfinal_mapping_path = os.path.join(WORK_DIR, \"category_mapping.json\")\nshutil.copy2(mapping_path, final_mapping_path)\nprint(f\"✅ Saved mapping: {final_mapping_path}\")\n\n# ⚡ AGGRESSIVE CLEANUP to free space\nprint(\"\\n🧹 Cleaning up to free space...\")\n\n# Remove training artifacts (keep only final model)\ncleanup_paths = [\n    os.path.join(WORK_DIR, \"yolo_train\"),\n    os.path.join(WORK_DIR, \"yolo_dataset\"),\n    \"/root/.cache\",\n]\n\nfor path in cleanup_paths:\n    if os.path.exists(path):\n        size = sum(os.path.getsize(os.path.join(dp, f)) \n                  for dp, dn, fn in os.walk(path) for f in fn)\n        shutil.rmtree(path, ignore_errors=True)\n        print(f\"   🗑️ Removed: {path} ({size/(1024*1024):.1f} MB)\")\n\n# Final disk check\nprint(\"\\n📊 Disk space after cleanup:\")\nsubprocess.run(['df', '-h', '/kaggle/working'])\n\n# List remaining files\nprint(\"\\n📁 Files in working directory:\")\nfor item in os.listdir(WORK_DIR):\n    path = os.path.join(WORK_DIR, item)\n    if os.path.isfile(path):\n        size = os.path.getsize(path) / (1024 * 1024)\n        print(f\"   📄 {item} ({size:.2f} MB)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.223180Z","iopub.status.idle":"2026-01-27T16:58:15.223757Z","shell.execute_reply.started":"2026-01-27T16:58:15.223267Z","shell.execute_reply":"2026-01-27T16:58:15.223275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 12: LOAD MODEL FOR INFERENCE\n# ============================================\n\nfrom ultralytics import YOLO\n\n# Load the trained model\nfinal_model_path = os.path.join(WORK_DIR, \"best_model.pt\")\n\nif os.path.exists(final_model_path):\n    model = YOLO(final_model_path)\n    print(f\"✅ Loaded model: {final_model_path}\")\nelse:\n    print(\"❌ Model not found! Using pretrained model...\")\n    model = YOLO(\"yolov8m.pt\")\n\n# Load category mapping\nfinal_mapping_path = os.path.join(WORK_DIR, \"category_mapping.json\")\nif os.path.exists(final_mapping_path):\n    with open(final_mapping_path, 'r') as f:\n        mapping_data = json.load(f)\n    yolo_to_cat_id = {int(k): v for k, v in mapping_data['yolo_to_cat_id'].items()}\n    print(f\"✅ Loaded mapping with {len(yolo_to_cat_id)} classes\")\nelse:\n    print(\"⚠️ Mapping not found, using identity mapping\")\n    yolo_to_cat_id = {i: i for i in range(100)}\n\nprint(f\"\\n📋 Sample mapping (YOLO → Original):\")\nfor k, v in list(yolo_to_cat_id.items())[:5]:\n    print(f\"   {k} → {v}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.224577Z","iopub.status.idle":"2026-01-27T16:58:15.224840Z","shell.execute_reply.started":"2026-01-27T16:58:15.224674Z","shell.execute_reply":"2026-01-27T16:58:15.224683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 13: FIND VALIDATION IMAGES\n# ============================================\n\n# Find validation images directory\npossible_val_dirs = [\n    VAL_IMG_DIR,\n    os.path.join(DATA_DIR, \"validation\", \"images\"),\n    os.path.join(DATA_DIR, \"validation\"),\n    os.path.join(DATA_DIR, \"val\", \"images\"),\n    os.path.join(DATA_DIR, \"val\"),\n    os.path.join(DATA_DIR, \"images\", \"validation\"),\n]\n\nVAL_IMG_DIR_FINAL = None\nfor path in possible_val_dirs:\n    if path and os.path.exists(path):\n        # Check if it contains images\n        files = os.listdir(path)\n        img_count = sum(1 for f in files if f.lower().endswith(('.jpg', '.jpeg', '.png')))\n        if img_count > 0:\n            VAL_IMG_DIR_FINAL = path\n            print(f\"✅ Found validation images: {path} ({img_count} images)\")\n            break\n\nif VAL_IMG_DIR_FINAL is None:\n    print(\"❌ Validation images not found!\")\n    print(\"\\n🔍 Searching entire dataset...\")\n    \n    # Search for any directory named 'validation' or 'val'\n    for root, dirs, files in os.walk(DATA_DIR):\n        dir_name = os.path.basename(root).lower()\n        if 'val' in dir_name:\n            img_count = sum(1 for f in files if f.lower().endswith(('.jpg', '.jpeg', '.png')))\n            if img_count > 0:\n                VAL_IMG_DIR_FINAL = root\n                print(f\"✅ Found: {root} ({img_count} images)\")\n                break\n\n# Get list of validation images\nif VAL_IMG_DIR_FINAL:\n    val_image_files = sorted([f for f in os.listdir(VAL_IMG_DIR_FINAL) \n                              if f.lower().endswith(('.jpg', '.jpeg', '.png'))])\n    print(f\"\\n📷 Total validation images: {len(val_image_files)}\")\n    print(f\"   Sample files: {val_image_files[:5]}\")\nelse:\n    print(\"❌ Could not find validation images!\")\n    val_image_files = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.225223Z","iopub.status.idle":"2026-01-27T16:58:15.225425Z","shell.execute_reply.started":"2026-01-27T16:58:15.225313Z","shell.execute_reply":"2026-01-27T16:58:15.225321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 14: RUN INFERENCE\n# ============================================\n\ndef extract_image_id(filename):\n    \"\"\"Extract numeric image ID from filename\"\"\"\n    # Try different patterns\n    name = os.path.splitext(filename)[0]\n    \n    # Pattern 1: Just numbers (e.g., \"123.jpg\")\n    if name.isdigit():\n        return int(name)\n    \n    # Pattern 2: Prefix + numbers (e.g., \"image_123.jpg\", \"img123.jpg\")\n    digits = ''.join(filter(str.isdigit, name))\n    if digits:\n        return int(digits)\n    \n    # Fallback: hash of filename\n    return hash(name) % 1000000\n\ndef run_inference(model, img_dir, image_files, yolo_to_cat_mapping, \n                  conf_threshold=0.25, iou_threshold=0.5, img_size=640):\n    \"\"\"Run inference on validation images\"\"\"\n    \n    predictions = []\n    \n    for img_file in tqdm(image_files, desc=\"🔮 Running inference\"):\n        img_path = os.path.join(img_dir, img_file)\n        \n        if not os.path.exists(img_path):\n            continue\n        \n        # Extract image ID\n        img_id = extract_image_id(img_file)\n        \n        # Run prediction\n        try:\n            results = model.predict(\n                img_path,\n                conf=conf_threshold,\n                iou=iou_threshold,\n                imgsz=img_size,\n                verbose=False,\n                device=0  # Use GPU\n            )[0]\n        except Exception as e:\n            print(f\"⚠️ Error on {img_file}: {e}\")\n            predictions.append({'image_id': img_id, 'categories': []})\n            continue\n        \n        # Extract predicted categories\n        pred_categories = []\n        \n        if len(results.boxes) > 0:\n            for cls_id in results.boxes.cls.cpu().numpy():\n                # Map YOLO class ID to original category ID\n                original_cat = yolo_to_cat_mapping.get(int(cls_id), int(cls_id))\n                pred_categories.append(int(original_cat))\n        \n        # Sort categories as required\n        pred_categories = sorted(pred_categories)\n        \n        predictions.append({\n            'image_id': img_id,\n            'categories': pred_categories\n        })\n    \n    # Sort by image_id\n    predictions = sorted(predictions, key=lambda x: x['image_id'])\n    \n    return predictions\n\n# Run inference\nif val_image_files:\n    print(f\"\\n🔮 Running inference on {len(val_image_files)} validation images...\")\n    print(f\"   Confidence threshold: 0.25\")\n    print(f\"   IOU threshold: 0.5\")\n    print(f\"   Image size: 640\")\n    \n    predictions = run_inference(\n        model=model,\n        img_dir=VAL_IMG_DIR_FINAL,\n        image_files=val_image_files,\n        yolo_to_cat_mapping=yolo_to_cat_id,\n        conf_threshold=0.25,\n        iou_threshold=0.5,\n        img_size=640\n    )\n    \n    print(f\"\\n✅ Generated {len(predictions)} predictions\")\n    \n    # Show sample predictions\n    print(\"\\n📋 Sample predictions:\")\n    for pred in predictions[:10]:\n        print(f\"   Image {pred['image_id']}: {pred['categories']}\")\nelse:\n    print(\"❌ No validation images to process!\")\n    predictions = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.225881Z","iopub.status.idle":"2026-01-27T16:58:15.226521Z","shell.execute_reply.started":"2026-01-27T16:58:15.225976Z","shell.execute_reply":"2026-01-27T16:58:15.225985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 15: CREATE SUBMISSION FILE\n# ============================================\n\ndef create_submission(predictions, output_path):\n    \"\"\"\n    Create submission CSV in required format:\n    \n    image_id,categories\n    1,\"[12,13,16]\"\n    2,\"[2,4,10,11]\"\n    \"\"\"\n    \n    rows = []\n    \n    for pred in predictions:\n        img_id = pred['image_id']\n        categories = pred['categories']\n        \n        # Convert to JSON string format\n        categories_str = json.dumps(categories)\n        \n        rows.append({\n            'image_id': img_id,\n            'categories': categories_str\n        })\n    \n    # Create DataFrame\n    df = pd.DataFrame(rows)\n    \n    # Ensure sorted by image_id\n    df = df.sort_values('image_id').reset_index(drop=True)\n    \n    # Save to CSV\n    df.to_csv(output_path, index=False)\n    \n    return df\n\n# Generate submission\nif predictions:\n    submission_path = os.path.join(WORK_DIR, \"submission.csv\")\n    submission_df = create_submission(predictions, submission_path)\n    \n    print(f\"✅ Submission saved: {submission_path}\")\n    print(f\"📊 Shape: {submission_df.shape}\")\n    print(f\"\\n📋 Submission preview:\")\n    print(submission_df.head(15))\n    \n    # Save file size\n    file_size = os.path.getsize(submission_path) / 1024\n    print(f\"\\n📦 File size: {file_size:.2f} KB\")\nelse:\n    print(\"❌ No predictions to submit!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.227156Z","iopub.status.idle":"2026-01-27T16:58:15.227930Z","shell.execute_reply.started":"2026-01-27T16:58:15.227260Z","shell.execute_reply":"2026-01-27T16:58:15.227269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 16: VALIDATE SUBMISSION\n# ============================================\n\ndef validate_submission(df):\n    \"\"\"Validate submission format\"\"\"\n    \n    print(\"🔍 Validating submission...\")\n    errors = []\n    warnings = []\n    \n    # Check 1: Required columns\n    required_cols = ['image_id', 'categories']\n    for col in required_cols:\n        if col not in df.columns:\n            errors.append(f\"Missing column: {col}\")\n    \n    if errors:\n        for e in errors:\n            print(f\"   ❌ {e}\")\n        return False\n    \n    # Check 2: No missing values\n    if df['image_id'].isna().any():\n        errors.append(\"Found NaN in image_id\")\n    if df['categories'].isna().any():\n        errors.append(\"Found NaN in categories\")\n    \n    # Check 3: No duplicate image_ids\n    duplicates = df['image_id'].duplicated().sum()\n    if duplicates > 0:\n        errors.append(f\"Found {duplicates} duplicate image_ids\")\n    \n    # Check 4: Image IDs are sorted\n    is_sorted = df['image_id'].is_monotonic_increasing\n    if not is_sorted:\n        warnings.append(\"Image IDs are not sorted (will be auto-sorted)\")\n    \n    # Check 5: Categories are valid JSON and sorted\n    for idx, row in df.iterrows():\n        try:\n            cats = json.loads(row['categories'])\n            if not isinstance(cats, list):\n                errors.append(f\"Row {idx}: categories is not a list\")\n            elif cats != sorted(cats):\n                warnings.append(f\"Row {idx}: categories not sorted\")\n        except json.JSONDecodeError:\n            errors.append(f\"Row {idx}: invalid JSON in categories\")\n    \n    # Print results\n    if errors:\n        print(\"   ❌ ERRORS:\")\n        for e in errors:\n            print(f\"      - {e}\")\n        return False\n    \n    if warnings:\n        print(\"   ⚠️ WARNINGS:\")\n        for w in warnings[:5]:\n            print(f\"      - {w}\")\n        if len(warnings) > 5:\n            print(f\"      ... and {len(warnings) - 5} more\")\n    \n    print(\"   ✅ Submission is valid!\")\n    return True\n\n# Validate\nif 'submission_df' in dir() and submission_df is not None:\n    is_valid = validate_submission(submission_df)\n    \n    if is_valid:\n        print(\"\\n🎉 SUBMISSION IS READY!\")\n        print(f\"   📁 File: {submission_path}\")\n        print(f\"   📊 Rows: {len(submission_df)}\")\nelse:\n    print(\"❌ No submission DataFrame found!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.228442Z","iopub.status.idle":"2026-01-27T16:58:15.228710Z","shell.execute_reply.started":"2026-01-27T16:58:15.228543Z","shell.execute_reply":"2026-01-27T16:58:15.228553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 17: SUBMISSION STATISTICS\n# ============================================\n\ndef analyze_submission(df):\n    \"\"\"Analyze submission statistics\"\"\"\n    \n    print(\"📊 SUBMISSION STATISTICS\")\n    print(\"=\" * 50)\n    \n    # Parse categories\n    all_objects = []\n    category_counts = defaultdict(int)\n    objects_per_image = []\n    \n    for _, row in df.iterrows():\n        cats = json.loads(row['categories'])\n        objects_per_image.append(len(cats))\n        all_objects.extend(cats)\n        for cat in cats:\n            category_counts[cat] += 1\n    \n    # Statistics\n    print(f\"\\n📈 Object Statistics:\")\n    print(f\"   Total images: {len(df)}\")\n    print(f\"   Total objects predicted: {len(all_objects)}\")\n    print(f\"   Objects per image:\")\n    print(f\"      - Min: {min(objects_per_image)}\")\n    print(f\"      - Max: {max(objects_per_image)}\")\n    print(f\"      - Mean: {np.mean(objects_per_image):.2f}\")\n    print(f\"      - Median: {np.median(objects_per_image):.2f}\")\n    print(f\"   Images with 0 objects: {objects_per_image.count(0)}\")\n    print(f\"   Unique categories used: {len(category_counts)}\")\n    \n    # Top categories\n    sorted_cats = sorted(category_counts.items(), key=lambda x: x[1], reverse=True)\n    print(f\"\\n🏆 Top 10 Predicted Categories:\")\n    for cat_id, count in sorted_cats[:10]:\n        name = category_id_to_name.get(cat_id, f\"Category {cat_id}\")\n        print(f\"   {cat_id} ({name}): {count} ({100*count/len(all_objects):.1f}%)\")\n    \n    return objects_per_image, category_counts\n\n# Analyze\nif 'submission_df' in dir() and submission_df is not None:\n    obj_per_img, cat_counts = analyze_submission(submission_df)\n    \n    # Visualize\n    fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n    \n    # Objects per image histogram\n    axes[0].hist(obj_per_img, bins=30, edgecolor='black', alpha=0.7, color='steelblue')\n    axes[0].axvline(np.mean(obj_per_img), color='red', linestyle='--', \n                    label=f'Mean: {np.mean(obj_per_img):.1f}')\n    axes[0].set_xlabel('Objects per Image')\n    axes[0].set_ylabel('Frequency')\n    axes[0].set_title('Distribution of Predicted Objects per Image')\n    axes[0].legend()\n    \n    # Top categories bar chart\n    sorted_cats = sorted(cat_counts.items(), key=lambda x: x[1], reverse=True)[:15]\n    if sorted_cats:\n        cat_ids, counts = zip(*sorted_cats)\n        y_pos = range(len(cat_ids))\n        axes[1].barh(y_pos, counts, color='steelblue', alpha=0.7)\n        axes[1].set_yticks(y_pos)\n        axes[1].set_yticklabels([f\"Cat {c}\" for c in cat_ids])\n        axes[1].set_xlabel('Count')\n        axes[1].set_title('Top 15 Predicted Categories')\n        axes[1].invert_yaxis()\n    \n    plt.tight_layout()\n    plt.savefig(os.path.join(WORK_DIR, 'submission_stats.png'), dpi=100, bbox_inches='tight')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.229595Z","iopub.status.idle":"2026-01-27T16:58:15.229855Z","shell.execute_reply.started":"2026-01-27T16:58:15.229689Z","shell.execute_reply":"2026-01-27T16:58:15.229698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 18: FINAL SUMMARY & DOWNLOAD\n# ============================================\n\nprint(\"=\" * 60)\nprint(\"🏆 VISTA'26 SUBMISSION COMPLETE\")\nprint(\"=\" * 60)\n\n# List all output files\nprint(\"\\n📁 OUTPUT FILES:\")\noutput_files = []\nfor item in os.listdir(WORK_DIR):\n    path = os.path.join(WORK_DIR, item)\n    if os.path.isfile(path):\n        size = os.path.getsize(path)\n        if size > 1024 * 1024:\n            size_str = f\"{size / (1024*1024):.2f} MB\"\n        else:\n            size_str = f\"{size / 1024:.2f} KB\"\n        output_files.append((item, size_str, path))\n        print(f\"   📄 {item} ({size_str})\")\n\n# Final disk usage\nprint(\"\\n📊 FINAL DISK USAGE:\")\nsubprocess.run(['df', '-h', '/kaggle/working'])\n\n# Instructions\nprint(\"\\n\" + \"=\" * 60)\nprint(\"📤 SUBMISSION INSTRUCTIONS:\")\nprint(\"=\" * 60)\nprint(\"\"\"\n1. Click on the \"Output\" tab on the right panel\n2. Find \"submission.csv\"\n3. Click the three dots (...) → Download\n4. Go to the competition page → \"Submit Predictions\"\n5. Upload the downloaded submission.csv file\n\nOR\n\n1. Save & Run All (Commit) this notebook\n2. After completion, go to the notebook output\n3. Submit directly from there\n\"\"\")\n\nprint(\"\\n💡 TIPS FOR BETTER RESULTS:\")\nprint(\"\"\"\n• Try different confidence thresholds (0.2, 0.3, 0.4)\n• Train for more epochs (100+) if time allows\n• Use YOLOv8l or YOLOv8x for better accuracy\n• Experiment with image size (768, 1024)\n• Use Test-Time Augmentation (TTA)\n\"\"\")\n\nprint(\"\\n🍀 Good luck with your submission!\")\nprint(\"=\" * 60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T16:58:15.230453Z","iopub.status.idle":"2026-01-27T16:58:15.230703Z","shell.execute_reply.started":"2026-01-27T16:58:15.230548Z","shell.execute_reply":"2026-01-27T16:58:15.230557Z"}},"outputs":[],"execution_count":null}]}