{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":117876,"databundleVersionId":14198377,"sourceType":"competition"},{"sourceId":13923389,"sourceType":"datasetVersion","datasetId":8872462},{"sourceId":666051,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":504148,"modelId":519193}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Environment setup","metadata":{}},{"cell_type":"code","source":"# Install Ultralytics\n!pip install -q \"ultralytics==8.2.103\"\n!pip uninstall -y ray ray[tune]\n!pip install ensemble_boxes","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Core libs\nimport os\nimport random\nfrom pathlib import Path\nimport shutil\nimport ast\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\n# Disable Weights & Biases everywhere\nos.environ[\"WANDB_DISABLED\"] = \"true\"\nos.environ[\"WANDB_MODE\"] = \"disabled\"\n\nfrom ensemble_boxes import weighted_boxes_fusion\n\nfrom ultralytics import YOLO, settings\n\nprint(\"Ultralytics version:\", YOLO.__module__.split('.')[0])\n\nsettings.update({\"wandb\": False})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:58:08.691503Z","iopub.execute_input":"2025-11-29T19:58:08.691850Z","iopub.status.idle":"2025-11-29T19:58:10.125373Z","shell.execute_reply.started":"2025-11-29T19:58:08.691816Z","shell.execute_reply":"2025-11-29T19:58:10.124416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"markdown","source":"### 1. Prepare editable working copy of the dataset\n\nThe original competition data is stored under `/kaggle/input`.  \nHere I copy it once into `/kaggle/working/CottonWeedDetection` so that I can safely modify labels and files (e.g. after 3LC corrections) without touching the read-only input directory.\n","metadata":{}},{"cell_type":"code","source":"SRC_ROOT = Path(\"/kaggle/input/the-3lc-cotton-weed-detection-challenge/cotton_weed_competition_dataset\")\nWORK_ROOT = Path(\"/kaggle/working/CottonWeedDetection\")  # editable copy\n\nif not WORK_ROOT.exists():\n    print(\"Creating working dataset...\")\n    WORK_ROOT.mkdir(parents=True, exist_ok=True)\n\n    for split in [\"train\", \"val\", \"test\"]:\n        src_split = SRC_ROOT / split\n        dst_split = WORK_ROOT / split\n        if src_split.exists():\n            print(f\"Copying {split} ...\")\n            shutil.copytree(src_split, dst_split)\n        else:\n            print(f\"WARNING: {src_split} not found\")\nelse:\n    print(\"Working copy already exists:\", WORK_ROOT)\n\nprint(\"SRC_ROOT:\", SRC_ROOT)\nprint(\"WORK_ROOT:\", WORK_ROOT)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T18:43:27.059502Z","iopub.execute_input":"2025-11-29T18:43:27.059927Z","iopub.status.idle":"2025-11-29T18:44:56.552055Z","shell.execute_reply.started":"2025-11-29T18:43:27.059898Z","shell.execute_reply":"2025-11-29T18:44:56.550806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Applying 3LC-Corrected Labels (YOLO Format)\n\nThis cell imports the high-fidelity, corrected bounding box labels from the 3LC dashboard (ref: 3LC Tutorial). The plant and weed identification was meticulously corrected by cross-referencing external weed science resources (Cornell, Missouri, Purdue). The code parses the bounding box JSON and converts it to the standard YOLO format (<class_id> <xc> <yc> <w> <h>) for the training splits.","metadata":{}},{"cell_type":"code","source":"DATA_ROOT = Path(\"/kaggle/working/CottonWeedDetection\")\n\nCSV_BY_SPLIT = {\n    \"train\": Path(\"/kaggle/input/3cl-cotton-weed-det-final-labels/train_labels.csv\"),\n    \"val\":   Path(\"/kaggle/input/3cl-cotton-weed-det-final-labels/val_labels.csv\"),\n}\n\nfor split, csv_path in CSV_BY_SPLIT.items():\n    img_dir = DATA_ROOT / split / \"images\"\n    labels_dir = DATA_ROOT / split / \"labels\"\n    labels_dir.mkdir(parents=True, exist_ok=True)\n\n    print(f\"\\n=== Processing {split.upper()} ===\")\n    print(\"CSV        :\", csv_path)\n    print(\"Images dir :\", img_dir)\n    print(\"Labels dir :\", labels_dir)\n\n    df = pd.read_csv(csv_path)\n\n    required_cols = {\"image\", \"width\", \"height\", \"bbs\", \"weight\"}\n    missing = required_cols - set(df.columns)\n    if missing:\n        raise ValueError(f\"{csv_path} missing required cols: {missing}\")\n\n    # Wipe old labels so we only keep 3LC-corrected ones\n    for txt in labels_dir.glob(\"*.txt\"):\n        txt.unlink()\n\n    num_files = 0\n    num_boxes = 0\n\n    # One image per row\n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        # Normalize to just filename so it matches dataset layout\n        file_name = Path(row[\"image\"]).name\n        img_path = img_dir / file_name\n        if not img_path.exists():\n            # Label for an image that is not in this split – skip\n            continue\n\n        bbs_raw = row[\"bbs\"]\n        if pd.isna(bbs_raw):\n            continue\n\n        try:\n            bbs_dict = ast.literal_eval(bbs_raw)\n        except Exception as e:\n            print(f\"  [WARN] Could not parse bbs for {file_name}: {e}\")\n            continue\n\n        bb_list = bbs_dict.get(\"bb_list\", [])\n        if not bb_list:\n            continue\n\n        lines = []\n        for bb in bb_list:\n            # x0,y0 are centre; x1,y1 are width,height (already normalized 0–1)\n            cls_id = int(bb[\"label\"])\n            xc = float(bb[\"x0\"])\n            yc = float(bb[\"y0\"])\n            w  = float(bb[\"x1\"])\n            h  = float(bb[\"y1\"])\n\n            # Clamp to [0, 1] just in case\n            xc = min(max(xc, 0.0), 1.0)\n            yc = min(max(yc, 0.0), 1.0)\n            w  = min(max(w,  0.0), 1.0)\n            h  = min(max(h,  0.0), 1.0)\n\n            lines.append(f\"{cls_id} {xc:.6f} {yc:.6f} {w:.6f} {h:.6f}\")\n            num_boxes += 1\n\n        if lines:\n            label_path = labels_dir / f\"{img_path.stem}.txt\"\n            label_path.write_text(\"\\n\".join(lines) + \"\\n\")\n            num_files += 1\n\n    print(f\"[{split}] wrote {num_files} label files, total boxes = {num_boxes}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T18:44:56.553276Z","iopub.execute_input":"2025-11-29T18:44:56.553558Z","iopub.status.idle":"2025-11-29T18:44:56.851490Z","shell.execute_reply.started":"2025-11-29T18:44:56.553526Z","shell.execute_reply":"2025-11-29T18:44:56.850641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport numpy as np\n\ndef set_seed(seed: int = 42):\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n\n    try:\n        import torch\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n    except ImportError:\n        pass \n\nset_seed(42) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:39:49.412389Z","iopub.execute_input":"2025-11-29T19:39:49.416583Z","iopub.status.idle":"2025-11-29T19:39:49.442928Z","shell.execute_reply.started":"2025-11-29T19:39:49.416461Z","shell.execute_reply":"2025-11-29T19:39:49.441879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT = Path(\"/kaggle/working/CottonWeedDetection\")\nprint(\"Dataset root:\", ROOT)\n\nyaml_path = Path(\"/kaggle/working/cotton_weed.yaml\")\n# for final model we also included validation set in training\nyaml_text = f\"\"\"\npath: {ROOT}          \n\ntrain:\n  - train/images      \n  - val/images        \n\nval: val/images      \ntest: test/images     \n\nnc: 3\nnames:\n  0: carpetweed\n  1: morningglory\n  2: palmer_amaranth\n\"\"\"\n\nyaml_path.write_text(yaml_text.strip() + \"\\n\")\nprint(\"==== cotton_weed.yaml ====\\n\")\nprint(yaml_path.read_text())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:41:01.109294Z","iopub.execute_input":"2025-11-29T19:41:01.110111Z","iopub.status.idle":"2025-11-29T19:41:01.121601Z","shell.execute_reply.started":"2025-11-29T19:41:01.110041Z","shell.execute_reply":"2025-11-29T19:41:01.120381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_images(p: Path):\n    return sum(1 for f in p.rglob(\"*\") if f.suffix.lower() in {\".jpg\", \".jpeg\", \".png\", \".bmp\"})\n\nfor split in [\"train\", \"val\", \"test\"]:\n    img_dir = ROOT / split / \"images\"\n    lbl_dir = ROOT / split / \"labels\"\n    print(f\"\\n== {split.upper()} ==\")\n    print(\" images:\", count_images(img_dir), \"in\", img_dir)\n    print(\" labels:\", len(list(lbl_dir.glob(\"*.txt\"))), \"in\", lbl_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:42:16.376062Z","iopub.execute_input":"2025-11-29T19:42:16.376916Z","iopub.status.idle":"2025-11-29T19:42:16.392544Z","shell.execute_reply.started":"2025-11-29T19:42:16.376882Z","shell.execute_reply":"2025-11-29T19:42:16.391437Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Hyperparameter search using YOLOv8 ```tune()```","metadata":{}},{"cell_type":"code","source":"# import os\n# os.environ[\"WANDB_DISABLED\"] = \"true\"\n# os.environ[\"WANDB_MODE\"] = \"disabled\"\n\n# from ultralytics import settings, YOLO\n# settings.update({\"wandb\": False})\n\n# DATASET = \"cotton_weed.yaml\"\n\n# search_space = {\n#     \"lr0\":            (1e-5, 1e-1),\n#     \"lrf\":            (0.01, 1.0),\n#     \"momentum\":       (0.6, 0.98),\n#     \"weight_decay\":   (0.0, 0.001),\n#     \"warmup_epochs\":   (0.0, 5.0),\n#     \"warmup_momentum\": (0.6, 0.98),\n#     \"warmup_bias_lr\":  (0.01, 0.5),\n#     \"box\":      (1.0, 12.0),\n#     \"cls\":      (0.1, 3.0),\n#     \"dfl\":      (0.5, 3.0),\n#     \"dropout\":  (0.0, 0.2),\n#     \"hsv_h\": (0.0, 0.1),\n#     \"hsv_s\": (0.0, 0.9),\n#     \"hsv_v\": (0.0, 0.9),\n#     \"degrees\":     (0.0, 15.0),\n#     \"translate\":   (0.0, 0.3),\n#     \"scale\":       (0.0, 0.9),\n#     \"shear\":       (0.0, 10.0),\n#     \"perspective\": (0.0, 0.001),\n#     \"flipud\":  (0.0, 0.2),\n#     \"fliplr\":  (0.0, 0.7),\n#     \"bgr\":     (0.0, 0.5),\n#     \"mosaic\":  (0.0, 1.0),\n#     \"mixup\":   (0.0, 0.7),\n# }\n\n# from pathlib import Path\n\n# def run_tune_and_get_yaml(model_path, optimizer_name, epochs=10, iterations=12, batch=32):\n#     model = YOLO(model_path)\n#     print(f\"\\n=== Tuning {optimizer_name}: {iterations} iters x {epochs} epochs, batch={batch} ===\\n\")\n#     model.tune(\n#         data=DATASET,\n#         epochs=epochs,\n#         iterations=iterations,\n#         optimizer=optimizer_name,\n#         batch=batch,\n#         imgsz=640,\n#         space=search_space,\n#         plots=False,\n#         save=False,\n#         val=True,\n#         project=\"cotton_weed\",\n#         name=f\"tune_{optimizer_name.lower()}\",\n#     )\n\n#     # Pick the latest best_hyperparameters.yaml\n#     candidates = list(Path(\"runs\").rglob(\"best_hyperparameters.yaml\"))\n#     assert candidates, \"No best_hyperparameters.yaml found; check runs/\"\n#     best_yaml = max(candidates, key=lambda p: p.stat().st_mtime)\n#     print(\"Best hyperparameters yaml:\", best_yaml)\n#     return str(best_yaml)\n\n# # Example: tune AdamW\n# best_yaml_adam = run_tune_and_get_yaml(\"yolov8n.pt\", optimizer_name=\"AdamW\",\n#                                        epochs=40, iterations=10, batch=64)\n# print(\"AdamW best hyperparameters saved at:\", best_yaml_adam)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:14:53.637923Z","iopub.execute_input":"2025-11-29T19:14:53.638309Z","iopub.status.idle":"2025-11-29T19:14:53.643936Z","shell.execute_reply.started":"2025-11-29T19:14:53.638279Z","shell.execute_reply":"2025-11-29T19:14:53.643012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training diverse YOLOv8n models for the ensemble","metadata":{}},{"cell_type":"code","source":"# # # MODEL A\n# !yolo task=detect mode=train \\\n#     model=yolov8n.pt \\\n#     data=/kaggle/working/cotton_weed.yaml \\\n#     epochs=200 \\\n#     patience=50 \\\n#     imgsz=640 \\\n#     batch=64 \\\n#     workers=4 \\\n#     optimizer=AdamW \\\n#     lr0=0.005 \\\n#     lrf=0.011 \\\n#     momentum=0.723 \\\n#     weight_decay=0.0004 \\\n#     warmup_epochs=3 \\\n#     warmup_momentum=0.66 \\\n#     warmup_bias_lr=0.096 \\\n#     box=6.345 \\\n#     cls=0.652 \\\n#     dfl=1.512 \\\n#     dropout=0.0 \\\n#     mosaic=0.721 \\\n#     close_mosaic=20 \\\n#     hsv_h=0.015 \\\n#     hsv_s=0.8 \\\n#     hsv_v=0.3 \\\n#     degrees=0.0 \\\n#     translate=0.094 \\\n#     scale=0.421 \\\n#     shear=0.0 \\\n#     perspective=0.0 \\\n#     flipud=0.0 \\\n#     fliplr=0.466 \\\n#     bgr=0.0 \\\n#     mixup=0.0 \\\n#     seed=42\n\n\n# # # MODEL B ************************************\n# !yolo task=detect mode=train \\\n#   model=yolov8n.pt \\\n#   data=/kaggle/working/cotton_weed.yaml \\\n#   epochs=200 patience=50 \\\n#   imgsz=640 batch=64 workers=4 \\\n#   optimizer=SGD \\\n#   lr0=0.01 lrf=0.01 momentum=0.937 weight_decay=0.0005 \\\n#   warmup_epochs=3 warmup_momentum=0.8 warmup_bias_lr=0.1 \\\n#   box=7.5 cls=0.5 dfl=1.5 \\\n#   dropout=0.0 \\\n#   mosaic=1.0 close_mosaic=10 \\\n#   hsv_h=0.02 hsv_s=0.8 hsv_v=0.45 \\\n#   degrees=10.0 translate=0.10 scale=0.50 shear=2.0 perspective=0.0 \\\n#   flipud=0.0 fliplr=0.5 bgr=0.0 mixup=0.0 \\\n#   seed=0\n\n# # # MODEL C ***********************************************************************************************************\n# !yolo task=detect mode=train \\\n#   model=yolov8n.pt \\\n#   data=/kaggle/working/cotton_weed.yaml \\\n#   epochs=150 patience=40 \\\n#   imgsz=640 batch=64 workers=4 \\\n#   optimizer=AdamW \\\n#   lr0=0.003 lrf=0.01 momentum=0.8 weight_decay=0.0006 \\\n#   warmup_epochs=2 warmup_momentum=0.7 warmup_bias_lr=0.08 \\\n#   box=5.5 cls=1.0 dfl=1.4 \\\n#   dropout=0.0 \\\n#   mosaic=0.3 close_mosaic=5 \\\n#   hsv_h=0.01 hsv_s=0.6 hsv_v=0.25 \\\n#   degrees=0.0 translate=0.05 scale=0.25 shear=0.0 perspective=0.0 \\\n#   flipud=0.0 fliplr=0.3 bgr=0.0 mixup=0.0 \\\n#   seed=123\n\n# # MODEL D\n# !yolo task=detect mode=train \\\n#   model=yolov8n.pt \\\n#   data=/kaggle/working/cotton_weed.yaml \\\n#   device=0,1 \\\n#   epochs=600 patience=80 \\\n#   imgsz=640 batch=128 workers=4 \\\n#   optimizer=AdamW \\\n#   lr0=0.004 lrf=0.01 momentum=0.80 weight_decay=0.0006 \\\n#   warmup_epochs=5 warmup_momentum=0.70 warmup_bias_lr=0.10 \\\n#   box=5.8 cls=1.2 dfl=1.5 \\\n#   label_smoothing=0.05 \\\n#   cos_lr=True \\\n#   dropout=0.0 \\\n#   mosaic=0.9 close_mosaic=30 \\\n#   hsv_h=0.03 hsv_s=0.90 hsv_v=0.50 \\\n#   degrees=10.0 translate=0.15 scale=0.60 shear=4.0 perspective=0.0005 \\\n#   flipud=0.10 fliplr=0.60 bgr=0.0 \\\n#   mixup=0.20 copy_paste=0.30 \\\n#   seed=7\n\n\n# MODEL E\n# !yolo task=detect mode=train \\\n#   model=yolov8n.pt \\\n#   data=/kaggle/working/cotton_weed.yaml \\\n#   device=0,1 \\\n#   epochs=900 patience=120 \\\n#   imgsz=640 batch=128 workers=4 \\\n#   optimizer=SGD \\\n#   lr0=0.009 lrf=0.01 momentum=0.94 weight_decay=0.0005 \\\n#   warmup_epochs=3 warmup_momentum=0.85 warmup_bias_lr=0.10 \\\n#   box=7.8 cls=0.45 dfl=1.6 \\\n#   label_smoothing=0.02 \\\n#   cos_lr=True \\\n#   dropout=0.0 \\\n#   mosaic=0.7 close_mosaic=40 \\\n#   hsv_h=0.02 hsv_s=0.80 hsv_v=0.40 \\\n#   degrees=5.0 translate=0.12 scale=0.70 shear=3.0 perspective=0.0005 \\\n#   flipud=0.05 fliplr=0.50 bgr=0.0 \\\n#   mixup=0.10 copy_paste=0.20 \\\n#   seed=2025","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction And Genrating Solution ","metadata":{}},{"cell_type":"code","source":"MODEL_PATHS = [\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_A.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_B.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_C.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_D1.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_D2.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_D3.pt\",\n    r\"/kaggle/input/3cl-cotton-weed-det-yolov8n-models/pytorch/default/1/yolov8n_model_E.pt\",\n]\n\n# Test dataset root (images/ + labels/)\nTEST_ROOT = r\"/kaggle/working/CottonWeedDetection/test\"\nTEST_IMAGES_DIR = os.path.join(TEST_ROOT, \"images\")\n\n# YOLO inference settings for raw predictions\nINFER_IMGSZ = 640\nINFER_CONF = 0.01  # very low, keep almost everything\nINFER_IOU = 0.9     # NMS IoU (just to remove exact dups per model)\n\n# WBF settings\nWBF_IOU_THR = 0.45\nWBF_SKIP_BOX_THR = 0.01\nWBF_WEIGHTS = [1, 1, 0.75, 0.5, 0.5, 1, 1] \n\n# Number of classes\nNUM_CLASSES = 3\n\nOUT_SUBMISSION_CSV = \"submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T20:00:03.979484Z","iopub.execute_input":"2025-11-29T20:00:03.980317Z","iopub.status.idle":"2025-11-29T20:00:03.986115Z","shell.execute_reply.started":"2025-11-29T20:00:03.980285Z","shell.execute_reply":"2025-11-29T20:00:03.985249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Utils Functions","metadata":{}},{"cell_type":"code","source":"# =========================\n# 1. LOAD MODELS\n# =========================\n\ndef load_models(model_paths):\n    \"\"\"Load YOLO models from given paths.\"\"\"\n    models = []\n    for p in model_paths:\n        print(f\"Loading model: {p}\")\n        models.append(YOLO(p))\n    return models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:44:39.294540Z","iopub.execute_input":"2025-11-29T19:44:39.295741Z","iopub.status.idle":"2025-11-29T19:44:39.301455Z","shell.execute_reply.started":"2025-11-29T19:44:39.295704Z","shell.execute_reply":"2025-11-29T19:44:39.300321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 2. RAW PREDICTIONS (ALL MODELS, LOW CONF/IOU)\n# =========================\n\ndef get_raw_predictions(models, images_dir, imgsz=640, conf=0.001, iou=0.7):\n    \"\"\"\n    Run all models on all images in images_dir with low conf/IoU and\n    return a DataFrame of raw predictions (normalized xyxy).\n    Columns: image_id, model_idx, class_id, score, x1,y1,x2,y2\n    \"\"\"\n    image_paths = sorted(\n        list(Path(images_dir).glob(\"*.jpg\")) + list(Path(images_dir).glob(\"*.png\"))\n    )\n    rows = []\n\n    for img_path in tqdm(image_paths, desc=\"Running models on test images\"):\n        image_id = img_path.stem\n        for m_idx, model in enumerate(models):\n            results = model.predict(\n                source=str(img_path),\n                conf=conf,\n                iou=iou,\n                imgsz=imgsz,\n                verbose=False\n            )[0]\n\n            if results.boxes is None or results.boxes.shape[0] == 0:\n                continue\n\n            # Normalized xyxy coordinates in [0, 1]\n            xyxyn = results.boxes.xyxyn.cpu().numpy()\n            scores = results.boxes.conf.cpu().numpy()\n            classes = results.boxes.cls.cpu().numpy().astype(int)\n\n            for b, s, c in zip(xyxyn, scores, classes):\n                rows.append(\n                    {\n                        \"image_id\": image_id,\n                        \"model_idx\": m_idx,\n                        \"class_id\": int(c),\n                        \"score\": float(s),\n                        \"x1\": float(b[0]),\n                        \"y1\": float(b[1]),\n                        \"x2\": float(b[2]),\n                        \"y2\": float(b[3]),\n                    }\n                )\n\n    df = pd.DataFrame(rows)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:44:52.177809Z","iopub.execute_input":"2025-11-29T19:44:52.178197Z","iopub.status.idle":"2025-11-29T19:44:52.192021Z","shell.execute_reply.started":"2025-11-29T19:44:52.178157Z","shell.execute_reply":"2025-11-29T19:44:52.191023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 3. APPLY WBF ON DATAFRAME\n# =========================\n\ndef apply_wbf_to_dataframe(df_preds, num_models, weights=None,\n                           iou_thr=0.55, skip_box_thr=0.01):\n    \"\"\"\n    Apply Weighted Boxes Fusion per image over predictions from multiple models.\n\n    df_preds: DataFrame with columns [image_id, model_idx, class_id, score, x1,y1,x2,y2]\n    Returns fused_df: DataFrame with columns [image_id, class_id, score, x1,y1,x2,y2]\n    \"\"\"\n    fused_rows = []\n\n    for image_id, df_img in tqdm(df_preds.groupby(\"image_id\"), desc=\"Applying WBF\"):\n        boxes_list = []\n        scores_list = []\n        labels_list = []\n\n        # Prepare list entries for each model 0..num_models-1\n        for m_idx in range(num_models):\n            df_m = df_img[df_img[\"model_idx\"] == m_idx]\n            if df_m.empty:\n                boxes_list.append([])\n                scores_list.append([])\n                labels_list.append([])\n                continue\n\n            boxes = df_m[[\"x1\", \"y1\", \"x2\", \"y2\"]].values.tolist()\n            scores = df_m[\"score\"].values.tolist()\n            labels = df_m[\"class_id\"].values.tolist()\n\n            boxes_list.append(boxes)\n            scores_list.append(scores)\n            labels_list.append(labels)\n\n        if all(len(b) == 0 for b in boxes_list):\n            # No boxes from any model\n            continue\n\n        fused_boxes, fused_scores, fused_labels = weighted_boxes_fusion(\n            boxes_list,\n            scores_list,\n            labels_list,\n            weights=weights,\n            iou_thr=iou_thr,\n            skip_box_thr=skip_box_thr,\n            conf_type='box_and_model_avg'\n        )\n\n        for b, s, c in zip(fused_boxes, fused_scores, fused_labels):\n            fused_rows.append(\n                {\n                    \"image_id\": image_id,\n                    \"class_id\": int(c),\n                    \"score\": float(s),\n                    \"x1\": float(b[0]),\n                    \"y1\": float(b[1]),\n                    \"x2\": float(b[2]),\n                    \"y2\": float(b[3]),\n                }\n            )\n\n    fused_df = pd.DataFrame(fused_rows)\n    return fused_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:45:05.215380Z","iopub.execute_input":"2025-11-29T19:45:05.215724Z","iopub.status.idle":"2025-11-29T19:45:05.229452Z","shell.execute_reply.started":"2025-11-29T19:45:05.215697Z","shell.execute_reply":"2025-11-29T19:45:05.228405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================\n# 5. BUILD SUBMISSION CSV\n# =========================\n\ndef build_submission_csv(df_preds, images_dir, out_csv_path):\n    \"\"\"\n    Convert final predictions (normalized xyxy) into Kaggle submission format:\n    image_id,prediction_string\n    \"\"\"\n    image_paths = sorted(\n        list(Path(images_dir).glob(\"*.jpg\")) + list(Path(images_dir).glob(\"*.png\"))\n    )\n    img_ids = [p.stem for p in image_paths]\n\n    grouped = {k: v for k, v in df_preds.groupby(\"image_id\")} if not df_preds.empty else {}\n\n    rows = []\n    for img_id in img_ids:\n        if img_id not in grouped:\n            rows.append({\"image_id\": img_id, \"prediction_string\": \"no box\"})\n            continue\n\n        df_img = grouped[img_id]\n        parts = []\n        for row in df_img.itertuples(index=False):\n            x1, y1, x2, y2 = row.x1, row.y1, row.x2, row.y2\n            w = max(0.0, x2 - x1)\n            h = max(0.0, y2 - y1)\n            xc = x1 + w / 2.0\n            yc = y1 + h / 2.0\n\n            xc = min(max(xc, 0.0), 1.0)\n            yc = min(max(yc, 0.0), 1.0)\n            w = min(max(w, 0.0), 1.0)\n            h = min(max(h, 0.0), 1.0)\n\n            parts.extend([\n                str(int(row.class_id)),\n                f\"{row.score:.6f}\",\n                f\"{xc:.6f}\",\n                f\"{yc:.6f}\",\n                f\"{w:.6f}\",\n                f\"{h:.6f}\",\n            ])\n\n        pred_str = \" \".join(parts) if parts else \"no box\"\n        rows.append({\"image_id\": img_id, \"prediction_string\": pred_str})\n\n    sub_df = pd.DataFrame(rows)\n    sub_df.to_csv(str(out_csv_path), index=False)\n    print(f\"Saved submission CSV to: {out_csv_path}\")\n    return sub_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:45:29.461919Z","iopub.execute_input":"2025-11-29T19:45:29.462255Z","iopub.status.idle":"2025-11-29T19:45:29.474749Z","shell.execute_reply.started":"2025-11-29T19:45:29.462233Z","shell.execute_reply":"2025-11-29T19:45:29.473587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. Prediction","metadata":{}},{"cell_type":"code","source":"# 1) load models\nmodels = load_models(MODEL_PATHS)\n\n# 2) raw predictions from all models on test set\nraw_df = get_raw_predictions(\n    models,\n    TEST_IMAGES_DIR,\n    imgsz=INFER_IMGSZ,\n    conf=INFER_CONF,\n    iou=INFER_IOU,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:47:07.273786Z","iopub.execute_input":"2025-11-29T19:47:07.274990Z","iopub.status.idle":"2025-11-29T19:53:10.458639Z","shell.execute_reply.started":"2025-11-29T19:47:07.274955Z","shell.execute_reply":"2025-11-29T19:53:10.457704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3) WBF fusion\nfused_df = apply_wbf_to_dataframe(\n    raw_df,\n    num_models=len(models),\n    weights=WBF_WEIGHTS,\n    iou_thr=WBF_IOU_THR,\n    skip_box_thr=WBF_SKIP_BOX_THR,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T19:58:31.053898Z","iopub.execute_input":"2025-11-29T19:58:31.054552Z","iopub.status.idle":"2025-11-29T19:58:33.669053Z","shell.execute_reply.started":"2025-11-29T19:58:31.054520Z","shell.execute_reply":"2025-11-29T19:58:33.668260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5) build submission CSV (for Kaggle)\n_ = build_submission_csv(fused_df, TEST_IMAGES_DIR, OUT_SUBMISSION_CSV)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T20:00:08.054955Z","iopub.execute_input":"2025-11-29T20:00:08.055299Z","iopub.status.idle":"2025-11-29T20:00:08.170307Z","shell.execute_reply.started":"2025-11-29T20:00:08.055271Z","shell.execute_reply":"2025-11-29T20:00:08.168982Z"}},"outputs":[],"execution_count":null}]}