{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":107469,"databundleVersionId":13058354,"sourceType":"competition"},{"sourceId":12474415,"sourceType":"datasetVersion","datasetId":7870336},{"sourceId":12548733,"sourceType":"datasetVersion","datasetId":7923011},{"sourceId":250085529,"sourceType":"kernelVersion"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# YOLO Baseline Object Detection\n\n**Author:** Mohamed Elnageeb  \n**Date:** July 12, 2025  \n\nThis notebook trains a YOLO model on the multi‑class object detection challenge dataset, runs inference across multiple checkpoints, applies weighted‑box fusion, and writes out the submission file.\n","metadata":{}},{"cell_type":"markdown","source":"## 1. Install Dependencies\n\nMake sure we have the right versions of ultralytics (YOLOv12) and the ensemble‑boxes library.\n","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\n!pip install ensemble-boxes","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:11:24.945195Z","iopub.execute_input":"2025-07-15T23:11:24.945422Z","iopub.status.idle":"2025-07-15T23:12:44.424472Z","shell.execute_reply.started":"2025-07-15T23:11:24.945403Z","shell.execute_reply":"2025-07-15T23:12:44.423713Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Import Libraries & Set Seeds\n\nLoad all necessary packages and fix the random seed for reproducibility.\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom ultralytics import YOLO\nfrom pathlib import Path\nimport csv\nimport os\nimport random\nimport torch\n# Set random seeds for reproducibility\nnp.random.seed(42)\nrandom.seed(42)\ntorch.manual_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:12:44.426494Z","iopub.execute_input":"2025-07-15T23:12:44.426738Z","iopub.status.idle":"2025-07-15T23:12:47.987992Z","shell.execute_reply.started":"2025-07-15T23:12:44.426713Z","shell.execute_reply":"2025-07-15T23:12:47.987436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Build `data.yaml` Configuration\n\nWe generate a YAML file pointing at your train/val/test folders and listing the 2 classes.\n","metadata":{}},{"cell_type":"code","source":"import yaml\nbase_path = \"/kaggle/input/multi-class-object-detection-challenge/Dataset\"\n\n# Build YAML dictionary\ndata_yaml = {\n    \"train\": [f\"{base_path}/train/images\",'/kaggle/input/sample-synthetic-data-generated/home/ubuntu/Output/2025-07-15-06-26-28/train/images',\n             '/kaggle/input/sample-synthetic-data-generated/home/ubuntu/Output/2025-07-15-06-26-28/val/images'],\n    \"val\":   f\"{base_path}/val/images\" ,\n    \"test\":  f\"{base_path}/TestImages\",\n    \"nc\":    2,\n    \"names\": [\"cheerios\",\"Soup\"]\n}\n\n# Save to data.yaml\nwith open(\"data.yaml\", \"w\") as f:\n    yaml.safe_dump(data_yaml, f, default_flow_style=False)\n\nprint(\"Created data.yaml with the following content:\")\nprint(yaml.safe_dump(data_yaml, default_flow_style=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:12:47.988693Z","iopub.execute_input":"2025-07-15T23:12:47.989047Z","iopub.status.idle":"2025-07-15T23:12:47.995886Z","shell.execute_reply.started":"2025-07-15T23:12:47.989029Z","shell.execute_reply":"2025-07-15T23:12:47.995172Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Train the YOLOv12 Model\n\nInitialize your YOLOv12 checkpoint and train for 30 epochs with a patience of 50.\n\n-----------CV score : 0.917\n\n-----------LB score : 0.880 ","metadata":{}},{"cell_type":"code","source":"TRAIN = False\n\nif TRAIN : \n    TRAIN = YOLO(\"yolo11x.pt\")\n    data_yaml = '/kaggle/working/data.yaml'\n    \n    model.train(\n        data=data_yaml,\n        epochs=60,                \n        batch=16,                   \n        imgsz=512,\n        patience=100,               \n        optimizer='SGD',        \n        lr0=0.001,\n        lrf = 0.0001,\n        dropout = 0.5,\n        weight_decay=0.0001,       \n        cos_lr=True,               \n        save_period=10,             \n        workers=4,\n        # Augmentations\n        close_mosaic=15,\n        hsv_h=0.025,\n        hsv_s=0.75,\n        hsv_v=0.45,\n        flipud=0.05,\n        fliplr=0.5,\n        translate=0.05,\n        scale=0.5,\n        shear=0.0,\n        warmup_epochs= 5,\n        warmup_momentum= 1,\n        \n        exist_ok = True,\n        project = 'runs2/train',\n        plots = True, \n        augment = True,\n        conf = 0.15,\n        iou = 0.45,\n        \n        # agnostic_nms=True,\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:12:47.996663Z","iopub.execute_input":"2025-07-15T23:12:47.996929Z","iopub.status.idle":"2025-07-15T23:14:02.236258Z","shell.execute_reply.started":"2025-07-15T23:12:47.996904Z","shell.execute_reply":"2025-07-15T23:14:02.234760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TRAIN: \n    # Plot Validation Losses & Confusion Matrix\n    \n    # %% Cell: plot_val_and_confusion_matrix\n    import pandas as pd\n    import matplotlib.pyplot as plt\n    from pathlib import Path\n    from PIL import Image\n    \n    # 1. Point to your training experiment directory\n    exp_dir = Path(\"/kaggle/working/runs2/train/train\")  # adjust if your run folder is different\n    results_csv = exp_dir / \"results.csv\"\n    \n    # 2. Load the epoch-by-epoch metrics\n    results = pd.read_csv(results_csv)\n    \n    # 3. Plot validation losses over epochs\n    plt.figure(figsize=(10, 5))\n    plt.plot(results['epoch'], results['val/box_loss'], label='val box loss')\n    plt.plot(results['epoch'], results['val/cls_loss'], label='val cls loss')\n    plt.plot(results['epoch'], results['val/dfl_loss'], label='val dfl loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.title('YOLO Validation Losses')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n    \n    # 4. Locate and display the confusion matrix image\n    #    (typically saved when you run `model.val()` in Ultralytics)\n    cm_paths = ['/kaggle/working/runs2/train/train/results.png',\n                '/kaggle/working/runs2/train/train/confusion_matrix.png']\n    print(cm_paths)\n    for p in cm_paths :\n        cm_img = Image.open(p)\n        plt.figure(figsize=(25, 10))\n        plt.imshow(cm_img)\n        plt.axis('off')\n        # plt.title('Validation Confusion Matrix')\n        plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:14:02.238616Z","iopub.status.idle":"2025-07-15T23:14:02.238900Z","shell.execute_reply.started":"2025-07-15T23:14:02.238783Z","shell.execute_reply":"2025-07-15T23:14:02.238796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Inference Helper Functions then Ensemble & Save Submission\n\nDefine two utilities:\n1. `filter_invalid_boxes` to drop zero‑area predictions  \n2. `run_inference` to batch through images at different sizes\n\nLoad both your “last” and “best” checkpoints, run inference, then apply WBF and write out a CSV.\n","metadata":{}},{"cell_type":"code","source":"def filter_invalid_boxes(boxes, scores, labels):\n    filtered_boxes, filtered_scores, filtered_labels = [], [], []\n    for b, s, l in zip(boxes, scores, labels):\n        if abs(b[2] - b[0]) > 1e-6 and abs(b[3] - b[1]) > 1e-6:\n            filtered_boxes.append(b)\n            filtered_scores.append(s)\n            filtered_labels.append(l)\n    return filtered_boxes, filtered_scores, filtered_labels\n    \ndef run_inference(models, image_sizes, test_images_path):\n    image_paths = [p for p in Path(test_images_path).glob(\"*\") if p.suffix.lower() in [\".jpg\", \".jpeg\", \".png\"]]\n    predictions = {}\n\n    for model_idx, model in enumerate(models):\n        model.eval()\n        predictions[model_idx] = {}\n        for size in image_sizes:\n            predictions[model_idx][size] = {}\n            pred = []\n            for img_path in image_paths:\n                image_id = img_path.stem\n                image = Image.open(img_path)\n                img_width, img_height = image.size\n\n                results = model.predict(source=str(img_path), conf=conf,iou=iou_thr, max_det=100, augment=True, imgsz=size, verbose=False)\n                boxes, scores, labels = [], [], []\n\n                for result in results:\n                    if result.boxes is None:\n                        continue\n                    boxes = result.boxes.xyxy.cpu().numpy().tolist()\n                    scores = result.boxes.conf.cpu().numpy().tolist()\n                    labels = result.boxes.cls.cpu().numpy().tolist()\n\n                    norm_boxes = [\n                        [x1 / img_width, y1 / img_height, x2 / img_width, y2 / img_height]\n                        for x1, y1, x2, y2 in boxes\n                    ]\n                    norm_boxes, scores, labels = filter_invalid_boxes(norm_boxes, scores, labels)\n\n                predictions[model_idx][size][image_id] = {\n                    \"boxes\": norm_boxes,\n                    \"scores\": scores,\n                    \"labels\": labels\n                }\n                \n                if boxes:\n                    prediction_string = \" \".join(\n                        f\"{int(lbl)} {score:.6f} {(b[0]+b[2])/2:.6f} {(b[1]+b[3])/2:.6f} {(b[2]-b[0]):.6f} {(b[3]-b[1]):.6f}\"\n                        for b, score, lbl in zip(norm_boxes, scores, labels)\n                    )\n                else:\n                    prediction_string = \"no boxes\"\n\n                pred.append({\n                    \"image_id\": image_id,\n                    \"prediction_string\": prediction_string\n                })\n\n            # Save CSV per model and size\n            df = pd.DataFrame(pred)\n            csv_path = f\"submission_{model_idx}_{size}.csv\"\n            df.to_csv(csv_path, index=False, quoting=csv.QUOTE_MINIMAL)\n            print(f\"[saved] {csv_path}\")\n            print(df.head(10))\n\n    return predictions\n\ndef apply_wbf_and_save_final_submission(predictions, image_ids, output_path=\"submission.csv\"):\n    wbf_results = []\n\n    for image_id in image_ids:\n        all_boxes, all_scores, all_labels = [], [], []\n\n        for model_preds in predictions.values():\n            for size_preds in model_preds.values():\n                if image_id not in size_preds:\n                    continue\n                pred = size_preds[image_id]\n                if not pred[\"boxes\"]:\n                    continue\n                all_boxes.append(pred[\"boxes\"])\n                all_scores.append(pred[\"scores\"])\n                all_labels.append(pred[\"labels\"])\n\n        if not all_boxes:\n            pred_str = \"no boxes\"\n        else:\n            fused_boxes, fused_scores, fused_labels = weighted_boxes_fusion(\n                all_boxes, all_scores, all_labels, iou_thr=iou_thr, skip_box_thr=skip_box_thr\n            )\n\n            pred_str = \" \".join(\n                f\"{int(lbl)} {score:.6f} {(b[0]+b[2])/2:.6f} {(b[1]+b[3])/2:.6f} {(b[2]-b[0]):.6f} {(b[3]-b[1]):.6f}\"\n                for b, score, lbl in zip(fused_boxes, fused_scores, fused_labels)\n            )\n\n        wbf_results.append({\n            \"image_id\": image_id,\n            \"prediction_string\": pred_str\n        })\n\n    wbf_df = pd.DataFrame(wbf_results)\n    wbf_df.to_csv(output_path, index=False, quoting=csv.QUOTE_MINIMAL)\n    print(f\"[notice] ✅ WBF submission saved to {output_path}\")\n    print(wbf_df.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:14:02.240371Z","iopub.status.idle":"2025-07-15T23:14:02.240656Z","shell.execute_reply.started":"2025-07-15T23:14:02.240539Z","shell.execute_reply":"2025-07-15T23:14:02.240552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport pandas as pd\nimport csv\nfrom ultralytics import YOLO\nfrom ensemble_boxes import weighted_boxes_fusion\nfrom PIL import Image\n\nmodel_paths = [\n    '/kaggle/input/3lc-yolo-baseline-submission/Duality-3LC-Kaggle/run-1/weights/best.pt',\n    '/kaggle/input/runs2-model/runs2/train/train/weights/best.pt',\n]\n\ntest_images_path = \"/kaggle/input/multi-class-object-detection-challenge/testImages/images\"\noutput_dir = \"/kaggle/working/predictions/labels\"\n\nconf = 0.05\niou_thr = 0.35\nskip_box_thr = 0.01\nimage_sizes = [640,800,864]\n\n\nmodels = [YOLO(path) for path in model_paths]\npredictions = run_inference(models, image_sizes, test_images_path)\n\nimage_ids = list(next(iter(next(iter(predictions.values())).values())).keys())\n\napply_wbf_and_save_final_submission(predictions, image_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T23:14:02.241955Z","iopub.status.idle":"2025-07-15T23:14:02.242240Z","shell.execute_reply.started":"2025-07-15T23:14:02.242093Z","shell.execute_reply":"2025-07-15T23:14:02.242110Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. General Tips for Boosting mAP<sub>50</sub>\n\n- Generate more data\n- Apply diverse augmentations (random flips, crops, rotations, color jitter)  \n- Tune key hyperparameters (learning rate, weight decay, batch size)  \n- Use test‑time augmentation (flips, scales, crops) with multiple image sizes\n- experiment with different model versions and complexities\n- Have Fun :)","metadata":{}}]}