{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":107469,"databundleVersionId":13058354,"sourceType":"competition"},{"sourceId":470538,"sourceType":"modelInstanceVersion","modelInstanceId":379586,"modelId":399493}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**# Sheyda_Asadi**\n\n# ##Step 1: Library and Model Setup\n*First, we install the necessary library, ultralytics, and import all required packages. Then, we define the paths to the pre-trained YOLO models and load them. I've renamed the model variables to make them more descriptive.*****","metadata":{}},{"cell_type":"code","source":"# Install the ultralytics library for YOLO models\n!pip install ultralytics > /dev/null\n\nimport os\nimport cv2\nimport csv\nimport random\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom ultralytics import YOLO\n\n# Define the file paths for the two object detection models\nmodel_path_a = '/kaggle/input/2-top-models/pytorch/default/1/habijabii.pt'\nmodel_path_b = '/kaggle/input/2-top-models/pytorch/default/1/nadiatriki.pt'\n\n# Load the models using the YOLO class\ndetector_a = YOLO(model_path_a, verbose=True)\ndetector_b = YOLO(model_path_b, verbose=False)\n\n# Set the directory for test images\ntest_image_folder = '/kaggle/input/multi-class-object-detection-challenge/testImages/images'\ntest_image_files = [f for f in os.listdir(test_image_folder) if f.endswith(('.jpg', '.png'))]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-11T16:27:34.166329Z","iopub.execute_input":"2025-08-11T16:27:34.166564Z","iopub.status.idle":"2025-08-11T16:29:07.210957Z","shell.execute_reply.started":"2025-08-11T16:27:34.166535Z","shell.execute_reply":"2025-08-11T16:29:07.210158Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 2: Bounding Box Formatting Function\nThis helper function processes the raw output from the YOLO models. It extracts the class ID, confidence score, and bounding box coordinates, then formats them into a single string. This specific format is required for the competition's submission file. I've renamed the function and its internal variables for clarity.","metadata":{}},{"cell_type":"code","source":"def serialize_predictions(yolo_results, class_id_offset=0):\n    \"\"\"\n    Converts YOLO detection results into a formatted string for submission.\n\n    Args:\n        yolo_results: The prediction results from a single image.\n        class_id_offset: An integer to add to the class ID, used to handle different\n                         class mappings between models.\n\n    Returns:\n        A string of space-separated predictions, or an empty string if no boxes are found.\n    \"\"\"\n    detected_boxes = yolo_results.boxes\n    img_width, img_height = yolo_results.orig_shape[1], yolo_results.orig_shape[0]\n\n    if detected_boxes is None or len(detected_boxes) == 0:\n        return \"\"\n\n    prediction_strings = []\n    for box_info in detected_boxes:\n        class_id = int(box_info.cls.cpu().numpy()) + class_id_offset\n        confidence = float(box_info.conf.cpu().numpy())\n        x_center, y_center, box_width, box_height = box_info.xywh[0].cpu().numpy()\n\n        # Normalize coordinates and dimensions to be between 0 and 1\n        x_norm = x_center / img_width\n        y_norm = y_center / img_height\n        w_norm = box_width / img_width\n        h_norm = box_height / img_height\n\n        prediction_strings.append(f\"{class_id} {confidence:.6f} {x_norm:.6f} {y_norm:.6f} {w_norm:.6f} {h_norm:.6f}\")\n\n    return \" \".join(prediction_strings)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-11T16:29:21.872466Z","iopub.execute_input":"2025-08-11T16:29:21.872759Z","iopub.status.idle":"2025-08-11T16:29:21.879127Z","shell.execute_reply.started":"2025-08-11T16:29:21.872703Z","shell.execute_reply":"2025-08-11T16:29:21.878380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 3: Inference and Ensembling\nThis is the core of the script. We iterate through each test image, run both models, and then combine their predictions. The class_id_offset is crucial here to ensure the class IDs are consistent across both models for the final submission.","metadata":{}},{"cell_type":"code","source":"import os\n\n# Prepare submission list\nsubmission_data = []\n\n# ✅ Proper path to image folder\ntest_image_folder = '/kaggle/input/multi-class-object-detection-challenge/testImages/images'\n\n# ✅ Get list of all image files (jpg/png only)\ntest_image_files = [f for f in os.listdir(test_image_folder) if f.endswith(('.jpg', '.png'))]\n\n# ✅ Loop through each image\nfor image_filename in test_image_files:\n    image_full_path = os.path.join(test_image_folder, image_filename)\n    print(f\"Processing: {image_full_path}\")  # optional debug print\n\n    # Run inference on both models\n    results_a = detector_a.predict(image_full_path, conf=1e-6, device=0, verbose=False)[0]\n    results_b = detector_b.predict(image_full_path, conf=1e-6, device=0, verbose=False)[0]\n\n    # Format predictions from each model\n    predictions_model_a = serialize_predictions(results_a, class_id_offset=1)\n    predictions_model_b = serialize_predictions(results_b, class_id_offset=0)\n\n    # Combine predictions\n    final_prediction_string = (predictions_model_a + \" \" + predictions_model_b).strip()\n\n    # If no detection at all\n    if final_prediction_string == \"\":\n        final_prediction_string = \"no boxes\"\n\n    # Image ID without .jpg\n    image_identifier = os.path.splitext(image_filename)[0]\n\n    # Add to final list\n    submission_data.append({\n        \"image_id\": image_identifier,\n        \"prediction_string\": final_prediction_string\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-11T16:29:26.179864Z","iopub.execute_input":"2025-08-11T16:29:26.180575Z","iopub.status.idle":"2025-08-11T16:34:47.879094Z","shell.execute_reply.started":"2025-08-11T16:29:26.180552Z","shell.execute_reply":"2025-08-11T16:34:47.878117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 4: Save Submission File\nAfter processing all images, we write the collected data to a submission.csv file, which is the required output format for the competition.","metadata":{}},{"cell_type":"code","source":"output_csv_path = \"submission.csv\"\nwith open(output_csv_path, 'w', newline='') as f:\n    writer = csv.DictWriter(f, fieldnames=[\"image_id\", \"prediction_string\"])\n    writer.writeheader()\n    writer.writerows(submission_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-11T17:06:48.657234Z","iopub.execute_input":"2025-08-11T17:06:48.657513Z","iopub.status.idle":"2025-08-11T17:06:48.841264Z","shell.execute_reply.started":"2025-08-11T17:06:48.657490Z","shell.execute_reply":"2025-08-11T17:06:48.840713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Step 5: Visualization\nThis section is for visual inspection and debugging. It randomly selects 10 images and displays the bounding boxes from both models, each with a different color and label prefix. This helps us see how the two models' predictions differ or complement each other.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\n\nfor image_filename in test_image_files[:5]:  # visualize first 5 images only\n    image_path = os.path.join(test_image_folder, image_filename)\n\n    results = detector_b.predict(image_path, conf=0.25)[0]\n    img = results.plot()  # Draw bounding boxes\n\n    # Display\n    plt.figure(figsize=(10, 8))\n    plt.imshow(img)\n    plt.title(f\"Predictions for {image_filename}\")\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T14:01:08.395177Z","iopub.execute_input":"2025-07-29T14:01:08.395863Z","iopub.status.idle":"2025-07-29T14:01:23.401962Z","shell.execute_reply.started":"2025-07-29T14:01:08.395836Z","shell.execute_reply":"2025-07-29T14:01:23.401266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!yolo task=detect mode=val model=your_model.pt data=data.yaml\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T14:03:03.574749Z","iopub.execute_input":"2025-07-29T14:03:03.575301Z","iopub.status.idle":"2025-07-29T14:03:04.720671Z","shell.execute_reply.started":"2025-07-29T14:03:03.575276Z","shell.execute_reply":"2025-07-29T14:03:04.719508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_code = \"\"\"\npath: /kaggle/input/multi-class-object-detection-challenge\ntrain: /kaggle/input/multi-class-object-detection-challenge/Starter_Dataset/train\nval: /kaggle/input/multi-class-object-detection-challenge/Starter_Dataset/val\nnc: 2\nnames:\n  0: 0\n  1: 1\n\"\"\"\n\nwith open(\"data.yaml\", \"w\") as f:\n    f.write(yaml_code)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T14:57:32.841287Z","iopub.execute_input":"2025-07-29T14:57:32.841559Z","iopub.status.idle":"2025-07-29T14:57:32.845975Z","shell.execute_reply.started":"2025-07-29T14:57:32.841537Z","shell.execute_reply":"2025-07-29T14:57:32.845269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# Load base model (YOLOv8n = nano, v8s = small)\nmodel = YOLO(\"yolov8s.pt\")\n\n# Start training\nmodel.train(\n    data=\"data.yaml\",\n    epochs=50,\n    imgsz=640,\n    batch=16,\n    device=0  # 0 for GPU, -1 for CPU\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T14:57:35.398364Z","iopub.execute_input":"2025-07-29T14:57:35.398639Z","iopub.status.idle":"2025-07-29T15:12:55.019017Z","shell.execute_reply.started":"2025-07-29T14:57:35.398615Z","shell.execute_reply":"2025-07-29T15:12:55.016865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Image\n\nImage(filename='runs/detect/train/results.png')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-29T15:13:14.025744Z","iopub.execute_input":"2025-07-29T15:13:14.026025Z","iopub.status.idle":"2025-07-29T15:13:14.042816Z","shell.execute_reply.started":"2025-07-29T15:13:14.026004Z","shell.execute_reply":"2025-07-29T15:13:14.041957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.val()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-08T12:33:52.038508Z","iopub.execute_input":"2025-08-08T12:33:52.038700Z","iopub.status.idle":"2025-08-08T12:33:52.112192Z","shell.execute_reply.started":"2025-08-08T12:33:52.038681Z","shell.execute_reply":"2025-08-08T12:33:52.111283Z"}},"outputs":[],"execution_count":null}]}