{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":117876,"databundleVersionId":14198377,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install -U ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:11.327204Z","iopub.execute_input":"2026-06-23T02:13:11.327473Z","iopub.status.idle":"2026-06-23T02:13:17.713217Z","shell.execute_reply.started":"2026-06-23T02:13:11.327449Z","shell.execute_reply":"2026-06-23T02:13:17.712299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nfrom ultralytics import YOLO \n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:17.715137Z","iopub.execute_input":"2026-06-23T02:13:17.715486Z","iopub.status.idle":"2026-06-23T02:13:26.805844Z","shell.execute_reply.started":"2026-06-23T02:13:17.715456Z","shell.execute_reply":"2026-06-23T02:13:26.805251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Explore Dataset Structure\n\n# Define dataset paths\nBASE_PATH = Path(\"/kaggle/input/the-3lc-cotton-weed-detection-challenge\")\nTRAIN_IMAGES = BASE_PATH / \"cotton_weed_competition_dataset/train/images\"\nTRAIN_LABELS = BASE_PATH / \"cotton_weed_competition_dataset/train/labels\"\nVAL_IMAGES = BASE_PATH / \"cotton_weed_competition_dataset/val/images\"\nVAL_LABELS = BASE_PATH / \"cotton_weed_competition_dataset/val/labels\"\nTEST_IMAGES = BASE_PATH / \"cotton_weed_competition_dataset/test/images\"\nprint(\"\\n=== Dataset Counts ===\")\ntrain_img_count = len(list(TRAIN_IMAGES.glob('*.jpg')))\ntrain_lbl_count = len(list(TRAIN_LABELS.glob('*.txt')))\nval_img_count = len(list(VAL_IMAGES.glob('*.jpg')))\nval_lbl_count = len(list(VAL_LABELS.glob('*.txt')))\ntest_img_count = len(list(TEST_IMAGES.glob('*.jpg')))\n\nprint(f\"Train: {train_img_count} images, {train_lbl_count} labels\")\nprint(f\"Val:   {val_img_count} images, {val_lbl_count} labels\")\nprint(f\"Test:  {test_img_count} images\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:26.806741Z","iopub.execute_input":"2026-06-23T02:13:26.807143Z","iopub.status.idle":"2026-06-23T02:13:27.036280Z","shell.execute_reply.started":"2026-06-23T02:13:26.807115Z","shell.execute_reply":"2026-06-23T02:13:27.035510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Class Distribution\n\ndef count_classes(labels_dir):\n    \"\"\"Count instances of each class in YOLO format labels\"\"\"\n    class_counts = {0: 0, 1: 0, 2: 0}\n    total_boxes = 0\n    images_with_labels = 0\n    \n    for label_file in Path(labels_dir).glob(\"*.txt\"):\n        has_label = False\n        with open(label_file, 'r') as f:\n            for line in f:\n                parts = line.strip().split()\n                if parts:\n                    class_id = int(parts[0])\n                    class_counts[class_id] += 1\n                    total_boxes += 1\n                    has_label = True\n        if has_label:\n            images_with_labels += 1\n    \n    return class_counts, total_boxes, images_with_labels\n\nprint(\"=== Class Distribution Analysis ===\\n\")\ntrain_classes, train_total, train_img_labeled = count_classes(TRAIN_LABELS)\nval_classes, val_total, val_img_labeled = count_classes(VAL_LABELS)\n\nclass_names = {0: 'Carpetweed', 1: 'Morning Glory', 2: 'Palmer Amaranth'}\n\nprint(\"Training Set:\")\nfor class_id, count in train_classes.items():\n    pct = (count/train_total*100) if train_total > 0 else 0\n    print(f\"  Class {class_id} ({class_names[class_id]}): {count} ({pct:.1f}%)\")\nprint(f\"  Total: {train_total} annotations in {train_img_labeled} images\")\n\nprint(\"\\nValidation Set:\")\nfor class_id, count in val_classes.items():\n    pct = (count/val_total*100) if val_total > 0 else 0\n    print(f\"  Class {class_id} ({class_names[class_id]}): {count} ({pct:.1f}%)\")\nprint(f\"  Total: {val_total} annotations in {val_img_labeled} images\")\n\n# Visualize distribution\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 5))\n\nclasses = ['Carpetweed', 'Morning Glory', 'Palmer Amaranth']\ntrain_counts = [train_classes[0], train_classes[1], train_classes[2]]\nval_counts = [val_classes[0], val_classes[1], val_classes[2]]\ncolors = ['#2ecc71', '#3498db', '#e74c3c']\n\nax1.bar(classes, train_counts, color=colors, alpha=0.8)\nax1.set_title('Training Set Class Distribution', fontsize=14, fontweight='bold')\nax1.set_ylabel('Number of Annotations', fontsize=12)\nax1.tick_params(axis='x', rotation=15)\nax1.grid(axis='y', alpha=0.3)\n\nax2.bar(classes, val_counts, color=colors, alpha=0.8)\nax2.set_title('Validation Set Class Distribution', fontsize=14, fontweight='bold')\nax2.set_ylabel('Number of Annotations', fontsize=12)\nax2.tick_params(axis='x', rotation=15)\nax2.grid(axis='y', alpha=0.3)\n\nplt.tight_layout()\nplt.savefig('class_distribution.png', dpi=100, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:27.037330Z","iopub.execute_input":"2026-06-23T02:13:27.037675Z","iopub.status.idle":"2026-06-23T02:13:29.598899Z","shell.execute_reply.started":"2026-06-23T02:13:27.037650Z","shell.execute_reply":"2026-06-23T02:13:29.598200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize Sample Images with Annotations\n\nprint(\"=== Sample Training Images ===\\n\")\n\n# Get 4 sample images\nsample_images = list(TRAIN_IMAGES.glob(\"*.jpg\"))[:4]\n\nfig, axes = plt.subplots(2, 2, figsize=(16, 16))\naxes = axes.ravel()\n\ncolors = ['#2ecc71', '#3498db', '#e74c3c']\nclass_names = ['Carpetweed', 'Morning Glory', 'Palmer Amaranth']\n\nfor idx, img_path in enumerate(sample_images):\n    label_path = TRAIN_LABELS / f\"{img_path.stem}.txt\"\n    \n    # Load image\n    img = Image.open(img_path)\n    img_w, img_h = img.size\n    axes[idx].imshow(img)\n    \n    # Read and plot bounding boxes\n    if label_path.exists():\n        box_count = 0\n        with open(label_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()\n                if len(parts) >= 5:\n                    class_id = int(parts[0])\n                    x_center, y_center, width, height = map(float, parts[1:5])\n                    \n                    # Convert from YOLO format to pixel coordinates\n                    x_center *= img_w\n                    y_center *= img_h\n                    width *= img_w\n                    height *= img_h\n                    \n                    x1 = x_center - width / 2\n                    y1 = y_center - height / 2\n                    \n                    # Draw rectangle\n                    from matplotlib.patches import Rectangle\n                    rect = Rectangle((x1, y1), width, height, \n                                   linewidth=3, edgecolor=colors[class_id], \n                                   facecolor='none')\n                    axes[idx].add_patch(rect)\n                    \n                    # Add label\n                    axes[idx].text(x1, y1-5, class_names[class_id], \n                                 color='white', fontsize=11, fontweight='bold',\n                                 bbox=dict(boxstyle='round,pad=0.3', \n                                         facecolor=colors[class_id], alpha=0.8))\n                    box_count += 1\n        \n        axes[idx].set_title(f\"{img_path.name}\\n({box_count} annotations)\", \n                          fontsize=12, fontweight='bold')\n    else:\n        axes[idx].set_title(f\"{img_path.name}\\n(No annotations)\", \n                          fontsize=12, fontweight='bold')\n    \n    axes[idx].axis('off')\n\nplt.tight_layout()\nplt.savefig('sample_annotations.png', dpi=100, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:29.600827Z","iopub.execute_input":"2026-06-23T02:13:29.601065Z","iopub.status.idle":"2026-06-23T02:13:42.658731Z","shell.execute_reply.started":"2026-06-23T02:13:29.601039Z","shell.execute_reply":"2026-06-23T02:13:42.657980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create YOLO Configuration File (UPDATED)\n\n# Create dataset YAML config with the correct subfolder paths\nyaml_content = f\"\"\"# Cotton Weed Detection Dataset\npath: {BASE_PATH}\ntrain: cotton_weed_competition_dataset/train/images\nval: cotton_weed_competition_dataset/val/images\ntest: cotton_weed_competition_dataset/test/images\n\n# Classes\nnames:\n  0: Carpetweed\n  1: Morning Glory\n  2: Palmer Amaranth\n\n# Number of classes\nnc: 3\n\"\"\"\n\nyaml_path = Path(\"/kaggle/working/cotton_weeds.yaml\")\nwith open(yaml_path, 'w') as f:\n    f.write(yaml_content)\n\nprint(f\"✓ Dataset config updated successfully: {yaml_path}\")\nprint(\"\\nNew Config contents:\")\nprint(yaml_content)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:42.659855Z","iopub.execute_input":"2026-06-23T02:13:42.660324Z","iopub.status.idle":"2026-06-23T02:13:42.668213Z","shell.execute_reply.started":"2026-06-23T02:13:42.660241Z","shell.execute_reply":"2026-06-23T02:13:42.667406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train Baseline YOLOv8n Model\n\nprint(\"=== Training Baseline YOLOv8n Model ===\\n\")\n\n# Load pretrained YOLOv8n\nmodel = YOLO(\"yolov8n.pt\")\n\nprint(\"Model loaded. Starting training...\")\nprint(\"This will take 10-20 minutes depending on dataset size.\\n\")\n\n# Train with optimized settings\nresults = model.train(\n    data=str(yaml_path),\n    epochs=50,\n    imgsz=800,\n    batch=16,\n    project=\"/kaggle/working/\",\n    name=\"baseline\",\n    verbose=True,\n    patience=10,\n    save=True,\n    plots=True,\n    device=0,\n    \n    # Optimized augmentations\n    hsv_h=0.015,\n    hsv_s=0.7,\n    hsv_v=0.4,\n    degrees=10.0,\n    translate=0.1,\n    scale=0.5,\n    fliplr=0.5,\n    mosaic=1.0,\n    mixup=0.1,\n    \n    # Training params\n    lr0=0.01,\n    lrf=0.01,\n    momentum=0.937,\n    weight_decay=0.0005,\n    warmup_epochs=3.0,\n)\n\nprint(\"\\n✓ Training complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:13:42.669408Z","iopub.execute_input":"2026-06-23T02:13:42.669616Z","iopub.status.idle":"2026-06-23T02:51:34.369885Z","shell.execute_reply.started":"2026-06-23T02:13:42.669593Z","shell.execute_reply":"2026-06-23T02:51:34.365546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate Baseline Model\n\nprint(\"=== Evaluating Baseline Model ===\\n\")\n\n\nbest_model_path = \"/kaggle/working/baseline/weights/best.pt\" \nbest_model = YOLO(best_model_path)\n\n# Validate\nmetrics = best_model.val(data=str(yaml_path))\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"BASELINE MODEL PERFORMANCE\")\nprint(\"=\"*70)\nprint(f\"mAP50:     {metrics.box.map50:.4f}\")\nprint(f\"mAP50-95:  {metrics.box.map:.4f}\")\nprint(f\"Precision: {metrics.box.mp:.4f}\")\nprint(f\"Recall:    {metrics.box.mr:.4f}\")\n\nprint(f\"\\nPer-Class mAP50:\")\nclass_names = ['Carpetweed', 'Morning Glory', 'Palmer Amaranth']\nfor i, class_name in enumerate(class_names):\n    if i < len(metrics.box.maps):\n        print(f\"  {class_name:20s}: {metrics.box.maps[i]:.4f}\")\n\nprint(\"=\"*70)\n\n# Display training plots\nresults_dir = Path(\"/kaggle/input/the-3lc-cotton-weed-detection-challenge\")\nif (results_dir / \"results.png\").exists():\n    img = Image.open(results_dir / \"results.png\")\n    plt.figure(figsize=(16, 10))\n    plt.imshow(img)\n    plt.axis('off')\n    plt.title(\"Training Results\", fontsize=16, fontweight='bold')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:51:34.373424Z","iopub.execute_input":"2026-06-23T02:51:34.373750Z","iopub.status.idle":"2026-06-23T02:51:57.875721Z","shell.execute_reply.started":"2026-06-23T02:51:34.373704Z","shell.execute_reply":"2026-06-23T02:51:57.874815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Analyze Validation Predictions\n\nprint(\"=== Analyzing Validation Set ===\\n\")\n\n# Generate predictions on validation set\nval_results = best_model.predict(\n    source=str(VAL_IMAGES),\n    conf=0.25,\n    iou=0.45,\n    save=True,\n    save_txt=True,\n    project=\"/kaggle/working/runs\",\n    name=\"val_predictions\",\n    verbose=False\n)\n\nprint(f\"✓ Predictions saved to: /kaggle/working/runs/val_predictions\")\n\n# Analyze prediction vs ground truth discrepancies\ndef analyze_predictions(image_dir, label_dir, pred_dir):\n    \"\"\"Find images with prediction/label mismatches\"\"\"\n    issues = []\n    \n    pred_label_dir = Path(pred_dir) / \"labels\"\n    if not pred_label_dir.exists():\n        print(f\"Warning: Prediction labels directory not found at {pred_label_dir}\")\n        return pd.DataFrame()\n    \n    for img_path in Path(image_dir).glob(\"*.jpg\"):\n        gt_path = Path(label_dir) / f\"{img_path.stem}.txt\"\n        pred_path = pred_label_dir / f\"{img_path.stem}.txt\"\n        \n        # Count ground truth\n        gt_boxes = 0\n        if gt_path.exists():\n            with open(gt_path, 'r') as f:\n                gt_boxes = sum(1 for line in f if line.strip())\n        \n        # Count predictions\n        pred_boxes = 0\n        max_conf = 0.0\n        if pred_path.exists():\n            with open(pred_path, 'r') as f:\n                for line in f:\n                    parts = line.strip().split()\n                    if parts:\n                        pred_boxes += 1\n                        if len(parts) >= 6:\n                            max_conf = max(max_conf, float(parts[5]))\n        \n        # Flag potential issues\n        issue_types = []\n        if pred_boxes > gt_boxes + 2:\n            issue_types.append(\"missing_annotations\")\n        if gt_boxes > pred_boxes + 2:\n            issue_types.append(\"false_annotations\")\n        if pred_boxes > 0 and max_conf > 0.7 and gt_boxes == 0:\n            issue_types.append(\"high_conf_no_gt\")\n        \n        if issue_types:\n            issues.append({\n                'image': img_path.name,\n                'gt_boxes': gt_boxes,\n                'pred_boxes': pred_boxes,\n                'max_conf': max_conf,\n                'issue_types': ', '.join(issue_types)\n            })\n    \n    return pd.DataFrame(issues)\n\nissues_df = analyze_predictions(VAL_IMAGES, VAL_LABELS, \"/kaggle/working/runs/val_predictions\")\n\nif len(issues_df) > 0:\n    print(f\"\\nFound {len(issues_df)} images with potential label issues:\")\n    print(\"\\nTop 10 issues by confidence:\")\n    print(issues_df.sort_values('max_conf', ascending=False).head(10))\n    \n    issues_df.to_csv('/kaggle/working/potential_issues.csv', index=False)\n    print(\"\\n✓ Full list saved to: /kaggle/working/potential_issues.csv\")\nelse:\n    print(\"\\nNo significant prediction/label discrepancies found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:51:57.877069Z","iopub.execute_input":"2026-06-23T02:51:57.877455Z","iopub.status.idle":"2026-06-23T02:52:38.969501Z","shell.execute_reply.started":"2026-06-23T02:51:57.877423Z","shell.execute_reply":"2026-06-23T02:52:38.968758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate Test Predictions\n\nprint(\"=== Generating Test Set Predictions ===\\n\")\n\n# Predict on test set\ntest_results = best_model.predict(\n    source=str(TEST_IMAGES),\n    conf=0.25,\n    iou=0.45,\n    save=True,\n    save_txt=True,\n    project=\"/kaggle/working/\",\n    name=\"test_predictions\",\n    verbose=False\n)\n\nprint(f\"✓ Test predictions generated\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T02:52:38.970834Z","iopub.execute_input":"2026-06-23T02:52:38.971215Z","iopub.status.idle":"2026-06-23T02:53:40.964645Z","shell.execute_reply.started":"2026-06-23T02:52:38.971185Z","shell.execute_reply":"2026-06-23T02:53:40.963922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create Submission File\n\nprint(\"=== Creating Submission File ===\\n\")\n\nsubmission_data = []\npred_labels_dir = Path(\"/kaggle/working/\")\n\nfor img_path in sorted(TEST_IMAGES.glob(\"*.jpg\")):\n    image_id = img_path.stem\n    pred_path = pred_labels_dir / f\"{image_id}.txt\"\n    \n    # Get image dimensions\n    img = Image.open(img_path)\n    img_w, img_h = img.size\n    \n    if pred_path.exists():\n        with open(pred_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()\n                if len(parts) >= 6:\n                    class_id = int(parts[0])\n                    x_center, y_center, width, height = map(float, parts[1:5])\n                    confidence = float(parts[5])\n                    \n                    # Convert to pixel coordinates\n                    x_center *= img_w\n                    y_center *= img_h\n                    width *= img_w\n                    height *= img_h\n                    \n                    xmin = x_center - width / 2\n                    ymin = y_center - height / 2\n                    xmax = x_center + width / 2\n                    ymax = y_center + height / 2\n                    \n                    submission_data.append({\n                        'image_id': image_id,\n                        'class': class_id,\n                        'confidence': confidence,\n                        'xmin': xmin,\n                        'ymin': ymin,\n                        'xmax': xmax,\n                        'ymax': ymax\n                    })\n\n# 1. ЗАСРАЛ: Жагсаалт хоосон байсан ч багануудыг бэлдэж өгнө\ncolumns = ['image_id', 'class', 'confidence', 'xmin', 'ymin', 'xmax', 'ymax']\nif submission_data:\n    submission_df = pd.DataFrame(submission_data)\nelse:\n    submission_df = pd.DataFrame(columns=columns)\n\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(f\"✓ Submission file created: submission.csv\")\nprint(f\"  Total predictions: {len(submission_df)}\")\n\n# 2. ЗАСРАЛ: Хэрэв хүснэгт хоосон бол 0 гэж хэвлэнэ\nif not submission_df.empty:\n    print(f\"  Unique images: {submission_df['image_id'].nunique()}\")\n    print(f\"\\nFirst few rows:\")\n    print(submission_df.head())\nelse:\n    print(f\"  Unique images: 0 (No objects detected in any test images)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T03:08:49.487879Z","iopub.execute_input":"2026-06-23T03:08:49.488599Z","iopub.status.idle":"2026-06-23T03:08:49.927728Z","shell.execute_reply.started":"2026-06-23T03:08:49.488560Z","shell.execute_reply":"2026-06-23T03:08:49.926962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary and Next Steps\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"BASELINE MODEL COMPLETE! 🎉\")\nprint(\"=\"*70)\n\nprint(f\"\\n📊 Final Metrics:\")\nprint(f\"  mAP50:     {metrics.box.map50:.4f}\")\nprint(f\"  mAP50-95:  {metrics.box.map:.4f}\")\nprint(f\"  Precision: {metrics.box.mp:.4f}\")\nprint(f\"  Recall:    {metrics.box.mr:.4f}\")\n\nprint(f\"\\n📁 Important Files:\")\nprint(f\"  Model:        /kaggle/working/runs/baseline/weights/best.pt\")\nprint(f\"  Submission:   /kaggle/working/submission.csv\")\nprint(f\"  Issues:       /kaggle/working/potential_issues.csv\")\n\nprint(f\"\\n🔄 Next Steps for Improvement:\")\nprint(f\"  1. Review potential_issues.csv for mislabeled data\")\nprint(f\"  2. Manually inspect high-confidence predictions without GT\")\nprint(f\"  3. Fix labels and create cleaned dataset\")\nprint(f\"  4. Retrain with cleaned data\")\nprint(f\"  5. Tune hyperparameters (learning rate, augmentations)\")\nprint(f\"  6. Try different confidence thresholds for submission\")\n\nprint(f\"\\n💡 Data-Centric Tips:\")\nprint(f\"  • Focus on 'high_conf_no_gt' issues - likely missing labels\")\nprint(f\"  • Look for systematic class confusion patterns\")\nprint(f\"  • Balance classes if severely imbalanced\")\nprint(f\"  • Remove truly ambiguous images\")\n\nprint(f\"\\n✅ Ready to submit! Upload submission.csv to Kaggle\")\nprint(\"=\"*70)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-23T03:08:58.550448Z","iopub.execute_input":"2026-06-23T03:08:58.551158Z","iopub.status.idle":"2026-06-23T03:08:58.558300Z","shell.execute_reply.started":"2026-06-23T03:08:58.551125Z","shell.execute_reply":"2026-06-23T03:08:58.557479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}