{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11772414,"sourceType":"datasetVersion","datasetId":7390992},{"sourceId":394291,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":324255,"modelId":345043},{"sourceId":395504,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":324858,"modelId":345693}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BYU Locating Flagellar Motors\n\n## last session\n[My preparing code](https://www.kaggle.com/code/hiranorm/let-s-examine-a-motor-and-prepare-training-data)\n\n## citation\n[train yolo](https://www.kaggle.com/code/andrewjdarley/train-yolo)\n\n## next I do\nanalyze train results and inferrence and submission!\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T00:17:41.317671Z","iopub.execute_input":"2025-05-16T00:17:41.317963Z","iopub.status.idle":"2025-05-16T00:17:46.668773Z","shell.execute_reply.started":"2025-05-16T00:17:41.317939Z","shell.execute_reply":"2025-05-16T00:17:46.667722Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# move dataset\ncopy my data to /kaggle/working/yolo_dataset","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/byu-yolo-t4-sp2-bs24-ts08 /kaggle/working/yolo_dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T00:17:46.670098Z","iopub.execute_input":"2025-05-16T00:17:46.670413Z","iopub.status.idle":"2025-05-16T00:19:02.593879Z","shell.execute_reply.started":"2025-05-16T00:17:46.670376Z","shell.execute_reply":"2025-05-16T00:19:02.593107Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"import os\nimport torch\nimport numpy as np\nimport random\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nfrom ultralytics import YOLO\nimport yaml\nimport pandas as pd\nimport json\n\n# Set random seeds for reproducibility\nnp.random.seed(42)\nrandom.seed(42)\ntorch.manual_seed(42)\n\n# Define paths for Kaggle environment\nyolo_dataset_dir = \"/kaggle/working/yolo_dataset\"\nyolo_weights_dir = \"/kaggle/working/yolo_weights\"\nyolo_pretrained_weights = \"/kaggle/input/yolo-v11-s/pytorch/default/1/yolo11s.pt\"  # Path to pre-downloaded weights\n\n# Create weights directory if it doesn't exist\nos.makedirs(yolo_weights_dir, exist_ok=True)\n\ndef fix_yaml_paths(yaml_path):\n    \"\"\"\n    Fix the paths in the YAML file to match the actual Kaggle directories\n    \n    Args:\n        yaml_path (str): Path to the original dataset YAML file\n        \n    Returns:\n        str: Path to the fixed YAML file\n    \"\"\"\n    print(f\"Fixing YAML paths in {yaml_path}\")\n    \n    # Read the original YAML\n    with open(yaml_path, 'r') as f:\n        yaml_data = yaml.safe_load(f)\n\n    # Update paths to use actual dataset location\n    if 'path' in yaml_data:\n        yaml_data['path'] = yolo_dataset_dir\n    \n    # Create a new fixed YAML in the working directory\n    fixed_yaml_path = \"/kaggle/working/fixed_dataset.yaml\"\n    with open(fixed_yaml_path, 'w') as f:\n        yaml.dump(yaml_data, f)\n    \n    print(f\"Created fixed YAML at {fixed_yaml_path} with path: {yaml_data.get('path')}\")\n    return fixed_yaml_path\n\ndef plot_dfl_loss_curve(run_dir):\n    \"\"\"\n    Plot the DFL loss curves for train and validation, marking the best model\n    \n    Args:\n        run_dir (str): Directory where the training results are stored\n    \"\"\"\n    # Path to the results CSV file\n    results_csv = os.path.join(run_dir, 'results.csv')\n    \n    if not os.path.exists(results_csv):\n        print(f\"Results file not found at {results_csv}\")\n        return\n    \n    # Read results CSV\n    results_df = pd.read_csv(results_csv)\n    \n    # Check if DFL loss columns exist\n    train_dfl_col = [col for col in results_df.columns if 'train/dfl_loss' in col]\n    val_dfl_col = [col for col in results_df.columns if 'val/dfl_loss' in col]\n    \n    if not train_dfl_col or not val_dfl_col:\n        print(\"DFL loss columns not found in results CSV\")\n        print(f\"Available columns: {results_df.columns.tolist()}\")\n        return\n    \n    train_dfl_col = train_dfl_col[0]\n    val_dfl_col = val_dfl_col[0]\n    \n    # Find the epoch with the best validation loss\n    best_epoch = results_df[val_dfl_col].idxmin()\n    best_val_loss = results_df.loc[best_epoch, val_dfl_col]\n    \n    # Create the plot\n    plt.figure(figsize=(10, 6))\n    \n    # Plot training and validation losses\n    plt.plot(results_df['epoch'], results_df[train_dfl_col], label='Train DFL Loss')\n    plt.plot(results_df['epoch'], results_df[val_dfl_col], label='Validation DFL Loss')\n    \n    # Mark the best model with a vertical line\n    plt.axvline(x=results_df.loc[best_epoch, 'epoch'], color='r', linestyle='--', \n                label=f'Best Model (Epoch {int(results_df.loc[best_epoch, \"epoch\"])}, Val Loss: {best_val_loss:.4f})')\n    \n    # Add labels and legend\n    plt.xlabel('Epoch')\n    plt.ylabel('DFL Loss')\n    plt.title('Training and Validation DFL Loss')\n    plt.legend()\n    plt.grid(True, linestyle='--', alpha=0.7)\n    \n    # Save the plot in the same directory as weights\n    plot_path = os.path.join(run_dir, 'dfl_loss_curve.png')\n    plt.savefig(plot_path)\n    \n    # Also save it to the working directory for easier access\n    plt.savefig(os.path.join('/kaggle/working', 'dfl_loss_curve.png'))\n    \n    print(f\"Loss curve saved to {plot_path}\")\n    plt.close()\n    \n    # Return the best epoch info\n    return best_epoch, best_val_loss\n\ndef train_yolo_model(yaml_path, pretrained_weights_path, epochs=30, batch_size=16, img_size=640):\n    \"\"\"\n    Train a YOLO model on the prepared dataset\n    \n    Args:\n        yaml_path (str): Path to the dataset YAML file\n        pretrained_weights_path (str): Path to pre-downloaded weights file\n        epochs (int): Number of training epochs\n        batch_size (int): Batch size for training\n        img_size (int): Image size for training\n    \"\"\"\n    print(f\"Loading pre-trained weights from: {pretrained_weights_path}\")\n    \n    # Load a pre-trained YOLOv11x model\n    model = YOLO(pretrained_weights_path)\n    \n    # Train the model with early stopping\n    results = model.train(\n        data=yaml_path,\n        epochs=epochs,\n        batch=batch_size,\n        imgsz=img_size,\n        project=yolo_weights_dir,\n        name='motor_detector',\n        mixup=0.5,\n        dropout=0.01,\n        cos_lr=True,\n        exist_ok=True,\n        patience=100,              # Early stopping if no improvement\n        save_period=5,\n        lr0=5e-4,# Save checkpoints every 5 epochs\n        val=True,                # Ensure validation is performed\n        verbose=True             # Show detailed output during training\n    )\n    \n    # Get the path to the run directory\n    run_dir = os.path.join(yolo_weights_dir, 'motor_detector')\n    \n    # Plot and save the loss curve\n    best_epoch_info = plot_dfl_loss_curve(run_dir)\n    \n    if best_epoch_info:\n        best_epoch, best_val_loss = best_epoch_info\n        print(f\"\\nBest model found at epoch {best_epoch} with validation DFL loss: {best_val_loss:.4f}\")\n    \n    return model, results\n\ndef predict_on_samples(model, num_samples=4):\n    \"\"\"\n    Run predictions on random validation samples and display results\n    \n    Args:\n        model: Trained YOLO model\n        num_samples (int): Number of random samples to test\n    \"\"\"\n    # Get validation images\n    val_dir = os.path.join(yolo_dataset_dir, 'images', 'val')\n    if not os.path.exists(val_dir):\n        print(f\"Validation directory not found at {val_dir}\")\n        # Try train directory instead if val doesn't exist\n        val_dir = os.path.join(yolo_dataset_dir, 'images', 'train')\n        print(f\"Using train directory for predictions instead: {val_dir}\")\n        \n    if not os.path.exists(val_dir):\n        print(\"No images directory found for predictions\")\n        return\n    \n    val_images = os.listdir(val_dir)\n    \n    if len(val_images) == 0:\n        print(\"No images found for prediction\")\n        return\n    \n    # Select random samples\n    num_samples = min(num_samples, len(val_images))\n    samples = random.sample(val_images, num_samples)\n    \n    # Create figure\n    fig, axes = plt.subplots(2, 2, figsize=(12, 12))\n    axes = axes.flatten()\n    \n    for i, img_file in enumerate(samples):\n        if i >= len(axes):\n            break\n            \n        img_path = os.path.join(val_dir, img_file)\n        \n        # Run prediction\n        results = model.predict(img_path, conf=0.25)[0]\n        \n        # Load and display the image\n        img = Image.open(img_path)\n        axes[i].imshow(np.array(img), cmap='gray')\n        \n        # Draw ground truth box if available (from filename)\n        try:\n            # This assumes your filenames contain coordinates in a specific format\n            parts = img_file.split('_')\n            y_part = [p for p in parts if p.startswith('y')]\n            x_part = [p for p in parts if p.startswith('x')]\n            \n            if y_part and x_part:\n                y_gt = int(y_part[0][1:])\n                x_gt = int(x_part[0][1:].split('.')[0])\n                \n                box_size = 24\n                rect_gt = Rectangle((x_gt - box_size//2, y_gt - box_size//2), \n                              box_size, box_size, \n                              linewidth=1, edgecolor='g', facecolor='none')\n                axes[i].add_patch(rect_gt)\n        except:\n            pass  # Skip ground truth if parsing fails\n        \n        # Draw predicted boxes (red)\n        if len(results.boxes) > 0:\n            boxes = results.boxes.xyxy.cpu().numpy()\n            confs = results.boxes.conf.cpu().numpy()\n            \n            for box, conf in zip(boxes, confs):\n                x1, y1, x2, y2 = box\n                rect_pred = Rectangle((x1, y1), x2-x1, y2-y1, \n                                     linewidth=1, edgecolor='r', facecolor='none')\n                axes[i].add_patch(rect_pred)\n                axes[i].text(x1, y1-5, f'{conf:.2f}', color='red')\n        \n        axes[i].set_title(f\"Image: {img_file}\\nGround Truth (green) vs Prediction (red)\")\n    \n    plt.tight_layout()\n    \n    # Save the predictions plot\n    plt.savefig(os.path.join('/kaggle/working', 'predictions.png'))\n    plt.show()\n\n# Check and create a dataset YAML if needed\ndef prepare_dataset():\n    \"\"\"\n    Check if dataset exists and create a proper YAML if needed\n    \n    Returns:\n        str: Path to the YAML file to use for training\n    \"\"\"\n    # Check if images exist\n    train_images_dir = os.path.join(yolo_dataset_dir, 'images', 'train')\n    val_images_dir = os.path.join(yolo_dataset_dir, 'images', 'val')\n    train_labels_dir = os.path.join(yolo_dataset_dir, 'labels', 'train')\n    val_labels_dir = os.path.join(yolo_dataset_dir, 'labels', 'val')\n    \n    # Print directory existence status\n    print(f\"Directory status:\")\n    print(f\"- Train images dir exists: {os.path.exists(train_images_dir)}\")\n    print(f\"- Val images dir exists: {os.path.exists(val_images_dir)}\")\n    print(f\"- Train labels dir exists: {os.path.exists(train_labels_dir)}\")\n    print(f\"- Val labels dir exists: {os.path.exists(val_labels_dir)}\")\n    \n    # Check for original YAML file\n    original_yaml_path = os.path.join(yolo_dataset_dir, 'dataset.yaml')\n    \n    if os.path.exists(original_yaml_path):\n        print(f\"Found original dataset.yaml at {original_yaml_path}\")\n        # Fix the paths in the YAML\n        return fix_yaml_paths(original_yaml_path)\n    else:\n        print(f\"Original dataset.yaml not found, creating a new one\")\n        \n        # Create a new YAML file\n        yaml_data = {\n            'path': yolo_dataset_dir,\n            'train': 'images/train',\n            'val': 'images/train' if not os.path.exists(val_images_dir) else 'images/val',\n            'names': {0: 'motor'}\n        }\n        \n        new_yaml_path = \"/kaggle/working/dataset.yaml\"\n        with open(new_yaml_path, 'w') as f:\n            yaml.dump(yaml_data, f)\n            \n        print(f\"Created new YAML at {new_yaml_path}\")\n        return new_yaml_path\n\n# Main execution\ndef main():\n    print(\"Starting YOLO training process...\")\n    \n    # Prepare dataset and get YAML path\n    yaml_path = prepare_dataset()\n    print(f\"Using YAML file: {yaml_path}\")\n    \n    # Print YAML file contents\n    with open(yaml_path, 'r') as f:\n        yaml_content = f.read()\n    print(f\"YAML file contents:\\n{yaml_content}\")\n    \n    # Train model\n    print(\"\\nStarting YOLO training...\")\n    model, results = train_yolo_model(\n        yaml_path,\n        pretrained_weights_path=yolo_pretrained_weights,\n        epochs=100  # Using 30 epochs instead of 100 for faster training\n    )\n    \n    print(\"\\nTraining complete!\")\n    \n    # Run predictions\n    print(\"\\nRunning predictions on sample images...\")\n    predict_on_samples(model, num_samples=4)\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T00:19:02.595553Z","iopub.execute_input":"2025-05-16T00:19:02.595781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -r /kaggle/working/yolo_dataset","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir /kaggle/working/train_result","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mv /kaggle/working/* /kaggle/working/train_result","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!tar -czvf /kaggle/working/train_result.tar.gz -C /kaggle/working/train_result .","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# train is complete\nMy data's training is complete :)\n\nDownload train_result.tar.gz and analyze train results by cpu notebook.\n\nAnd next, let's try inference...","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}