{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":226008626,"sourceType":"kernelVersion"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BYU Locating Flagellar Motors\n\n## Data Visualization Notebook\n\nThis is the second notebook in a series for the BYU Locating Bacterial Flagellar Motors 2025 Kaggle challenge. This notebook focuses on visualizing the preprocessed data to validate our preparation steps and better understand the dataset characteristics.\n\n### Notebook Series:\n1. **[Parse Data](https://www.kaggle.com/code/andrewjdarley/parse-data)**: Extracting and preparing 2D slices containing motors to make a YOLO dataset\n2. **Visualize Data (Current)**: Exploratory data analysis and visualization of annotated motor locations\n3. **[Train YOLO](https://www.kaggle.com/code/andrewjdarley/train-yolo)**: Fine tuning an YOLOv8 object detection model on the prepared dataset\n4. **[Submission Notebook](https://www.kaggle.com/code/andrewjdarley/submission-notebook)**: Running inference and generating submission files\n\n## About this Notebook\n\nThis visualization notebook displays random annotations, for ease of understanding of the dataset","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import random\nimport matplotlib.pyplot as plt\nfrom PIL import Image, ImageDraw\nimport os\nimport numpy as np\nimport glob\n\n# Define base_dir - this was missing in the original code\n# In Kaggle, we can use the working directory as base or remove it completely\n# since we're using absolute paths\nbase_dir = \"/kaggle/working\"  # or simply use \"\" if using absolute paths\n\n# Updated paths without concatenating with base_dir since they're already absolute\nimages_train_dir = \"/kaggle/input/parse-data/yolo_dataset/images/train/\"\nlabels_train_dir = \"/kaggle/input/parse-data/yolo_dataset/labels/train/\"\n\n# Box size for highlighting the motor\nBOX_SIZE = 24\n\ndef visualize_random_training_samples(num_samples=4):\n    \"\"\"\n    Visualize random training samples with YOLO annotations\n    \n    Args:\n        num_samples (int): Number of random images to display\n    \"\"\"\n    # Get all image files from the train directory\n    image_files = []\n    for ext in ['*.jpg', '*.jpeg', '*.png']:\n        image_files.extend(glob.glob(os.path.join(images_train_dir, \"**\", ext), recursive=True))\n    \n    # Make sure we have enough images\n    if len(image_files) == 0:\n        print(\"No image files found in the train directory!\")\n        return\n        \n    num_samples = min(num_samples, len(image_files))\n    \n    # Select random images\n    random_images = random.sample(image_files, num_samples)\n    \n    # Create a figure with subplots\n    rows = int(np.ceil(num_samples / 2))\n    cols = min(num_samples, 2)\n    fig, axes = plt.subplots(rows, cols, figsize=(14, 5 * rows))\n    \n    # Handle the case of a single subplot\n    if num_samples == 1:\n        axes = np.array([axes])\n    \n    # Flatten axes array for easy indexing\n    axes = axes.flatten()\n    \n    # Process each selected image\n    for i, img_path in enumerate(random_images):\n        try:\n            # Get corresponding label file\n            # YOLO labels have same name but .txt extension instead of image extension\n            relative_path = os.path.relpath(img_path, images_train_dir)\n            label_path = os.path.join(labels_train_dir, os.path.splitext(relative_path)[0] + '.txt')\n            \n            # Load the image\n            img = Image.open(img_path)\n            img_width, img_height = img.size\n            \n            # Normalize image using percentiles for better visualization\n            img_array = np.array(img)\n            p2 = np.percentile(img_array, 2)\n            p98 = np.percentile(img_array, 98)\n            normalized = np.clip(img_array, p2, p98)\n            normalized = 255 * (normalized - p2) / (p98 - p2)\n            img_normalized = Image.fromarray(np.uint8(normalized))\n            \n            # Convert image to RGB for colored box\n            img_rgb = img_normalized.convert('RGB')\n            \n            # Create a transparent overlay\n            overlay = Image.new('RGBA', img_rgb.size, (0, 0, 0, 0))\n            draw = ImageDraw.Draw(overlay)\n            \n            # Load YOLO format annotations if they exist\n            annotations = []\n            if os.path.exists(label_path):\n                with open(label_path, 'r') as f:\n                    for line in f:\n                        # YOLO format: class x_center y_center width height\n                        # All values are normalized from 0 to 1\n                        values = line.strip().split()\n                        class_id = int(values[0])\n                        x_center = float(values[1]) * img_width\n                        y_center = float(values[2]) * img_height\n                        width = float(values[3]) * img_width\n                        height = float(values[4]) * img_height\n                        \n                        annotations.append({\n                            'class_id': class_id,\n                            'x_center': x_center,\n                            'y_center': y_center,\n                            'width': width,\n                            'height': height\n                        })\n            \n            # Draw all annotations\n            for ann in annotations:\n                x_center = ann['x_center']\n                y_center = ann['y_center']\n                width = ann['width']\n                height = ann['height']\n                \n                # Calculate bounding box coordinates\n                x1 = max(0, int(x_center - width/2))\n                y1 = max(0, int(y_center - height/2))\n                x2 = min(img_width, int(x_center + width/2))\n                y2 = min(img_height, int(y_center + height/2))\n                \n                # Draw semi-transparent red rectangle\n                draw.rectangle([x1, y1, x2, y2], fill=(255, 0, 0, 64), outline=(255, 0, 0, 200))\n                \n                # Draw label\n                label_text = f\"Class {ann['class_id']}\"\n                draw.text((x1, y1-10), label_text, fill=(255, 0, 0, 255))\n            \n            # If no annotations found, indicate this\n            if not annotations:\n                draw.text((10, 10), \"No annotations found\", fill=(255, 0, 0, 255))\n            \n            # Composite the overlay onto the original image\n            img_rgb = Image.alpha_composite(img_rgb.convert('RGBA'), overlay).convert('RGB')\n            \n            # Display the image with annotations\n            axes[i].imshow(np.array(img_rgb))\n            img_name = os.path.basename(img_path)\n            axes[i].set_title(f\"Image: {img_name}\\nAnnotations: {len(annotations)}\")\n            axes[i].axis('on')\n            \n        except Exception as e:\n            print(f\"Error processing image {img_path}: {e}\")\n            axes[i].text(0.5, 0.5, f\"Error loading image: {os.path.basename(img_path)}\", \n                       horizontalalignment='center', verticalalignment='center')\n            axes[i].axis('off')\n    \n    # Handle extra subplots if any\n    for j in range(i + 1, len(axes)):\n        axes[j].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print summary\n    print(f\"Displayed {num_samples} random images with YOLO annotations\")\n\n# Run the visualization\nvisualize_random_training_samples(4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T01:03:26.655927Z","iopub.execute_input":"2025-03-06T01:03:26.656259Z","iopub.status.idle":"2025-03-06T01:03:30.761894Z","shell.execute_reply.started":"2025-03-06T01:03:26.656234Z","shell.execute_reply":"2025-03-06T01:03:30.760799Z"}},"outputs":[],"execution_count":null}]}