{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":405704,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":331527,"modelId":352413},{"sourceId":625645,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":470913,"modelId":479072},{"sourceId":627931,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":472870,"modelId":479072},{"sourceId":627943,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":472882,"modelId":479072},{"sourceId":625643,"sourceType":"modelInstanceVersion","modelInstanceId":470911,"modelId":479072},{"sourceId":627974,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":472913,"modelId":479072},{"sourceId":627975,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":472914,"modelId":479072}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:28:09.936178Z","iopub.execute_input":"2025-11-03T04:28:09.936708Z","iopub.status.idle":"2025-11-03T04:28:09.940832Z","shell.execute_reply.started":"2025-11-03T04:28:09.936686Z","shell.execute_reply":"2025-11-03T04:28:09.940027Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Install and import libraries","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:28:10.641246Z","iopub.execute_input":"2025-11-03T04:28:10.641511Z","iopub.status.idle":"2025-11-03T04:29:22.000206Z","shell.execute_reply.started":"2025-11-03T04:28:10.641461Z","shell.execute_reply":"2025-11-03T04:29:21.999283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport cv2 as cv\nimport os\nimport shutil\nimport time\nimport yaml\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm  # Use tqdm.notebook for Jupyter/Kaggle environments\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image, ImageDraw\nimport glob\n\nimport torch\nfrom matplotlib.patches import Rectangle\nfrom ultralytics import YOLO\nimport yaml\nimport json\n\nimport warnings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:22.001928Z","iopub.execute_input":"2025-11-03T04:29:22.002223Z","iopub.status.idle":"2025-11-03T04:29:25.854483Z","shell.execute_reply.started":"2025-11-03T04:29:22.002196Z","shell.execute_reply":"2025-11-03T04:29:25.853686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# import datasets","metadata":{}},{"cell_type":"code","source":"EXTRA_DATA = '/kaggle/input/byu-2025-cryoet-dataset-part-2/dataset'\nTRAIN = '/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/train'\nTEST = '/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/test'\n\ntrain_labels = pd.read_csv('/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/train_labels.csv')\n# extra_data_labels = pd.read_csv('/kaggle/input/byu-2025-cryoet-dataset-part-2/labels.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:25.855319Z","iopub.execute_input":"2025-11-03T04:29:25.855711Z","iopub.status.idle":"2025-11-03T04:29:25.877066Z","shell.execute_reply.started":"2025-11-03T04:29:25.855691Z","shell.execute_reply":"2025-11-03T04:29:25.876397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# missing_samples = [sample for sample in extra_data_labels if sample not in os.listdir(EXTRA_DATA)]\n# print(f\"No of missing Samples in Cryoet dataset part-2: {len(missing_samples)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:25.879259Z","iopub.execute_input":"2025-11-03T04:29:25.879442Z","iopub.status.idle":"2025-11-03T04:29:29.186908Z","shell.execute_reply.started":"2025-11-03T04:29:25.879427Z","shell.execute_reply":"2025-11-03T04:29:29.186232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set random seed for reproducibility\nnp.random.seed(42)\n\n# Define YOLO dataset structure\nyolo_dataset_dir = \"/kaggle/working/yolo_dataset\"\nyolo_images_train = os.path.join(yolo_dataset_dir, \"images\", \"train\")\nyolo_images_test = os.path.join(yolo_dataset_dir,\"images\",\"test\")\nyolo_images_val = os.path.join(yolo_dataset_dir, \"images\", \"val\")\nyolo_labels_train = os.path.join(yolo_dataset_dir, \"labels\", \"train\")\nyolo_labels_test = os.path.join(yolo_dataset_dir,\"labels\",\"test\")\nyolo_labels_val = os.path.join(yolo_dataset_dir, \"labels\", \"val\")\n\n# Create directories\nfor dir_path in [yolo_images_train, yolo_images_val, yolo_images_test,\n                 yolo_labels_train, yolo_labels_val, yolo_labels_test]:\n    os.makedirs(dir_path, exist_ok=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:29.187661Z","iopub.execute_input":"2025-11-03T04:29:29.188170Z","iopub.status.idle":"2025-11-03T04:29:29.201828Z","shell.execute_reply.started":"2025-11-03T04:29:29.188143Z","shell.execute_reply":"2025-11-03T04:29:29.201248Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Define constants\nTRUST = 3  # Number of slices above and below center slice (total 2*TRUST + 1 slices)\nBOX_SIZE = 80  # Bounding box size for annotations (in pixels)\nTRAIN_SPLIT = 0.7  # 80% for training, 20% for validation\nTEST_SPLIT = 0.20\n\n# Image processing functions\ndef normalize_slice(slice_data):\n    \"\"\"\n    Normalize slice data using 2nd and 98th percentiles\n    \"\"\"\n    # Calculate percentiles\n    p2 = np.percentile(slice_data, 2)\n    p98 = np.percentile(slice_data, 98)\n    \n    # Clip the data to the percentile range\n    clipped_data = np.clip(slice_data, p2, p98)\n    \n    # Normalize to [0, 255] range\n    normalized = 255 * (clipped_data - p2) / (p98 - p2)\n    \n    return np.uint8(normalized)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:29.202428Z","iopub.execute_input":"2025-11-03T04:29:29.202670Z","iopub.status.idle":"2025-11-03T04:29:29.213828Z","shell.execute_reply.started":"2025-11-03T04:29:29.202652Z","shell.execute_reply":"2025-11-03T04:29:29.213162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef prepare_yolo_dataset(labels_df,train_dir,trust=TRUST, train_split=TRAIN_SPLIT , test_split=TEST_SPLIT):\n    \"\"\"\n    Extract slices containing motors from tomograms and save to YOLO structure with annotations\n    \"\"\"\n    # Load the labels CSV\n    # labels_df = pd.read_csv(os.path.join(data_path, \"train_labels.csv\"))\n    \n    # Count total number of motors\n    total_motors = labels_df['Number of motors'].sum()\n    print(f\"Total number of motors in the dataset: {total_motors}\")\n    \n    # Get unique tomograms that have motors\n    tomo_df = labels_df[labels_df['Number of motors'] > 0].copy()\n    unique_tomos = tomo_df['tomo_id'].unique()\n    \n    print(f\"Found {len(unique_tomos)} unique tomograms with motors\")\n    \n    # Perform the train-val split at the tomogram level (not motor level)\n    # This ensures all slices from a single tomogram go to either train or val\n    np.random.shuffle(unique_tomos)  # Shuffle the tomograms\n    split_idx = int(len(unique_tomos) * train_split)\n    test_split_idx = split_idx + int(len(unique_tomos) * test_split)\n    train_tomos = unique_tomos[:split_idx]\n    test_tomos = unique_tomos[split_idx:test_split_idx]\n    val_tomos = unique_tomos[test_split_idx:]\n    \n    print(f\"Split: {len(train_tomos)} tomograms for training, {len(val_tomos)} tomograms for validation, {len(test_tomos)}\")\n    \n    # Function to process a set of tomograms\n    def process_tomogram_set(tomogram_ids, images_dir, labels_dir, set_name):\n        motor_counts = []\n        for tomo_id in tomogram_ids:\n            # Get all motors for this tomogram\n            tomo_motors = labels_df[labels_df['tomo_id'] == tomo_id]\n            for _, motor in tomo_motors.iterrows():\n                if pd.isna(motor['Motor axis 0']):\n                    continue\n                motor_counts.append(\n                    (tomo_id, \n                     int(motor['Motor axis 0']), \n                     int(motor['Motor axis 1']), \n                     int(motor['Motor axis 2']),\n                     int(motor['Array shape (axis 0)']))\n                )\n        \n        print(f\"Will process approximately {len(motor_counts) * (2 * trust + 1)} slices for {set_name}\")\n        \n        # Process each motor\n        processed_slices = 0\n        \n        for tomo_id, z_center, y_center, x_center, z_max in tqdm(motor_counts, desc=f\"Processing {set_name} motors\"):\n            # Calculate range of slices to include\n            z_min = max(0, z_center - trust)\n            z_max = min(z_max - 1, z_center + trust)\n            \n            # Process each slice in the range\n            for z in range(z_min, z_max + 1):\n                # Create slice filename\n                slice_filename = f\"slice_{z:04d}.jpg\"\n                \n                # Source path for the slice\n                src_path = os.path.join(train_dir, tomo_id, slice_filename)\n                \n                if not os.path.exists(src_path):\n                    print(f\"Warning: {src_path} does not exist, skipping.\")\n                    continue\n                \n                # Load and normalize the slice\n                img = Image.open(src_path)\n                img_array = np.array(img)\n                \n                # Normalize the image\n                normalized_img = normalize_slice(img_array)\n                \n                # Create destination filename (with unique identifier)\n                dest_filename = f\"{tomo_id}_z{z:04d}_y{y_center:04d}_x{x_center:04d}.jpg\"\n                dest_path = os.path.join(images_dir, dest_filename)\n                \n                # Save the normalized image\n                Image.fromarray(normalized_img).save(dest_path)\n                \n                # Get image dimensions\n                img_width, img_height = img.size\n                \n                # Create YOLO format label\n                # YOLO format: <class> <x_center> <y_center> <width> <height>\n                # Values are normalized to [0, 1]\n                x_center_norm = x_center / img_width\n                y_center_norm = y_center / img_height\n                box_width_norm = BOX_SIZE / img_width\n                box_height_norm = BOX_SIZE / img_height\n                \n                # Write label file\n                label_path = os.path.join(labels_dir, dest_filename.replace('.jpg', '.txt'))\n                with open(label_path, 'w') as f:\n                    f.write(f\"0 {x_center_norm} {y_center_norm} {box_width_norm} {box_height_norm}\\n\")\n                \n                processed_slices += 1\n        \n        return processed_slices, len(motor_counts)\n    \n    # Process training tomograms\n    train_slices, train_motors = process_tomogram_set(train_tomos, yolo_images_train, yolo_labels_train, \"training\")\n\n    # Process test tomograms\n    test_slices, test_motors = process_tomogram_set(test_tomos, yolo_images_test, yolo_labels_test, \"test\")\n    # Process validation tomograms\n    val_slices, val_motors = process_tomogram_set(val_tomos, yolo_images_val, yolo_labels_val, \"validation\")\n    \n    # Create YAML configuration file for YOLO\n    yaml_content = {\n        'path': yolo_dataset_dir,\n        'train': 'images/train',\n        'val': 'images/val',\n        'test': 'images/test',\n        'names': {0: 'motor'}\n    }\n    \n    with open(os.path.join(yolo_dataset_dir, 'dataset.yaml'), 'w') as f:\n        yaml.dump(yaml_content, f, default_flow_style=False)\n    \n    print(f\"\\nProcessing Summary:\")\n    print(f\"- Train set: {len(train_tomos)} tomograms, {train_motors} motors, {train_slices} slices\")\n    print(f\"- Test set: {len(test_tomos)} tomograms, {test_motors} motors, {test_slices} slices\")\n    print(f\"- Validation set: {len(val_tomos)} tomograms, {val_motors} motors, {val_slices} slices\")\n    print(f\"- Total: {len(train_tomos) + len(val_tomos) + len(test_tomos)} tomograms, {train_motors + val_motors + test_motors} motors, {train_slices + val_slices + test_slices} slices\")\n    \n    # Return summary info\n    return {\n        \"dataset_dir\": yolo_dataset_dir,\n        \"yaml_path\": os.path.join(yolo_dataset_dir, 'dataset.yaml'),\n        \"train_tomograms\": len(train_tomos),\n        \"test_tomograms\": len(test_tomos),\n        \"val_tomograms\": len(val_tomos),\n        \"train_motors\": train_motors,\n        \"test_motors\": test_motors,\n        \"val_motors\": val_motors,\n        \"train_slices\": train_slices,\n        \"test_slices\": test_slices,\n        \"val_slices\": val_slices\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:29.214608Z","iopub.execute_input":"2025-11-03T04:29:29.214870Z","iopub.status.idle":"2025-11-03T04:29:29.228943Z","shell.execute_reply.started":"2025-11-03T04:29:29.214845Z","shell.execute_reply":"2025-11-03T04:29:29.228415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the preprocessing\nsummary = prepare_yolo_dataset(labels_df=train_labels,train_dir=TRAIN,trust=TRUST)\nprint(f\"\\nPreprocessing Complete:\")\nprint(f\"- Training data: {summary['train_tomograms']} tomograms, {summary['train_motors']} motors, {summary['train_slices']} slices\")\nprint(f\"- Testing data: {summary['test_tomograms']} tomograms, {summary['test_motors']} motors, {summary['test_slices']} slices\")\nprint(f\"- Validation data: {summary['val_tomograms']} tomograms, {summary['val_motors']} motors, {summary['val_slices']} slices\")\nprint(f\"- Dataset directory: {summary['dataset_dir']}\")\nprint(f\"- YAML configuration: {summary['yaml_path']}\")\nprint(f\"\\nReady for YOLO training!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:29:29.229725Z","iopub.execute_input":"2025-11-03T04:29:29.229943Z","iopub.status.idle":"2025-11-03T04:32:11.745042Z","shell.execute_reply.started":"2025-11-03T04:29:29.229923Z","shell.execute_reply":"2025-11-03T04:32:11.744449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"No of train samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/train'))}\")\nprint(f\"No of test samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/test'))}\")\nprint(f\"No of valid samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/val'))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.745740Z","iopub.execute_input":"2025-11-03T04:32:11.745998Z","iopub.status.idle":"2025-11-03T04:32:11.752692Z","shell.execute_reply.started":"2025-11-03T04:32:11.745970Z","shell.execute_reply":"2025-11-03T04:32:11.751996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(extra_data_labels.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.755034Z","iopub.execute_input":"2025-11-03T04:32:11.755266Z","iopub.status.idle":"2025-11-03T04:32:11.764195Z","shell.execute_reply.started":"2025-11-03T04:32:11.755251Z","shell.execute_reply":"2025-11-03T04:32:11.763606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra_data_tomo_id = os.listdir(os.path.join(EXTRA_DATA))\n# print(len([_id for _id in extra_data_labels['tomo_id'] if _id not in extra_data_tomo_id]))\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.764892Z","iopub.execute_input":"2025-11-03T04:32:11.765137Z","iopub.status.idle":"2025-11-03T04:32:11.779050Z","shell.execute_reply.started":"2025-11-03T04:32:11.765118Z","shell.execute_reply":"2025-11-03T04:32:11.778447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra_data_tomo_id = os.listdir(os.path.join(EXTRA_DATA))\n# print(len([_id for _id in extra_data_labels['tomo_id'] if _id not in extra_data_tomo_id]))\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.779624Z","iopub.execute_input":"2025-11-03T04:32:11.779846Z","iopub.status.idle":"2025-11-03T04:32:11.793751Z","shell.execute_reply.started":"2025-11-03T04:32:11.779827Z","shell.execute_reply":"2025-11-03T04:32:11.793234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Run the preprocessing\n# summary = prepare_yolo_dataset(labels_df=extra_data_labels,train_dir=EXTRA_DATA,trust=TRUST)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.794354Z","iopub.execute_input":"2025-11-03T04:32:11.794548Z","iopub.status.idle":"2025-11-03T04:32:11.806447Z","shell.execute_reply.started":"2025-11-03T04:32:11.794534Z","shell.execute_reply":"2025-11-03T04:32:11.805890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(f\"\\nPreprocessing Complete:\")\n# print(f\"- Training data: {summary['train_tomograms']} tomograms, {summary['train_motors']} motors, {summary['train_slices']} slices\")\n# print(f\"- Testing data: {summary['test_tomograms']} tomograms, {summary['test_motors']} motors, {summary['test_slices']} slices\")\n# print(f\"- Validation data: {summary['val_tomograms']} tomograms, {summary['val_motors']} motors, {summary['val_slices']} slices\")\n# print(f\"- Dataset directory: {summary['dataset_dir']}\")\n# print(f\"- YAML configuration: {summary['yaml_path']}\")\n# print(f\"\\nReady for YOLO training!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.807063Z","iopub.execute_input":"2025-11-03T04:32:11.807270Z","iopub.status.idle":"2025-11-03T04:32:11.819821Z","shell.execute_reply.started":"2025-11-03T04:32:11.807257Z","shell.execute_reply":"2025-11-03T04:32:11.819269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(f\"No of train samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/train'))}\")\n# print(f\"No of test samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/test'))}\")\n# print(f\"No of valid samples: {len(os.listdir('/kaggle/working/yolo_dataset/images/val'))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.820561Z","iopub.execute_input":"2025-11-03T04:32:11.821001Z","iopub.status.idle":"2025-11-03T04:32:11.836756Z","shell.execute_reply.started":"2025-11-03T04:32:11.820980Z","shell.execute_reply":"2025-11-03T04:32:11.836195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"# Define base_dir - this was missing in the original code\n# In Kaggle, we can use the working directory as base or remove it completely\n# since we're using absolute paths\nbase_dir = \"/kaggle/working\"  # or simply use \"\" if using absolute paths\n\n# Updated paths without concatenating with base_dir since they're already absolute\nimages_train_dir = \"/kaggle/working/yolo_dataset/images/train/\"\nlabels_train_dir = \"/kaggle/working/yolo_dataset/labels/train\"\n\n# Box size for highlighting the motor\nBOX_SIZE = 80\n\ndef visualize_random_training_samples(num_samples=4):\n    \"\"\"\n    Visualize random training samples with YOLO annotations\n    \n    Args:\n        num_samples (int): Number of random images to display\n    \"\"\"\n    # Get all image files from the train directory\n    image_files = []\n    for ext in ['*.jpg', '*.jpeg', '*.png']:\n        image_files.extend(glob.glob(os.path.join(images_train_dir, \"**\", ext), recursive=True))\n    \n    # Make sure we have enough images\n    if len(image_files) == 0:\n        print(\"No image files found in the train directory!\")\n        return\n        \n    num_samples = min(num_samples, len(image_files))\n    \n    # Select random images\n    random_images = random.sample(image_files, num_samples)\n    \n    # Create a figure with subplots\n    rows = int(np.ceil(num_samples / 2))\n    cols = min(num_samples, 2)\n    fig, axes = plt.subplots(rows, cols, figsize=(14, 5 * rows))\n    \n    # Handle the case of a single subplot\n    if num_samples == 1:\n        axes = np.array([axes])\n    \n    # Flatten axes array for easy indexing\n    axes = axes.flatten()\n    \n    # Process each selected image\n    for i, img_path in enumerate(random_images):\n        try:\n            # Get corresponding label file\n            # YOLO labels have same name but .txt extension instead of image extension\n            relative_path = os.path.relpath(img_path, images_train_dir)\n            label_path = os.path.join(labels_train_dir, os.path.splitext(relative_path)[0] + '.txt')\n            \n            # Load the image\n            img = Image.open(img_path)\n            img_width, img_height = img.size\n            \n            # Normalize image using percentiles for better visualization\n            img_array = np.array(img)\n            p2 = np.percentile(img_array, 2)\n            p98 = np.percentile(img_array, 98)\n            normalized = np.clip(img_array, p2, p98)\n            normalized = 255 * (normalized - p2) / (p98 - p2)\n            img_normalized = Image.fromarray(np.uint8(normalized))\n            \n            # Convert image to RGB for colored box\n            img_rgb = img_normalized.convert('RGB')\n            \n            # Create a transparent overlay\n            overlay = Image.new('RGBA', img_rgb.size, (0, 0, 0, 0))\n            draw = ImageDraw.Draw(overlay)\n            \n            # Load YOLO format annotations if they exist\n            annotations = []\n            if os.path.exists(label_path):\n                with open(label_path, 'r') as f:\n                    for line in f:\n                        # YOLO format: class x_center y_center width height\n                        # All values are normalized from 0 to 1\n                        values = line.strip().split()\n                        class_id = int(values[0])\n                        x_center = float(values[1]) * img_width\n                        y_center = float(values[2]) * img_height\n                        width = float(values[3]) * img_width\n                        height = float(values[4]) * img_height\n                        \n                        annotations.append({\n                            'class_id': class_id,\n                            'x_center': x_center,\n                            'y_center': y_center,\n                            'width': width,\n                            'height': height\n                        })\n            \n            # Draw all annotations\n            for ann in annotations:\n                x_center = ann['x_center']\n                y_center = ann['y_center']\n                width = ann['width']\n                height = ann['height']\n                \n                # Calculate bounding box coordinates\n                x1 = max(0, int(x_center - width/2))\n                y1 = max(0, int(y_center - height/2))\n                x2 = min(img_width, int(x_center + width/2))\n                y2 = min(img_height, int(y_center + height/2))\n                \n                # Draw semi-transparent red rectangle\n                draw.rectangle([x1, y1, x2, y2], fill=(255, 0, 0, 64), outline=(255, 0, 0, 200))\n                \n                # Draw label\n                label_text = f\"Class {ann['class_id']}\"\n                draw.text((x1, y1-10), label_text, fill=(255, 0, 0, 255))\n            \n            # If no annotations found, indicate this\n            if not annotations:\n                draw.text((10, 10), \"No annotations found\", fill=(255, 0, 0, 255))\n            \n            # Composite the overlay onto the original image\n            img_rgb = Image.alpha_composite(img_rgb.convert('RGBA'), overlay).convert('RGB')\n            \n            # Display the image with annotations\n            axes[i].imshow(np.array(img_rgb))\n            img_name = os.path.basename(img_path)\n            axes[i].set_title(f\"Image: {img_name}\\nAnnotations: {len(annotations)}\")\n            axes[i].axis('on')\n            \n        except Exception as e:\n            print(f\"Error processing image {img_path}: {e}\")\n            axes[i].text(0.5, 0.5, f\"Error loading image: {os.path.basename(img_path)}\", \n                       horizontalalignment='center', verticalalignment='center')\n            axes[i].axis('off')\n    \n    # Handle extra subplots if any\n    for j in range(i + 1, len(axes)):\n        axes[j].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print summary\n    print(f\"Displayed {num_samples} random images with YOLO annotations\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.837347Z","iopub.execute_input":"2025-11-03T04:32:11.837558Z","iopub.status.idle":"2025-11-03T04:32:11.851612Z","shell.execute_reply.started":"2025-11-03T04:32:11.837541Z","shell.execute_reply":"2025-11-03T04:32:11.851034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_random_training_samples(4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T04:32:11.852361Z","iopub.execute_input":"2025-11-03T04:32:11.852623Z","iopub.status.idle":"2025-11-03T04:32:13.305839Z","shell.execute_reply.started":"2025-11-03T04:32:11.852602Z","shell.execute_reply":"2025-11-03T04:32:13.305073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# YOLO Model Training","metadata":{}},{"cell_type":"code","source":"# Set random seeds for reproducibility\nnp.random.seed(42)\nrandom.seed(42)\ntorch.manual_seed(42)\n\n# Define paths for Kaggle environment\nyolo_dataset_dir = \"/kaggle/working/yolo_dataset\"\nyolo_weights_dir = \"/kaggle/working/yolo_weights\"\n# yolo_pretrained_weights = \"yolov8n.pt\"  # Path to pre-downloaded weights\n# yolo_pretrained_weights = \"/kaggle/input/yolo11/pytorch/default/1/yolo11n.pt\"\n# yolo_pretrained_weights = 'yolo11x.yaml'\nyolo_pretrained_weights = 'yolo11n.yaml'\n# Create weights directory if it doesn't exist\nos.makedirs(yolo_weights_dir, exist_ok=True)\n\ndef fix_yaml_paths(yaml_path):\n    \"\"\"\n    Fix the paths in the YAML file to match the actual Kaggle directories\n    \n    Args:\n        yaml_path (str): Path to the original dataset YAML file\n        \n    Returns:\n        str: Path to the fixed YAML file\n    \"\"\"\n    print(f\"Fixing YAML paths in {yaml_path}\")\n    \n    # Read the original YAML\n    with open(yaml_path, 'r') as f:\n        yaml_data = yaml.safe_load(f)\n    \n    # Update paths to use actual dataset location\n    if 'path' in yaml_data:\n        yaml_data['path'] = yolo_dataset_dir\n    \n    # Create a new fixed YAML in the working directory\n    fixed_yaml_path = \"/kaggle/working/fixed_dataset.yaml\"\n    with open(fixed_yaml_path, 'w') as f:\n        yaml.dump(yaml_data, f)\n    \n    print(f\"Created fixed YAML at {fixed_yaml_path} with path: {yaml_data.get('path')}\")\n    return fixed_yaml_path\n\ndef plot_dfl_loss_curve(run_dir):\n    \"\"\"\n    Plot the DFL loss curves for train and validation, marking the best model\n    \n    Args:\n        run_dir (str): Directory where the training results are stored\n    \"\"\"\n    # Path to the results CSV file\n    results_csv = os.path.join(run_dir, 'results.csv')\n    \n    if not os.path.exists(results_csv):\n        print(f\"Results file not found at {results_csv}\")\n        return\n    \n    # Read results CSV\n    results_df = pd.read_csv(results_csv)\n    \n    # Check if DFL loss columns exist\n    train_dfl_col = [col for col in results_df.columns if 'train/dfl_loss' in col]\n    val_dfl_col = [col for col in results_df.columns if 'val/dfl_loss' in col]\n    \n    if not train_dfl_col or not val_dfl_col:\n        print(\"DFL loss columns not found in results CSV\")\n        print(f\"Available columns: {results_df.columns.tolist()}\")\n        return\n    \n    train_dfl_col = train_dfl_col[0]\n    val_dfl_col = val_dfl_col[0]\n    \n    # Find the epoch with the best validation loss\n    best_epoch = results_df[val_dfl_col].idxmin()\n    best_val_loss = results_df.loc[best_epoch, val_dfl_col]\n    \n    # Create the plot\n    plt.figure(figsize=(10, 6))\n    \n    # Plot training and validation losses\n    plt.plot(results_df['epoch'], results_df[train_dfl_col], label='Train DFL Loss')\n    plt.plot(results_df['epoch'], results_df[val_dfl_col], label='Validation DFL Loss')\n    \n    # Mark the best model with a vertical line\n    plt.axvline(x=results_df.loc[best_epoch, 'epoch'], color='r', linestyle='--', \n                label=f'Best Model (Epoch {int(results_df.loc[best_epoch, \"epoch\"])}, Val Loss: {best_val_loss:.4f})')\n    \n    # Add labels and legend\n    plt.xlabel('Epoch')\n    plt.ylabel('DFL Loss')\n    plt.title('Training and Validation DFL Loss')\n    plt.legend()\n    plt.grid(True, linestyle='--', alpha=0.7)\n    \n    # Save the plot in the same directory as weights\n    plot_path = os.path.join(run_dir, 'dfl_loss_curve.png')\n    plt.savefig(plot_path)\n    \n    # Also save it to the working directory for easier access\n    plt.savefig(os.path.join('/kaggle/working', 'dfl_loss_curve.png'))\n    \n    print(f\"Loss curve saved to {plot_path}\")\n    plt.close()\n    \n    # Return the best epoch info\n    return best_epoch, best_val_loss\n\n# def train_yolo_model(yaml_path, pretrained_weights_path, epochs=30, batch_size=16, img_size=640,model_file = '/kaggle/input/custom-yolo/pytorch/custom_yolo11_6/1/custome_YOLOv11.yaml'):\n#     \"\"\"\n#     Train a YOLO model on the prepared dataset\n    \n#     Args:\n#         yaml_path (str): Path to the dataset YAML file\n#         pretrained_weights_path (str): Path to pre-downloaded weights file\n#         epochs (int): Number of training epochs\n#         batch_size (int): Batch size for training\n#         img_size (int): Image size for training\n#     \"\"\"\n#     print(f\"Loading pre-trained weights from: {pretrained_weights_path}\")\n    \n#     # Load a pre-trained YOLOv8 model\n#     # model = YOLO(pretrained_weights_path)\n#     # model = YOLO(pretrained_weights_path) # Training Model from scratch\n#     model = YOLO(model_file)\n#     model.load(pretrained_weights_path,strict=False)\n#     # Train the model with early stopping\n#     results = model.train(\n#         data=yaml_path,\n#         epochs=epochs,\n#         batch=batch_size,\n#         imgsz=img_size,\n#         project=yolo_weights_dir,\n#         name='motor_detector',\n#         exist_ok=True,\n#         patience=5,              # Early stopping if no improvement for 5 epochs\n#         save_period=5,           # Save checkpoints every 5 epochs\n#         val=True,                # Ensure validation is performed\n#         verbose=True             # Show detailed output during training\n#     )\n    \n#     # Get the path to the run directory\n#     run_dir = os.path.join(yolo_weights_dir, 'motor_detector')\n    \n#     # Plot and save the loss curve\n#     best_epoch_info = plot_dfl_loss_curve(run_dir)\n    \n#     if best_epoch_info:\n#         best_epoch, best_val_loss = best_epoch_info\n#         print(f\"\\nBest model found at epoch {best_epoch} with validation DFL loss: {best_val_loss:.4f}\")\n    \n#     return model, results\ndef train_yolo_model(yaml_path, pretrained_weights_path, epochs=30, batch_size=16, img_size=640, model_file='/kaggle/input/custom-yolo/pytorch/custom_yolo11_19/1/coustome_yolov11-2.yaml'):\n    \"\"\"\n    Train a YOLO model on the prepared dataset\n    \n    Args:\n        yaml_path (str): Path to the dataset YAML file\n        pretrained_weights_path (str): Path to pre-downloaded weights file\n        epochs (int): Number of training epochs\n        batch_size (int): Batch size for training\n        img_size (int): Image size for training\n        model_file (str): Path to custom YOLO architecture YAML\n    \"\"\"\n    print(f\"Loading custom architecture from: {model_file}\")\n    print(f\"Loading pre-trained weights from: {pretrained_weights_path}\")\n    \n    # Create model from custom architecture\n    model = YOLO(model_file)\n    # model = YOLO(\"yolo11.pt\")\n    # Load pretrained weights if it's a .pt file\n    # if pretrained_weights_path.endswith('.pt'):\n    #     try:\n    #         # Load weights with strict=False to allow partial loading\n    #         import torch\n    #         checkpoint = torch.load(pretrained_weights_path, map_location='cpu')\n            \n    #         # Try to load the state dict with strict=False\n    #         if 'model' in checkpoint:\n    #             model.model.load_state_dict(checkpoint['model'].float().state_dict(), strict=False)\n    #         else:\n    #             model.model.load_state_dict(checkpoint, strict=False)\n            \n    #         print(\"Successfully loaded pretrained weights (partial loading)\")\n    #     except Exception as e:\n    #         print(f\"Warning: Could not load pretrained weights: {e}\")\n    #         print(\"Training from scratch with custom architecture\")\n    # else:\n    #     print(f\"Pretrained weights path is a YAML file, training from scratch\")\n    \n    # Train the model with early stopping\n    results = model.train(\n        data=yaml_path,\n        epochs=epochs,\n        batch=batch_size,\n        imgsz=img_size,\n        project=yolo_weights_dir,\n        name='motor_detector',\n        exist_ok=True,\n        patience=10,              # Early stopping if no improvement for 5 epochs\n        save_period=10,           # Save checkpoints every 5 epochs\n        val=True,                # Ensure validation is performed\n        verbose=True             # Show detailed output during training\n    )\n    \n    # Get the path to the run directory\n    run_dir = os.path.join(yolo_weights_dir, 'motor_detector')\n    \n    # Plot and save the loss curve\n    best_epoch_info = plot_dfl_loss_curve(run_dir)\n    \n    if best_epoch_info:\n        best_epoch, best_val_loss = best_epoch_info\n        print(f\"\\nBest model found at epoch {best_epoch} with validation DFL loss: {best_val_loss:.4f}\")\n    \n    return model, results\ndef predict_on_samples(model, num_samples=4):\n    \"\"\"\n    Run predictions on random validation samples and display results\n    \n    Args:\n        model: Trained YOLO model\n        num_samples (int): Number of random samples to test\n    \"\"\"\n    # Get validation images\n    val_dir = os.path.join(yolo_dataset_dir, 'images', 'val')\n    if not os.path.exists(val_dir):\n        print(f\"Validation directory not found at {val_dir}\")\n        # Try train directory instead if val doesn't exist\n        val_dir = os.path.join(yolo_dataset_dir, 'images', 'train')\n        print(f\"Using train directory for predictions instead: {val_dir}\")\n        \n    if not os.path.exists(val_dir):\n        print(\"No images directory found for predictions\")\n        return\n    \n    val_images = os.listdir(val_dir)\n    \n    if len(val_images) == 0:\n        print(\"No images found for prediction\")\n        return\n    \n    # Select random samples\n    num_samples = min(num_samples, len(val_images))\n    samples = random.sample(val_images, num_samples)\n    \n    # Create figure\n    fig, axes = plt.subplots(2, 2, figsize=(12, 12))\n    axes = axes.flatten()\n    \n    for i, img_file in enumerate(samples):\n        if i >= len(axes):\n            break\n            \n        img_path = os.path.join(val_dir, img_file)\n        \n        # Run prediction\n        results = model.predict(img_path, conf=0.25)[0]\n        \n        # Load and display the image\n        img = Image.open(img_path)\n        axes[i].imshow(np.array(img), cmap='gray')\n        \n        # Draw ground truth box if available (from filename)\n        try:\n            # This assumes your filenames contain coordinates in a specific format\n            parts = img_file.split('_')\n            y_part = [p for p in parts if p.startswith('y')]\n            x_part = [p for p in parts if p.startswith('x')]\n            \n            if y_part and x_part:\n                y_gt = int(y_part[0][1:])\n                x_gt = int(x_part[0][1:].split('.')[0])\n                \n                box_size = 24\n                rect_gt = Rectangle((x_gt - box_size//2, y_gt - box_size//2), \n                              box_size, box_size, \n                              linewidth=1, edgecolor='g', facecolor='none')\n                axes[i].add_patch(rect_gt)\n        except:\n            pass  # Skip ground truth if parsing fails\n        \n        # Draw predicted boxes (red)\n        if len(results.boxes) > 0:\n            boxes = results.boxes.xyxy.cpu().numpy()\n            confs = results.boxes.conf.cpu().numpy()\n            \n            for box, conf in zip(boxes, confs):\n                x1, y1, x2, y2 = box\n                rect_pred = Rectangle((x1, y1), x2-x1, y2-y1, \n                                     linewidth=1, edgecolor='r', facecolor='none')\n                axes[i].add_patch(rect_pred)\n                axes[i].text(x1, y1-5, f'{conf:.2f}', color='red')\n        \n        axes[i].set_title(f\"Image: {img_file}\\nGround Truth (green) vs Prediction (red)\")\n    \n    plt.tight_layout()\n    \n    # Save the predictions plot\n    plt.savefig(os.path.join('/kaggle/working', 'predictions.png'))\n    plt.show()\n\n# Check and create a dataset YAML if needed\ndef prepare_dataset():\n    \"\"\"\n    Check if dataset exists and create a proper YAML if needed\n    \n    Returns:\n        str: Path to the YAML file to use for training\n    \"\"\"\n    # Check if images exist\n    train_images_dir = os.path.join(yolo_dataset_dir, 'images', 'train')\n    val_images_dir = os.path.join(yolo_dataset_dir, 'images', 'val')\n    train_labels_dir = os.path.join(yolo_dataset_dir, 'labels', 'train')\n    val_labels_dir = os.path.join(yolo_dataset_dir, 'labels', 'val')\n    \n    # Print directory existence status\n    print(f\"Directory status:\")\n    print(f\"- Train images dir exists: {os.path.exists(train_images_dir)}\")\n    print(f\"- Val images dir exists: {os.path.exists(val_images_dir)}\")\n    print(f\"- Train labels dir exists: {os.path.exists(train_labels_dir)}\")\n    print(f\"- Val labels dir exists: {os.path.exists(val_labels_dir)}\")\n    \n    # Check for original YAML file\n    original_yaml_path = os.path.join(yolo_dataset_dir, 'dataset.yaml')\n    \n    if os.path.exists(original_yaml_path):\n        print(f\"Found original dataset.yaml at {original_yaml_path}\")\n        # Fix the paths in the YAML\n        return fix_yaml_paths(original_yaml_path)\n    else:\n        print(f\"Original dataset.yaml not found, creating a new one\")\n        \n        # Create a new YAML file\n        yaml_data = {\n            'path': yolo_dataset_dir,\n            'train': 'images/train',\n            'val': 'images/train' if not os.path.exists(val_images_dir) else 'images/val',\n            'names': {0: 'motor'}\n        }\n        \n        new_yaml_path = \"/kaggle/working/dataset.yaml\"\n        with open(new_yaml_path, 'w') as f:\n            yaml.dump(yaml_data, f)\n            \n        print(f\"Created new YAML at {new_yaml_path}\")\n        return new_yaml_path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T05:42:25.685373Z","iopub.execute_input":"2025-11-03T05:42:25.685995Z","iopub.status.idle":"2025-11-03T05:42:25.711750Z","shell.execute_reply.started":"2025-11-03T05:42:25.685964Z","shell.execute_reply":"2025-11-03T05:42:25.711015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(\"Starting YOLO training process...\")\n\n# Prepare dataset and get YAML path\nyaml_path = prepare_dataset()\nprint(f\"Using YAML file: {yaml_path}\")\n\n# Print YAML file contents\nwith open(yaml_path, 'r') as f:\n    yaml_content = f.read()\nprint(f\"YAML file contents:\\n{yaml_content}\")\n\n# Train model\nprint(\"\\nStarting YOLO training...\")\nmodel, results = train_yolo_model(\n    yaml_path,\n    pretrained_weights_path=yolo_pretrained_weights,\n    epochs=80  # Using 30 epochs instead of 100 for faster training\n)\n\nprint(\"\\nTraining complete!\")\n\n# Run predictions\nprint(\"\\nRunning predictions on sample images...\")\npredict_on_samples(model, num_samples=4)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T05:42:25.888680Z","iopub.execute_input":"2025-11-03T05:42:25.889384Z","iopub.status.idle":"2025-11-03T05:58:24.610174Z","shell.execute_reply.started":"2025-11-03T05:42:25.889359Z","shell.execute_reply":"2025-11-03T05:58:24.608752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"metrics = model.val(data='/kaggle/working/yolo_dataset/dataset.yaml', split='test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T05:58:28.542166Z","iopub.execute_input":"2025-11-03T05:58:28.542531Z","iopub.status.idle":"2025-11-03T05:58:37.518707Z","shell.execute_reply.started":"2025-11-03T05:58:28.542497Z","shell.execute_reply":"2025-11-03T05:58:37.517924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print metrics\nprint(f\"Precision: {metrics.box.p[0]:.2f}\")\nprint(f\"Recall: {metrics.box.r[0]:.2f}\")\nprint(f\"mAP@0.5: {metrics.box.map50:.3f}\")\nprint(f\"mAP@0.5:0.95: {metrics.box.map:.3f}\")\nprint(f\"F_1 Score: {2*(metrics.box.p[0]*metrics.box.r[0])/(metrics.box.p[0]+ metrics.box.r[0]):.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T05:58:37.520617Z","iopub.execute_input":"2025-11-03T05:58:37.520868Z","iopub.status.idle":"2025-11-03T05:58:37.526151Z","shell.execute_reply.started":"2025-11-03T05:58:37.520844Z","shell.execute_reply":"2025-11-03T05:58:37.525399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Image\nImage(filename='/kaggle/working/yolo_weights/motor_detector/results.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-03T05:58:37.526898Z","iopub.execute_input":"2025-11-03T05:58:37.527420Z","iopub.status.idle":"2025-11-03T05:58:37.551450Z","shell.execute_reply.started":"2025-11-03T05:58:37.527401Z","shell.execute_reply":"2025-11-03T05:58:37.550843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}