{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport shutil\nimport time\nimport yaml\nimport cv2\nfrom pathlib import Path\nfrom sklearn.model_selection import KFold\nfrom tqdm.notebook import tqdm\n\n# Set random seed for reproducibility\nnp.random.seed(42)\n\n# Define Kaggle paths\ndata_path = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/\"\ntrain_dir = os.path.join(data_path, \"train\")\n\n# Define YOLO dataset structure\nyolo_dataset_dir = \"/kaggle/working/yolo_dataset_5fold\"\nos.makedirs(yolo_dataset_dir, exist_ok=True)\n\n# Define constants\nTRUST = 4  # Number of slices above and below center slice\nBOX_SIZE = 24  # Bounding box size for annotations (in pixels)\nIMG_SIZE = 640\nN_FOLDS = 5  # Number of folds for cross-validation\n\n# Image processing functions\ndef normalize_slice(slice_data):\n    \"\"\"Normalize slice data using 2nd and 98th percentiles\"\"\"\n    p2 = np.percentile(slice_data, 2)\n    p98 = np.percentile(slice_data, 98)\n    clipped_data = np.clip(slice_data, p2, p98)\n    normalized = 255 * (clipped_data - p2) / (p98 - p2)\n    return np.uint8(normalized)\n\ndef prepare_fold_datasets(trust=TRUST, n_folds=N_FOLDS):\n    \"\"\"\n    Prepare 5-fold cross validation datasets for YOLO training\n    \"\"\"\n    # Load the labels CSV\n    labels_df = pd.read_csv(os.path.join(data_path, \"train_labels.csv\"))\n    \n    # Filter tomograms with motors\n    tomo_df = labels_df[labels_df['Number of motors'] > 0].copy()\n    unique_tomos = tomo_df['tomo_id'].unique()\n    np.random.shuffle(unique_tomos)  # Shuffle the tomograms\n    \n    print(f\"Found {len(unique_tomos)} unique tomograms with motors\")\n    print(f\"Preparing {n_folds}-fold cross validation datasets...\")\n    \n    # Initialize KFold\n    kf = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n    \n    fold_info = []\n    \n    for fold, (train_idx, val_idx) in enumerate(kf.split(unique_tomos), 1):\n        print(f\"\\nProcessing Fold {fold}/{n_folds}\")\n        \n        # Create fold directories\n        fold_dir = os.path.join(yolo_dataset_dir, f\"fold_{fold}\")\n        yolo_images_train = os.path.join(fold_dir, \"images\", \"train\")\n        yolo_images_val = os.path.join(fold_dir, \"images\", \"val\")\n        yolo_labels_train = os.path.join(fold_dir, \"labels\", \"train\")\n        yolo_labels_val = os.path.join(fold_dir, \"labels\", \"val\")\n        \n        for dir_path in [yolo_images_train, yolo_images_val, yolo_labels_train, yolo_labels_val]:\n            os.makedirs(dir_path, exist_ok=True)\n        \n        train_tomos = unique_tomos[train_idx]\n        val_tomos = unique_tomos[val_idx]\n        \n        # Function to process a set of tomograms\n        def process_tomogram_set(tomogram_ids, images_dir, labels_dir, set_name):\n            motor_counts = []\n            for tomo_id in tomogram_ids:\n                tomo_motors = labels_df[labels_df['tomo_id'] == tomo_id]\n                for _, motor in tomo_motors.iterrows():\n                    if pd.isna(motor['Motor axis 0']):\n                        continue\n                    motor_counts.append(\n                        (tomo_id, \n                         int(motor['Motor axis 0']), \n                         int(motor['Motor axis 1']), \n                         int(motor['Motor axis 2']),\n                         int(motor['Array shape (axis 0)']))\n                    )\n            \n            processed_slices = 0\n            \n            for tomo_id, z_center, y_center, x_center, z_max in tqdm(motor_counts, desc=f\"Processing {set_name} motors\"):\n                z_min = max(0, z_center - trust)\n                z_max = min(z_max - 1, z_center + trust)\n                \n                for z in range(z_min, z_max + 1):\n                    slice_filename = f\"slice_{z:04d}.jpg\"\n                    src_path = os.path.join(train_dir, tomo_id, slice_filename)\n                    \n                    if not os.path.exists(src_path):\n                        continue\n                    \n                    # Load and process image\n                    img = Image.open(src_path)\n                    orig_width, orig_height = img.size\n                    img_array = np.array(img)\n                    scale = min(IMG_SIZE/orig_width, IMG_SIZE/orig_height)\n                    new_width = int(orig_width * scale)\n                    new_height = int(orig_height * scale)\n                    resized_img = cv2.resize(img_array, (new_width, new_height), interpolation=cv2.INTER_AREA)\n                    canvas = np.zeros((IMG_SIZE, IMG_SIZE), dtype=np.uint8) + 114\n                    x_offset = (IMG_SIZE - new_width) // 2\n                    y_offset = (IMG_SIZE - new_height) // 2\n                    canvas[y_offset:y_offset+new_height, x_offset:x_offset+new_width] = resized_img\n                    normalized_img = normalize_slice(canvas)\n                    \n                    # Create destination filename\n                    dest_filename = f\"{tomo_id}_z{z:04d}.jpg\"\n                    dest_path = os.path.join(images_dir, dest_filename)\n                    Image.fromarray(normalized_img).save(dest_path)\n                    \n                    # Calculate and save annotations\n                    scaled_x = x_center * scale\n                    scaled_y = y_center * scale\n                    new_x_center = scaled_x + x_offset\n                    new_y_center = scaled_y + y_offset\n                    scaled_box_size = BOX_SIZE * scale\n                    x_center_norm = new_x_center / IMG_SIZE\n                    y_center_norm = new_y_center / IMG_SIZE\n                    box_width_norm = scaled_box_size / IMG_SIZE\n                    box_height_norm = scaled_box_size / IMG_SIZE\n                    \n                    label_path = os.path.join(labels_dir, dest_filename.replace('.jpg', '.txt'))\n                    with open(label_path, 'a') as f:\n                        f.write(f\"0 {x_center_norm} {y_center_norm} {box_width_norm} {box_height_norm}\\n\")\n                    \n                    processed_slices += 1\n            \n            return processed_slices, len(motor_counts)\n        \n        # Process training and validation sets\n        train_slices, train_motors = process_tomogram_set(train_tomos, yolo_images_train, yolo_labels_train, \"training\")\n        val_slices, val_motors = process_tomogram_set(val_tomos, yolo_images_val, yolo_labels_val, \"validation\")\n        \n        # Create YAML configuration file for this fold\n        yaml_content = {\n            'path': fold_dir,\n            'train': 'images/train',\n            'val': 'images/val',\n            'names': {0: 'motor'}\n        }\n        \n        yaml_path = os.path.join(fold_dir, 'dataset.yaml')\n        with open(yaml_path, 'w') as f:\n            yaml.dump(yaml_content, f, default_flow_style=False)\n        \n        fold_info.append({\n            \"fold\": fold,\n            \"dir\": fold_dir,\n            \"yaml_path\": yaml_path,\n            \"train_tomograms\": len(train_tomos),\n            \"val_tomograms\": len(val_tomos),\n            \"train_motors\": train_motors,\n            \"val_motors\": val_motors,\n            \"train_slices\": train_slices,\n            \"val_slices\": val_slices\n        })\n        \n        print(f\"Fold {fold} Summary:\")\n        print(f\"- Train: {len(train_tomos)} tomograms, {train_motors} motors, {train_slices} slices\")\n        print(f\"- Val: {len(val_tomos)} tomograms, {val_motors} motors, {val_slices} slices\")\n    \n    # Save overall summary\n    summary_path = os.path.join(yolo_dataset_dir, \"cross_validation_summary.yaml\")\n    with open(summary_path, 'w') as f:\n        yaml.dump({\"folds\": fold_info}, f, default_flow_style=False)\n    \n    print(f\"\\n5-fold Cross Validation Preparation Complete!\")\n    print(f\"Summary saved to: {summary_path}\")\n    \n    return fold_info\n\n# Run the preprocessing\nfold_info = prepare_fold_datasets()\n\n# Print final summary\nprint(\"\\nFinal Cross Validation Dataset Summary:\")\nfor fold in fold_info:\n    print(f\"\\nFold {fold['fold']}:\")\n    print(f\"- Directory: {fold['dir']}\")\n    print(f\"- Train: {fold['train_tomograms']} tomograms, {fold['train_motors']} motors, {fold['train_slices']} slices\")\n    print(f\"- Val: {fold['val_tomograms']} tomograms, {fold['val_motors']} motors, {fold['val_slices']} slices\")\n    print(f\"- Config: {fold['yaml_path']}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}