{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport shutil\nimport time\nimport yaml\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm  # Use tqdm.notebook for Jupyter/Kaggle environments\n\n# Set random seed for reproducibility\nnp.random.seed(42)\n\n# Define Kaggle paths\ndata_path = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/\"\ntrain_dir = os.path.join(data_path, \"train\")\n\n\n# Define constants\nTRUST = 4  # Number of slices above and below center slice (total 2*TRUST + 1 slices)\nBOX_SIZE = 24  # Bounding box size for annotations (in pixels)\nTRAIN_SPLIT = 0.8  # 80% for training, 20% for validation\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport json\nfrom tqdm import tqdm\n\ndef normalize_slice(slice_data):\n    \"\"\"\n    Normalize slice data using 2nd and 98th percentiles\n    \"\"\"\n    p2 = np.percentile(slice_data, 2)\n    p98 = np.percentile(slice_data, 98)\n    clipped_data = np.clip(slice_data, p2, p98)\n    normalized = 255 * (clipped_data - p2) / (p98 - p2)\n    return np.uint8(normalized)\n\ndef prepare_json_dataset(trust=TRUST, train_split=TRAIN_SPLIT, data_path=\"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025\", output_dir=\"dataset_json\"):\n    \"\"\"\n    Extract slices containing motors and save to JSON format dataset structure\n    \"\"\"\n    # Define directory structure\n    os.makedirs(output_dir, exist_ok=True)\n    train_img_dir = os.path.join(output_dir, \"train2017\")\n    val_img_dir = os.path.join(output_dir, \"valid2017\")\n    os.makedirs(train_img_dir, exist_ok=True)\n    os.makedirs(val_img_dir, exist_ok=True)\n\n    # Load the labels CSV\n    labels_df = pd.read_csv(os.path.join(data_path, \"train_labels.csv\"))\n    \n    total_motors = labels_df['Number of motors'].sum()\n    print(f\"Total number of motors in the dataset: {total_motors}\")\n    \n    # Get unique tomograms with motors\n    tomo_df = labels_df[labels_df['Number of motors'] >= 1].copy()\n    unique_tomos = tomo_df['tomo_id'].unique()\n    \n    print(f\"Found {len(unique_tomos)} unique tomograms with motors\")\n    \n    # Train-val split at tomogram level\n    np.random.shuffle(unique_tomos)\n    split_idx = int(len(unique_tomos) * train_split)\n    train_tomos = unique_tomos[:split_idx]\n    val_tomos = unique_tomos[split_idx:]\n    \n    print(f\"Split: {len(train_tomos)} tomograms for training, {len(val_tomos)} tomograms for validation\")\n    \n    def process_tomogram_set(tomogram_ids, images_dir, set_name):\n        images = []\n        annotations = []\n        image_id = 0\n        annotation_id = 0\n        processed_slices = 0\n        \n        motor_counts = []\n        for tomo_id in tomogram_ids:\n            tomo_motors = labels_df[labels_df['tomo_id'] == tomo_id]\n            for _, motor in tomo_motors.iterrows():\n                if pd.isna(motor['Motor axis 0']):\n                    continue\n                motor_counts.append(\n                    (tomo_id, \n                     int(motor['Motor axis 0']), \n                     int(motor['Motor axis 1']), \n                     int(motor['Motor axis 2']),\n                     int(motor['Array shape (axis 0)']))\n                )\n        \n        print(f\"Will process approximately {len(motor_counts) * (2 * trust + 1)} slices for {set_name}\")\n        \n        for tomo_id, z_center, y_center, x_center, z_max in tqdm(motor_counts, desc=f\"Processing {set_name} motors\"):\n            z_min = max(0, z_center - trust)\n            z_max = min(z_max - 1, z_center + trust)\n            \n            for z in range(z_min, z_max + 1):\n                slice_filename = f\"slice_{z:04d}.jpg\"\n                src_path = os.path.join(train_dir, tomo_id, slice_filename)\n                \n                if not os.path.exists(src_path):\n                    print(f\"Warning: {src_path} does not exist, skipping.\")\n                    continue\n                \n                # Load and normalize image\n                img = Image.open(src_path)\n                img_array = np.array(img)\n                normalized_img = normalize_slice(img_array)\n                \n                # Generate filename\n                dest_filename = f\"{tomo_id}_z{z:04d}_y{y_center:04d}_x{x_center:04d}.jpg\"\n                dest_path = os.path.join(images_dir, dest_filename)\n                \n                # Save image\n                Image.fromarray(normalized_img).save(dest_path)\n                \n                # Image metadata\n                img_width, img_height = img.size\n                images.append({\n                    \"id\": image_id,\n                    \"file_name\": dest_filename,\n                    \"width\": img_width,\n                    \"height\": img_height\n                })\n                \n                # Annotation (bounding box centered on motor)\n                x_min = max(0, x_center - BOX_SIZE//2)\n                y_min = max(0, y_center - BOX_SIZE//2)\n                width = min(BOX_SIZE, img_width - x_min)\n                height = min(BOX_SIZE, img_height - y_min)\n                \n                segmentation = [\n                    x_min, y_min,              \n                    x_min + width, y_min,      \n                    x_min + width, y_min + height,  \n                    x_min, y_min + height    \n                ]\n                \n                annotations.append({\n                    \"id\": annotation_id,\n                    \"image_id\": image_id,\n                    \"category_id\": 0,  # 0 for motor\n                    \"bbox\": [x_min, y_min, width, height],\n                    \"area\": width * height,\n                    \"iscrowd\": 0,\n                    \"segmentation\": [segmentation]\n                })\n                \n                image_id += 1\n                annotation_id += 1\n                processed_slices += 1\n        \n        return images, annotations, processed_slices, len(motor_counts)\n    \n    # Process train and validation sets\n    train_images, train_annotations, train_slices, train_motors = process_tomogram_set(train_tomos, train_img_dir, \"training\")\n    val_images, val_annotations, val_slices, val_motors = process_tomogram_set(val_tomos, val_img_dir, \"validation\")\n    \n    # Create JSON dataset structure\n    dataset_train = {\n        \"info\": {\"description\": \"Motor Detection Dataset\"},\n        \"categories\": [{\"id\": 0, \"name\": \"motor\"}],\n        \"images\": train_images,\n        \"annotations\": train_annotations\n    }\n\n    dataset_valid = {\n        \"info\": {\"description\": \"Motor Detection Dataset\"},\n        \"categories\": [{\"id\": 0, \"name\": \"motor\"}],\n        \"images\": val_images,\n        \"annotations\": val_annotations\n    }\n    \n    # Save JSON file\n    json_path = os.path.join(output_dir, \"annotations_train.json\")\n    with open(json_path, 'w') as f:\n        json.dump(dataset_train, f)\n\n    json_path = os.path.join(output_dir, \"annotations_valid.json\")\n    with open(json_path, 'w') as f:\n        json.dump(dataset_valid, f)\n    \n    print(f\"\\nProcessing Summary:\")\n    print(f\"- Train set: {len(train_tomos)} tomograms, {train_motors} motors, {train_slices} slices\")\n    print(f\"- Validation set: {len(val_tomos)} tomograms, {val_motors} motors, {val_slices} slices\")\n    print(f\"- Total: {len(train_tomos) + len(val_tomos)} tomograms, {train_motors + val_motors} motors, {train_slices + val_slices} slices\")\n    \n    return {\n        \"dataset_dir\": output_dir,\n        \"json_path\": json_path,\n        \"train_tomograms\": len(train_tomos),\n        \"val_tomograms\": len(val_tomos),\n        \"train_motors\": train_motors,\n        \"val_motors\": val_motors,\n        \"train_slices\": train_slices,\n        \"val_slices\": val_slices\n    }\n\n# Run the preprocessing\nsummary = prepare_json_dataset(TRUST)\nprint(f\"\\nPreprocessing Complete:\")\nprint(f\"- Training data: {summary['train_tomograms']} tomograms, {summary['train_motors']} motors, {summary['train_slices']} slices\")\nprint(f\"- Validation data: {summary['val_tomograms']} tomograms, {summary['val_motors']} motors, {summary['val_slices']} slices\")\nprint(f\"- Dataset directory: {summary['dataset_dir']}\")\nprint(f\"- JSON annotations: {summary['json_path']}\")\nprint(f\"\\nReady for training with JSON dataset!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T07:19:14.724441Z","iopub.execute_input":"2025-03-29T07:19:14.724844Z","iopub.status.idle":"2025-03-29T07:21:33.430909Z","shell.execute_reply.started":"2025-03-29T07:19:14.724809Z","shell.execute_reply":"2025-03-29T07:21:33.429735Z"}},"outputs":[],"execution_count":null}]}