{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":11153971,"sourceType":"datasetVersion","datasetId":6959173}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef clean_working(directory_path: str = \"/kaggle/working/\"):\n    \"\"\"\n    Clean kaggle output directory.\n    \"\"\"\n    if os.path.exists(directory_path):\n        for item in os.listdir(directory_path):\n            if item == \"submission.csv\":\n                continue\n            item_path = os.path.join(directory_path, item)\n            os.remove(item_path) if os.path.isfile(item_path) else shutil.rmtree(item_path)\n        print(f\"All items in '{directory_path}' have been removed.\")\n    else:\n        print(f\"'{directory_path}' does not exist.\")\n        \nclean_working()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T00:16:40.79564Z","iopub.execute_input":"2025-03-25T00:16:40.796095Z","iopub.status.idle":"2025-03-25T00:16:41.206496Z","shell.execute_reply.started":"2025-03-25T00:16:40.796048Z","shell.execute_reply":"2025-03-25T00:16:41.20556Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install zarr cryoet_data_portal -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T00:16:48.363271Z","iopub.execute_input":"2025-03-25T00:16:48.363571Z","iopub.status.idle":"2025-03-25T00:17:00.040124Z","shell.execute_reply.started":"2025-03-25T00:16:48.363541Z","shell.execute_reply":"2025-03-25T00:17:00.038878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd \npd.read_csv(\"/kaggle/input/cryoet-flagellar-motors-dataset/labels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T11:02:04.268824Z","iopub.execute_input":"2025-04-01T11:02:04.269175Z","iopub.status.idle":"2025-04-01T11:02:05.494552Z","shell.execute_reply.started":"2025-04-01T11:02:04.269140Z","shell.execute_reply":"2025-04-01T11:02:05.493500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport yaml\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\n\n# Set paths\ndata_path = \"/kaggle/input/cryoet-flagellar-motors-dataset/\"\nvolumes_dir = os.path.join(data_path, \"volumes\")\nlabels_path = os.path.join(data_path, \"labels.csv\")\n\n# YOLO dataset structure\nyolo_dataset_dir = \"/kaggle/working/yolo_dataset\"\nyolo_images_train = os.path.join(yolo_dataset_dir, \"images\", \"train\")\nyolo_images_val = os.path.join(yolo_dataset_dir, \"images\", \"val\")\nyolo_labels_train = os.path.join(yolo_dataset_dir, \"labels\", \"train\")\nyolo_labels_val = os.path.join(yolo_dataset_dir, \"labels\", \"val\")\n\n# Create directories\nfor dir_path in [yolo_images_train, yolo_images_val, yolo_labels_train, yolo_labels_val]:\n    os.makedirs(dir_path, exist_ok=True)\n\n# Constants\nTRUST = 4  # Number of slices above and below the center slice\nBOX_SIZE = 24  # Bounding box size in pixels\nTRAIN_SPLIT = 0.8  # 80% training, 20% validation\n\ndef normalize_slice(slice_data):\n    \"\"\"Normalize slice data using the 2nd and 98th percentiles.\"\"\"\n    p2, p98 = np.percentile(slice_data, [2, 98])\n    clipped = np.clip(slice_data, p2, p98)\n    normalized = 255 * (clipped - p2) / (p98 - p2)\n    return np.uint8(normalized)\n\ndef prepare_yolo_dataset():\n    \"\"\"Prepare dataset for YOLO training.\"\"\"\n    labels_df = pd.read_csv(labels_path)\n    unique_tomos = labels_df[\"tomo_id\"].unique()\n    np.random.shuffle(unique_tomos)\n    split_idx = int(len(unique_tomos) * TRAIN_SPLIT)\n    train_tomos, val_tomos = unique_tomos[:split_idx], unique_tomos[split_idx:]\n\n    def process_tomograms(tomogram_ids, images_dir, labels_dir, set_name):\n        processed_slices = 0\n        for tomo_id in tqdm(tomogram_ids, desc=f\"Processing {set_name} set\"):\n            volume_path = os.path.join(volumes_dir, f\"{tomo_id}.npy\")\n            if not os.path.exists(volume_path):\n                print(f\"Warning: {volume_path} not found, skipping.\")\n                continue\n            volume = np.load(volume_path)  # Load 3D volume\n            tomo_motors = labels_df[labels_df[\"tomo_id\"] == tomo_id]\n            for _, motor in tomo_motors.iterrows():\n                z_center, y_center, x_center = int(motor[\"z\"]), int(motor[\"y\"]), int(motor[\"x\"])\n                z_min, z_max = max(0, z_center - TRUST), min(volume.shape[0] - 1, z_center + TRUST)\n                for z in range(z_min, z_max + 1):\n                    slice_data = volume[z]\n                    normalized_img = normalize_slice(slice_data)\n                    dest_filename = f\"{tomo_id}_z{z:04d}_y{y_center:04d}_x{x_center:04d}.jpg\"\n                    Image.fromarray(normalized_img).save(os.path.join(images_dir, dest_filename))\n                    img_height, img_width = slice_data.shape\n                    x_norm, y_norm = x_center / img_width, y_center / img_height\n                    box_w_norm, box_h_norm = BOX_SIZE / img_width, BOX_SIZE / img_height\n                    with open(os.path.join(labels_dir, dest_filename.replace('.jpg', '.txt')), 'w') as f:\n                        f.write(f\"0 {x_norm} {y_norm} {box_w_norm} {box_h_norm}\\n\")\n                    processed_slices += 1\n        return processed_slices\n\n    train_slices = process_tomograms(train_tomos, yolo_images_train, yolo_labels_train, \"training\")\n    val_slices = process_tomograms(val_tomos, yolo_images_val, yolo_labels_val, \"validation\")\n    yaml_content = {\n        'path': yolo_dataset_dir,\n        'train': 'images/train',\n        'val': 'images/val',\n        'names': {0: 'motor'}\n    }\n    with open(os.path.join(yolo_dataset_dir, 'dataset.yaml'), 'w') as f:\n        yaml.dump(yaml_content, f, default_flow_style=False)\n    print(f\"\\nDataset ready: {train_slices} training slices, {val_slices} validation slices.\")\n\nprepare_yolo_dataset()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-01T11:04:07.001916Z","iopub.execute_input":"2025-04-01T11:04:07.002296Z","iopub.status.idle":"2025-04-01T11:13:09.560148Z","shell.execute_reply.started":"2025-04-01T11:04:07.002269Z","shell.execute_reply":"2025-04-01T11:13:09.558947Z"}},"outputs":[],"execution_count":null}]}