{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.8"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"2dee0a45","cell_type":"markdown","source":"# BYU - Locating Bacterial Flagellar Motors 2025 \n","metadata":{}},{"id":"727355f0","cell_type":"markdown","source":"## 1) Setup & Config","metadata":{}},{"id":"db626788","cell_type":"code","source":"import os, sys, gc, math, random, json, time, warnings\nfrom pathlib import Path\nfrom typing import Tuple, Optional, List\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings(\"ignore\")\n\n# Detect Kaggle\nON_KAGGLE = Path(\"/kaggle/input\").exists()\nprint(\"Running on Kaggle:\", ON_KAGGLE)\n\n# Paths\nCOMPETITION_SLUG = \"byu-locating-bacterial-flagellar-motors-2025\"\nBASE_DIR = Path(f\"/kaggle/input/{COMPETITION_SLUG}\") if ON_KAGGLE else Path(f\"./data/{COMPETITION_SLUG}\")\nOUTPUT_DIR = Path(\"/kaggle/working\") if ON_KAGGLE else Path(\"./outputs\")\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\nprint(\"BASE_DIR:\", BASE_DIR.resolve())\nprint(\"OUTPUT_DIR:\", OUTPUT_DIR.resolve())\n\n# Filenames (will auto-detect, but you can override)\nTRAIN_LABELS = BASE_DIR / \"train_labels.csv\"\nSAMPLE_SUB = BASE_DIR / \"sample_submission.csv\"\nTRAIN_DIR = BASE_DIR / \"train\"\nTEST_DIR  = BASE_DIR / \"test\"\n\ndef log(msg: str):\n    print(time.strftime(\"[%Y-%m-%d %H:%M:%S]\"), msg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:55:15.104813Z","iopub.execute_input":"2025-09-19T02:55:15.105433Z","iopub.status.idle":"2025-09-19T02:55:15.128254Z","shell.execute_reply.started":"2025-09-19T02:55:15.105399Z","shell.execute_reply":"2025-09-19T02:55:15.126397Z"}},"outputs":[],"execution_count":null},{"id":"c280fa7b","cell_type":"markdown","source":"## 2) Data Inspection / Auto-Detection","metadata":{}},{"id":"c62d1792","cell_type":"code","source":"# Check that folders/files exist\nfor p in [TRAIN_LABELS, SAMPLE_SUB, TRAIN_DIR, TEST_DIR]:\n    log(f\"{p} exists? {p.exists()}\")\n\n# Load sample submission if present\nsample_sub = None\nif SAMPLE_SUB.exists():\n    sample_sub = pd.read_csv(SAMPLE_SUB)\n    log(f\"sample_submission shape: {sample_sub.shape}\")\n    display(sample_sub.head())\nelse:\n    log(\"No sample_submission.csv found. We'll build IDs by scanning TEST_DIR.\")\n\n# Helper: get list of test tomo_ids\ndef discover_test_ids() -> List[str]:\n    ids = []\n    if sample_sub is not None and 'tomo_id' in sample_sub.columns:\n        return sample_sub['tomo_id'].astype(str).tolist()\n    # else discover by scanning test/\n    if TEST_DIR.exists():\n        # Case A: .npy volumes directly under test/\n        for f in sorted(TEST_DIR.glob(\"*.npy\")):\n            ids.append(f.stem)\n        # Case B: per-tomo subdirectories (with jpg slices)\n        for sub in sorted(TEST_DIR.iterdir()):\n            if sub.is_dir():\n                jpgs = list(sub.glob(\"*.jpg\")) + list(sub.glob(\"*.jpeg\")) + list(sub.glob(\"*.png\"))\n                if jpgs:\n                    ids.append(sub.name)\n    return sorted(list(dict.fromkeys(ids)))\n\ntest_ids = discover_test_ids()\nlog(f\"Discovered {len(test_ids)} test tomo_ids (first 5): {test_ids[:5]}\")\nassert len(test_ids) > 0, \"No test tomograms found. Place files under BASE_DIR/test/ or provide sample_submission.csv.\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:55:25.108811Z","iopub.execute_input":"2025-09-19T02:55:25.109230Z","iopub.status.idle":"2025-09-19T02:55:25.138199Z","shell.execute_reply.started":"2025-09-19T02:55:25.109197Z","shell.execute_reply":"2025-09-19T02:55:25.136941Z"}},"outputs":[],"execution_count":null},{"id":"b7732c37","cell_type":"markdown","source":"## 3) Volume Loader (supports `.npy` and JPG slices)","metadata":{}},{"id":"e2769c3b","cell_type":"code","source":"from PIL import Image\n\ndef load_volume(tomo_id: str) -> np.ndarray:\n    # Try to load a 3D volume for given tomo_id.\n    # Priority: test/{tomo_id}.npy -> test/{tomo_id}/*.jpg as z-stack.\n    # Returns float32 array (Z,Y,X) normalized to [0,1].\n    npy_path = TEST_DIR / f\"{tomo_id}.npy\"\n    if npy_path.exists():\n        vol = np.load(npy_path)\n        vol = np.asarray(vol, dtype=np.float32)\n    else:\n        slice_dir = TEST_DIR / tomo_id\n        assert slice_dir.exists(), f\"Cannot find npy or slice dir for {tomo_id}\"\n        slices = sorted(list(slice_dir.glob(\"*.jpg\")) + list(slice_dir.glob(\"*.jpeg\")) + list(slice_dir.glob(\"*.png\")))\n        assert len(slices) > 0, f\"No slices found for {tomo_id}\"\n        imgs = []\n        for p in slices:\n            img = Image.open(p).convert(\"F\")  # 32-bit float grayscale\n            imgs.append(np.asarray(img, dtype=np.float32))\n        vol = np.stack(imgs, axis=0)  # (Z,Y,X)\n\n    # Normalize per volume robustly\n    vmin, vmax = np.percentile(vol, [1, 99])\n    if not np.isfinite(vmin) or not np.isfinite(vmax) or vmax <= vmin:\n        vmin = float(np.nanmin(vol)) if np.isfinite(np.nanmin(vol)) else 0.0\n        vmax = float(np.nanmax(vol)) if np.isfinite(np.nanmax(vol)) else 1.0\n        if vmax <= vmin:\n            vmax = vmin + 1.0\n    vol = np.clip((vol - vmin) / (vmax - vmin), 0.0, 1.0).astype(np.float32)\n    return vol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:55:30.779774Z","iopub.execute_input":"2025-09-19T02:55:30.780203Z","iopub.status.idle":"2025-09-19T02:55:30.790091Z","shell.execute_reply.started":"2025-09-19T02:55:30.780172Z","shell.execute_reply":"2025-09-19T02:55:30.788779Z"}},"outputs":[],"execution_count":null},{"id":"5738e227","cell_type":"markdown","source":"## 4) Baseline Predictor: Brightest Voxel","metadata":{}},{"id":"aeb13144","cell_type":"code","source":"def predict_brightest_voxel(vol: np.ndarray, none_threshold: Optional[float]=None) -> Tuple[float,float,float]:\n    # Returns (z, y, x) of brightest voxel. If max < threshold -> (-1,-1,-1).\n    m = float(vol.max())\n    if none_threshold is not None and m < none_threshold:\n        return (-1.0, -1.0, -1.0)\n    idx = int(np.argmax(vol))  # flattened index\n    z, y, x = np.unravel_index(idx, vol.shape)\n    return (float(z), float(y), float(x))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:55:33.894816Z","iopub.execute_input":"2025-09-19T02:55:33.895208Z","iopub.status.idle":"2025-09-19T02:55:33.901592Z","shell.execute_reply.started":"2025-09-19T02:55:33.895178Z","shell.execute_reply":"2025-09-19T02:55:33.900177Z"}},"outputs":[],"execution_count":null},{"id":"36fc3960","cell_type":"markdown","source":"## 5) Run Inference over Test Set & Create `submission.csv`","metadata":{}},{"id":"da13d896","cell_type":"code","source":"PRED_NONE_THRESHOLD = None  # Example: set to 0.02 to abstain on very dark volumes\n\nrows = []\nfor i, tid in enumerate(test_ids):\n    if (i % 25) == 0:\n        log(f\"Inferencing {i}/{len(test_ids)}...\")\n    vol = load_volume(tid)\n    z, y, x = predict_brightest_voxel(vol, none_threshold=PRED_NONE_THRESHOLD)\n    rows.append({\"tomo_id\": tid, \"Motor axis 0\": z, \"Motor axis 1\": y, \"Motor axis 2\": x})\n\nsubmission = pd.DataFrame(rows)\n\n# Align to sample submission column order if available\nif 'sample_sub' in globals() and sample_sub is not None:\n    expected_cols = list(sample_sub.columns)\n    for c in [\"tomo_id\", \"Motor axis 0\", \"Motor axis 1\", \"Motor axis 2\"]:\n        assert c in submission.columns, f\"Missing column {c}\"\n    submission = submission[expected_cols]\n\nsub_path = (Path('/kaggle/working') if ON_KAGGLE else Path('./outputs')) / \"submission.csv\"\nsubmission.to_csv(sub_path, index=False)\nlog(f\"Saved submission to: {sub_path}\")\ndisplay(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:55:36.789069Z","iopub.execute_input":"2025-09-19T02:55:36.789444Z","iopub.status.idle":"2025-09-19T02:57:40.781167Z","shell.execute_reply.started":"2025-09-19T02:55:36.789411Z","shell.execute_reply":"2025-09-19T02:57:40.780046Z"}},"outputs":[],"execution_count":null},{"id":"f13e3ac4","cell_type":"markdown","source":"## 6) (Optional) Quick Visualization for a Few Tomograms","metadata":{}},{"id":"fed3796b","cell_type":"code","source":"N_SHOW = min(3, len(test_ids))\n\nfor tid in test_ids[:N_SHOW]:\n    vol = load_volume(tid)\n    z, y, x = predict_brightest_voxel(vol, none_threshold=PRED_NONE_THRESHOLD)\n    z_int = int(z) if z >= 0 else vol.shape[0] // 2\n    slice_img = vol[z_int]\n    import matplotlib.pyplot as plt\n    plt.figure()\n    plt.imshow(slice_img, cmap=\"gray\")\n    if z >= 0:\n        plt.scatter([x], [y], s=60, marker=\"x\")\n        plt.title(f\"{tid}  z={z:.0f}, y={y:.0f}, x={x:.0f}\")\n    else:\n        plt.title(f\"{tid}  (no motor predicted)\")\n    plt.axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-19T02:58:48.113325Z","iopub.execute_input":"2025-09-19T02:58:48.113910Z","iopub.status.idle":"2025-09-19T03:00:36.084529Z","shell.execute_reply.started":"2025-09-19T02:58:48.113876Z","shell.execute_reply":"2025-09-19T03:00:36.083338Z"}},"outputs":[],"execution_count":null}]}