{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":12070702,"sourceType":"datasetVersion","datasetId":6959173}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef clean_working(directory_path: str = \"/kaggle/working/\"):\n    \"\"\"\n    Clean kaggle output directory.\n    \"\"\"\n    if os.path.exists(directory_path):\n        for item in os.listdir(directory_path):\n            if item == \"submission.csv\":\n                continue\n            item_path = os.path.join(directory_path, item)\n            os.remove(item_path) if os.path.isfile(item_path) else shutil.rmtree(item_path)\n        print(f\"All items in '{directory_path}' have been removed.\")\n    else:\n        print(f\"'{directory_path}' does not exist.\")\n        \nclean_working()\n\n\nif not os.path.exists('/tmp/'):\n    os.mkdir('/tmp/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:00.743314Z","iopub.execute_input":"2025-06-05T18:38:00.743793Z","iopub.status.idle":"2025-06-05T18:38:02.204464Z","shell.execute_reply.started":"2025-06-05T18:38:00.743746Z","shell.execute_reply":"2025-06-05T18:38:02.203192Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Flaggellar Extra Training Data\n\nThis notebook showcases the [CryoET Flagellar Motors Dataset](https://www.kaggle.com/datasets/brendanartley/cryoet-flagellar-motors-dataset).\n\nThe dataset includes 1324 manually annotated tomograms from the CryoET Data Portal [here](https://cryoetdataportal.czscience.com/). The raw tomograms are sourced from Wei Chang, Ariane Briegel and Morgan Beeby. Thank you for sharing these datasets with the community.\n\nThe tomograms were manually annotated using [Napari](https://napari.org/dev/index.html).\n\n### Version Notes\n\nV2: Remove unnecessary downloads (3x faster for full collection)\n\nV3: Added processing logic used in 1st place solution.\n+ Pixel anomaly correction from [@Bilzard](https://www.kaggle.com/tatamikenn). Thanks for sharing!\n+ scipy.ndimage.zoom()\n+ Added a few more tomograms","metadata":{}},{"cell_type":"code","source":"DATA_DIR= \"/kaggle/input/cryoet-flagellar-motors-dataset/\"\nSEED= 0\n\ndf= pd.read_csv(os.path.join(DATA_DIR, \"labels_new.csv\"))\ndf= df[~df[\"tomo_id\"].str.startswith(\"tomo_\")]\ndf[\"coordinates\"] = df[\"coordinates\"].apply(lambda x: eval(x))\ndf= df.reset_index(drop=True)\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:02.206258Z","iopub.execute_input":"2025-06-05T18:38:02.206802Z","iopub.status.idle":"2025-06-05T18:38:02.307570Z","shell.execute_reply.started":"2025-06-05T18:38:02.206769Z","shell.execute_reply":"2025-06-05T18:38:02.306217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique Tomograms: {}\".format(df[\"tomo_id\"].nunique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:02.309706Z","iopub.execute_input":"2025-06-05T18:38:02.310031Z","iopub.status.idle":"2025-06-05T18:38:02.320444Z","shell.execute_reply.started":"2025-06-05T18:38:02.309996Z","shell.execute_reply":"2025-06-05T18:38:02.319208Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization\n\nThe original tomograms came in different sizes, voxel spacings, and pixel scales. To ensure consistency in the final dataset, we preprocessed the tomograms using the following transformations.\n\nIn the latest version, we use @Bilzards pixel anomaly correction code. \n\nPlease see their notebook [here](https://www.kaggle.com/code/tatamikenn/byu-download-correct-pixel-annomaly) for more information.","metadata":{}},{"cell_type":"code","source":"idx= 123\nrow= df.iloc[idx].to_dict()\nrow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:02.322023Z","iopub.execute_input":"2025-06-05T18:38:02.322404Z","iopub.status.idle":"2025-06-05T18:38:02.345431Z","shell.execute_reply.started":"2025-06-05T18:38:02.322369Z","shell.execute_reply":"2025-06-05T18:38:02.344199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fpath= os.path.join(DATA_DIR, \"volumes_704\", row[\"tomo_id\"] + \".npy\")\narr= np.load(fpath)\nprint(arr.shape)\n\nfig, ax = plt.subplots(figsize=(6, 6))\nax.set_title(row[\"tomo_id\"])\nax.imshow(arr[int(row[\"z\"]), ...], cmap=\"gray\")\nax.scatter(row[\"coordinates\"][0][2]*704, row[\"coordinates\"][0][1]*704, c=\"red\", s=50)\n\nax.set_xticks([])\nax.set_yticks([])\nax.set_frame_on(False)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:02.346558Z","iopub.execute_input":"2025-06-05T18:38:02.346949Z","iopub.status.idle":"2025-06-05T18:38:03.506818Z","shell.execute_reply.started":"2025-06-05T18:38:02.346901Z","shell.execute_reply":"2025-06-05T18:38:03.505257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lets visualize a few more.","metadata":{}},{"cell_type":"code","source":"# Sample rows\ntmp= df.sample(frac=1, random_state=SEED)\n\n# Create figure\nfig, axes = plt.subplots(2, 8, figsize=(24, 6))\naxes= axes.flatten()\n\nfor idx in range(len(axes)):\n    row= tmp.iloc[idx].to_dict()\n\n    # Load tomo\n    fpath= os.path.join(DATA_DIR, \"volumes_704\", row[\"tomo_id\"] + \".npy\")\n    arr= np.load(fpath)\n\n    # Visualize\n    frame= int(row[\"coordinates\"][0][0]*128)\n    axes[idx].imshow(arr[frame, ...], cmap=\"gray\")\n    axes[idx].scatter(row[\"coordinates\"][0][2]*704, row[\"coordinates\"][0][1]*704, c=\"red\", s=50)\n    axes[idx].set_xticks([])\n    axes[idx].set_yticks([])\n    axes[idx].set_frame_on(False)\n    \nplt.tight_layout()\nplt.show()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:03.508220Z","iopub.execute_input":"2025-06-05T18:38:03.508630Z","iopub.status.idle":"2025-06-05T18:38:11.777418Z","shell.execute_reply.started":"2025-06-05T18:38:03.508591Z","shell.execute_reply":"2025-06-05T18:38:11.775932Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optional: CryoET Data Portal Crawling\n\nIf this processing pipeline does not work for your models, you can always recollect the data and perform your own processing steps. \n\nThis is the preprocessing code we used, though you can easily adapt the `process_volume` function as needed. You will need sufficient disk space and a speedy connection to collect all the tomograms.\n\nAPI Docs [here](https://chanzuckerberg.github.io/cryoet-data-portal/stable/).","metadata":{}},{"cell_type":"code","source":"try:\n    from cryoet_data_portal import Client\nexcept:\n    !pip install zarr cryoet_data_portal -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:11.779104Z","iopub.execute_input":"2025-06-05T18:38:11.779606Z","iopub.status.idle":"2025-06-05T18:38:23.638951Z","shell.execute_reply.started":"2025-06-05T18:38:11.779558Z","shell.execute_reply":"2025-06-05T18:38:23.637365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy\n\n\ndef calc_irregular_score(x, bins=256, lb=50, ub=200):\n    hist, _ = np.histogram(x, bins=bins)\n    return (hist[:lb].sum() + hist[ub:].sum()) / x.size\n\n\ndef process_volume(x, target_shape=(128, 512, 512), lb=1, ub=99, bins=512, resample_z=False):\n    # Approximate percentile normalization\n    hist, bins = np.histogram(x, bins=bins)\n    cdf = np.cumsum(hist) / hist.sum()\n    l_idx = np.searchsorted(cdf, lb/100)\n    u_idx = np.searchsorted(cdf, ub/100)\n    lower, upper = bins[l_idx], bins[u_idx]\n    x = np.clip(x, lower, upper)\n    x = (x - lower) / (upper - lower)\n\n    if resample_z:\n        # Resample\n        indices = np.linspace(0, x.shape[0]-1, target_shape[0]).astype(int)\n        x= x[indices]    \n\n    # Resize\n    zoom_factor = tuple(ts / xs for ts, xs in zip(target_shape, x.shape))\n    x = scipy.ndimage.zoom(x, zoom_factor, order=1)\n\n    # Quantize\n    x = np.round(x.clip(0, 1) * 255).astype(np.uint8)\n\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:23.642251Z","iopub.execute_input":"2025-06-05T18:38:23.642628Z","iopub.status.idle":"2025-06-05T18:38:23.670208Z","shell.execute_reply.started":"2025-06-05T18:38:23.642597Z","shell.execute_reply":"2025-06-05T18:38:23.668857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Tuple\nfrom pathlib import Path\nimport shutil\nimport traceback\n\nimport zarr\nfrom cryoet_data_portal import Client, Dataset, Run\nimport numpy as np\n\n\nclass CziiCollector():\n    def __init__(\n        self,\n        tmp_dir: str = \"/tmp/\", \n        out_dir: str = \"/kaggle/working/volumes/\", \n        img_size: Tuple[int] = (128, 704, 704),\n    ):\n        super().__init__()\n\n        self.client = Client()\n        self.tmp_dir = Path(tmp_dir)\n        self.img_size = img_size\n\n        # Tmp dir\n        self.tmp_dir.mkdir(parents=True, exist_ok=True)\n\n    def cleanup(self):\n        shutil.rmtree(self.tmp_dir)\n        self.tmp_dir.mkdir(parents=True, exist_ok=True)\n\n    def process_tomogram(self, x):\n        print(\"original_shape: {}\".format(x.shape))        \n        x = process_volume(x, target_shape=self.img_size)\n        print(\"final_shape: {}\".format(x.shape))\n        return x\n        \n    def run(self, tomo_ids, irregular_threshold=0.6):\n        client = self.client\n\n        for tomo_id in tomo_ids:            \n            run = Run.find(client, query_filters=[Run.name == tomo_id])\n            if len(run) == 0:\n                print(\"MISSING: \", tomo_id)\n                continue\n            else:\n                run = run[0]\n\n            zarr_path = Path(self.tmp_dir) / f\"{run.name}.zarr\"\n            if not zarr_path.exists():\n                # download\n                tomo = run.tomograms[0]\n                tomo.download_omezarr(dest_path=self.tmp_dir)    \n\n            try:\n                # Load tomo\n                tomo_pixels= tomo.size_z * tomo.size_x * tomo.size_y\n                print(\"tomo_pixels: {:_}\".format(tomo_pixels))\n\n                # Take smaller version of XL tomograms\n                if tomo_pixels < 2_000_000_000:\n                    x = zarr.open(zarr_path / \"0\", mode='r')\n                else:\n                    x = zarr.open(zarr_path / \"1\", mode='r')\n\n                # irregular check\n                x = x[:]\n                irregular_score = calc_irregular_score(x)\n                if irregular_score > irregular_threshold:\n                    x = x.astype(\"uint8\").astype(\"float32\")\n\n                # Preprocess\n                x = self.process_tomogram(x)\n\n                # Save\n                out_path = f\"{run.name}.npy\"\n                np.save(out_path, x)\n                print(f\"Success: {tomo_id}\")\n            except Exception as e:\n                print(traceback.format_exc())\n                print(e)\n                print(f\"Failed: {tomo_id}\")\n            finally:\n                print(\"-\"*25)\n\n            self.cleanup()\n        return","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:23.671697Z","iopub.execute_input":"2025-06-05T18:38:23.672129Z","iopub.status.idle":"2025-06-05T18:38:24.871870Z","shell.execute_reply.started":"2025-06-05T18:38:23.672092Z","shell.execute_reply":"2025-06-05T18:38:24.870581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample a few\nIDXS= df[\"tomo_id\"].unique()\nIDXS= IDXS[:2]\n\n# Run collection\np= CziiCollector()\np.run(tomo_ids=IDXS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T18:38:24.873006Z","iopub.execute_input":"2025-06-05T18:38:24.873653Z","iopub.status.idle":"2025-06-05T18:41:32.979548Z","shell.execute_reply.started":"2025-06-05T18:38:24.873605Z","shell.execute_reply":"2025-06-05T18:41:32.978276Z"}},"outputs":[],"execution_count":null}]}