{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":11315004,"sourceType":"datasetVersion","datasetId":6959173}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nfrom pathlib import Path\n\nimport numpy as np\nimport polars as pl\nimport matplotlib.pyplot as plt\n\n\ndef clean_working(directory_path: str = \"/kaggle/working/\"):\n    \"\"\"\n    Clean kaggle output directory.\n    \"\"\"\n    if os.path.exists(directory_path):\n        for item in os.listdir(directory_path):\n            if item == \"submission.csv\":\n                continue\n            item_path = os.path.join(directory_path, item)\n            os.remove(item_path) if os.path.isfile(item_path) else shutil.rmtree(item_path)\n        print(f\"All items in '{directory_path}' have been removed.\")\n    else:\n        print(f\"'{directory_path}' does not exist.\")\n        \nclean_working()\n\n\nif not os.path.exists('/tmp/'):\n    os.mkdir('/tmp/')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:29.089503Z","iopub.execute_input":"2025-04-24T16:34:29.090381Z","iopub.status.idle":"2025-04-24T16:34:29.392101Z","shell.execute_reply.started":"2025-04-24T16:34:29.090346Z","shell.execute_reply":"2025-04-24T16:34:29.389129Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n\nIS_INTERACTIVE = os.environ.get('KAGGLE_KERNEL_RUN_TYPE') == 'Interactive'\n\n\nIS_INTERACTIVE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:29.394840Z","iopub.execute_input":"2025-04-24T16:34:29.395675Z","iopub.status.idle":"2025-04-24T16:34:29.409375Z","shell.execute_reply.started":"2025-04-24T16:34:29.395640Z","shell.execute_reply":"2025-04-24T16:34:29.406897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile irregular_labels.csv\n\ntomo_id,z\naba2014-02-21-14,18.0\nmba2011-02-16-103,62.0\nmba2011-02-16-106,66.0\nmba2011-02-16-108,42.0\nmba2011-02-16-111,56.0\nmba2011-02-16-115,70.0\nmba2011-02-16-116,68.0\nmba2011-02-16-11,66.0\nmba2011-02-16-122,42.0\nmba2011-02-16-123,62.0\nmba2011-02-16-129,64.0\nmba2011-02-16-12,60.0\nmba2011-02-16-133,59.0\nmba2011-02-16-139,60.0\nmba2011-02-16-141,66.0\nmba2011-02-16-143,57.0\nmba2011-02-16-145,48.0\nmba2011-02-16-145,51.0\nmba2011-02-16-147,51.0\nmba2011-02-16-150,62.0\nmba2011-02-16-153,55.0\nmba2011-02-16-153,63.0\nmba2011-02-16-155,33.0\nmba2011-02-16-157,60.0\nmba2011-02-16-15,56.0\nmba2011-02-16-15,50.0\nmba2011-02-16-160,60.0\nmba2011-02-16-160,51.0\nmba2011-02-16-162,62.0\nmba2011-02-16-170,62.0\nmba2011-02-16-173,52.0\nmba2011-02-16-176,59.0\nmba2011-02-16-17,68.0\nmba2011-02-16-19,70.0\nmba2011-02-16-1,66.0\nmba2011-02-16-1,46.0\nmba2011-02-16-20,65.0\nmba2011-02-16-23,59.0\nmba2011-02-16-26,59.0\nmba2011-02-16-27,63.0\nmba2011-02-16-28,59.0\nmba2011-02-16-28,68.0\nmba2011-02-16-29,65.0\nmba2011-02-16-30,66.0\nmba2011-02-16-32,75.0\nmba2011-02-16-33,56.0\nmba2011-02-16-34,54.0\nmba2011-02-16-35,57.0\nmba2011-02-16-37,70.0\nmba2011-02-16-3,63.0\nmba2011-02-16-40,59.0\nmba2011-02-16-40,55.0\nmba2011-02-16-42,44.0\nmba2011-02-16-42,26.0\nmba2011-02-16-46,64.0\nmba2011-02-16-48,63.0\nmba2011-02-16-52,59.0\nmba2011-02-16-53,54.0\nmba2011-02-16-55,46.0\nmba2011-02-16-60,54.0\nmba2011-02-16-64,56.0\nmba2011-02-16-65,66.0\nmba2011-02-16-67,60.0\nmba2011-02-16-68,60.0\nmba2011-02-16-71,59.0\nmba2011-02-16-75,47.0\nmba2011-02-16-79,43.0\nmba2011-02-16-79,44.0\nmba2011-02-16-88,69.0\nmba2011-02-16-90,60.0\nmba2011-02-16-95,63.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:29.593559Z","iopub.execute_input":"2025-04-24T16:34:29.594081Z","iopub.status.idle":"2025-04-24T16:34:29.602658Z","shell.execute_reply.started":"2025-04-24T16:34:29.594023Z","shell.execute_reply":"2025-04-24T16:34:29.601349Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_df = pl.read_csv(\"/kaggle/input/cryoet-flagellar-motors-dataset/labels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:29.869837Z","iopub.execute_input":"2025-04-24T16:34:29.870232Z","iopub.status.idle":"2025-04-24T16:34:29.958698Z","shell.execute_reply.started":"2025-04-24T16:34:29.870200Z","shell.execute_reply":"2025-04-24T16:34:29.957513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"irregular_df = pl.read_csv(\"irregular_labels.csv\")\nprint(irregular_df.shape)\nirregular_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:30.356116Z","iopub.execute_input":"2025-04-24T16:34:30.356500Z","iopub.status.idle":"2025-04-24T16:34:30.366462Z","shell.execute_reply.started":"2025-04-24T16:34:30.356474Z","shell.execute_reply":"2025-04-24T16:34:30.365231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Unique Tomograms: {}\".format(irregular_df[\"tomo_id\"].n_unique()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:30.864567Z","iopub.execute_input":"2025-04-24T16:34:30.864913Z","iopub.status.idle":"2025-04-24T16:34:30.871111Z","shell.execute_reply.started":"2025-04-24T16:34:30.864888Z","shell.execute_reply":"2025-04-24T16:34:30.869779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    from cryoet_data_portal import Client\nexcept:\n    ! pip install 'zarr<3.0' cryoet_data_portal -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:32.822548Z","iopub.execute_input":"2025-04-24T16:34:32.823029Z","iopub.status.idle":"2025-04-24T16:34:33.443212Z","shell.execute_reply.started":"2025-04-24T16:34:32.822974Z","shell.execute_reply":"2025-04-24T16:34:33.441901Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tomo_ids = irregular_df[\"tomo_id\"].unique(maintain_order=True).to_list()\nprint(\"N_TOMOS: {:_}\".format(len(tomo_ids)))\ntomo_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:34:39.925591Z","iopub.execute_input":"2025-04-24T16:34:39.925937Z","iopub.status.idle":"2025-04-24T16:34:39.935647Z","shell.execute_reply.started":"2025-04-24T16:34:39.925912Z","shell.execute_reply":"2025-04-24T16:34:39.934493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy\n\n\ndef calc_irregular_score(x, bins=256, lb=50, ub=200):\n    hist, _ = np.histogram(x, bins=bins)\n    return (hist[:lb].sum() + hist[ub:].sum()) / x.size\n\n\ndef process_volume(x, target_shape=(128, 512, 512), lb=1, ub=99, bins=512, resample_z=True):\n    # Approximate percentile normalization\n    hist, bins = np.histogram(x, bins=bins)\n    cdf = np.cumsum(hist) / hist.sum()\n    l_idx = np.searchsorted(cdf, lb/100)\n    u_idx = np.searchsorted(cdf, ub/100)\n    lower, upper = bins[l_idx], bins[u_idx]\n    x = np.clip(x, lower, upper)\n    x = (x - lower) / (upper - lower)\n\n    if resample_z:\n        # Resample\n        indices = np.linspace(0, x.shape[0]-1, target_shape[0]).astype(int)\n        x= x[indices]    \n\n    # Resize\n    zoom_factor = tuple(ts / xs for ts, xs in zip(target_shape, x.shape))\n    x = scipy.ndimage.zoom(x, zoom_factor, order=1)\n\n    # Quantize\n    x = np.round(x.clip(0, 1) * 255).astype(np.uint8)\n\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:38:58.513838Z","iopub.execute_input":"2025-04-24T16:38:58.514253Z","iopub.status.idle":"2025-04-24T16:38:58.524060Z","shell.execute_reply.started":"2025-04-24T16:38:58.514215Z","shell.execute_reply":"2025-04-24T16:38:58.522978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Tuple\nimport os\nimport traceback\nimport glob\nimport shutil\nimport scipy\n\nimport zarr\nfrom cryoet_data_portal import Client, Dataset, Run\nimport numpy as np\nfrom tqdm import tqdm\n\n\nclass CziiCollector():\n    def __init__(\n        self,\n        tmp_dir: str = \"/tmp/\", \n        out_dir: str = \"/kaggle/working/volumes/\", \n        img_size: Tuple[int] = (128, 512, 512),\n    ):\n        super().__init__()\n\n        self.client = Client()\n        self.tomo_ids = tomo_ids\n        self.tmp_dir = Path(tmp_dir)\n        self.out_dir = Path(out_dir)\n        self.img_size = img_size\n\n        # Tmp dir\n        self.tmp_dir.mkdir(parents=True, exist_ok=True)\n        self.out_dir.mkdir(parents=True, exist_ok=True)\n\n    def cleanup(self):\n        shutil.rmtree(self.tmp_dir)\n        self.tmp_dir.mkdir(parents=True, exist_ok=True)\n\n    def process_tomogram(self, x):\n        print(\"original_shape: {}\".format(x.shape))        \n        x = process_volume(x, target_shape=self.img_size)\n        print(\"final_shape: {}\".format(x.shape))\n        return x\n        \n    def run(self, tomo_ids, irregular_threshold=0.6):\n        client = self.client\n\n        for tomo_id in tomo_ids:            \n            run = Run.find(client, query_filters=[Run.name == tomo_id])\n            if len(run) == 0:\n                print(\"MISSING: \", tomo_id)\n                continue\n            else:\n                run = run[0]\n\n            zarr_path = Path(self.tmp_dir) / f\"{run.name}.zarr\"\n            if not zarr_path.exists():\n                # download\n                tomo = run.tomograms[0]\n                tomo.download_omezarr(dest_path=self.tmp_dir)                \n\n            try:\n                # Load tomo\n                x = zarr.open(zarr_path / \"0\", mode='r')\n\n                # irregular check\n                x = x[:]\n                irregular_score = calc_irregular_score(x)\n                if irregular_score > irregular_threshold:\n                    x = x.astype(\"uint8\").astype(\"float32\")\n\n                # Preprocess\n                x = self.process_tomogram(x)\n\n                # Save\n                out_path = Path(self.out_dir) / f\"{run.name}.npy\"\n                np.save(out_path, x)\n                print(f\"Success: {tomo_id}\")\n            except Exception as e:\n                print(traceback.format_exc())\n                print(e)\n                print(f\"Failed: {tomo_id}\")\n\n            if not IS_INTERACTIVE:\n                self.cleanup()\n        return","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:40:34.954455Z","iopub.execute_input":"2025-04-24T16:40:34.954880Z","iopub.status.idle":"2025-04-24T16:40:34.967876Z","shell.execute_reply.started":"2025-04-24T16:40:34.954845Z","shell.execute_reply":"2025-04-24T16:40:34.966373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nif IS_INTERACTIVE:\n    run_tomo_ids = [\"mba2011-02-16-27\", \"mba2011-02-16-123\", \"mba2011-02-16-45\"]\n    run_tomo_ids = [\"mba2011-02-16-27\", \"mba2011-02-16-45\"]\nelse:\n    run_tomo_ids = tomo_ids\n\n\np = CziiCollector()\np.run(run_tomo_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:40:35.739155Z","iopub.execute_input":"2025-04-24T16:40:35.739573Z","iopub.status.idle":"2025-04-24T16:41:29.496962Z","shell.execute_reply.started":"2025-04-24T16:40:35.739543Z","shell.execute_reply":"2025-04-24T16:41:29.495899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"for i, tomo_id in enumerate(run_tomo_ids):\n    if i > 3:\n        break\n    df = label_df.filter(pl.col(\"tomo_id\") == tomo_id)\n\n    path = Path(\"volumes\") / f\"{tomo_id}.npy\"\n    x = np.load(path)\n    for row in df.iter_rows(named=True):\n        z = int(row[\"z\"])\n        plt.imshow(x[z], cmap=\"gray\")\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:41:54.308279Z","iopub.execute_input":"2025-04-24T16:41:54.308655Z","iopub.status.idle":"2025-04-24T16:41:54.835340Z","shell.execute_reply.started":"2025-04-24T16:41:54.308628Z","shell.execute_reply":"2025-04-24T16:41:54.834208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!du -sh volumes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-24T16:42:02.756437Z","iopub.execute_input":"2025-04-24T16:42:02.756830Z","iopub.status.idle":"2025-04-24T16:42:02.889770Z","shell.execute_reply.started":"2025-04-24T16:42:02.756802Z","shell.execute_reply":"2025-04-24T16:42:02.888262Z"}},"outputs":[],"execution_count":null}]}