{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9862305,"sourceType":"datasetVersion","datasetId":6052780},{"sourceId":9867543,"sourceType":"datasetVersion","datasetId":6040935},{"sourceId":10445850,"sourceType":"datasetVersion","datasetId":6465904},{"sourceId":10471985,"sourceType":"datasetVersion","datasetId":6484063},{"sourceId":206640467,"sourceType":"kernelVersion"},{"sourceId":211097053,"sourceType":"kernelVersion"},{"sourceId":236245,"sourceType":"modelInstanceVersion","modelInstanceId":201776,"modelId":223549},{"sourceId":236249,"sourceType":"modelInstanceVersion","modelInstanceId":201779,"modelId":223552},{"sourceId":236964,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":202377,"modelId":224123}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### ℹ️ Info¶\n\n* **forked original great work kernels**\n    * (YOLO) https://www.kaggle.com/code/itsuki9180/czii-yolo11-submission-baseline    \n    * (YOLO) https://www.kaggle.com/code/sersasj/czii-yolo11-submission-baseline-with-kdtree-update\n    * (Unet3D) https://www.kaggle.com/code/ahsuna123/3d-u-net-training-only\n    * (Unet3D) https://www.kaggle.com/code/fnands/baseline-unet-train-submit\n    * (Unet3D) https://www.kaggle.com/code/linheshen/esemble-2d-and-3d\n    \n* **2025/01/15 My Additional**\n    * Unet3D-Monai local train model add. Model dataset is here.\n    * https://www.kaggle.com/datasets/hideyukizushi/cziials-a-230-unet/data\n    ```\n    # train validation score&loss\n    val_metric=0.450\n    train_loss=0.593\n    ```","metadata":{}},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# **《《《 YOLO 》》》**","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output\n!tar xfvz /kaggle/input/ultralytics-for-offline-install/archive.tar.gz\n!pip install --no-index --find-links=./packages ultralytics\n!rm -rf ./packages\ntry:\n    import zarr\nexcept: \n    !cp -r '/kaggle/input/hengck-czii-cryo-et-01/wheel_file' '/kaggle/working/'\n    !pip install /kaggle/working/wheel_file/asciitree-0.3.3/asciitree-0.3.3\n    !pip install --no-index --find-links=/kaggle/working/wheel_file zarr\n    !pip install --no-index --find-links=/kaggle/working/wheel_file connected-components-3d\nfrom typing import List, Tuple, Union\ndeps_path = '/kaggle/input/czii-cryoet-dependencies'\n! pip install -q --no-index --find-links {deps_path} --requirement {deps_path}/requirements.txt\nimport lightning.pytorch as pl\nfrom datetime import datetime\nimport pytz\nimport sys\nsys.path.append('/kaggle/input/hengck-czii-cryo-et-01')\nfrom czii_helper import *\nfrom dataset import *\nfrom model2 import *\nclear_output()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:54:23.163066Z","iopub.execute_input":"2025-01-20T16:54:23.163598Z","iopub.status.idle":"2025-01-20T16:55:41.190789Z","shell.execute_reply.started":"2025-01-20T16:54:23.16357Z","shell.execute_reply":"2025-01-20T16:55:41.190036Z"}},"outputs":[],"execution_count":1},{"cell_type":"code","source":"import os\nimport glob\nimport time\nimport sys\nimport warnings\nimport math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport torch\nfrom tqdm import tqdm\nfrom ultralytics import YOLO\nimport zarr\nfrom scipy.spatial import cKDTree\nfrom collections import defaultdict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:08.762809Z","iopub.execute_input":"2025-01-20T16:56:08.763129Z","iopub.status.idle":"2025-01-20T16:56:11.093023Z","shell.execute_reply.started":"2025-01-20T16:56:08.763107Z","shell.execute_reply":"2025-01-20T16:56:11.09218Z"}},"outputs":[{"name":"stdout","text":"Creating new Ultralytics Settings v0.0.6 file ✅ \nView Ultralytics Settings with 'yolo settings' or at '/root/.config/Ultralytics/settings.json'\nUpdate Settings with 'yolo settings key=value', i.e. 'yolo settings runs_dir=path/to/dir'. For help see https://docs.ultralytics.com/quickstart/#ultralytics-settings.\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"model_path = '/kaggle/input/czii-yolo-l-trained-with-synthetic-data/best_synthetic.pt'\nmodel = YOLO(model_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:14.848947Z","iopub.execute_input":"2025-01-20T16:56:14.849272Z","iopub.status.idle":"2025-01-20T16:56:15.385231Z","shell.execute_reply.started":"2025-01-20T16:56:14.849244Z","shell.execute_reply":"2025-01-20T16:56:15.384152Z"}},"outputs":[],"execution_count":3},{"cell_type":"code","source":"runs_path = '/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/*'\nruns = sorted(glob.glob(runs_path))\nruns = [os.path.basename(run) for run in runs]\nsp = len(runs)//2\nruns1 = runs[:sp]\nruns1[:5]\n\n#add by @minfuka\nruns2 = runs[sp:]\nruns2[:5]\n\n#add by @minfuka\nassert torch.cuda.device_count() == 2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:17.022863Z","iopub.execute_input":"2025-01-20T16:56:17.023184Z","iopub.status.idle":"2025-01-20T16:56:17.059697Z","shell.execute_reply.started":"2025-01-20T16:56:17.023156Z","shell.execute_reply":"2025-01-20T16:56:17.058844Z"}},"outputs":[],"execution_count":4},{"cell_type":"code","source":"particle_names = [\n    'apo-ferritin',\n    'beta-amylase',\n    'beta-galactosidase',\n    'ribosome',\n    'thyroglobulin',\n    'virus-like-particle'\n]\n\nparticle_to_index = {\n    'apo-ferritin': 0,\n    'beta-amylase': 1,\n    'beta-galactosidase': 2,\n    'ribosome': 3,\n    'thyroglobulin': 4,\n    'virus-like-particle': 5\n}\n\nindex_to_particle = {index: name for name, index in particle_to_index.items()}\n\nparticle_radius = {\n    'apo-ferritin': 60,\n    'beta-amylase': 65,\n    'beta-galactosidase': 90,\n    'ribosome': 150,\n    'thyroglobulin': 130,\n    'virus-like-particle': 135,\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:19.862046Z","iopub.execute_input":"2025-01-20T16:56:19.862463Z","iopub.status.idle":"2025-01-20T16:56:19.867747Z","shell.execute_reply.started":"2025-01-20T16:56:19.862418Z","shell.execute_reply":"2025-01-20T16:56:19.866478Z"}},"outputs":[],"execution_count":5},{"cell_type":"code","source":"# add by @sesasj\nclass UnionFind:\n    def __init__(self, size):\n        self.parent = np.arange(size)\n        self.rank = np.zeros(size, dtype=int)\n\n    def find(self, u):\n        if self.parent[u] != u:\n            self.parent[u] = self.find(self.parent[u])  \n        return self.parent[u]\n\n    def union(self, u, v):\n        u_root = self.find(u)\n        v_root = self.find(v)\n        if u_root == v_root:\n            return\n            \n        if self.rank[u_root] < self.rank[v_root]:\n            self.parent[u_root] = v_root\n        else:\n            self.parent[v_root] = u_root\n            if self.rank[u_root] == self.rank[v_root]:\n                self.rank[u_root] += 1\n\nclass PredictionAggregator:\n    def __init__(self, first_conf=0.2, conf_coef=0.75):\n        self.first_conf = first_conf\n        self.conf_coef = conf_coef\n        self.particle_confs = np.array([0.5, 0.0, 0.2, 0.5, 0.2, 0.5])\n        \n    def convert_to_8bit(self, volume):\n        lower, upper = np.percentile(volume, (0.5, 99.5))\n        clipped = np.clip(volume, lower, upper)\n        scaled = ((clipped - lower) / (upper - lower + 1e-12) * 255).astype(np.uint8)\n        return scaled\n\n    def make_predictions(self, run_id, model, device_no):\n        volume_path = f'/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/{run_id}/VoxelSpacing10.000/denoised.zarr'\n        volume = zarr.open(volume_path, mode='r')[0]\n        volume_8bit = self.convert_to_8bit(volume)\n        num_slices = volume_8bit.shape[0]\n\n        detections = {\n            'particle_type': [],\n            'confidence': [],\n            'x': [],\n            'y': [],\n            'z': []\n        }\n\n        for slice_idx in range(num_slices):\n            \n            img = volume_8bit[slice_idx]\n            input_image = cv2.resize(np.stack([img]*3, axis=-1), (640, 640))\n\n            results = model.predict(\n                input_image,\n                save=False,\n                imgsz=640,\n                conf=self.first_conf,\n                device=device_no,\n                batch=1,\n                verbose=False,\n            )\n\n            for result in results:\n                boxes = result.boxes\n                if boxes is None:\n                    continue\n                cls = boxes.cls.cpu().numpy().astype(int)\n                conf = boxes.conf.cpu().numpy()\n                xyxy = boxes.xyxy.cpu().numpy()\n\n                xc = ((xyxy[:, 0] + xyxy[:, 2]) / 2.0) * 10 * (63/64) # 63/64 because of the resize\n                yc = ((xyxy[:, 1] + xyxy[:, 3]) / 2.0) * 10 * (63/64)\n                zc = np.full(xc.shape, slice_idx * 10 + 5)\n\n                particle_types = [index_to_particle[c] for c in cls]\n\n                detections['particle_type'].extend(particle_types)\n                detections['confidence'].extend(conf)\n                detections['x'].extend(xc)\n                detections['y'].extend(yc)\n                detections['z'].extend(zc)\n\n        if not detections['particle_type']:\n            return pd.DataFrame()  \n\n        particle_types = np.array(detections['particle_type'])\n        confidences = np.array(detections['confidence'])\n        xs = np.array(detections['x'])\n        ys = np.array(detections['y'])\n        zs = np.array(detections['z'])\n\n        aggregated_data = []\n\n        for idx, particle in enumerate(particle_names):\n            if particle == 'beta-amylase':\n                continue \n\n            mask = (particle_types == particle)\n            if not np.any(mask):\n                continue  \n                \n            particle_confidences = confidences[mask]\n            particle_xs = xs[mask]\n            particle_ys = ys[mask]\n            particle_zs = zs[mask]\n            # -------------modified by @sersasj ------------------------\n            coords = np.vstack((particle_xs, particle_ys, particle_zs)).T\n\n           \n            z_distance = 30 # How many slices can you \"jump\" to aggregate predictions 10 = 1, 20 = 2...\n            xy_distance = 20 # xy_tol_p2 in original code by ITK8191\n            \n            max_distance = math.sqrt(z_distance**2 + xy_distance**2)\n            tree = cKDTree(coords)            \n            pairs = tree.query_pairs(r=max_distance, p=2)\n\n            \n            uf = UnionFind(len(coords))\n            \n            coords_xy = coords[:, :2]\n            coords_z = coords[:, 2]\n            for u, v in pairs:\n                z_diff = abs(coords_z[u] - coords_z[v])\n                if z_diff > z_distance:\n                    continue  \n\n                xy_diff = np.linalg.norm(coords_xy[u] - coords_xy[v])\n                if xy_diff > xy_distance:\n                    continue  \n\n                uf.union(u, v)\n\n            roots = np.array([uf.find(i) for i in range(len(coords))])\n            unique_roots, inverse_indices, counts = np.unique(roots, return_inverse=True, return_counts=True)\n            conf_sums = np.bincount(inverse_indices, weights=particle_confidences)\n            \n            aggregated_confidences = conf_sums / (counts ** self.conf_coef)\n            cluster_per_particle = [4,1,2,9,4,8]\n            valid_clusters = (counts >= cluster_per_particle[idx]) & (aggregated_confidences > self.particle_confs[idx])\n\n            if not np.any(valid_clusters):\n                continue  \n\n            cluster_ids = unique_roots[valid_clusters]\n\n            centers_x = np.bincount(inverse_indices, weights=particle_xs) / counts\n            centers_y = np.bincount(inverse_indices, weights=particle_ys) / counts\n            centers_z = np.bincount(inverse_indices, weights=particle_zs) / counts\n\n            centers_x = centers_x[valid_clusters]\n            centers_y = centers_y[valid_clusters]\n            centers_z = centers_z[valid_clusters]\n\n            aggregated_df = pd.DataFrame({\n                'experiment': [run_id] * len(centers_x),\n                'particle_type': [particle] * len(centers_x),\n                'x': centers_x,\n                'y': centers_y,\n                'z': centers_z\n            })\n\n            aggregated_data.append(aggregated_df)\n\n        if aggregated_data:\n            return pd.concat(aggregated_data, axis=0)\n        else:\n            return pd.DataFrame()  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:27.403887Z","iopub.execute_input":"2025-01-20T16:56:27.404363Z","iopub.status.idle":"2025-01-20T16:56:27.42564Z","shell.execute_reply.started":"2025-01-20T16:56:27.404323Z","shell.execute_reply":"2025-01-20T16:56:27.424856Z"}},"outputs":[],"execution_count":6},{"cell_type":"code","source":"# # instance main class\n# aggregator = PredictionAggregator(first_conf=0.19,  conf_coef=0.34) #Update\n# aggregated_results = []\n# #add by @minfuka\n# from concurrent.futures import ProcessPoolExecutor #add by @minfuka\n\n# #add by @minfuka\n# def inference(runs, model, device_no):\n#     subs = []\n#     for r in tqdm(runs, total=len(runs)):\n#         df = aggregator.make_predictions(r, model, device_no)\n#         subs.append(df)\n    \n#     return subs\n# start_time = time.time()\n\n# with ProcessPoolExecutor(max_workers=2) as executor:\n#     results = list(executor.map(inference, (runs1, runs2), (model, model), (\"0\", \"1\")))\n\n\n# end_time = time.time()\n\n# estimated_total_time = (end_time - start_time) / len(runs) * 500  \n# print(f'estimated total prediction time for 500 runs: {estimated_total_time:.4f} seconds')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-14T23:02:41.30962Z","iopub.execute_input":"2025-01-14T23:02:41.309916Z","iopub.status.idle":"2025-01-14T23:03:08.832131Z","shell.execute_reply.started":"2025-01-14T23:02:41.309888Z","shell.execute_reply":"2025-01-14T23:03:08.831276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #change by @minfuka\n# submission0 = pd.concat(results[0])\n# submission1 = pd.concat(results[1])\n# submission_ = pd.concat([submission0, submission1]).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-14T23:03:08.833061Z","iopub.execute_input":"2025-01-14T23:03:08.833368Z","iopub.status.idle":"2025-01-14T23:03:08.840434Z","shell.execute_reply.started":"2025-01-14T23:03:08.833333Z","shell.execute_reply":"2025-01-14T23:03:08.83974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission_.insert(0, 'id', range(len(submission_)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-14T23:03:08.841196Z","iopub.execute_input":"2025-01-14T23:03:08.841533Z","iopub.status.idle":"2025-01-14T23:03:08.88363Z","shell.execute_reply.started":"2025-01-14T23:03:08.8415Z","shell.execute_reply":"2025-01-14T23:03:08.882888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **《《《 Unet3D(Monai) 》》》**","metadata":{}},{"cell_type":"code","source":"# !pip install torch==2.2.0 torchvision==0.17.0 --index-url https://download.pytorch.org/whl/cu118\n# !pip install einops matplotlib mrcfile pandas tqdm\n# !pip install git+https://github.com/facebookresearch/segment-anything.git\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T18:57:06.194094Z","iopub.execute_input":"2025-01-19T18:57:06.19439Z","iopub.status.idle":"2025-01-19T18:57:39.105373Z","shell.execute_reply.started":"2025-01-19T18:57:06.194369Z","shell.execute_reply":"2025-01-19T18:57:39.104368Z"}},"outputs":[{"name":"stdout","text":"Collecting git+https://github.com/facebookresearch/segment-anything.git\n  Cloning https://github.com/facebookresearch/segment-anything.git to /tmp/pip-req-build-id6wunqq\n  Running command git clone --filter=blob:none --quiet https://github.com/facebookresearch/segment-anything.git /tmp/pip-req-build-id6wunqq\n  fatal: unable to access 'https://github.com/facebookresearch/segment-anything.git/': Could not resolve host: github.com\n  \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\n  \n  \u001b[31m×\u001b[0m \u001b[32mgit clone --\u001b[0m\u001b[32mfilter\u001b[0m\u001b[32m=\u001b[0m\u001b[32mblob\u001b[0m\u001b[32m:none --quiet \u001b[0m\u001b[4;32mhttps://github.com/facebookresearch/segment-anything.git\u001b[0m\u001b[32m \u001b[0m\u001b[32m/tmp/\u001b[0m\u001b[32mpip-req-build-id6wunqq\u001b[0m did not run successfully.\n  \u001b[31m│\u001b[0m exit code: \u001b[1;36m128\u001b[0m\n  \u001b[31m╰─>\u001b[0m See above for output.\n  \n  \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\n\u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\n\n\u001b[31m×\u001b[0m \u001b[32mgit clone --\u001b[0m\u001b[32mfilter\u001b[0m\u001b[32m=\u001b[0m\u001b[32mblob\u001b[0m\u001b[32m:none --quiet \u001b[0m\u001b[4;32mhttps://github.com/facebookresearch/segment-anything.git\u001b[0m\u001b[32m \u001b[0m\u001b[32m/tmp/\u001b[0m\u001b[32mpip-req-build-id6wunqq\u001b[0m did not run successfully.\n\u001b[31m│\u001b[0m exit code: \u001b[1;36m128\u001b[0m\n\u001b[31m╰─>\u001b[0m See above for output.\n\n\u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"# !wget https://github.com/xulabs/aitom/archive/refs/heads/master.zip -O cryosam.zip\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T20:12:17.58634Z","iopub.execute_input":"2025-01-19T20:12:17.58668Z","iopub.status.idle":"2025-01-19T20:12:18.285261Z","shell.execute_reply.started":"2025-01-19T20:12:17.586658Z","shell.execute_reply":"2025-01-19T20:12:18.284182Z"}},"outputs":[{"name":"stdout","text":"--2025-01-19 20:12:17--  https://github.com/xulabs/aitom/archive/refs/heads/master.zip\nResolving github.com (github.com)... 140.82.112.3\nConnecting to github.com (github.com)|140.82.112.3|:443... connected.\nHTTP request sent, awaiting response... 302 Found\nLocation: https://codeload.github.com/xulabs/aitom/zip/refs/heads/master [following]\n--2025-01-19 20:12:17--  https://codeload.github.com/xulabs/aitom/zip/refs/heads/master\nResolving codeload.github.com (codeload.github.com)... 140.82.114.9\nConnecting to codeload.github.com (codeload.github.com)|140.82.114.9|:443... connected.\nHTTP request sent, awaiting response... 200 OK\nLength: unspecified [application/zip]\nSaving to: ‘cryosam.zip’\n\ncryosam.zip             [ <=>                ] 875.92K  --.-KB/s    in 0.1s    \n\n2025-01-19 20:12:18 (5.91 MB/s) - ‘cryosam.zip’ saved [896938]\n\n","output_type":"stream"}],"execution_count":23},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/cry-default-v1/segment-anything')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:56:49.128085Z","iopub.execute_input":"2025-01-20T16:56:49.128434Z","iopub.status.idle":"2025-01-20T16:56:49.132069Z","shell.execute_reply.started":"2025-01-20T16:56:49.128404Z","shell.execute_reply":"2025-01-20T16:56:49.131112Z"}},"outputs":[],"execution_count":7},{"cell_type":"code","source":"import sys\nsys.path.append('/kaggle/input/sam_mod/pytorch/default/1/cryosam')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:57:35.881168Z","iopub.execute_input":"2025-01-20T16:57:35.881517Z","iopub.status.idle":"2025-01-20T16:57:35.885246Z","shell.execute_reply.started":"2025-01-20T16:57:35.881489Z","shell.execute_reply":"2025-01-20T16:57:35.884401Z"}},"outputs":[],"execution_count":9},{"cell_type":"code","source":"from segment_anything import sam_model_registry\n\n# Initialize CryoSAM\nsam_checkpoint = '/kaggle/input/cryosam/pytorch/default/1/sam_vit_h_4b8939.pth'  # Update if checkpoint is in a different location\nmodel_type = \"vit_h\"  # Assuming the model type is \"vit_h\" (update if different)\n\n# Load the SAM model\nsam = sam_model_registry[model_type](checkpoint=sam_checkpoint)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T16:59:41.339658Z","iopub.execute_input":"2025-01-20T16:59:41.340011Z","iopub.status.idle":"2025-01-20T17:00:05.944764Z","shell.execute_reply.started":"2025-01-20T16:59:41.339982Z","shell.execute_reply":"2025-01-20T17:00:05.943804Z"}},"outputs":[],"execution_count":11},{"cell_type":"code","source":"import json\nimport copick\n\n# Path to the configuration file\nTRAIN_DATA_DIR = \"/kaggle/input/create-numpy-dataset-exp-name\"\ncopick_config_path = TRAIN_DATA_DIR + \"/copick.config\"\n\n# Load and modify the configuration\nwith open(copick_config_path) as f:\n    copick_config = json.load(f)\n\ncopick_config['static_root'] = '/kaggle/input/czii-cryo-et-object-identification/test/static'\n\n# Save the modified configuration\ncopick_test_config_path = 'copick_test.config'\nwith open(copick_test_config_path, 'w') as outfile:\n    json.dump(copick_config, outfile)\n\n# Initialize `root` using Copick\nroot = copick.from_file(copick_test_config_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T19:52:17.297988Z","iopub.execute_input":"2025-01-19T19:52:17.298288Z","iopub.status.idle":"2025-01-19T19:52:19.763664Z","shell.execute_reply.started":"2025-01-19T19:52:17.298267Z","shell.execute_reply":"2025-01-19T19:52:19.762741Z"}},"outputs":[{"name":"stderr","text":"<ipython-input-15-9cce5f4b33e0>:20: DeprecationWarning: config_type not found in config file, defaulting to filesystem\n  root = copick.from_file(copick_test_config_path)\n","output_type":"stream"}],"execution_count":15},{"cell_type":"code","source":"# Import required libraries\nimport json\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport cc3d\n\n\n\n\n# Parameters\nid_to_name = {1: \"apo-ferritin\", 2: \"beta-amylase\", 3: \"beta-galactosidase\", \n              4: \"ribosome\", 5: \"thyroglobulin\", 6: \"virus-like-particle\"}\nBLOB_THRESHOLD = 200\nCERTAINTY_THRESHOLD = 0.05\nclasses = [1, 2, 3, 4, 5, 6]\n\ndef dict_to_df(coord_dict, experiment_name):\n    all_coords = []\n    all_labels = []\n    for label, coords in coord_dict.items():\n        all_coords.append(coords)\n        all_labels.extend([label] * len(coords))\n    all_coords = np.vstack(all_coords)\n    df = pd.DataFrame({\n        'experiment': experiment_name,\n        'particle_type': all_labels,\n        'x': all_coords[:, 0],\n        'y': all_coords[:, 1],\n        'z': all_coords[:, 2]\n    })\n    return df\n\n# Run inference with CryoSAM\nlocation_df = []\nfor run in root.runs:  # Iterate over all runs\n    # Load tomogram (3D volume)\n    tomo = run.get_voxel_spacing(10).get_tomogram(\"denoised\").numpy()\n\n    # Generate prompts for CryoSAM (replace with meaningful prompts based on prior knowledge)\n    D, H, W = tomo.shape\n    num_prompts = 10\n    input_prompts = np.column_stack((\n        np.random.randint(0, W, size=num_prompts),  # Random x-coordinates\n        np.random.randint(0, H, size=num_prompts),  # Random y-coordinates\n        np.random.randint(0, D, size=num_prompts)   # Random z-coordinates\n    ))\n\n    # Perform inference\n    output = sam.infer(voxel=tomo, input_prompts=input_prompts)\n\n    # Post-process the output to extract particle coordinates\n    location = {}\n    for c in classes:\n        cc = cc3d.connected_components(output == c)\n        stats = cc3d.statistics(cc)\n        zyx = stats['centroids'][1:] * 10.012444\n        zyx_large = zyx[stats['voxel_counts'][1:] > BLOB_THRESHOLD]\n        xyz = np.ascontiguousarray(zyx_large[:, ::-1])\n        location[id_to_name[c]] = xyz\n\n    # Convert results to DataFrame\n    df = dict_to_df(location, run.name)\n    location_df.append(df)\n\n# Combine results from all runs\nlocation_df = pd.concat(location_df)\nlocation_df.insert(loc=0, column='id', value=np.arange(len(location_df)))\n\n# Sort and save results to CSV\nlocation_df = location_df.sort_values(by=['experiment', 'particle_type']).reset_index(drop=True)\nlocation_df.to_csv('submission.csv', index=False)\nprint(f\"Submission saved to 'submission.csv' with {len(location_df)} entries.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T20:14:08.218396Z","iopub.execute_input":"2025-01-19T20:14:08.218706Z","iopub.status.idle":"2025-01-19T20:14:09.02671Z","shell.execute_reply.started":"2025-01-19T20:14:08.218685Z","shell.execute_reply":"2025-01-19T20:14:09.025557Z"}},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;32m<ipython-input-25-1dc0f51fb1b6>\u001b[0m in \u001b[0;36m<cell line: 36>\u001b[0;34m()\u001b[0m\n\u001b[1;32m     48\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     49\u001b[0m     \u001b[0;31m# Perform inference\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 50\u001b[0;31m     \u001b[0moutput\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0msam\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0minfer\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mvoxel\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mtomo\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0minput_prompts\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0minput_prompts\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     51\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     52\u001b[0m     \u001b[0;31m# Post-process the output to extract particle coordinates\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.10/dist-packages/torch/nn/modules/module.py\u001b[0m in \u001b[0;36m__getattr__\u001b[0;34m(self, name)\u001b[0m\n\u001b[1;32m   1727\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0mname\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mmodules\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1728\u001b[0m                 \u001b[0;32mreturn\u001b[0m \u001b[0mmodules\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mname\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 1729\u001b[0;31m         \u001b[0;32mraise\u001b[0m \u001b[0mAttributeError\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34mf\"'{type(self).__name__}' object has no attribute '{name}'\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m   1730\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   1731\u001b[0m     \u001b[0;32mdef\u001b[0m \u001b[0m__setattr__\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mname\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mstr\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mvalue\u001b[0m\u001b[0;34m:\u001b[0m \u001b[0mUnion\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mTensor\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m'Module'\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m->\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mAttributeError\u001b[0m: 'Sam' object has no attribute 'infer'"],"ename":"AttributeError","evalue":"'Sam' object has no attribute 'infer'","output_type":"error"}],"execution_count":25},{"cell_type":"code","source":"# class Model(pl.LightningModule):\n#     def __init__(\n#         self, \n#         spatial_dims: int = 3,\n#         in_channels: int = 1,\n#         out_channels: int = 7,\n#         channels: Union[Tuple[int, ...], List[int]] = (48, 64, 80, 80),\n#         strides: Union[Tuple[int, ...], List[int]] = (2, 2, 1),\n#         num_res_units: int = 1,\n#     ):\n#         super().__init__()\n#         self.save_hyperparameters()\n#         self.model = UNet(\n#             spatial_dims=self.hparams.spatial_dims,\n#             in_channels=self.hparams.in_channels,\n#             out_channels=self.hparams.out_channels,\n#             channels=self.hparams.channels,\n#             strides=self.hparams.strides,\n#             num_res_units=self.hparams.num_res_units,\n#         )\n#     def forward(self, x):\n#         return self.model(x)\n\n# channels = (48, 64, 80, 80)\n# strides_pattern = (2, 2, 1)\n# num_res_units = 1\n# def extract_3d_patches_minimal_overlap(arrays: List[np.ndarray], patch_size: int) -> Tuple[List[np.ndarray], List[Tuple[int, int, int]]]:\n#     if not arrays or not isinstance(arrays, list):\n#         raise ValueError(\"Input must be a non-empty list of arrays\")\n    \n#     # Verify all arrays have the same shape\n#     shape = arrays[0].shape\n#     if not all(arr.shape == shape for arr in arrays):\n#         raise ValueError(\"All input arrays must have the same shape\")\n    \n#     if patch_size > min(shape):\n#         raise ValueError(f\"patch_size ({patch_size}) must be smaller than smallest dimension {min(shape)}\")\n    \n#     m, n, l = shape\n#     patches = []\n#     coordinates = []\n    \n#     # Calculate starting positions for each dimension\n#     x_starts = calculate_patch_starts(m, patch_size)\n#     y_starts = calculate_patch_starts(n, patch_size)\n#     z_starts = calculate_patch_starts(l, patch_size)\n    \n#     # Extract patches from each array\n#     for arr in arrays:\n#         for x in x_starts:\n#             for y in y_starts:\n#                 for z in z_starts:\n#                     patch = arr[\n#                         x:x + patch_size,\n#                         y:y + patch_size,\n#                         z:z + patch_size\n#                     ]\n#                     patches.append(patch)\n#                     coordinates.append((x, y, z))\n    \n#     return patches, coordinates\n# def reconstruct_array(patches: List[np.ndarray], \n#                      coordinates: List[Tuple[int, int, int]], \n#                      original_shape: Tuple[int, int, int]) -> np.ndarray:\n#     reconstructed = np.zeros(original_shape, dtype=np.int64)  # To track overlapping regions\n    \n#     patch_size = patches[0].shape[0]\n    \n#     for patch, (x, y, z) in zip(patches, coordinates):\n#         reconstructed[\n#             x:x + patch_size,\n#             y:y + patch_size,\n#             z:z + patch_size\n#         ] = patch\n        \n    \n#     return reconstructed\n# def calculate_patch_starts(dimension_size: int, patch_size: int) -> List[int]:\n#     if dimension_size <= patch_size:\n#         return [0]\n        \n#     # Calculate number of patches needed\n#     n_patches = np.ceil(dimension_size / patch_size)\n    \n#     if n_patches == 1:\n#         return [0]\n    \n#     # Calculate overlap\n#     total_overlap = (n_patches * patch_size - dimension_size) / (n_patches - 1)\n    \n#     # Generate starting positions\n#     positions = []\n#     for i in range(int(n_patches)):\n#         pos = int(i * (patch_size - total_overlap))\n#         if pos + patch_size > dimension_size:\n#             pos = dimension_size - patch_size\n#         if pos not in positions:  # Avoid duplicates\n#             positions.append(pos)\n    \n#     return positions\n# import pandas as pd\n\n# def dict_to_df(coord_dict, experiment_name):\n#     # Create lists to store data\n#     all_coords = []\n#     all_labels = []\n    \n#     # Process each label and its coordinates\n#     for label, coords in coord_dict.items():\n#         all_coords.append(coords)\n#         all_labels.extend([label] * len(coords))\n    \n#     # Concatenate all coordinates\n#     all_coords = np.vstack(all_coords)\n    \n#     df = pd.DataFrame({\n#         'experiment': experiment_name,\n#         'particle_type': all_labels,\n#         'x': all_coords[:, 0],\n#         'y': all_coords[:, 1],\n#         'z': all_coords[:, 2]\n#     })\n\n    \n#     return df\n# from typing import List, Tuple, Union\n# import numpy as np\n# import torch\n# from monai.data import DataLoader, Dataset, CacheDataset, decollate_batch\n# from monai.transforms import (\n#     Compose, \n#     EnsureChannelFirstd, \n#     Orientationd,  \n#     AsDiscrete,  \n#     RandFlipd, \n#     RandRotate90d, \n#     NormalizeIntensityd,\n#     RandCropByLabelClassesd,\n# )\n# TRAIN_DATA_DIR = \"/kaggle/input/create-numpy-dataset-exp-name\"\n# import json\n# copick_config_path = TRAIN_DATA_DIR + \"/copick.config\"\n\n# with open(copick_config_path) as f:\n#     copick_config = json.load(f)\n\n# copick_config['static_root'] = '/kaggle/input/czii-cryo-et-object-identification/test/static'\n\n# copick_test_config_path = 'copick_test.config'\n\n# with open(copick_test_config_path, 'w') as outfile:\n#     json.dump(copick_config, outfile)\n# import copick\n\n# root = copick.from_file(copick_test_config_path)\n\n# copick_user_name = \"copickUtils\"\n# copick_segmentation_name = \"paintedPicks\"\n# voxel_size = 10\n# tomo_type = \"denoised\"\n# inference_transforms = Compose([\n#     EnsureChannelFirstd(keys=[\"image\"], channel_dim=\"no_channel\"),\n#     NormalizeIntensityd(keys=\"image\"),\n#     Orientationd(keys=[\"image\"], axcodes=\"RAS\")\n# ])\n# import cc3d\n\n# id_to_name = {1: \"apo-ferritin\", \n#               2: \"beta-amylase\",\n#               3: \"beta-galactosidase\", \n#               4: \"ribosome\", \n#               5: \"thyroglobulin\", \n#               6: \"virus-like-particle\"}\n# BLOB_THRESHOLD = 200\n# CERTAINTY_THRESHOLD = 0.05\n\n# classes = [1, 2, 3, 4, 5, 6]\n# import torch\n# import numpy as np\n# import pandas as pd\n# import cc3d\n# from monai.data import CacheDataset\n# from monai.transforms import Compose, EnsureType\n# from torch import nn\n# from tqdm import tqdm\n# from monai.networks.nets import UNet\n# from monai.losses import TverskyLoss\n# from monai.metrics import DiceMetric\n\n# def load_models(model_paths):\n#     models = []\n#     for model_path in model_paths:\n#         channels = (48, 64, 80, 80)\n#         strides_pattern = (2, 2, 1)       \n#         num_res_units = 1\n#         learning_rate = 1e-3\n#         num_epochs = 100\n#         model = Model(channels=channels, strides=strides_pattern, num_res_units=num_res_units)\n        \n#         weights =torch.load(model_path)['state_dict']\n#         model.load_state_dict(weights)\n#         model.to('cuda')\n#         model.eval()\n#         models.append(model)\n#     return models\n\n\n# model_paths = [\n#     '/kaggle/input/cziials-a-230-unet/UNet-Model-val_metric0.450.ckpt',\n# ]\n\n\n# models = load_models(model_paths)\n# def ensemble_prediction_tta(models, input_tensor, threshold=0.5):\n#     probs_list = []\n#     data_copy0 = input_tensor.clone()\n#     data_copy0=torch.flip(data_copy0, dims=[2])\n#     data_copy1 = input_tensor.clone()\n#     data_copy1=torch.flip(data_copy1, dims=[3])\n#     data_copy2 = input_tensor.clone()\n#     data_copy2=torch.flip(data_copy2, dims=[4])\n#     data_copy3 = input_tensor.clone()\n#     data_copy3 = data_copy3.rot90(1, dims=[3, 4])\n#     with torch.no_grad():\n#         model_output0 = model(input_tensor)\n#         model_output1 = model(data_copy0)\n#         model_output1=torch.flip(model_output1, dims=[2])\n#         model_output2 = model(data_copy1)\n#         model_output2=torch.flip(model_output2, dims=[3])\n#         model_output3 = model(data_copy2)\n#         model_output3=torch.flip(model_output3, dims=[4])\n#         probs0 = torch.softmax(model_output0[0], dim=0)\n#         probs1 = torch.softmax(model_output1[0], dim=0)\n#         probs2 = torch.softmax(model_output2[0], dim=0)\n#         probs3 = torch.softmax(model_output3[0], dim=0)\n#         probs_list.append(probs0)\n#         probs_list.append(probs1)\n#         probs_list.append(probs2)\n#         probs_list.append(probs3)\n#     avg_probs = torch.mean(torch.stack(probs_list), dim=0)\n#     thresh_probs = avg_probs > threshold\n#     _, max_classes = thresh_probs.max(dim=0)\n#     return max_classes\n# sub=[]\n# for model in models:\n#     with torch.no_grad():\n#         location_df = []\n#         for run in root.runs:\n#             tomo = run.get_voxel_spacing(10)\n#             tomo = tomo.get_tomogram(tomo_type).numpy()\n#             tomo_patches, coordinates = extract_3d_patches_minimal_overlap([tomo], 96)\n#             tomo_patched_data = [{\"image\": img} for img in tomo_patches]\n#             tomo_ds = CacheDataset(data=tomo_patched_data, transform=inference_transforms, cache_rate=1.0)\n#             pred_masks = []\n#             for i in tqdm(range(len(tomo_ds))):\n#                 input_tensor = tomo_ds[i]['image'].unsqueeze(0).to(\"cuda\")\n#                 max_classes = ensemble_prediction_tta(models, input_tensor, threshold=CERTAINTY_THRESHOLD)\n#                 pred_masks.append(max_classes.cpu().numpy())\n#             reconstructed_mask = reconstruct_array(pred_masks, coordinates, tomo.shape)\n#             location = {}\n#             for c in classes:\n#                 cc = cc3d.connected_components(reconstructed_mask == c)\n#                 stats = cc3d.statistics(cc)\n#                 zyx = stats['centroids'][1:] * 10.012444  # 转换单位\n#                 zyx_large = zyx[stats['voxel_counts'][1:] > BLOB_THRESHOLD]\n#                 xyz = np.ascontiguousarray(zyx_large[:, ::-1])\n#                 location[id_to_name[c]] = xyz\n#             df = dict_to_df(location, run.name)\n#             location_df.append(df)\n#         location_df = pd.concat(location_df)\n#         location_df.insert(loc=0, column='id', value=np.arange(len(location_df)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T08:10:53.703813Z","iopub.execute_input":"2025-01-17T08:10:53.704685Z","iopub.status.idle":"2025-01-17T08:11:49.84576Z","shell.execute_reply.started":"2025-01-17T08:10:53.704659Z","shell.execute_reply":"2025-01-17T08:11:49.84505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **《《《 Finaly Blend 》》》**","metadata":{}},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n# from sklearn.cluster import DBSCAN\n\n\n# df = pd.concat([submission_,location_df], ignore_index=True)\n\n# particle_names = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']\n# particle_radius = {\n#     'apo-ferritin': 60,\n#     'beta-amylase': 65,\n#     'beta-galactosidase': 90,\n#     'ribosome': 150,\n#     'thyroglobulin': 130,\n#     'virus-like-particle': 135,\n# }\n\n# final = []\n# for pidx, p in enumerate(particle_names):\n#     pdf = df[df['particle_type'] == p].reset_index(drop=True)\n#     p_rad = particle_radius[p]\n    \n#     grouped = pdf.groupby(['experiment'])\n    \n#     for exp, group in grouped:\n#         group = group.reset_index(drop=True)\n        \n#         coords = group[['x', 'y', 'z']].values\n#         db = DBSCAN(eps=p_rad, min_samples=2, metric='euclidean').fit(coords)\n#         labels = db.labels_\n        \n#         group['cluster'] = labels\n        \n#         for cluster_id in np.unique(labels):\n#             if cluster_id == -1:\n#                 continue\n            \n#             cluster_points = group[group['cluster'] == cluster_id]\n            \n#             avg_x = cluster_points['x'].mean()\n#             avg_y = cluster_points['y'].mean()\n#             avg_z = cluster_points['z'].mean()\n            \n#             group.loc[group['cluster'] == cluster_id, ['x', 'y', 'z']] = avg_x, avg_y, avg_z\n#             group = group.drop_duplicates(subset=['x', 'y', 'z'])\n#         final.append(group)\n\n# df_save = pd.concat(final, ignore_index=True)\n# df_save = df_save.drop(columns=['cluster'])\n# df_save = df_save.sort_values(by=['experiment', 'particle_type']).reset_index(drop=True)\n# df_save['id'] = np.arange(0, len(df_save))\n# df_save.to_csv('submission.csv', index=False)\n\n\nimport pandas as pd\nimport numpy as np\n\n# Assuming location_df is the DataFrame generated by the UNet predictions\n# location_df contains columns: ['id', 'experiment', 'particle_type', 'x', 'y', 'z']\n\n# Sort and reset the index for the output\nlocation_df = location_df.sort_values(by=['experiment', 'particle_type']).reset_index(drop=True)\n\n# Add unique ID to each row\nlocation_df['id'] = np.arange(len(location_df))\n\n# Save to CSV\nlocation_df.to_csv('submission.csv', index=False)\n\n# Output success message\nprint(f\"Submission saved to 'submission.csv' with {len(location_df)} entries.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T08:15:21.703777Z","iopub.execute_input":"2025-01-17T08:15:21.704218Z","iopub.status.idle":"2025-01-17T08:15:21.733825Z","shell.execute_reply.started":"2025-01-17T08:15:21.704168Z","shell.execute_reply":"2025-01-17T08:15:21.733101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}