{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9867543,"sourceType":"datasetVersion","datasetId":6040935},{"sourceId":10671260,"sourceType":"datasetVersion","datasetId":6609543},{"sourceId":220858915,"sourceType":"kernelVersion"}],"dockerImageVersionId":30839,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Install and Import modules","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"!tar xfvz /kaggle/input/czii-ultralytics-packages/archieve.tar.gz\n!pip install --no-index --find-links=./packages ultralytics\n!rm -rf ,/packages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:25:28.820503Z","iopub.execute_input":"2025-02-13T09:25:28.820838Z","iopub.status.idle":"2025-02-13T09:26:16.351216Z","shell.execute_reply.started":"2025-02-13T09:25:28.820813Z","shell.execute_reply":"2025-02-13T09:26:16.349931Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cp -r '/kaggle/input/hengck-czii-cryo-et-01/wheel_file' '/kaggle/working/'\n!pip install /kaggle/working/wheel_file/asciitree-0.3.3/asciitree-0.3.3\n!pip install --no-index --find-links=/kaggle/working/wheel_file zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:16.352475Z","iopub.execute_input":"2025-02-13T09:26:16.352823Z","iopub.status.idle":"2025-02-13T09:26:26.877946Z","shell.execute_reply.started":"2025-02-13T09:26:16.352792Z","shell.execute_reply":"2025-02-13T09:26:26.876811Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zarr\nfrom ultralytics import YOLO\nfrom tqdm import tqdm\nimport glob, os\nimport torch\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\n\nimport sys\nsys.setrecursionlimit(10000)\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:26.879030Z","iopub.execute_input":"2025-02-13T09:26:26.879284Z","iopub.status.idle":"2025-02-13T09:26:33.528724Z","shell.execute_reply.started":"2025-02-13T09:26:26.879262Z","shell.execute_reply":"2025-02-13T09:26:33.528019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prepare trained YOLO model","metadata":{}},{"cell_type":"code","source":"model = YOLO(\"/kaggle/input/czii-yolo-model/runs/detect/train/weights/best.pt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:33.529541Z","iopub.execute_input":"2025-02-13T09:26:33.530164Z","iopub.status.idle":"2025-02-13T09:26:34.609476Z","shell.execute_reply.started":"2025-02-13T09:26:33.530133Z","shell.execute_reply":"2025-02-13T09:26:34.608539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"runs = sorted(glob.glob('/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/*'))\nruns = [os.path.basename(x) for x in runs]\nruns[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.610298Z","iopub.execute_input":"2025-02-13T09:26:34.610595Z","iopub.status.idle":"2025-02-13T09:26:34.622820Z","shell.execute_reply.started":"2025-02-13T09:26:34.610571Z","shell.execute_reply":"2025-02-13T09:26:34.622113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Information of labels\n\nparticle_names = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']\n\np2i_dict = {\n        'apo-ferritin': 0,\n        'beta-amylase': 1,\n        'beta-galactosidase': 2,\n        'ribosome': 3,\n        'thyroglobulin': 4,\n        'virus-like-particle': 5\n    }\n\ni2p = {v:k for k, v in p2i_dict.items()}\n\n#Radius\nparticle_radius = {\n        'apo-ferritin': 60,\n        'beta-amylase': 65,\n        'beta-galactosidase': 90,\n        'ribosome': 150,\n        'thyroglobulin': 130,\n        'virus-like-particle': 135,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.625009Z","iopub.execute_input":"2025-02-13T09:26:34.625212Z","iopub.status.idle":"2025-02-13T09:26:34.629682Z","shell.execute_reply.started":"2025-02-13T09:26:34.625195Z","shell.execute_reply":"2025-02-13T09:26:34.629006Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Main process Class\nto manage variables","metadata":{}},{"cell_type":"code","source":"class PredAggForYOLO:\n    def __init__(self,first_conf=0.2, final_conf=0.3,conf_coef=0.75):\n        self.first_conf = first_conf\n        self.final_conf = final_conf\n        self.conf_coef = conf_coef\n        self.particle_confs = [0.5,0,0.2,0.5,0.2,0.5]\n\n    def convert_to_8bit(self,x):\n        lower,upper = np.percentile(x,(0.5,99.5))\n        x = np.clip(x,lower,upper)\n        x = (x-x.min())/(x.max()-x.min() + 1e-12) * 255\n        return x.round().astype(\"uint8\")\n\n    # depth first search.\n    # aggregate the coordinates and confidence scores of connected graphs.\n    def dfs(self,v):\n        self.passed[v]=True\n        self.conf_sum += self.pdf.iloc[v].confidence\n        self.cx += self.pdf.iloc[v].x\n        self.cy += self.pdf.iloc[v].y\n        self.cz += self.pdf.iloc[v].z\n        self.nv += 1\n        for next_v in self.adjacency_list[v]:\n            if (self.passed[next_v]): \n                continue\n            self.dfs(next_v)\n\n    def make_predict_yolo(self,r,model):\n        vol = zarr.open(f'/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/{r}/VoxelSpacing10.000/denoised.zarr',mode='r')\n        vol = vol[0]\n        vol2 = self.convert_to_8bit(vol)\n        n_imgs = vol2.shape[0]\n\n        df = pd.DataFrame()\n\n        pts = []\n        confs = []\n        xs = []\n        ys = []\n        zs = []\n\n        for i in range(n_imgs):\n            tmp_img = np.zeros((630,630))\n            tmp_img[:] = vol2[i]\n\n            inp_arr = np.stack([tmp_img]*3,axis=-1)\n            inp_arr = cv2.resize(inp_arr, (640,640))\n            res = model.predict(inp_arr, save=False, imgsz=640, conf=self.first_conf, device=\"0\", batch=1, verbose=False)\n            for j ,result in enumerate(res):\n                boxes = result.boxes # Boxes object for bounding box outputs\n                for k in range(len(boxes.cls)):\n                    ptype = i2p[boxes.cls.cpu().numpy()[k]] # particle type\n                    conf = boxes.conf.cpu().numpy()[k] # confidence score\n                    # YOLO can infer (start_x, end_x, start_y, end_y)\n                    xc = (boxes.xyxy[k,0] + boxes.xyxy[k,2])/2.0*10*(63/64)\n                    yc = (boxes.xyxy[k,1] + boxes.xyxy[k,3])/2.0*10*(63/64)\n                    zc = i*10+5\n\n                    pts.append(ptype)\n                    confs.append(conf)\n                    xs.append(xc.cpu().numpy())\n                    ys.append(yc.cpu().numpy())\n                    zs.append(zc)\n\n        df['particle_type'] = pts\n        df['confidence'] = confs\n        df['x'] = xs\n        df['y'] = ys\n        df['z'] = zs\n        \n        df= df.sort_values(['particle_type','z'],ascending=[True,True])\n        agg_df=[]\n        for pidx,p in enumerate(particle_names):\n            if p =='beta-amylase':\n                continue\n            pdf = df[df['particle_type']==p].reset_index(drop = True)\n            self.pdf = pdf\n            p_rad = particle_radius[p]\n            # The distance between the x and y coordinates of adjacent slices is expected to be very small.\n            xy_tol = p_rad/16.0\n            xy_tol_p2 = xy_tol**2\n            # Define the graph\n            self.adjacency_list = [[]for _ in range(len(pdf))]\n            # which already passed in dfs\n            self.passed = [False for _ in range(len(pdf))]\n            #Connect two points when they are close enough\n            for i in range(len(pdf)):\n                x1=pdf['x'].iloc[i]\n                y1=pdf['y'].iloc[i]\n                z1=pdf['z'].iloc[i]\n                for j in range(i+1,len(pdf),1):\n                    x2 = pdf['x'].iloc[j]\n                    y2 = pdf['y'].iloc[j]\n                    z2 = pdf['z'].iloc[j]\n                    if abs(z1-z2)>20:\n                        break\n                    dist_p2 = (x1-x2)**2 + (y1-y2)**2\n                    if dist_p2<xy_tol_p2 and dist_p2+(z1-z2)**2 < p_rad**2 and abs(z1-z2)<=20:\n                        self.adjacency_list[i].append(j)\n                        self.adjacency_list[j].append(i)\n\n            rdf = pd.DataFrame()\n            cxs = []\n            cys = []\n            czs = []\n\n            # Perform DFS on all points and find the center of the sphere from the average of the coordinates\n            for i in range(len(pdf)):\n                self.conf_sum = 0\n                self.nv = 0\n                self.cx = 0\n                self.cy = 0\n                self.cz = 0\n                if not self.passed[i]:\n                    self.dfs(i)\n\n                # Different confidence for different particle types\n                if self.nv >= 2 and self.conf_sum/(self.nv**self.conf_coef)> self.particle_confs[pidx]:\n                    cxs.append(self.cx/self.nv)\n                    cys.append(self.cy/self.nv)\n                    czs.append(self.cz/self.nv)\n            \n            rdf['experiment'] = [r]* len(cxs)\n            rdf['particle_type'] = [p]*len(cys)\n            rdf['x'] = cxs\n            rdf['y'] = cys\n            rdf['z'] = czs\n            agg_df.append(rdf)\n                        \n\n        return pd.concat(agg_df, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.630759Z","iopub.execute_input":"2025-02-13T09:26:34.631009Z","iopub.status.idle":"2025-02-13T09:26:34.648737Z","shell.execute_reply.started":"2025-02-13T09:26:34.630982Z","shell.execute_reply":"2025-02-13T09:26:34.648011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# instance main class\nagent = PredAggForYOLO(first_conf=0.15, final_conf=0.2, conf_coef=0.5) \n# final_conf is not used after version 14\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.649541Z","iopub.execute_input":"2025-02-13T09:26:34.649865Z","iopub.status.idle":"2025-02-13T09:26:34.666790Z","shell.execute_reply.started":"2025-02-13T09:26:34.649834Z","shell.execute_reply":"2025-02-13T09:26:34.666080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs=[]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.667457Z","iopub.execute_input":"2025-02-13T09:26:34.667697Z","iopub.status.idle":"2025-02-13T09:26:34.683204Z","shell.execute_reply.started":"2025-02-13T09:26:34.667633Z","shell.execute_reply":"2025-02-13T09:26:34.682379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.683931Z","iopub.execute_input":"2025-02-13T09:26:34.684126Z","iopub.status.idle":"2025-02-13T09:26:34.698443Z","shell.execute_reply.started":"2025-02-13T09:26:34.684109Z","shell.execute_reply":"2025-02-13T09:26:34.697619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Main loop for inference","metadata":{}},{"cell_type":"code","source":"%%time\ntick = time.time()\nfor r in tqdm(runs,total=len(runs)):\n    df =  agent.make_predict_yolo(r, model)\n    subs.append(df)\ntock = time.time()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:26:34.699205Z","iopub.execute_input":"2025-02-13T09:26:34.699485Z","iopub.status.idle":"2025-02-13T09:27:31.231872Z","shell.execute_reply.started":"2025-02-13T09:26:34.699464Z","shell.execute_reply":"2025-02-13T09:27:31.230758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'estimated predict time is {(tock-tick)/3*500:.4f} seconds')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:27:31.232735Z","iopub.execute_input":"2025-02-13T09:27:31.232965Z","iopub.status.idle":"2025-02-13T09:27:31.237719Z","shell.execute_reply.started":"2025-02-13T09:27:31.232944Z","shell.execute_reply":"2025-02-13T09:27:31.236596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.concat(subs).reset_index(drop=True)\nsubmission.insert(0, 'id', range(len(submission)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:27:31.238531Z","iopub.execute_input":"2025-02-13T09:27:31.238820Z","iopub.status.idle":"2025-02-13T09:27:31.259945Z","shell.execute_reply.started":"2025-02-13T09:27:31.238800Z","shell.execute_reply":"2025-02-13T09:27:31.259157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-13T09:27:31.260627Z","iopub.execute_input":"2025-02-13T09:27:31.260857Z","iopub.status.idle":"2025-02-13T09:27:31.297566Z","shell.execute_reply.started":"2025-02-13T09:27:31.260839Z","shell.execute_reply":"2025-02-13T09:27:31.296952Z"}},"outputs":[],"execution_count":null}]}