{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9867543,"sourceType":"datasetVersion","datasetId":6040935},{"sourceId":211097053,"sourceType":"kernelVersion"},{"sourceId":139474,"sourceType":"modelInstanceVersion","modelInstanceId":118113,"modelId":141350}],"dockerImageVersionId":30840,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r '/kaggle/input/hengck-czii-cryo-et-01/wheel_file' '/kaggle/working/'\n!pip install /kaggle/working/wheel_file/asciitree-0.3.3/asciitree-0.3.3\n!pip install --no-index --find-links=/kaggle/working/wheel_file zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T04:34:35.857912Z","iopub.execute_input":"2025-01-20T04:34:35.858185Z","iopub.status.idle":"2025-01-20T04:34:45.952824Z","shell.execute_reply.started":"2025-01-20T04:34:35.858164Z","shell.execute_reply":"2025-01-20T04:34:45.951862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset making","metadata":{}},{"cell_type":"code","source":"import json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport zarr\nfrom tqdm import tqdm\nimport glob, os\nimport cv2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-20T04:34:46.993189Z","iopub.execute_input":"2025-01-20T04:34:46.993534Z","iopub.status.idle":"2025-01-20T04:34:48.296630Z","shell.execute_reply.started":"2025-01-20T04:34:46.993503Z","shell.execute_reply":"2025-01-20T04:34:48.295998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"runs = sorted(glob.glob('/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns/*'))\nruns = [os.path.basename(x) for x in runs]\ni2r_dict = {i:r for i, r in zip(range(len(runs)), runs)}\nr2t_dict = {r:i for i, r in zip(range(len(runs)), runs)}\ni2r_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.520088Z","iopub.execute_input":"2025-01-17T07:06:07.520702Z","iopub.status.idle":"2025-01-17T07:06:07.532842Z","shell.execute_reply.started":"2025-01-17T07:06:07.520669Z","shell.execute_reply":"2025-01-17T07:06:07.532042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_to_8bit(x):\n    lower, upper = np.percentile(x, (0.5, 99.5))\n    x = np.clip(x, lower, upper)\n    x = (x - x.min()) / (x.max() - x.min() + 1e-12) * 255\n    return x.round().astype(\"uint8\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.533677Z","iopub.execute_input":"2025-01-17T07:06:07.533949Z","iopub.status.idle":"2025-01-17T07:06:07.538323Z","shell.execute_reply.started":"2025-01-17T07:06:07.533916Z","shell.execute_reply":"2025-01-17T07:06:07.537694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"p2i_dict = {\n        'apo-ferritin': 0,\n        'beta-amylase': 1,\n        'beta-galactosidase': 2,\n        'ribosome': 3,\n        'thyroglobulin': 4,\n        'virus-like-particle': 5\n    }\n\ni2p = {v:k for k, v in p2i_dict.items()}\n\nparticle_radius = {\n        'apo-ferritin': 60,\n        'beta-amylase': 65,\n        'beta-galactosidase': 90,\n        'ribosome': 150,\n        'thyroglobulin': 130,\n        'virus-like-particle': 135,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.539102Z","iopub.execute_input":"2025-01-17T07:06:07.539408Z","iopub.status.idle":"2025-01-17T07:06:07.553307Z","shell.execute_reply.started":"2025-01-17T07:06:07.539379Z","shell.execute_reply":"2025-01-17T07:06:07.552454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_names = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.555593Z","iopub.execute_input":"2025-01-17T07:06:07.555789Z","iopub.status.idle":"2025-01-17T07:06:07.564083Z","shell.execute_reply.started":"2025-01-17T07:06:07.555773Z","shell.execute_reply":"2025-01-17T07:06:07.563272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def make_annotate_yolo(run_name, is_train_path=True):\n    # to split validation\n    is_train_path = 'train' if is_train_path else 'val'\n\n    # read a volume\n    vol = zarr.open(f'/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns/{run_name}/VoxelSpacing10.000/denoised.zarr', mode='r') #bug fixed. Thanks to @pratyushh\n    # use largest images\n    vol = vol[0]\n    # normalize [0, 255]\n    vol2 = convert_to_8bit(vol)\n    \n    n_imgs = vol2.shape[0]\n    # process each slices\n    for j in range(n_imgs):\n        newvol = vol2[j]\n        newvolf = np.stack([newvol]*3, axis=-1)\n        # YOLO requires image_size is multiple of 32\n        newvolf = cv2.resize(newvolf, (640,640))\n        # save as 1 slice\n        cv2.imwrite(f'images/{is_train_path}/{run_name}_{j*10}.png', newvolf)\n        # make txt file for annotation\n        with open(f'labels/{is_train_path}/{run_name}_{j*10}.txt', 'w'):\n            pass # make empty file\n            \n    # process each paticle types\n    for p, particle in enumerate(tqdm(particle_names)):\n        # we do not have to detect beta-amylase which weight is 0\n        if particle==\"beta-amylase\":\n            continue\n        json_each_paticle = f\"/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns/{run_name}/Picks/{particle}.json\"\n        df = pd.read_json(json_each_paticle) \n        # pick each coordinate of particles\n        for axis in \"x\", \"y\", \"z\":\n            df[axis] = df.points.apply(lambda x: x[\"location\"][axis])\n\n        \n        radius = particle_radius[particle]\n        for i, row in df.iterrows():\n            # The radius from the center of the particle is used to determine the slices present.\n            start_z = np.round(row['z'] - radius).astype(np.int32)\n            start_z = max(0, start_z//10) # 10 means pixelspacing\n            end_z = np.round(row['z'] + radius).astype(np.int32)\n            end_z = min(n_imgs, end_z//10) # 10 means pixelspacing\n            \n            for j in range(start_z+1, end_z+1-1, 1):\n                # white the results of annotation\n                with open(f'labels/{is_train_path}/{run_name}_{j*10}.txt', 'a') as f:\n                    f.write(f'{p2i_dict[particle]} {row[\"x\"]/10/vol2.shape[1]} {row[\"y\"]/10/vol2.shape[2]} {radius/10/vol2.shape[1]*2} {radius/10/vol2.shape[2]*2} \\n')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.565169Z","iopub.execute_input":"2025-01-17T07:06:07.565449Z","iopub.status.idle":"2025-01-17T07:06:07.576891Z","shell.execute_reply.started":"2025-01-17T07:06:07.565421Z","shell.execute_reply":"2025-01-17T07:06:07.576111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs(\"images/train\", exist_ok=True)\nos.makedirs(\"images/val\", exist_ok=True)\nos.makedirs(\"labels/val\", exist_ok=True)\nos.makedirs(\"labels/train\", exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.577732Z","iopub.execute_input":"2025-01-17T07:06:07.578016Z","iopub.status.idle":"2025-01-17T07:06:07.591319Z","shell.execute_reply.started":"2025-01-17T07:06:07.577988Z","shell.execute_reply":"2025-01-17T07:06:07.590486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# use TS_5_4 as validation\nfor i, r in enumerate(runs):\n    make_annotate_yolo(r, is_train_path=False if i==0 else True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:06:07.592150Z","iopub.execute_input":"2025-01-17T07:06:07.592458Z","iopub.status.idle":"2025-01-17T07:07:11.941596Z","shell.execute_reply.started":"2025-01-17T07:06:07.592429Z","shell.execute_reply":"2025-01-17T07:07:11.940830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nos.makedirs('datasets/czii_det2d', exist_ok=True)\nshutil.move('images/train', 'datasets/czii_det2d/images/train')\nshutil.move('images/val', 'datasets/czii_det2d/images')\nshutil.move('labels/train', 'datasets/czii_det2d/labels/train')\nshutil.move('labels/val', 'datasets/czii_det2d/labels')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:07:11.942395Z","iopub.execute_input":"2025-01-17T07:07:11.942623Z","iopub.status.idle":"2025-01-17T07:07:13.263900Z","shell.execute_reply.started":"2025-01-17T07:07:11.942603Z","shell.execute_reply":"2025-01-17T07:07:13.262833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile czii_conf.yaml\n\npath: /kaggle/working/datasets/czii_det2d # dataset root dir\ntrain: images/train # train images (relative to 'path') \nval: images/val # val images (relative to 'path') \n\n# Classes\nnames:\n  0: apo-ferritin\n  1: beta-amylase\n  2: beta-galactosidase\n  3: ribosome\n  4: thyroglobulin\n  5: virus-like-particle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:07:13.264815Z","iopub.execute_input":"2025-01-17T07:07:13.265171Z","iopub.status.idle":"2025-01-17T07:07:13.270701Z","shell.execute_reply.started":"2025-01-17T07:07:13.265127Z","shell.execute_reply":"2025-01-17T07:07:13.269969Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"!tar xfvz /kaggle/input/ultralytics-for-offline-install/archive.tar.gz\n!pip install --no-index --find-links=./packages ultralytics\n!rm -rf ./packages","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-17T07:07:13.271560Z","iopub.execute_input":"2025-01-17T07:07:13.271797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nimport glob, os\nfrom ultralytics import YOLO","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a pretrained model\nmodel = YOLO(\"/kaggle/input/yolo11/pytorch/default/1/yolo11l.pt\")  # load a pretrained model (recommended for training)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\n_ = model.train(\n    data=\"/kaggle/working/czii_conf.yaml\",\n    epochs=120,\n    warmup_epochs=10,\n    optimizer='AdamW',\n    cos_lr=True,\n    lr0=3e-4,\n    lrf=0.03,\n    imgsz=640,\n    device=\"0,1\",\n    weight_decay=0.005,\n    batch=32,\n    scale=0,\n    flipud=0.5,\n    fliplr=0.5,\n    degrees=45,\n    shear=5,\n    mixup=0.2,\n    copy_paste=0.25,\n    seed=8620, # (｡•◡•｡)\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = YOLO(\"/kaggle/working/runs/detect/train/weights/best.pt\")\nmetrics = model.val(data=\"/kaggle/working/czii_conf.yaml\", imgsz=640, batch=16, conf=0.25, iou=0.6, device=\"0\", save_json=True)  # no arguments needed, dataset and settings remembered\nprint(metrics.box.map)  # map50-95\nprint(metrics.box.map50)  # map50\nprint(metrics.box.map75)  # map75\nprint(metrics.box.maps)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = model(\"/kaggle/working/datasets/czii_det2d/images/val/TS_5_4_920.png\")\nresults[0].show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Making Submission","metadata":{}},{"cell_type":"code","source":"import zarr\nfrom ultralytics import YOLO\nfrom tqdm import tqdm\nimport glob, os\nimport torch","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\nsys.setrecursionlimit(10000)\n\nimport warnings\nwarnings.simplefilter('ignore')\nnp.warnings = warnings","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"runs = sorted(glob.glob('/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/*'))\nruns = [os.path.basename(x) for x in runs]\n#change by @minfuka\n# runs[:5]\nsp = len(runs)//2\nruns1 = runs[:sp]\nruns1[:5]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"runs2 = runs[sp:]\nruns2[:5]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert torch.cuda.device_count() == 2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_names = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"p2i_dict = {\n        'apo-ferritin': 0,\n        'beta-amylase': 1,\n        'beta-galactosidase': 2,\n        'ribosome': 3,\n        'thyroglobulin': 4,\n        'virus-like-particle': 5\n    }\n\ni2p = {v:k for k, v in p2i_dict.items()}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_radius = {\n        'apo-ferritin': 60,\n        'beta-amylase': 65,\n        'beta-galactosidase': 90,\n        'ribosome': 150,\n        'thyroglobulin': 130,\n        'virus-like-particle': 135,\n    }","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class PredAggForYOLO:\n    def __init__(self, first_conf=0.2, final_conf=0.3, conf_coef=0.75):\n        self.first_conf = first_conf # threshold of confidence yolo\n        self.final_conf = final_conf # final threshold score (not be used in version 14)\n        self.conf_coef = conf_coef # if found many points, give bonus\n        self.particle_confs = [0.5, 0.0, 0.2, 0.5, 0.2, 0.5] # be strict to easy labels \n\n    def convert_to_8bit(self, x):\n        lower, upper = np.percentile(x, (0.5, 99.5))\n        x = np.clip(x, lower, upper)\n        x = (x - x.min()) / (x.max() - x.min() + 1e-12) * 255\n        return x.round().astype(\"uint8\")\n\n    # depth first search.\n    # aggregate the coordinates and confidence scores of connected graphs.\n    def dfs(self, v):\n        self.passed[v] = True\n        self.conf_sum += self.pdf.iloc[v].confidence\n        self.cx += self.pdf.iloc[v].x\n        self.cy += self.pdf.iloc[v].y\n        self.cz += self.pdf.iloc[v].z\n        self.nv += 1\n        for next_v in self.adjacency_list[v]:\n            if (self.passed[next_v]): continue\n            self.dfs(next_v)\n\n    # main routine.\n    # change by @minfuka\n    # def make_predict_yolo(self, r, model):\n    def make_predict_yolo(self, r, model, device_no):\n        vol = zarr.open(f'/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns/{r}/VoxelSpacing10.000/denoised.zarr', mode='r')\n        vol = vol[0]\n        vol2 = self.convert_to_8bit(vol)\n        n_imgs = vol2.shape[0]\n    \n        df = pd.DataFrame()\n    \n        pts = []\n        confs = []\n        xs = []\n        ys = []\n        zs = []\n        \n        for i in range(n_imgs):\n            # Unfortunately the image size needs to be a multiple of 32.\n            tmp_img = np.zeros((630, 630))\n            tmp_img[:] = vol2[i]\n    \n            inp_arr = np.stack([tmp_img]*3,axis=-1)\n            inp_arr = cv2.resize(inp_arr, (640,640))\n\n            # change by @minfuka\n            # res = model.predict(inp_arr, save=False, imgsz=640, conf=self.first_conf, device=\"0\", batch=1, verbose=False)\n            res = model.predict(inp_arr, save=False, imgsz=640, conf=self.first_conf, device=device_no, batch=1, verbose=False)\n            for j, result in enumerate(res):\n                boxes = result.boxes # Boxes object for bounding box outputs    \n                for k in range(len(boxes.cls)):\n                    ptype = i2p[boxes.cls.cpu().numpy()[k]] # particle type\n                    conf = boxes.conf.cpu().numpy()[k] # confidence score\n                    # YOLO can infer (start_x, end_x, start_y, end_y)\n                    xc = (boxes.xyxy[k,0] + boxes.xyxy[k,2]) / 2.0 * 10 * (63/64)\n                    yc = (boxes.xyxy[k,1] + boxes.xyxy[k,3]) / 2.0 * 10 * (63/64)\n                    zc = i * 10 + 0\n    \n                    pts.append(ptype)\n                    confs.append(conf)\n                    xs.append(xc.cpu().numpy())\n                    ys.append(yc.cpu().numpy())\n                    zs.append(zc)           \n                \n        df['particle_type'] = pts\n        df['confidence'] = confs\n        df['x'] = xs\n        df['y'] = ys\n        df['z'] = zs\n\n        # df includes overall canditate of CIRCLE. \n        df = df.sort_values(['particle_type', 'z'], ascending=[True, True])\n    \n        agg_df = []\n\n        # infer center of sphere each particle types\n        for pidx, p in enumerate(particle_names):\n            if p == 'beta-amylase':\n                continue\n            pdf = df[df['particle_type']==p].reset_index(drop=True)\n            self.pdf = pdf\n            p_rad = particle_radius[p]\n\n            # The distance between the x and y coordinates of adjacent slices is expected to be very small.\n            xy_tol = p_rad / 16.0\n            xy_tol_p2 = xy_tol ** 2\n\n            # define the graph\n            self.adjacency_list = [[] for _ in range(len(pdf))]\n            # which already passed in dfs\n            self.passed = [False for _ in range(len(pdf))]\n\n            # Connect two points when they are close enough\n            for i in range(len(pdf)):\n                x1 = pdf['x'].iloc[i]\n                y1 = pdf['y'].iloc[i]\n                z1 = pdf['z'].iloc[i]\n                for j in range(i+1, len(pdf), 1):\n                    x2 = pdf['x'].iloc[j]\n                    y2 = pdf['y'].iloc[j]\n                    z2 = pdf['z'].iloc[j]\n                    # Can be pruned. thanks to min fuka (@minfuka)\n                    if abs(z1-z2)>20:\n                        break\n    \n                    dist_p2 = (x1-x2)**2 + (y1-y2)**2\n                    if dist_p2<xy_tol_p2 and dist_p2+(z1-z2)**2 < p_rad**2 and abs(z1-z2)<=20:\n                        self.adjacency_list[i].append(j)\n                        self.adjacency_list[j].append(i)\n\n            rdf = pd.DataFrame()\n            cxs = []\n            cys = []\n            czs = []\n\n            # Perform DFS on all points and find the center of the sphere from the average of the coordinates\n            for i in range(len(pdf)):\n                self.conf_sum = 0\n                self.nv = 0\n                self.cx = 0\n                self.cy = 0\n                self.cz = 0\n                if not self.passed[i]:\n                    self.dfs(i)\n\n                # Different confidence for different particle types\n                if self.nv>=2 and self.conf_sum / (self.nv**self.conf_coef) > self.particle_confs[pidx]:\n                    cxs.append(self.cx / self.nv)\n                    cys.append(self.cy / self.nv)\n                    czs.append(self.cz / self.nv)\n\n            rdf['experiment'] = [r] * len(cxs)\n            rdf['particle_type'] = [p] * len(cys)\n            rdf['x'] = cxs\n            rdf['y'] = cys\n            rdf['z'] = czs\n\n            agg_df.append(rdf)\n\n       \n        return pd.concat(agg_df, axis=0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# instance main class\nagent = PredAggForYOLO(first_conf=0.15, final_conf=0.2, conf_coef=0.5) # final_conf is not used after version 14","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n#add by @minfuka\nfrom concurrent.futures import ProcessPoolExecutor #add by @minfuka","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#add by @minfuka\ndef inference(runs, model, device_no):\n    subs = []\n    for r in tqdm(runs, total=len(runs)):\n        df = agent.make_predict_yolo(r, model, device_no)\n        subs.append(df)\n    \n    return subs","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs = inference(runs, model, \"0\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(f'estimated predict time is {(tock-tick)/3*500:.4f} seconds')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#submission = pd.concat(subs).reset_index(drop=True)\n#change by @minfuka\nsubmission = pd.concat(subs).reset_index(drop=True)\nsubmission.insert(0, 'id', range(len(submission)))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/czii-cryo-et-object-identification/sample_submission.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /kaggle/working/*","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample.drop(sample.index, inplace=True)  # Удаляем все строки\nsample = pd.concat([sample, submission], ignore_index=True)  # Добавляем строки submission\nsample.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}