{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":104.820086,"end_time":"2024-12-05T09:07:06.020274","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-05T09:05:21.200188","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CZII making datasets for YOLO\n\nThis is a challenging competition in which participants must identify the location of particles contained in a 3D volumetric image.\n\nThere are already some great baselines published, but most of them focus on 3D volumetric images.\n\nHowever, using 3D images directly is difficult: for example, we always have to be careful about VRAM consumption: even a small 3D image uses a lot of memory.\n\nTherefore, I propose to decompose the 3D data provided by the host into 2D image slices and reduce it to an object detection problem.\n\nThis method allows us to treat just 7 3D images as more than 1k 2D images, mitigating the data scarcity problem.","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":15.976063,"end_time":"2024-12-05T09:05:40.010098","exception":false,"start_time":"2024-12-05T09:05:24.034035","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-12-05T21:23:37.316944Z","iopub.execute_input":"2024-12-05T21:23:37.317946Z","iopub.status.idle":"2024-12-05T21:23:54.039570Z","shell.execute_reply.started":"2024-12-05T21:23:37.317856Z","shell.execute_reply":"2024-12-05T21:23:54.038202Z"}}},{"cell_type":"markdown","source":"## My other notebooks\n\n- [CZII making datasets for YOLO](https://www.kaggle.com/code/itsuki9180/czii-making-datasets-for-yolo) <- now you're reading\n- [CZII YOLO11 Training Baseline](https://www.kaggle.com/code/itsuki9180/czii-yolo11-training-baseline)\n- [CZII YOLO11 Submission Baseline](https://www.kaggle.com/code/itsuki9180/czii-yolo11-submission-baseline)\n\nIf you're interested in my baseline, check these out too!","metadata":{}},{"cell_type":"markdown","source":"# Install and Import modules","metadata":{}},{"cell_type":"code","source":"!pip install zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:22.556073Z","iopub.execute_input":"2024-12-06T00:50:22.556484Z","iopub.status.idle":"2024-12-06T00:50:37.473062Z","shell.execute_reply.started":"2024-12-06T00:50:22.556449Z","shell.execute_reply":"2024-12-06T00:50:37.471804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport zarr\nfrom tqdm import tqdm\nimport glob, os\nimport cv2","metadata":{"papermill":{"duration":1.485698,"end_time":"2024-12-05T09:05:41.501599","exception":false,"start_time":"2024-12-05T09:05:40.015901","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:37.475185Z","iopub.execute_input":"2024-12-06T00:50:37.475532Z","iopub.status.idle":"2024-12-06T00:50:38.123849Z","shell.execute_reply.started":"2024-12-06T00:50:37.475498Z","shell.execute_reply":"2024-12-06T00:50:38.122932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"runs = sorted(glob.glob('/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns/*'))\nruns = [os.path.basename(x) for x in runs]\ni2r_dict = {i:r for i, r in zip(range(len(runs)), runs)}\nr2t_dict = {r:i for i, r in zip(range(len(runs)), runs)}\ni2r_dict","metadata":{"papermill":{"duration":0.029085,"end_time":"2024-12-05T09:05:41.536029","exception":false,"start_time":"2024-12-05T09:05:41.506944","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.124972Z","iopub.execute_input":"2024-12-06T00:50:38.125556Z","iopub.status.idle":"2024-12-06T00:50:38.142319Z","shell.execute_reply.started":"2024-12-06T00:50:38.125519Z","shell.execute_reply":"2024-12-06T00:50:38.141479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normalize Function\nTo treat it as an image, normalize it to a value between 0 and 255.\n\n1e-12 is very small and has the meaning of epsilon.","metadata":{}},{"cell_type":"code","source":"def convert_to_8bit(x):\n    lower, upper = np.percentile(x, (0.5, 99.5))\n    x = np.clip(x, lower, upper)\n    x = (x - x.min()) / (x.max() - x.min() + 1e-12) * 255\n    return x.round().astype(\"uint8\")","metadata":{"papermill":{"duration":0.015646,"end_time":"2024-12-05T09:05:41.635797","exception":false,"start_time":"2024-12-05T09:05:41.620151","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.144166Z","iopub.execute_input":"2024-12-06T00:50:38.144482Z","iopub.status.idle":"2024-12-06T00:50:38.149389Z","shell.execute_reply.started":"2024-12-06T00:50:38.144451Z","shell.execute_reply":"2024-12-06T00:50:38.148405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Information about labels","metadata":{}},{"cell_type":"code","source":"p2i_dict = {\n        'apo-ferritin': 0,\n        'beta-amylase': 1,\n        'beta-galactosidase': 2,\n        'ribosome': 3,\n        'thyroglobulin': 4,\n        'virus-like-particle': 5\n    }\n\ni2p = {v:k for k, v in p2i_dict.items()}\n\nparticle_radius = {\n        'apo-ferritin': 60,\n        'beta-amylase': 65,\n        'beta-galactosidase': 90,\n        'ribosome': 150,\n        'thyroglobulin': 130,\n        'virus-like-particle': 135,\n    }","metadata":{"papermill":{"duration":0.014787,"end_time":"2024-12-05T09:05:41.674147","exception":false,"start_time":"2024-12-05T09:05:41.659360","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.150621Z","iopub.execute_input":"2024-12-06T00:50:38.150987Z","iopub.status.idle":"2024-12-06T00:50:38.159769Z","shell.execute_reply.started":"2024-12-06T00:50:38.150945Z","shell.execute_reply":"2024-12-06T00:50:38.158794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_names = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']","metadata":{"papermill":{"duration":0.01351,"end_time":"2024-12-05T09:05:41.692783","exception":false,"start_time":"2024-12-05T09:05:41.679273","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.160835Z","iopub.execute_input":"2024-12-06T00:50:38.161149Z","iopub.status.idle":"2024-12-06T00:50:38.168756Z","shell.execute_reply.started":"2024-12-06T00:50:38.161087Z","shell.execute_reply":"2024-12-06T00:50:38.167813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main function for making datasets for YOLO\nThis is the main function.\n\nWatch that YOLO annotation requires normalized 0 to 1 value range and (center_x, center_y, width, height) coordinate format.","metadata":{}},{"cell_type":"code","source":"def make_annotate_yolo(run_name, is_train_path=True):\n    # to split validation\n    is_train_path = 'train' if is_train_path else 'val'\n\n    # read a volume\n    vol = zarr.open(f'/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns/{run_name}/VoxelSpacing10.000/denoised.zarr', mode='r') #bug fixed. Thanks to @pratyushh\n    # use largest images\n    vol = vol[0]\n    # normalize [0, 255]\n    vol2 = convert_to_8bit(vol)\n    \n    n_imgs = vol2.shape[0]\n    # process each slices\n    for j in range(n_imgs):\n        newvol = vol2[j]\n        newvolf = np.stack([newvol]*3, axis=-1)\n        # YOLO requires image_size is multiple of 32\n        newvolf = cv2.resize(newvolf, (640,640))\n        # save as 1 slice\n        cv2.imwrite(f'images/{is_train_path}/{run_name}_{j*10}.png', newvolf)\n        # make txt file for annotation\n        with open(f'labels/{is_train_path}/{run_name}_{j*10}.txt', 'w'):\n            pass # make empty file\n            \n    # process each paticle types\n    for p, particle in enumerate(tqdm(particle_names)):\n        # we do not have to detect beta-amylase which weight is 0\n        if particle==\"beta-amylase\":\n            continue\n        json_each_paticle = f\"/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns/{run_name}/Picks/{particle}.json\"\n        df = pd.read_json(json_each_paticle) \n        # pick each coordinate of particles\n        for axis in \"x\", \"y\", \"z\":\n            df[axis] = df.points.apply(lambda x: x[\"location\"][axis])\n\n        \n        radius = particle_radius[particle]\n        for i, row in df.iterrows():\n            # The radius from the center of the particle is used to determine the slices present.\n            start_z = np.round(row['z'] - radius).astype(np.int32)\n            start_z = max(0, start_z//10) # 10 means pixelspacing\n            end_z = np.round(row['z'] + radius).astype(np.int32)\n            end_z = min(n_imgs, end_z//10) # 10 means pixelspacing\n            \n            for j in range(start_z+1, end_z+1-1, 1):\n                # white the results of annotation\n                with open(f'labels/{is_train_path}/{run_name}_{j*10}.txt', 'a') as f:\n                    f.write(f'{p2i_dict[particle]} {row[\"x\"]/10/vol2.shape[1]} {row[\"y\"]/10/vol2.shape[2]} {radius/10/vol2.shape[1]*2} {radius/10/vol2.shape[2]*2} \\n')\n    ","metadata":{"papermill":{"duration":0.019182,"end_time":"2024-12-05T09:05:41.717008","exception":false,"start_time":"2024-12-05T09:05:41.697826","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.170265Z","iopub.execute_input":"2024-12-06T00:50:38.170740Z","iopub.status.idle":"2024-12-06T00:50:38.180866Z","shell.execute_reply.started":"2024-12-06T00:50:38.170697Z","shell.execute_reply":"2024-12-06T00:50:38.179928Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prepare Folders","metadata":{}},{"cell_type":"code","source":"os.makedirs(\"images/train\", exist_ok=True)\nos.makedirs(\"images/val\", exist_ok=True)\nos.makedirs(\"labels/val\", exist_ok=True)\nos.makedirs(\"labels/train\", exist_ok=True)","metadata":{"papermill":{"duration":0.015137,"end_time":"2024-12-05T09:05:41.737480","exception":false,"start_time":"2024-12-05T09:05:41.722343","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.182040Z","iopub.execute_input":"2024-12-06T00:50:38.182448Z","iopub.status.idle":"2024-12-06T00:50:38.193302Z","shell.execute_reply.started":"2024-12-06T00:50:38.182387Z","shell.execute_reply":"2024-12-06T00:50:38.192539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main loop to make slice images and annotations","metadata":{}},{"cell_type":"code","source":"# use TS_5_4 as validation\nfor i, r in enumerate(runs):\n    make_annotate_yolo(r, is_train_path=False if i==0 else True)","metadata":{"papermill":{"duration":81.881721,"end_time":"2024-12-05T09:07:03.624350","exception":false,"start_time":"2024-12-05T09:05:41.742629","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:50:38.194428Z","iopub.execute_input":"2024-12-06T00:50:38.194719Z","iopub.status.idle":"2024-12-06T00:51:47.879777Z","shell.execute_reply.started":"2024-12-06T00:50:38.194690Z","shell.execute_reply":"2024-12-06T00:51:47.878698Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Put them all in one folder.","metadata":{}},{"cell_type":"code","source":"import shutil\nos.makedirs('datasets/czii_det2d', exist_ok=True)\nshutil.move('images/train', 'datasets/czii_det2d/images/train')\nshutil.move('images/val', 'datasets/czii_det2d/images')\nshutil.move('labels/train', 'datasets/czii_det2d/labels/train')\nshutil.move('labels/val', 'datasets/czii_det2d/labels')","metadata":{"papermill":{"duration":1.721015,"end_time":"2024-12-05T09:07:05.351974","exception":false,"start_time":"2024-12-05T09:07:03.630959","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:51:47.882897Z","iopub.execute_input":"2024-12-06T00:51:47.883622Z","iopub.status.idle":"2024-12-06T00:51:49.216537Z","shell.execute_reply.started":"2024-12-06T00:51:47.883588Z","shell.execute_reply":"2024-12-06T00:51:49.215428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# make yaml file for Training \nWe need to create a yaml configuration file for training, the format of which will not be detailed here.","metadata":{}},{"cell_type":"code","source":"%%writefile czii_conf.yaml\n\npath: /kaggle/input/czii-yolo-datasets/datasets/czii_det2d # dataset root dir\ntrain: images/train # train images (relative to 'path') \nval: images/val # val images (relative to 'path') \n\n# Classes\nnames:\n  0: apo-ferritin\n  1: beta-amylase\n  2: beta-galactosidase\n  3: ribosome\n  4: thyroglobulin\n  5: virus-like-particle","metadata":{"papermill":{"duration":0.018337,"end_time":"2024-12-05T09:07:05.377214","exception":false,"start_time":"2024-12-05T09:07:05.358877","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T00:51:49.217718Z","iopub.execute_input":"2024-12-06T00:51:49.218100Z","iopub.status.idle":"2024-12-06T00:51:49.224987Z","shell.execute_reply.started":"2024-12-06T00:51:49.218057Z","shell.execute_reply":"2024-12-06T00:51:49.223834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Continue to [Training Baseline...](https://www.kaggle.com/code/itsuki9180/czii-yolo11-training-baseline)","metadata":{}}]}