{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9902245,"sourceType":"datasetVersion","datasetId":6083037}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install git+https://github.com/copick/copick-utils.git matplotlib tqdm copick","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-09T13:12:05.547062Z","iopub.execute_input":"2025-02-09T13:12:05.547394Z","iopub.status.idle":"2025-02-09T13:12:26.115311Z","shell.execute_reply.started":"2025-02-09T13:12:05.547358Z","shell.execute_reply":"2025-02-09T13:12:26.113951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import copick\nimport fileinput\nimport json\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport shutil\nimport zarr\n\nfrom glob import glob\nfrom tqdm import tqdm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def ndjson_to_pick(run, particle, src_path, dest_path):\n    pick = {}\n    pick['pickable_object_name'] = particle\n    pick['user_id'] = 'curation'\n    pick['session_id'] = '0'\n    pick['run_name'] = run\n    pick['voxel_spacing'] = None\n    pick['unit'] = 'angstrom'\n    pick['points'] = []\n\n    lines = fileinput.input(files=[src_path])\n    for line in lines:\n        nd_point = json.loads(line)\n        point = {}\n        point['location'] = {}\n        point['location']['x'] = 10.012*nd_point['location']['x']\n        point['location']['y'] = 10.012*nd_point['location']['y']\n        point['location']['z'] = 10.012*nd_point['location']['z']\n        point['transformation_'] = [\n            [1.0, 0.0, 0.0, 0.0],\n            [0.0, 1.0, 0.0, 0.0],\n            [0.0, 0.0, 1.0, 0.0],\n            [0.0, 0.0, 0.0, 1.0],\n        ]\n        point['instance_id'] = 0\n        pick['points'].append(point)\n        \n    lines.close()\n    \n    with open(dest_path, 'w') as f:\n        f.write(json.dumps(pick))\n\nfrom scipy.ndimage import gaussian_filter, median_filter\n\ndef denoise_tomogram(tomogram, method='gaussian', **kwargs):\n    \"\"\"\n    Apply denoising to a tomogram.\n\n    Parameters:\n        tomogram (np.ndarray): The input tomogram to denoise.\n        method (str): The denoising method ('gaussian' or 'median').\n        kwargs: Parameters for the respective method.\n    \n    Returns:\n        np.ndarray: The denoised tomogram.\n    \"\"\"\n    if method == 'gaussian':\n        return gaussian_filter(tomogram, sigma=kwargs.get('sigma', 1))\n    elif method == 'median':\n        return median_filter(tomogram, size=kwargs.get('size', 3))\n    else:\n        raise ValueError(f\"Unsupported denoising method: {method}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"config_blob = \"\"\"{\n    \"name\": \"czii_cryoet_mlchallenge_2024\",\n    \"description\": \"2024 CZII CryoET ML Challenge training data.\",\n    \"version\": \"1.0.0\",\n\n    \"pickable_objects\": [\n        {\n            \"name\": \"apo-ferritin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"4V1W\",\n            \"label\": 1,\n            \"color\": [  0, 117, 220, 128],\n            \"radius\": 60,\n            \"map_threshold\": 0.0418\n        },\n        {\n            \"name\": \"beta-amylase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"1FA2\",\n            \"label\": 2,\n            \"color\": [153,  63,   0, 128],\n            \"radius\": 65,\n            \"map_threshold\": 0.035\n        },\n        {\n            \"name\": \"beta-galactosidase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6X1Q\",\n            \"label\": 3,\n            \"color\": [ 76,   0,  92, 128],\n            \"radius\": 90,\n            \"map_threshold\": 0.0578\n        },\n        {\n            \"name\": \"ribosome\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6EK0\",\n            \"label\": 4,\n            \"color\": [  0,  92,  49, 128],\n            \"radius\": 150,\n            \"map_threshold\": 0.0374\n        },\n        {\n            \"name\": \"thyroglobulin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6SCJ\",\n            \"label\": 5,\n            \"color\": [ 43, 206,  72, 128],\n            \"radius\": 130,\n            \"map_threshold\": 0.0278\n        },\n        {\n            \"name\": \"virus-like-particle\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6N4V\",            \n            \"label\": 6,\n            \"color\": [255, 204, 153, 128],\n            \"radius\": 135,\n            \"map_threshold\": 0.201\n        }\n    ],\n\n    \"overlay_root\": \"/kaggle/working/overlay\",\n\n    \"overlay_fs_args\": {\n        \"auto_mkdir\": true\n    }\n\n}\"\"\"\n# \"static_root\": \"/kaggle/input/czii-cryo-et-object-identification/train/static\"\n\ncopick_config_path = \"/kaggle/working/copick.config\"\noutput_overlay = \"/kaggle/working/overlay\"\n\nwith open(copick_config_path, \"w\") as f:\n    f.write(config_blob)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pick_map = {\n    'ferritin_complex': 'apo-ferritin',\n    'beta_amylase': 'beta-amylase',\n    'beta_galactosidase': 'beta-galactosidase',\n    'cytosolic_ribosome': 'ribosome',\n    'thyroglobulin': 'thyroglobulin',\n    'pp7_vlp': 'virus-like-particle',\n}\n\n#/kaggle/input/czii10441/10441/**/*.ndjson\n#ndjson_files = glob('/kaggle/input/czii-cryoet-simulated-training-data-ts*/**/*.ndjson',\n\nndjson_files = glob('/kaggle/input/czii10441/10441/**/*.ndjson',\n                    recursive=True)\n\nprint(f\"found {len(ndjson_files)} files\")\n\nfor file in ndjson_files:\n    print(file)\n    run = file.split('/')[5]\n    print(run)\n    if \"_\" not in run:\n        continue\n    particle = pick_map[file.split('/')[10].split('-')[0]]\n    dest_dir = f'/kaggle/working/overlay/ExperimentRuns/{run}/Picks'\n    dest_path = f'{dest_dir}/curation_0_{particle}.json'\n    os.makedirs(dest_dir, exist_ok=True)\n    ndjson_to_pick(run, particle, file, dest_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root = copick.from_file(copick_config_path)\n\ncopick_user_name = \"copickUtils\"\ncopick_segmentation_name = \"paintedPicks\"\nvoxel_size = 10\ntomo_type = \"denoised\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from copick_utils.segmentation import segmentation_from_picks\nimport copick_utils.writers.write as write\nfrom collections import defaultdict\n\n# Just do this once\ngenerate_masks = True\n\nif generate_masks:\n    target_objects = defaultdict(dict)\n    for object in root.pickable_objects:\n        if object.is_particle:\n            target_objects[object.name]['label'] = object.label\n            target_objects[object.name]['radius'] = object.radius\n\n\n    for run in tqdm(root.runs):\n        # tomo = run.get_voxel_spacing(10)\n        # tomo = tomo.get_tomogram(tomo_type).numpy()\n        # target = np.zeros(tomo.shape, dtype=np.uint8)\n        target = np.zeros((200, 630, 630), dtype=np.uint8)\n        for pickable_object in root.pickable_objects:\n            pick = run.get_picks(object_name=pickable_object.name, user_id=\"curation\")\n            if len(pick):  \n                target = segmentation_from_picks.from_picks(pick[0], \n                                                            target, \n                                                            target_objects[pickable_object.name]['radius'] * 0.8,\n                                                            target_objects[pickable_object.name]['label']\n                                                            )\n        write.segmentation(run, target, copick_user_name, name=copick_segmentation_name)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#tomograms = glob('/kaggle/input/czii-cryoet-simulated-training-data-ts*/**/Tomograms/**/*.zarr',\n\n#/kaggle/input/czii10441/10441/**/*.ndjson\n\ntomograms = glob('/kaggle/input/czii10441/10441/**/Tomograms/**/*.zarr',\n                 recursive=True)\ntomogram_map = {}\nfor t in tomograms:\n    print(t.split('/')[5])\n    tomogram_map[t.split('/')[5]] = t","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(tomogram_map)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_tomograms = len(root.runs)\nfirst_third = int(num_tomograms) / 3\nsecond_third = 2 * int(num_tomograms) / 3\nchunk_boundaries = [\n    (0, first_third),\n    (first_third, second_third),\n    (second_third, num_tomograms),\n]\n\nchunk_index = 1\nstart_idx, end_idx = chunk_boundaries[chunk_index]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts = []\nprint(root.runs)\ncount = 0\nfor i, run in tqdm(enumerate(root.runs), total=num_tomograms):\n    if not (start_idx <= i < end_idx):\n        continue\n    # if count == 7:\n    #     break\n    print(run.name)\n    if run.name not in tomogram_map:\n        print(f\"Run '{run.name}' does not have a corresponding tomogram. Skipping.\")\n        continue\n    tomogram_path = tomogram_map[run.name]\n    tomogram = np.array(zarr.open(tomogram_path, mode='r')[0])[:184]\n    tomogram_denoised = denoise_tomogram(tomogram, method='gaussian', sigma=1)  # Apply denoising\n\n    segmentation = run.get_segmentations(name=copick_segmentation_name, user_id=copick_user_name, voxel_size=voxel_size, is_multilabel=True)[0].numpy()[:184]\n    data_dicts.append({\n        \"name\": run.name, \n        \"original_image\": tomogram,          # Storing original image\n        \"denoised_image\": tomogram_denoised, # Storing denoised image\n        \"label\": segmentation\n    })\n    count +=1\n    \nprint(np.unique(data_dicts[0]['label']))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['label'].shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['original_image'].shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(len(data_dicts)):\n    with open(f\"train_original_image_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['original_image'])\n        \n    with open(f\"train_image_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['denoised_image'])\n        \n    with open(f\"train_label_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['label'])","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}