{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":10104241,"sourceType":"datasetVersion","datasetId":6232544},{"sourceId":10123719,"sourceType":"datasetVersion","datasetId":6247131},{"sourceId":10123770,"sourceType":"datasetVersion","datasetId":6247176},{"sourceId":10124414,"sourceType":"datasetVersion","datasetId":6247596},{"sourceId":10124445,"sourceType":"datasetVersion","datasetId":6247613}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis notebook illustrates how to generate labels from simulated data.  It is based on the notebook from @fnands located here: [https://www.kaggle.com/code/fnands/create-numpy-dataset-exp-name](https://www.kaggle.com/code/fnands/create-numpy-dataset-exp-name) that does the same for the non-simulated data.  Be forewarned, when I trained a model incorporating this data with @fnands other notebook, [https://www.kaggle.com/code/fnands/baseline-unet-train-submit](https://www.kaggle.com/code/fnands/baseline-unet-train-submit), the score dropped from 486 to 450.  **If you are able to use this notebook in a way that improves your model, it would be great (though not required) if you would share some of the details.**\n\nNote some of the related discussion:  \n* [https://www.kaggle.com/competitions/czii-cryo-et-object-identification/discussion/550481](https://www.kaggle.com/competitions/czii-cryo-et-object-identification/discussion/550481)  \n* [https://www.kaggle.com/competitions/czii-cryo-et-object-identification/discussion/549744](https://www.kaggle.com/competitions/czii-cryo-et-object-identification/discussion/549744)\n","metadata":{}},{"cell_type":"markdown","source":"# Install Packages","metadata":{}},{"cell_type":"code","source":"!pip install git+https://github.com/copick/copick-utils.git matplotlib tqdm copick","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T01:22:57.045537Z","iopub.execute_input":"2024-12-09T01:22:57.045925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import copick\nimport fileinput\nimport json\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport shutil\nimport zarr\n\nfrom glob import glob\nfrom tqdm import tqdm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ND JSON to Pick Conversion\n\nIn the simulated code the particle coordinates are stored in .ndjson files instead of the Picks .json files.  The following method converts the .ndjson files to the Picks .json files.","metadata":{}},{"cell_type":"code","source":"def ndjson_to_pick(run, particle, src_path, dest_path):\n    pick = {}\n    pick['pickable_object_name'] = particle\n    pick['user_id'] = 'curation'\n    pick['session_id'] = '0'\n    pick['run_name'] = run\n    pick['voxel_spacing'] = None\n    pick['unit'] = 'angstrom'\n    pick['points'] = []\n\n    lines = fileinput.input(files=[src_path])\n    for line in lines:\n        nd_point = json.loads(line)\n        point = {}\n        point['location'] = {}\n        point['location']['x'] = 10.012*nd_point['location']['x']\n        point['location']['y'] = 10.012*nd_point['location']['y']\n        point['location']['z'] = 10.012*nd_point['location']['z']\n        point['transformation_'] = [\n            [1.0, 0.0, 0.0, 0.0],\n            [0.0, 1.0, 0.0, 0.0],\n            [0.0, 0.0, 1.0, 0.0],\n            [0.0, 0.0, 0.0, 1.0],\n        ]\n        point['instance_id'] = 0\n        pick['points'].append(point)\n        \n    lines.close()\n    \n    with open(dest_path, 'w') as f:\n        f.write(json.dumps(pick))\n        ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setup Copick","metadata":{}},{"cell_type":"code","source":"\nconfig_blob = \"\"\"{\n    \"name\": \"czii_cryoet_mlchallenge_2024\",\n    \"description\": \"2024 CZII CryoET ML Challenge training data.\",\n    \"version\": \"1.0.0\",\n\n    \"pickable_objects\": [\n        {\n            \"name\": \"apo-ferritin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"4V1W\",\n            \"label\": 1,\n            \"color\": [  0, 117, 220, 128],\n            \"radius\": 60,\n            \"map_threshold\": 0.0418\n        },\n        {\n            \"name\": \"beta-amylase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"1FA2\",\n            \"label\": 2,\n            \"color\": [153,  63,   0, 128],\n            \"radius\": 65,\n            \"map_threshold\": 0.035\n        },\n        {\n            \"name\": \"beta-galactosidase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6X1Q\",\n            \"label\": 3,\n            \"color\": [ 76,   0,  92, 128],\n            \"radius\": 90,\n            \"map_threshold\": 0.0578\n        },\n        {\n            \"name\": \"ribosome\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6EK0\",\n            \"label\": 4,\n            \"color\": [  0,  92,  49, 128],\n            \"radius\": 150,\n            \"map_threshold\": 0.0374\n        },\n        {\n            \"name\": \"thyroglobulin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6SCJ\",\n            \"label\": 5,\n            \"color\": [ 43, 206,  72, 128],\n            \"radius\": 130,\n            \"map_threshold\": 0.0278\n        },\n        {\n            \"name\": \"virus-like-particle\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6N4V\",            \n            \"label\": 6,\n            \"color\": [255, 204, 153, 128],\n            \"radius\": 135,\n            \"map_threshold\": 0.201\n        }\n    ],\n\n    \"overlay_root\": \"/kaggle/working/overlay\",\n\n    \"overlay_fs_args\": {\n        \"auto_mkdir\": true\n    }\n\n}\"\"\"\n# \"static_root\": \"/kaggle/input/czii-cryo-et-object-identification/train/static\"\n\ncopick_config_path = \"/kaggle/working/copick.config\"\noutput_overlay = \"/kaggle/working/overlay\"\n\nwith open(copick_config_path, \"w\") as f:\n    f.write(config_blob)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Do the Conversion\n\nFind all the .ndjson files and convert them.","metadata":{}},{"cell_type":"code","source":"pick_map = {\n    'ferritin_complex': 'apo-ferritin',\n    'beta_amylase': 'beta-amylase',\n    'beta_galactosidase': 'beta-galactosidase',\n    'cytosolic_ribosome': 'ribosome',\n    'thyroglobulin': 'thyroglobulin',\n    'pp7_vlp': 'virus-like-particle',\n}\nndjson_files = glob('/kaggle/input/czii-cryoet-simulated-training-data-ts*/**/*.ndjson',\n                    recursive=True)\nfor file in ndjson_files:\n    run = file.split('/')[4]\n    particle = pick_map[file.split('/')[9].split('-')[0]]\n    dest_dir = f'/kaggle/working/overlay/ExperimentRuns/{run}/Picks'\n    dest_path = f'{dest_dir}/curation_0_{particle}.json'\n    os.makedirs(dest_dir, exist_ok=True)\n    ndjson_to_pick(run, particle, file, dest_path)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create the Labels\n\nThe following code mostly from [https://www.kaggle.com/code/fnands/create-numpy-dataset-exp-name](https://www.kaggle.com/code/fnands/create-numpy-dataset-exp-name) generates the labels.  Note the simulated data already has its own segmentation layers that could alternatively be used, but this code creates the spherical labels used in [https://www.kaggle.com/code/fnands/baseline-unet-train-submit](https://www.kaggle.com/code/fnands/baseline-unet-train-submit).","metadata":{}},{"cell_type":"code","source":"root = copick.from_file(copick_config_path)\n\ncopick_user_name = \"copickUtils\"\ncopick_segmentation_name = \"paintedPicks\"\nvoxel_size = 10\ntomo_type = \"denoised\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from copick_utils.segmentation import segmentation_from_picks\nimport copick_utils.writers.write as write\nfrom collections import defaultdict\n\n# Just do this once\ngenerate_masks = True\n\nif generate_masks:\n    target_objects = defaultdict(dict)\n    for object in root.pickable_objects:\n        if object.is_particle:\n            target_objects[object.name]['label'] = object.label\n            target_objects[object.name]['radius'] = object.radius\n\n\n    for run in tqdm(root.runs):\n        # tomo = run.get_voxel_spacing(10)\n        # tomo = tomo.get_tomogram(tomo_type).numpy()\n        # target = np.zeros(tomo.shape, dtype=np.uint8)\n        target = np.zeros((200, 630, 630), dtype=np.uint8)\n        for pickable_object in root.pickable_objects:\n            pick = run.get_picks(object_name=pickable_object.name, user_id=\"curation\")\n            if len(pick):  \n                target = segmentation_from_picks.from_picks(pick[0], \n                                                            target, \n                                                            target_objects[pickable_object.name]['radius'] * 0.8,\n                                                            target_objects[pickable_object.name]['label']\n                                                            )\n        write.segmentation(run, target, copick_user_name, name=copick_segmentation_name)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the Tomograms\n\nLoad the tomograms and truncate the size to be that of the phantom training data (i.e. TS_6_4, etc.)","metadata":{}},{"cell_type":"code","source":"tomograms = glob('/kaggle/input/czii-cryoet-simulated-training-data-ts*/**/Tomograms/**/*.zarr',\n                 recursive=True)\ntomogram_map = {}\nfor t in tomograms:\n    tomogram_map[t.split('/')[4]] = t","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts = []\nfor run in tqdm(root.runs):\n    tomogram = np.array(zarr.open(tomogram_map[run.name], mode='r')[0])[:184]\n    segmentation = run.get_segmentations(name=copick_segmentation_name, user_id=copick_user_name, voxel_size=voxel_size, is_multilabel=True)[0].numpy()[:184]\n    data_dicts.append({\"name\": run.name, \"image\": tomogram, \"label\": segmentation})\n    \nprint(np.unique(data_dicts[0]['label']))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['label'].shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['image'].shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(len(tomogram_map)):\n    with open(f\"train_image_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['image'])\n        \n    with open(f\"train_label_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['label'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls -lh","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Image vs. Label","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.subplot(2,2,1)\nplt.xticks([])\nplt.yticks([])\nplt.title('Simulated Data')\nplt.imshow(data_dicts[0]['image'][92], cmap='gray', vmin=-2, vmax=2)\nplt.subplot(2,2,2)\nplt.xticks([])\nplt.yticks([])\nplt.title('Labels')\nplt.imshow(data_dicts[0]['label'][92])\nplt.subplot(2,2,3)\nplt.xticks([])\nplt.yticks([])\nplt.title('Data Overlaid With Labels')\nplt.imshow(data_dicts[0]['image'][92], cmap='gray', vmin=-2, vmax=2)\n_ = plt.imshow(data_dicts[0]['label'][92], cmap='Greens', alpha=0.3)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}