{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":165828,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":141094,"modelId":163703}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install copick git+https://github.com/copick/copick-utils.git git+https://github.com/copick/DeepFindET.git","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-05T16:30:38.565733Z","iopub.execute_input":"2024-11-05T16:30:38.566019Z","iopub.status.idle":"2024-11-05T16:30:38.570538Z","shell.execute_reply.started":"2024-11-05T16:30:38.565986Z","shell.execute_reply":"2024-11-05T16:30:38.569708Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install -q copick","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:38.572328Z","iopub.execute_input":"2024-11-05T16:30:38.572695Z","iopub.status.idle":"2024-11-05T16:30:38.580665Z","shell.execute_reply.started":"2024-11-05T16:30:38.572652Z","shell.execute_reply":"2024-11-05T16:30:38.579830Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(\"Available CPUs:\", os.cpu_count())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T17:45:23.073863Z","iopub.execute_input":"2024-11-05T17:45:23.074261Z","iopub.status.idle":"2024-11-05T17:45:23.085368Z","shell.execute_reply.started":"2024-11-05T17:45:23.074221Z","shell.execute_reply":"2024-11-05T17:45:23.084398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip list","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:38.582197Z","iopub.execute_input":"2024-11-05T16:30:38.582557Z","iopub.status.idle":"2024-11-05T16:30:44.146683Z","shell.execute_reply.started":"2024-11-05T16:30:38.582517Z","shell.execute_reply":"2024-11-05T16:30:44.145559Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a copick project\n\nconfig_blob = \"\"\"{\n    \"name\": \"czii_cryoet_mlchallenge_2024\",\n    \"description\": \"2024 CZII CryoET ML Challenge training data.\",\n    \"version\": \"1.0.0\",\n\n    \"pickable_objects\": [\n        {\n            \"name\": \"apo-ferritin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"4V1W\",\n            \"label\": 1,\n            \"color\": [  0, 117, 220, 128],\n            \"radius\": 60,\n            \"map_threshold\": 0.0418\n        },\n        {\n            \"name\": \"beta-amylase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"1FA2\",\n            \"label\": 2,\n            \"color\": [153,  63,   0, 128],\n            \"radius\": 65,\n            \"map_threshold\": 0.035\n        },\n        {\n            \"name\": \"beta-galactosidase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6X1Q\",\n            \"label\": 3,\n            \"color\": [ 76,   0,  92, 128],\n            \"radius\": 90,\n            \"map_threshold\": 0.0578\n        },\n        {\n            \"name\": \"ribosome\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6EK0\",\n            \"label\": 4,\n            \"color\": [  0,  92,  49, 128],\n            \"radius\": 150,\n            \"map_threshold\": 0.0374\n        },\n        {\n            \"name\": \"thyroglobulin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6SCJ\",\n            \"label\": 5,\n            \"color\": [ 43, 206,  72, 128],\n            \"radius\": 130,\n            \"map_threshold\": 0.0278\n        },\n        {\n            \"name\": \"virus-like-particle\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6N4V\",            \n            \"label\": 6,\n            \"color\": [255, 204, 153, 128],\n            \"radius\": 135,\n            \"map_threshold\": 0.201\n        }\n    ],\n\n    \"overlay_root\": \"/kaggle/working/test/overlay\",\n\n    \"overlay_fs_args\": {\n        \"auto_mkdir\": true\n    },\n\n    \"static_root\": \"/kaggle/input/czii-cryo-et-object-identification/test/static\"\n}\"\"\"\n\ncopick_config_path = \"/kaggle/working/copick.config\"\n\nwith open(copick_config_path, \"w\") as f:\n    f.write(config_blob)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.148284Z","iopub.execute_input":"2024-11-05T16:30:44.148662Z","iopub.status.idle":"2024-11-05T16:30:44.159344Z","shell.execute_reply.started":"2024-11-05T16:30:44.148627Z","shell.execute_reply":"2024-11-05T16:30:44.158429Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deepfindET.entry_points import step3\nfrom deepfindET.utils import copick_tools\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport copick\n\n%matplotlib inline\n\n################## Input Parameters #################\n\n# Config File\nconfig = '/kaggle/working/copick.config'\n\n# Model Parameters\nn_class = 8                 # Number of classes to predict.\npatch_size = 160            # Size of the input patch fed into the model for inference.\nmodel_name = 'res_unet'     # The model architecture used for training.\nfilters = [48, 64, 80]      # Number of filters for U-Net (same parameter as used for training).\ndropout = 0                 # Dropout rate applied during inference.\n\n# Path to the pre-trained model weights for the chosen architecture.\nmodel_weights = '/kaggle/input/deepfindetv3/keras/default/1/net_weights_FINAL.h5'\n\n# Query for Tomogram\nvoxel_size = 10             # Resolution of the tomogram in voxel size.\ntomogram_algorithm = 'denoised'  # Reconstruction algorithm used for generating the tomogram\n\n# Output Segmentation Write Name\nsegmentation_name = 'predict'\nsession_id = '0'\nuser_id = 'deepfindET'\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.161973Z","iopub.execute_input":"2024-11-05T16:30:44.162361Z","iopub.status.idle":"2024-11-05T16:30:44.538889Z","shell.execute_reply.started":"2024-11-05T16:30:44.162320Z","shell.execute_reply":"2024-11-05T16:30:44.537219Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the train DeepFindET 3D U-Net model on the copick directory. (run on all runs)\n# step3.inference_tomogram_segmentation(\n#     config,                                 # Copick Configuration File\n#     n_class,                                # Number of classes to predict.\n#     model_name,                             # The model architecture used for training.\n#     model_weights,                          # Path to the pre-trained model weights for the chosen architecture.\n#     patch_size,                             # Size of the input patch fed into the model for inference.\n#     user_id,                                # Identifier of the user or project running the segmentation.\n#     session_id,                             # Session identifier for tracking purposes.\n#     segmentation_name=segmentation_name,     # Identifier for the output segmentation file.\n#     voxel_size = voxel_size,                # Voxel Size of Tomogram to Run Inference On       \n#     model_filters = filters,                # Number of filters for U-Net\n#     model_dropout = dropout,                # Dropout rate\n#     tomogram_algorithm= tomogram_algorithm, # Reconstruction algorithm used for generating the tomogram    \n# )","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.539808Z","iopub.status.idle":"2024-11-05T16:30:44.540302Z","shell.execute_reply.started":"2024-11-05T16:30:44.540043Z","shell.execute_reply":"2024-11-05T16:30:44.540067Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import deepfindET.utils.copick_tools as tools\nfrom deepfindET.inference import Segment\nimport click, copick, json\nimport tensorflow as tf\nfrom typing import List\n\ndef store_segmentation_parameters(predict_config: str,\n    n_class: int,\n    model_name: str,\n    path_weights: str,\n    patch_size: int,\n    user_id: str,\n    session_id: str,\n    voxel_size: float = 10,\n    model_filters: List[int] = [48, 64, 80],\n    model_dropout: float = 0,\n    tomogram_algorithm: str = \"denoised\",\n    tomo_ids: str = None,\n    output_scoremap: bool = False,\n    scoremap_name: str = \"scoremap\",\n    segmentation_name: str = \"segmentation\"):\n\n    parameters = {\n        \"input\": {\n            \"predict_config\": predict_config,\n            \"voxel_size\": voxel_size,\n            \"tomogram_algorithm\": tomogram_algorithm\n        },\n        \"model_architecture\": {\n            \"n_class\": n_class,\n            \"model_name\": model_name,\n            \"path_weights\": path_weights,\n            \"patch_size\": patch_size,\n            \"model_filters\": model_filters,\n            \"model_dropout\": model_dropout\n        },\n        \"output\": {\n            \"user_id\": user_id,\n            \"session_id\": session_id,\n            \"output_scoremap\": output_scoremap,\n            \"scoremap_name\": scoremap_name,\n            \"segmentation_name\": segmentation_name,\n            \"tomo_ids\": tomo_ids\n        }\n    }\n    print('\\nSegmentation Parameters: ', json.dumps(parameters,indent=4),'\\n')\n\n    # Save to JSON file\n    output_file = f'{user_id}_{session_id}_{segmentation_name}_seg_params.json'\n    with open(output_file, 'w') as json_file:\n        json.dump(parameters, json_file, indent=4)\n\ndef inference_tomogram_segmentation(\n    predict_config: str,\n    n_class: int,\n    model_name: str,\n    path_weights: str,\n    patch_size: int,\n    user_id: str,\n    session_id: str,\n    voxel_size: float = 10,\n    model_filters: List[int] = [48, 64, 80],\n    model_dropout: float = 0,\n    tomogram_algorithm: str = \"denoised\",\n    parallel_mpi: bool = False,\n    tomo_ids: str = None,\n    output_scoremap: bool = False,\n    scoremap_name: str = \"scoremap\",\n    segmentation_name: str = \"segmentation\",        \n    ):\n\n    # Determine if Using MPI or Sequential Processing\n    if parallel_mpi:\n        from mpi4py import MPI\n\n        # Initialize MPI (Get Rank and nProc)\n        comm = MPI.COMM_WORLD\n        rank = comm.Get_rank()\n        nProcess = comm.Get_size()\n\n        locGPU = rank % len(tf.config.list_physical_devices('GPU'))\n    else:\n        nProcess = 1\n        rank = 0\n        locGPU = None\n\n    ############## (Step 1) Initialize segmentation task: ##############\n\n    # Load CoPick root\n    copickRoot = copick.from_file(predict_config)\n\n    seg = Segment(n_class, model_name, path_weights=path_weights, \n                  patch_size=patch_size, model_filters = model_filters, model_dropout = model_dropout, \n                  gpuID = locGPU)\n\n    # Load Evaluate TomoIDs\n    evalTomos = tomo_ids.split(\",\") if tomo_ids is not None else [run.name for run in copickRoot.runs]\n    # Hack to just evaluate every 10th tomo\n    evalTomos = evalTomos[::10]\n\n    # Print Segmenation Parameters\n    store_segmentation_parameters(predict_config, n_class, model_name, path_weights, patch_size, user_id,\n                                  session_id, voxel_size, model_filters, model_dropout, tomogram_algorithm,\n                                  tomo_ids, output_scoremap, scoremap_name, segmentation_name)    \n\n    # Create Temporary Empty Folder\n    for tomoInd in range(len(evalTomos)):\n        if (tomoInd + 1) % nProcess == rank:\n            # Extract TomoID and Associated Run\n            tomoID = evalTomos[tomoInd]\n            print(f'\\nProcessing Run: {tomoID} ({tomoInd}/{len(evalTomos)})')\n\n            # Load data:\n            tomo = tools.get_copick_tomogram(\n                copickRoot,\n                voxelSize=voxel_size,\n                tomoAlgorithm=tomogram_algorithm,\n                tomoID=tomoID,\n            )\n\n            # Segment tomogram:\n            scoremaps = seg.launch(tomo[:])\n\n            # Query Copick Runs\n            copickRun = copickRoot.get_run(tomoID)\n\n            # Write scoremaps to file:\n            if output_scoremap:\n                tools.write_ome_zarr_scoremap(\n                    copickRun,\n                    scoremaps,\n                    voxelSize=voxel_size,\n                    scoremapName=scoremap_name,\n                    userID=user_id,\n                    sessionID=session_id,\n                    tomo_type=tomogram_algorithm,\n                )\n\n            # Get labelmap from scoremaps:\n            labelmap = seg.to_labelmap(scoremaps)\n            tools.write_ome_zarr_segmentation(\n                copickRun,\n                labelmap,\n                voxelSize=voxel_size,\n                segmentationName=segmentation_name,\n                userID=user_id,\n                sessionID=session_id,\n            )\n\n    print(\"Segmentations Complete!\")\n\ninference_tomogram_segmentation(\n    config,                                 # Copick Configuration File\n    n_class,                                # Number of classes to predict.\n    model_name,                             # The model architecture used for training.\n    model_weights,                          # Path to the pre-trained model weights for the chosen architecture.\n    patch_size,                             # Size of the input patch fed into the model for inference.\n    user_id,                                # Identifier of the user or project running the segmentation.\n    session_id,                             # Session identifier for tracking purposes.\n    segmentation_name=segmentation_name,     # Identifier for the output segmentation file.\n    voxel_size = voxel_size,                # Voxel Size of Tomogram to Run Inference On       \n    model_filters = filters,                # Number of filters for U-Net\n    model_dropout = dropout,                # Dropout rate\n    tomogram_algorithm= tomogram_algorithm, # Reconstruction algorithm used for generating the tomogram    \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T16:06:15.631794Z","iopub.execute_input":"2024-11-13T16:06:15.632099Z","iopub.status.idle":"2024-11-13T16:06:16.009478Z","shell.execute_reply.started":"2024-11-13T16:06:15.632064Z","shell.execute_reply":"2024-11-13T16:06:16.007984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deepfindET.entry_points import step4\n\ncopick_root = copick.from_file(config)\n\n# Session ID for the Output Picks\npicks_session_id = '0'\n\nsegmentation_session_id = '0'\n\nsegmentation_name = 'predict' \n\nmin_protein_size = 0.4  \n\npath_output = f\"{copick_root.root_overlay}/ExperimentRuns\"","metadata":{"execution":{"iopub.status.busy":"2024-11-13T16:06:16.010324Z","iopub.status.idle":"2024-11-13T16:06:16.010697Z","shell.execute_reply.started":"2024-11-13T16:06:16.010518Z","shell.execute_reply":"2024-11-13T16:06:16.010537Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"[obj.radius for obj in copick_root.pickable_objects]","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.543963Z","iopub.status.idle":"2024-11-05T16:30:44.544484Z","shell.execute_reply.started":"2024-11-05T16:30:44.544201Z","shell.execute_reply":"2024-11-05T16:30:44.544233Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Derived from https://github.com/copick/DeepFindET/blob/main/deepfindET/entry_points/step4.py\n\nfrom skimage.measure import label\nimport deepfindET.utils.copick_tools as tools\nimport deepfindET.utils.evaluate as evaluate\nimport scipy.ndimage as ndimage\nfrom tqdm import tqdm\n\n# Currently Filtering Process always finds coordinate at (cx,cy,cz) - center coordinate\n# This seems to always be at the first row, so we can remove it \nremove_index = 0\n\n# Extract Protein Coordinates from the Segmentation Masks\ndef extract_coords(pickable_object, copick_run):\n    labelmap = np.asarray(tools.get_copick_segmentation(copick_run, segmentation_name, user_id, segmentation_session_id)[:])\n    label_id = pickable_object.label\n    protein_name = pickable_object.name\n    # ndimage version\n    label_objs, _ = ndimage.label(labelmap == label_id)\n    # label_objs, _ = label(labelmap == label_id)\n\n    # Filter Candidates based on Object Size\n    # Get the sizes of all objects\n    object_sizes = np.bincount(label_objs.flat)\n\n    # Filter the objects based on size\n    min_object_size = 4/3 * np.pi * ((pickable_object.radius/voxel_size)**2) * min_protein_size\n    valid_objects = np.where(object_sizes > min_object_size)[0]                          \n\n    # Estimate Coordiantes from CoM for LabelMaps\n    deepFinderCoords = []\n    for object_num in tqdm(valid_objects):\n        com = ndimage.center_of_mass(label_objs == object_num)\n        swapped_com = (com[2], com[1], com[0])\n        deepFinderCoords.append(swapped_com)\n    deepFinderCoords = np.array(deepFinderCoords)   \n\n    # For some reason, consistently extracting center coordinate\n    # Remove the row with the closest index\n    deepFinderCoords = np.delete(deepFinderCoords, remove_index, axis=0)                    \n\n    # Estimate Distance Threshold Based on 1/2 of Particle Diameter\n    threshold = np.ceil(  pickable_object.radius / (voxel_size * 3) )\n\n    try: \n        # Remove Double Counted Coordinates\n        deepFinderCoords = evaluate.remove_repeated_picks(deepFinderCoords, threshold)\n\n        # Append Euler Angles to Coordinates [ Expand Dimensions from Nx3 -> Nx6 ]\n        deepFinderCoords = np.concatenate((deepFinderCoords, np.zeros(deepFinderCoords.shape)),axis=1)\n\n        # Convert from Voxel to Physical Units\n        deepFinderCoords *= voxel_size\n\n    except Exception as e:\n        print(f\"Error processing label {label} in tomo {copick_run}: {e}\")\n        deepFinderCoords = np.array([]).reshape(0,6)\n\n    # Save Picks in Copick Format / Directory \n    tools.write_copick_output(protein_name, copick_run.meta.name, deepFinderCoords, path_output, pickMethod=user_id, sessionID = picks_session_id)\n    \n    \n# for run in copick_root.runs:\nfor run in copick_root.runs[::10]:    \n    print(f\"Run {run}\")\n    for pickable_object in copick_root.pickable_objects:\n        print(pickable_object.name)\n        if pickable_object.is_particle:\n            extract_coords(pickable_object, run)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.546298Z","iopub.status.idle":"2024-11-05T16:30:44.546721Z","shell.execute_reply.started":"2024-11-05T16:30:44.546516Z","shell.execute_reply":"2024-11-05T16:30:44.546551Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv\nimport os\n\nos.listdir(\"/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns\")\n\nresults = []\npick_id = 0\n\n# id,experiment,particle_type,x,y,z\nfor run in copick_root.runs:\n    run_id = run.meta.name\n    for particle_type in copick_root.pickable_objects:\n        picks = run.get_picks(particle_type.name, user_id=\"deepfindET\")\n        if picks:\n            picks = picks[0]\n            points = picks.points\n            for point in points:                \n                row = [pick_id, run_id, particle_type.name, point.location.x, point.location.y, point.location.z]\n                results.append(row)\n                pick_id += 1\n\nprint(f\"Found {len(results)} picks\")\n\n# Define CSV output file path\noutput_csv_path = \"/kaggle/working/submission.csv\"\n\n# Write results to CSV\nwith open(output_csv_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write header\n    writer.writerow([\"id\", \"experiment\", \"particle_type\", \"x\", \"y\", \"z\"])\n    # Write data rows\n    writer.writerows(results)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T16:30:44.547941Z","iopub.status.idle":"2024-11-05T16:30:44.548283Z","shell.execute_reply.started":"2024-11-05T16:30:44.548100Z","shell.execute_reply":"2024-11-05T16:30:44.548116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}