{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":152689,"sourceType":"modelInstanceVersion","modelInstanceId":129676,"modelId":152535}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install copick git+https://github.com/copick/copick-utils.git git+https://github.com/copick/DeepFindET.git","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-01-13T20:23:21.934456Z","iopub.execute_input":"2025-01-13T20:23:21.934681Z","iopub.status.idle":"2025-01-13T20:25:40.244994Z","shell.execute_reply.started":"2025-01-13T20:23:21.93466Z","shell.execute_reply":"2025-01-13T20:25:40.24412Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q copick","metadata":{"execution":{"iopub.status.busy":"2025-01-13T16:32:53.87836Z","iopub.execute_input":"2025-01-13T16:32:53.878646Z","iopub.status.idle":"2025-01-13T16:32:57.336235Z","shell.execute_reply.started":"2025-01-13T16:32:53.878623Z","shell.execute_reply":"2025-01-13T16:32:57.335298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip list","metadata":{"execution":{"iopub.status.busy":"2024-11-04T13:51:14.033311Z","iopub.execute_input":"2024-11-04T13:51:14.034097Z","iopub.status.idle":"2024-11-04T13:51:16.639771Z","shell.execute_reply.started":"2024-11-04T13:51:14.034055Z","shell.execute_reply":"2024-11-04T13:51:16.638797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a copick project\n\nconfig_blob = \"\"\"{\n    \"name\": \"czii_cryoet_mlchallenge_2024\",\n    \"description\": \"2024 CZII CryoET ML Challenge training data.\",\n    \"version\": \"1.0.0\",\n\n    \"pickable_objects\": [\n        {\n            \"name\": \"apo-ferritin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"4V1W\",\n            \"label\": 1,\n            \"color\": [  0, 117, 220, 128],\n            \"radius\": 60,\n            \"map_threshold\": 0.0418\n        },\n        {\n            \"name\": \"beta-amylase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"1FA2\",\n            \"label\": 2,\n            \"color\": [153,  63,   0, 128],\n            \"radius\": 65,\n            \"map_threshold\": 0.035\n        },\n        {\n            \"name\": \"beta-galactosidase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6X1Q\",\n            \"label\": 3,\n            \"color\": [ 76,   0,  92, 128],\n            \"radius\": 90,\n            \"map_threshold\": 0.0578\n        },\n        {\n            \"name\": \"ribosome\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6EK0\",\n            \"label\": 4,\n            \"color\": [  0,  92,  49, 128],\n            \"radius\": 150,\n            \"map_threshold\": 0.0374\n        },\n        {\n            \"name\": \"thyroglobulin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6SCJ\",\n            \"label\": 5,\n            \"color\": [ 43, 206,  72, 128],\n            \"radius\": 130,\n            \"map_threshold\": 0.0278\n        },\n        {\n            \"name\": \"virus-like-particle\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6N4V\",            \n            \"label\": 6,\n            \"color\": [255, 204, 153, 128],\n            \"radius\": 135,\n            \"map_threshold\": 0.201\n        }\n    ],\n\n    \"overlay_root\": \"/kaggle/working/test/overlay\",\n\n    \"overlay_fs_args\": {\n        \"auto_mkdir\": true\n    },\n\n    \"static_root\": \"/kaggle/input/czii-cryo-et-object-identification/test/static\"\n}\"\"\"\n\ncopick_config_path = \"/kaggle/working/copick.config\"\n\nwith open(copick_config_path, \"w\") as f:\n    f.write(config_blob)\n    \n","metadata":{"execution":{"iopub.status.busy":"2025-01-13T16:33:36.930597Z","iopub.execute_input":"2025-01-13T16:33:36.930927Z","iopub.status.idle":"2025-01-13T16:33:36.935397Z","shell.execute_reply.started":"2025-01-13T16:33:36.930899Z","shell.execute_reply":"2025-01-13T16:33:36.934402Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deepfindET.entry_points import step3\nfrom deepfindET.utils import copick_tools\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport copick\n\n%matplotlib inline\n\n################## Input Parameters #################\n\n# Config File\nconfig = '/kaggle/working/copick.config'\n\n# Model Parameters\nn_class = 8                 # Number of classes to predict.\npatch_size = 160            # Size of the input patch fed into the model for inference.\nmodel_name = 'res_unet'     # The model architecture used for training.\nfilters = [48, 64, 128]      # Number of filters for U-Net (same parameter as used for training).\ndropout = 0                 # Dropout rate applied during inference.\n\n# Path to the pre-trained model weights for the chosen architecture.\nmodel_weights = '/kaggle/input/deepfindetv002/keras/default/1/net_weights_epoch70.h5'\n\n# Query for Tomogram\nvoxel_size = 10             # Resolution of the tomogram in voxel size.\ntomogram_algorithm = 'denoised'  # Reconstruction algorithm used for generating the tomogram\n\n# Output Segmentation Write Name\nsegmentation_name = 'predict'\nsession_id = '0'\nuser_id = 'deepfindET'\n","metadata":{"execution":{"iopub.status.busy":"2025-01-13T16:33:51.641187Z","iopub.execute_input":"2025-01-13T16:33:51.64149Z","iopub.status.idle":"2025-01-13T16:34:02.404297Z","shell.execute_reply.started":"2025-01-13T16:33:51.641468Z","shell.execute_reply":"2025-01-13T16:34:02.403439Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the train DeepFindET 3D U-Net model on the copick directory.\nstep3.inference_tomogram_segmentation(\n    config,                                 # Copick Configuration File\n    n_class,                                # Number of classes to predict.\n    model_name,                             # The model architecture used for training.\n    model_weights,                          # Path to the pre-trained model weights for the chosen architecture.\n    patch_size,                             # Size of the input patch fed into the model for inference.\n    user_id,                                # Identifier of the user or project running the segmentation.\n    session_id,                             # Session identifier for tracking purposes.\n    segmentation_name=segmentation_name,     # Identifier for the output segmentation file.\n    voxel_size = voxel_size,                # Voxel Size of Tomogram to Run Inference On       \n    model_filters = filters,                # Number of filters for U-Net\n    model_dropout = dropout,                # Dropout rate\n    tomogram_algorithm= tomogram_algorithm, # Reconstruction algorithm used for generating the tomogram\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from deepfindET.entry_points import step4\n\ncopick_root = copick.from_file(config)\n\n# Session ID for the Output Picks\npicks_session_id = '0'\n\nsegmentation_session_id = '0'\n\nsegmentation_name = 'predict' \n\nmin_protein_size = 0.4  \n\npath_output = f\"{copick_root.root_overlay}/ExperimentRuns\"","metadata":{"execution":{"iopub.status.busy":"2024-11-04T13:56:59.847672Z","iopub.execute_input":"2024-11-04T13:56:59.848676Z","iopub.status.idle":"2024-11-04T13:57:00.063256Z","shell.execute_reply.started":"2024-11-04T13:56:59.848634Z","shell.execute_reply":"2024-11-04T13:57:00.06245Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"[obj.radius for obj in copick_root.pickable_objects]","metadata":{"execution":{"iopub.status.busy":"2024-11-04T13:57:01.766186Z","iopub.execute_input":"2024-11-04T13:57:01.767714Z","iopub.status.idle":"2024-11-04T13:57:01.775055Z","shell.execute_reply.started":"2024-11-04T13:57:01.76767Z","shell.execute_reply":"2024-11-04T13:57:01.774027Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Derived from https://github.com/copick/DeepFindET/blob/main/deepfindET/entry_points/step4.py\n\nimport deepfindET.utils.copick_tools as tools\nimport deepfindET.utils.evaluate as evaluate\nimport scipy.ndimage as ndimage\nfrom tqdm import tqdm\n\n# Currently Filtering Process always finds coordinate at (cx,cy,cz) - center coordinate\n# This seems to always be at the first row, so we can remove it \nremove_index = 0\n\n# Extract Protein Coordinates from the Segmentation Masks\ndef extract_coords(pickable_object, copick_run):\n    labelmap = tools.get_copick_segmentation(copick_run, segmentation_name, user_id, segmentation_session_id)[:]\n    label = pickable_object.label\n    protein_name = pickable_object.name\n    label_objs, _ = ndimage.label(labelmap == label)\n\n    # Filter Candidates based on Object Size\n    # Get the sizes of all objects\n    object_sizes = np.bincount(label_objs.flat)\n\n    # Filter the objects based on size\n    min_object_size = 4/3 * np.pi * ((pickable_object.radius/voxel_size)**2) * min_protein_size\n    valid_objects = np.where(object_sizes > min_object_size)[0]                          \n\n    # Estimate Coordiantes from CoM for LabelMaps\n    deepFinderCoords = []\n    for object_num in tqdm(valid_objects):\n        com = ndimage.center_of_mass(label_objs == object_num)\n        swapped_com = (com[2], com[1], com[0])\n        deepFinderCoords.append(swapped_com)\n    deepFinderCoords = np.array(deepFinderCoords)   \n\n    # For some reason, consistently extracting center coordinate\n    # Remove the row with the closest index\n    deepFinderCoords = np.delete(deepFinderCoords, remove_index, axis=0)                    \n\n    # Estimate Distance Threshold Based on 1/2 of Particle Diameter\n    threshold = np.ceil(  pickable_object.radius / (voxel_size * 3) )\n\n    try: \n        # Remove Double Counted Coordinates\n        deepFinderCoords = evaluate.remove_repeated_picks(deepFinderCoords, threshold)\n\n        # Append Euler Angles to Coordinates [ Expand Dimensions from Nx3 -> Nx6 ]\n        deepFinderCoords = np.concatenate((deepFinderCoords, np.zeros(deepFinderCoords.shape)),axis=1)\n\n        # Convert from Voxel to Physical Units\n        deepFinderCoords *= voxel_size\n\n    except Exception as e:\n        print(f\"Error processing label {label} in tomo {copick_run}: {e}\")\n        deepFinderCoords = np.array([]).reshape(0,6)\n\n    # Save Picks in Copick Format / Directory \n    tools.write_copick_output(protein_name, copick_run.meta.name, deepFinderCoords, path_output, pickMethod=user_id, sessionID = picks_session_id)\n    \n    \nfor run in copick_root.runs:\n    print(f\"Run {run}\")\n    for pickable_object in copick_root.pickable_objects:\n        print(pickable_object.name)\n        if pickable_object.is_particle:\n            extract_coords(pickable_object, run)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T13:57:02.387194Z","iopub.execute_input":"2024-11-04T13:57:02.388014Z","iopub.status.idle":"2024-11-04T14:07:05.911382Z","shell.execute_reply.started":"2024-11-04T13:57:02.387973Z","shell.execute_reply":"2024-11-04T14:07:05.910329Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv\nimport os\n\nos.listdir(\"/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns\")\n\nresults = []\npick_id = 0\n\n# id,experiment,particle_type,x,y,z\nfor run in copick_root.runs:\n    run_id = run.meta.name\n    for particle_type in copick_root.pickable_objects:\n        picks = run.get_picks(particle_type.name, user_id=\"deepfindET\")\n        if picks:\n            picks = picks[0]\n            points = picks.points\n            for point in points:                \n                row = [pick_id, run_id, particle_type.name, point.location.x, point.location.y, point.location.z]\n                results.append(row)\n                pick_id += 1\n\nprint(f\"Found {len(results)} picks\")\n\n# Define CSV output file path\noutput_csv_path = \"/kaggle/working/submission.csv\"\n\n# Write results to CSV\nwith open(output_csv_path, mode='w', newline='') as file:\n    writer = csv.writer(file)\n    # Write header\n    writer.writerow([\"id\", \"experiment\", \"particle_type\", \"x\", \"y\", \"z\"])\n    # Write data rows\n    writer.writerows(results)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T14:07:05.998142Z","iopub.execute_input":"2024-11-04T14:07:05.998843Z","iopub.status.idle":"2024-11-04T14:07:06.016577Z","shell.execute_reply.started":"2024-11-04T14:07:05.998779Z","shell.execute_reply":"2024-11-04T14:07:06.015345Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}