{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9867543,"sourceType":"datasetVersion","datasetId":6040935},{"sourceId":10104520,"sourceType":"datasetVersion","datasetId":6232734}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install git+https://github.com/copick/copick-utils.git matplotlib tqdm copick \n!pip install -q \"monai-weekly[mlflow]\"\n!pip install zarr\n!pip install copick\nimport os\nimport shutil\n\nconfig_blob = \"\"\"{\n    \"name\": \"czii_cryoet_mlchallenge_2024\",\n    \"description\": \"2024 CZII CryoET ML Challenge training data.\",\n    \"version\": \"1.0.0\",\n\n    \"pickable_objects\": [\n        {\n            \"name\": \"apo-ferritin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"4V1W\",\n            \"label\": 1,\n            \"color\": [  0, 117, 220, 128],\n            \"radius\": 60,\n            \"map_threshold\": 0.0418\n        },\n        {\n            \"name\": \"beta-amylase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"1FA2\",\n            \"label\": 2,\n            \"color\": [153,  63,   0, 128],\n            \"radius\": 65,\n            \"map_threshold\": 0.035\n        },\n        {\n            \"name\": \"beta-galactosidase\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6X1Q\",\n            \"label\": 3,\n            \"color\": [ 76,   0,  92, 128],\n            \"radius\": 90,\n            \"map_threshold\": 0.0578\n        },\n        {\n            \"name\": \"ribosome\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6EK0\",\n            \"label\": 4,\n            \"color\": [  0,  92,  49, 128],\n            \"radius\": 150,\n            \"map_threshold\": 0.0374\n        },\n        {\n            \"name\": \"thyroglobulin\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6SCJ\",\n            \"label\": 5,\n            \"color\": [ 43, 206,  72, 128],\n            \"radius\": 130,\n            \"map_threshold\": 0.0278\n        },\n        {\n            \"name\": \"virus-like-particle\",\n            \"is_particle\": true,\n            \"pdb_id\": \"6N4V\",            \n            \"label\": 6,\n            \"color\": [255, 204, 153, 128],\n            \"radius\": 135,\n            \"map_threshold\": 0.201\n        }\n    ],\n\n    \"overlay_root\": \"/kaggle/working/overlay\",\n\n    \"overlay_fs_args\": {\n        \"auto_mkdir\": true\n    },\n\n    \"static_root\": \"/kaggle/input/extra-data/czii\"\n}\"\"\"\n\ncopick_config_path = \"/kaggle/working/copick.config\"\noutput_overlay = \"/kaggle/working/overlay\"\n\nwith open(copick_config_path, \"w\") as f:\n    f.write(config_blob)\n    \n# Update the overlay\n# Define source and destination directories\nsource_dir = '/kaggle/input/extra-data/czii_static'\ndestination_dir = '/kaggle/working/overlay'\n\n# Walk through the source directory\nfor root, dirs, files in os.walk(source_dir):\n    # Create corresponding subdirectories in the destination\n    relative_path = os.path.relpath(root, source_dir)\n    target_dir = os.path.join(destination_dir, relative_path)\n    os.makedirs(target_dir, exist_ok=True)\n    \n    # Copy and rename each file\n    for file in files:\n        if file.startswith(\"curation_0_\"):\n            new_filename = file\n        else:\n            new_filename = f\"curation_0_{file}\"\n            \n        \n        # Define full paths for the source and destination files\n        source_file = os.path.join(root, file)\n        destination_file = os.path.join(target_dir, new_filename)\n        \n        # Copy the file with the new name\n        shutil.copy2(source_file, destination_file)\n        print(f\"Copied {source_file} to {destination_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-28T15:20:39.692443Z","iopub.execute_input":"2025-01-28T15:20:39.692881Z","iopub.status.idle":"2025-01-28T15:22:12.545928Z","shell.execute_reply.started":"2025-01-28T15:20:39.692823Z","shell.execute_reply":"2025-01-28T15:22:12.544546Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.ndimage import gaussian_filter, median_filter\n\ndef denoise_tomogram(tomogram, method='gaussian', **kwargs):\n    \"\"\"\n    Apply denoising to a tomogram.\n\n    Parameters:\n        tomogram (np.ndarray): The input tomogram to denoise.\n        method (str): The denoising method ('gaussian' or 'median').\n        kwargs: Parameters for the respective method.\n    \n    Returns:\n        np.ndarray: The denoised tomogram.\n    \"\"\"\n    if method == 'gaussian':\n        return gaussian_filter(tomogram, sigma=kwargs.get('sigma', 1))\n    elif method == 'median':\n        return median_filter(tomogram, size=kwargs.get('size', 3))\n    else:\n        raise ValueError(f\"Unsupported denoising method: {method}\")\nimport os\nimport numpy as np\nfrom pathlib import Path\nimport torch\nimport torchinfo\nimport zarr, copick\nfrom tqdm import tqdm\nfrom monai.data import DataLoader, Dataset, CacheDataset, decollate_batch\nfrom monai.transforms import (\n    Compose, \n    EnsureChannelFirstd, \n    Orientationd,  \n    AsDiscrete,  \n    RandFlipd, \n    RandRotate90d, \n    NormalizeIntensityd,\n    RandCropByLabelClassesd,\n)\nfrom monai.networks.nets import UNet\nfrom monai.losses import DiceLoss, FocalLoss, TverskyLoss\nfrom monai.metrics import DiceMetric, ConfusionMatrixMetric\nimport mlflow\nimport mlflow.pytorch\nroot = copick.from_file(copick_config_path)\n\ncopick_user_name = \"copickUtils\"\ncopick_segmentation_name = \"paintedPicks\"\nvoxel_size = 10\ntomo_type = \"denoised\"\nfrom copick_utils.segmentation import segmentation_from_picks\nimport copick_utils.writers.write as write\nfrom collections import defaultdict\n\n# Just do this once\ngenerate_masks = True\n\nif generate_masks:\n    target_objects = defaultdict(dict)\n    for object in root.pickable_objects:\n        if object.is_particle:\n            target_objects[object.name]['label'] = object.label\n            target_objects[object.name]['radius'] = object.radius\n\n\n    for run in tqdm(root.runs):\n        tomo = run.get_voxel_spacing(10)\n        tomo = tomo.get_tomogram(tomo_type).numpy()\n        \n        target = np.zeros(tomo.shape, dtype=np.uint8)\n        for pickable_object in root.pickable_objects:\n            pick = run.get_picks(object_name=pickable_object.name, user_id=\"curation\")\n            if len(pick):  \n                target = segmentation_from_picks.from_picks(pick[0], \n                                                            target, \n                                                            target_objects[pickable_object.name]['radius'] * 0.8,\n                                                            target_objects[pickable_object.name]['label']\n                                                            )\n        write.segmentation(run, target, copick_user_name, name=copick_segmentation_name)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-28T15:22:21.971170Z","iopub.execute_input":"2025-01-28T15:22:21.971575Z","iopub.status.idle":"2025-01-28T15:24:51.453272Z","shell.execute_reply.started":"2025-01-28T15:22:21.971536Z","shell.execute_reply":"2025-01-28T15:24:51.451663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts = []\nfor run in tqdm(root.runs):\n    tomogram = run.get_voxel_spacing(voxel_size).get_tomogram(tomo_type).numpy()\n    tomogram = denoise_tomogram(tomogram, method='gaussian', sigma=1)\n    segmentation = run.get_segmentations(name=copick_segmentation_name, user_id=copick_user_name, voxel_size=voxel_size, is_multilabel=True)[0].numpy()\n    data_dicts.append({\"name\": run.name, \"image\": tomogram, \"label\": segmentation})\n    \nprint(np.unique(data_dicts[0]['label']))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-28T15:25:23.778398Z","iopub.execute_input":"2025-01-28T15:25:23.782119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['label'].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T02:21:29.523642Z","iopub.execute_input":"2024-12-05T02:21:29.524165Z","iopub.status.idle":"2024-12-05T02:21:29.532089Z","shell.execute_reply.started":"2024-12-05T02:21:29.524121Z","shell.execute_reply":"2024-12-05T02:21:29.530868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dicts[0]['image'].shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T02:21:31.603034Z","iopub.execute_input":"2024-12-05T02:21:31.604360Z","iopub.status.idle":"2024-12-05T02:21:31.613266Z","shell.execute_reply.started":"2024-12-05T02:21:31.604272Z","shell.execute_reply":"2024-12-05T02:21:31.611623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(27):\n    with open(f\"train_image_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['image'])\n        \n    with open(f\"train_label_{data_dicts[i]['name']}.npy\", 'wb') as f:\n        np.save(f, data_dicts[i]['label'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T02:21:38.418391Z","iopub.execute_input":"2024-12-05T02:21:38.418913Z","iopub.status.idle":"2024-12-05T02:22:10.887169Z","shell.execute_reply.started":"2024-12-05T02:21:38.418873Z","shell.execute_reply":"2024-12-05T02:22:10.885451Z"}},"outputs":[],"execution_count":null}]}