{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-20T20:53:06.641671Z","iopub.execute_input":"2024-12-20T20:53:06.642088Z","iopub.status.idle":"2024-12-20T20:53:08.653651Z","shell.execute_reply.started":"2024-12-20T20:53:06.642055Z","shell.execute_reply":"2024-12-20T20:53:08.652778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install zarr\nimport os\nimport numpy as np\nimport zarr\nimport json\nimport matplotlib.pyplot as plt\nfrom scipy.ndimage import rotate\n# from skimage.draw import sphere","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T20:53:08.654855Z","iopub.execute_input":"2024-12-20T20:53:08.655293Z","iopub.status.idle":"2024-12-20T20:53:17.885368Z","shell.execute_reply.started":"2024-12-20T20:53:08.655267Z","shell.execute_reply":"2024-12-20T20:53:17.884364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tomogram_path = \"/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns\"\ntrain_json_path = \"/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T20:53:17.886948Z","iopub.execute_input":"2024-12-20T20:53:17.887724Z","iopub.status.idle":"2024-12-20T20:53:17.892524Z","shell.execute_reply.started":"2024-12-20T20:53:17.887692Z","shell.execute_reply":"2024-12-20T20:53:17.891564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# initial data preprocessing\n# subtask 1: normalization\ndef normalize(tomogram):\n    # range [0, 1]\n    tomogram_np = tomogram[:]\n    tomogram_min = tomogram_np.min()\n    tomogram_max = tomogram_np.max()\n    return (tomogram_np - tomogram_min) / (tomogram_max - tomogram_min)\n\n#subtask 2: augmentation\ndef augment(tomogram):\n    # random rotation\n    angle = np.random.uniform(0, 360)\n    tomogram = rotate(tomogram, angle, axes=(1, 2), reshape=False, mode=\"reflect\")\n    \n    # random flip\n    if np.random.rand() > 0.5:\n        tomogram = np.flip(tomogram, axis=1)\n    if np.random.rand() > 0.5:\n        tomogram = np.flip(tomogram, axis=2)\n    \n    # random intensity\n    scaling_factor = np.random.uniform(0.8, 1.2)\n    tomogram = tomogram * scaling_factor\n    tomogram = np.clip(tomogram, 0, 1)  \n    return tomogram\n\ndef segment(shape, coords, radius):\n    mask = np.zeros(shape, dtype=np.uint8)\n    for coord in coords:\n        sphere_mask = create_sphere(coord, radius, shape)\n        mask[sphere_mask] = 1\n    return mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T20:53:17.893791Z","iopub.execute_input":"2024-12-20T20:53:17.894469Z","iopub.status.idle":"2024-12-20T20:53:17.916377Z","shell.execute_reply.started":"2024-12-20T20:53:17.894435Z","shell.execute_reply":"2024-12-20T20:53:17.914916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(experiment):\n    tomogram_file = os.path.join(train_tomogram_path, experiment, \"VoxelSpacing10.000/denoised.zarr\")\n    json_file = os.path.join(train_json_path, experiment, \"Picks/ribosome.json\")\n    \n    tomogram = zarr.open(tomogram_file, mode=\"r\")[0]  # Load highest-resolution data\n    normalized_tomogram = normalize(tomogram)\n    augmented_tomogram = augment(normalized_tomogram)\n    \n    if not os.path.exists(json_file):\n        print(f\"JSON file not found: {json_file}\")\n        print(\"Available directories in train_json_path:\")\n        print(os.listdir(os.path.join(train_json_path, experiment)))\n        return None, None\n\n    # load particle data and create segment mask\n    with open(json_file, \"r\") as f:\n        particle_data = json.load(f)\n    particle_coords = [(p['x'], p['y'], p['z']) for p in particle_data['points']]\n    segmentation_mask = segment(tomogram.shape, particle_coords, radius=5)\n    \n    return augmented_tomogram, segmentation_mask\n\n\ndef visualize(tomogram, mask, slice_idx):\n    plt.figure(figsize=(12, 6))\n    plt.subplot(1, 2, 1)\n    plt.title(\"Tomogram Slice\")\n    plt.imshow(tomogram[slice_idx], cmap=\"gray\")\n    plt.subplot(1, 2, 2)\n    plt.title(\"Segmentation Mask\")\n    plt.imshow(mask[slice_idx], cmap=\"gray\")\n    plt.show()\n\nexperiment_name = \"TS_5_4\"\naugmented_tomogram, segmentation_mask = preprocess(experiment_name)\n\nif augmented_tomogram is None or segmentation_mask is None:\n    print(\"Preprocessing failed. Visualization skipped.\")\nelse:\n    visualize(augmented_tomogram, segmentation_mask, slice_idx=50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-20T20:53:17.917622Z","iopub.execute_input":"2024-12-20T20:53:17.918239Z","iopub.status.idle":"2024-12-20T20:53:33.415509Z","shell.execute_reply.started":"2024-12-20T20:53:17.918203Z","shell.execute_reply":"2024-12-20T20:53:33.414480Z"}},"outputs":[],"execution_count":null}]}