{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-30T12:29:55.016818Z","iopub.execute_input":"2024-11-30T12:29:55.017267Z","iopub.status.idle":"2024-11-30T12:29:56.835250Z","shell.execute_reply.started":"2024-11-30T12:29:55.017219Z","shell.execute_reply":"2024-11-30T12:29:56.834265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis\r\nThe dataset for this competition consists of 3D tomograms containing biological structures. Our task is to identify and classify these structures within the tomograms. To better understand the dataset and prepare for modeling, the following data analysis steps are performed:\r\n\r\n1. **Data Visualization**: Visualize 3D tomograms and overlay annotations to explore the spatial distribution of structures and identify any patterns or anomalies.\r\n2. **Class Distribution Analysis**: Analyze the frequency of each structure type to detect class imbalance and visualize their proportions.\r\n3. **Spatial Distribution Analysis**: Plot the locations of structures in 3D space and generate density maps to study spatial patterns or clustering tendencies.","metadata":{}},{"cell_type":"code","source":"!pip install napari -q\n!pip install napari-ome-zarr -q\n# Library imports\nimport os\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport napari\nimport json\n\n# Constants\nROOT = os.getcwd()\nDATA_DIR = '/kaggle/input/czii-cryo-et-object-identification'\nTRAIN_ZARRS = '/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T12:38:43.189555Z","iopub.execute_input":"2024-11-30T12:38:43.189993Z","iopub.status.idle":"2024-11-30T12:39:06.464096Z","shell.execute_reply.started":"2024-11-30T12:38:43.189955Z","shell.execute_reply":"2024-11-30T12:39:06.462788Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Visualization \r\n\r\nThe first block of code below will open the 3D tomogram images using Napari. It will open a Napari viewer for each experiment. I also added some point layers with the coordinates of each type of structure. Not much information can be extracted from this visualization but it's always good to see what you are dealing with. \r\nIf you want to visualize the 4 different types of zarr files in the training set then change the **tomogram type** (denoised, isonetcorrected, wbp, ctfdeconvolved).","metadata":{}},{"cell_type":"code","source":"# Choose the type of tomogram to visualize (e.g., 'denoised')\ntomogram_type = 'denoised'\n\nstructure_colors = {\n    'apo-ferritin': 'red',\n    'beta-amylase': 'blue',\n    'beta-galactosidase': 'green',\n    'ribosome': 'yellow',\n    'thyroglobulin': 'magenta',\n    'virus-like-particle': 'cyan'\n}\n\nexperiments = os.listdir(TRAIN_ZARRS)\n\nfor exp in experiments:\n    print(f'Processing experiment: {exp}')\n    \n    zarr_path = os.path.join(TRAIN_ZARRS, exp, 'VoxelSpacing10.000', f'{tomogram_type}.zarr')\n    \n    if not os.path.exists(zarr_path):\n        print(f'Zarr file not found for experiment {exp}: {zarr_path}')\n        continue\n    \n    viewer = napari.Viewer(title=f'Experiment: {exp}')\n    viewer.open(zarr_path)\n    \n    try:\n        viewer.open(zarr_path, plugin='napari-ome-zarr', layer_type='image')\n    except Exception as e:\n        print(f'Failed to open Zarr file for experiment {exp}: {e}')\n        continue\n    \n    if 'Tomogram' not in viewer.layers:\n        for layer in viewer.layers:\n            layer.name = 'Tomogram'\n    \n    structure_coords = {key: [] for key in structure_colors.keys()}\n    \n    annotations_dir = os.path.join(DATA_DIR, 'train', 'overlay', 'ExperimentRuns', exp, 'Picks')\n    \n    if not os.path.exists(annotations_dir):\n        print(f'Annotations directory not found for experiment {exp}: {annotations_dir}')\n        viewer.close()\n        continue\n    \n    annotation_files = os.listdir(annotations_dir)\n    \n    for annotation_file in annotation_files:\n        structure_name = os.path.splitext(annotation_file)[0]\n        annotation_path = os.path.join(annotations_dir, annotation_file)\n        \n        with open(annotation_path, 'r') as f:\n            annotation_data = json.load(f)\n            points = annotation_data.get('points', [])\n            \n            coords = []\n            for point in points:\n                location = point.get('location', {})\n                x = location.get('x')\n                y = location.get('y')\n                z = location.get('z')\n                \n                if x is not None and y is not None and z is not None:\n                    coords.append([z, y, x])  \n                else:\n                    print(f'Incomplete location data in {annotation_file}: {point}')\n        \n        if structure_name in structure_coords:\n            structure_coords[structure_name].extend(coords)\n        else:\n            print(f'Unknown structure type: {structure_name}')\n    \n    for key in structure_coords:\n        structure_coords[key] = np.array(structure_coords[key])\n    \n    for structure_name, coords in structure_coords.items():\n        if coords.size == 0:\n            continue  \n        viewer.add_points(\n            coords,\n            ndim = 3,\n            name=structure_name,\n            text=structure_name,\n            size=30,  \n            face_color=structure_colors[structure_name],\n            symbol='o',\n            opacity=1.0,\n            blending='opaque'\n        )\n\n    # napari.run()\n    # We can;t run it here since we need to install the plugin from inside the Napari GUI\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T12:39:11.248103Z","iopub.execute_input":"2024-11-30T12:39:11.248533Z","iopub.status.idle":"2024-11-30T12:39:14.206116Z","shell.execute_reply.started":"2024-11-30T12:39:11.248498Z","shell.execute_reply":"2024-11-30T12:39:14.202551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class Distribution \r\nThe results of the code block below will show a histogram of the number of structures contained in each experiment as well as the total number in the training dataset. We can see that the dataset is highly imbalanced. The most prominent structures are apo-ferritin, ribosomes and thyroglobulin.The beta-amylase has the least amount of examples in the dataset and is not scored for the competition.","metadata":{}},{"cell_type":"code","source":"experiments = os.listdir(TRAIN_ZARRS)\n\n# Initialize samples_dict with counts as numbers (not lists)\nsamples_dict = {exp: {key: 0 for key in structure_colors.keys()} for exp in experiments}\n\n# Iterate over each experiment to populate samples_dict\nfor exp in experiments:\n    annotations_dir = os.path.join(DATA_DIR, 'train', 'overlay', 'ExperimentRuns', exp, 'Picks') \n\n    annotation_files = os.listdir(annotations_dir)\n\n    for annotation_file in annotation_files:\n        structure_name = os.path.splitext(annotation_file)[0]\n        annotation_path = os.path.join(annotations_dir, annotation_file)\n\n        with open(annotation_path, 'r') as f:\n            annotation_data = json.load(f)\n            num_of_structures = len(annotation_data.get(\"points\", []))\n\n        samples_dict[exp][structure_name] += num_of_structures\n\nstructure_counts = pd.DataFrame.from_dict(samples_dict, orient='index')\n\ntotal_counts = structure_counts.sum().to_frame().T\ntotal_counts.index = ['Total']\n\nstructure_counts = pd.concat([structure_counts, total_counts], ignore_index=False)\n\nplt.figure(figsize=(15, 7))\nstructure_counts.plot(kind='bar', figsize=(15, 7), colormap='tab10')\nplt.xlabel('Experiments')\nplt.ylabel('Number of Structures')\nplt.title('Structure Counts per Experiment')\nplt.legend(title='Structure Types', bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T12:39:31.063728Z","iopub.execute_input":"2024-11-30T12:39:31.064092Z","iopub.status.idle":"2024-11-30T12:39:31.807049Z","shell.execute_reply.started":"2024-11-30T12:39:31.064059Z","shell.execute_reply":"2024-11-30T12:39:31.805841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Spatial Distribution Analysis\r\nThe final block belw calculated the spatial distribution of each structure using the kde plot from seaborn. We can see that most of these structures tend to group together with apo-ferritin being the best example of this phenomenon. Beta-galactosidase has an almost even spread. All structures are most probable to be found in the center of the tomogram.","metadata":{}},{"cell_type":"code","source":"structure_coords = {key: [] for key in structure_colors.keys()}\nexperiments = os.listdir(TRAIN_ZARRS)\n\nfor exp in experiments:    \n    zarr_path = os.path.join(TRAIN_ZARRS, exp, 'VoxelSpacing10.000', f'{tomogram_type}.zarr')\n    \n    if not os.path.exists(zarr_path):\n        print(f'Zarr file not found for experiment {exp}: {zarr_path}')\n        continue\n    \n    annotations_dir = os.path.join(DATA_DIR, 'train', 'overlay', 'ExperimentRuns', exp, 'Picks')\n    \n    annotation_files = os.listdir(annotations_dir)\n    \n    for annotation_file in annotation_files:\n        structure_name = os.path.splitext(annotation_file)[0]\n        annotation_path = os.path.join(annotations_dir, annotation_file)\n        \n        with open(annotation_path, 'r') as f:\n            annotation_data = json.load(f)\n            points = annotation_data.get('points', [])\n            \n            coords = []\n            for point in points:\n                location = point.get('location', {})\n                x = location.get('x')\n                y = location.get('y')\n                z = location.get('z')\n                \n                if x is not None and y is not None and z is not None:\n                    coords.append([x, y, z])  \n                else:\n                    print(f'Incomplete location data in {annotation_file}: {point}')\n        \n        if structure_name in structure_coords:\n            structure_coords[structure_name].extend(coords)\n\nplt.figure(figsize=(10, 15))\n\nfor i, (structure_name, coords) in enumerate(structure_coords.items()):\n    coords = np.array(coords)\n    x = coords[:, 0]\n    y = coords[:, 1]\n\n    plt.subplot(2, 3, i+1)\n    plt.title(f\"{structure_name} density plot\")\n    sns.kdeplot(x=x, y=y, cmap='flare', fill=True)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-30T12:39:36.037792Z","iopub.execute_input":"2024-11-30T12:39:36.038810Z","iopub.status.idle":"2024-11-30T12:39:38.285676Z","shell.execute_reply.started":"2024-11-30T12:39:36.038766Z","shell.execute_reply":"2024-11-30T12:39:38.284516Z"}},"outputs":[],"execution_count":null}]}