{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qq zarr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-12T10:28:38.52585Z","iopub.execute_input":"2024-11-12T10:28:38.526807Z","iopub.status.idle":"2024-11-12T10:28:57.869985Z","shell.execute_reply.started":"2024-11-12T10:28:38.526726Z","shell.execute_reply":"2024-11-12T10:28:57.86828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\n\nimport glob\nimport zarr\nimport json\nimport lzma\nimport pickle","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:28:57.873031Z","iopub.execute_input":"2024-11-12T10:28:57.874131Z","iopub.status.idle":"2024-11-12T10:28:58.570932Z","shell.execute_reply.started":"2024-11-12T10:28:57.874069Z","shell.execute_reply":"2024-11-12T10:28:58.569612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matplotlib Configuration","metadata":{}},{"cell_type":"code","source":"plt.rcParams['font.size'] = 12         # Default font size for all text\nplt.rcParams['axes.titlesize'] = 24    # Title font size\nplt.rcParams['axes.labelsize'] = 14    # Axis label font size\nplt.rcParams['xtick.labelsize'] = 12   # X-tick label font size\nplt.rcParams['ytick.labelsize'] = 12   # Y-tick label font size\nplt.rcParams['legend.fontsize'] = 16   # Legend font size","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:28:58.572457Z","iopub.execute_input":"2024-11-12T10:28:58.573303Z","iopub.status.idle":"2024-11-12T10:28:58.581076Z","shell.execute_reply.started":"2024-11-12T10:28:58.573259Z","shell.execute_reply":"2024-11-12T10:28:58.579527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration File","metadata":{}},{"cell_type":"code","source":"class Config():\n    ROOT = '/kaggle/input/czii-cryo-et-object-identification'\n    LABELS = (\n        'apo-ferritin', 'ribosome', 'thyroglobulin',\n        'virus-like-particle', 'beta-amylase',\n    )\n    \nCONFIG = Config()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:38:18.443365Z","iopub.execute_input":"2024-11-12T10:38:18.445222Z","iopub.status.idle":"2024-11-12T10:38:18.456885Z","shell.execute_reply.started":"2024-11-12T10:38:18.445132Z","shell.execute_reply":"2024-11-12T10:38:18.455274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Submission","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv(f'{CONFIG.ROOT}/sample_submission.csv')\n\ndisplay(sample_submission.info())\ndisplay(sample_submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:28:58.594832Z","iopub.execute_input":"2024-11-12T10:28:58.595272Z","iopub.status.idle":"2024-11-12T10:28:58.660592Z","shell.execute_reply.started":"2024-11-12T10:28:58.595232Z","shell.execute_reply":"2024-11-12T10:28:58.659485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Labels","metadata":{}},{"cell_type":"code","source":"TRAIN_FILE_PATHS = glob.glob(f'{CONFIG.ROOT}/train/static/ExperimentRuns/*/VoxelSpacing10.000/denoised.zarr')\n\nprint(f'Found {len(TRAIN_FILE_PATHS)} samples')","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:28:58.662307Z","iopub.execute_input":"2024-11-12T10:28:58.662807Z","iopub.status.idle":"2024-11-12T10:28:58.700263Z","shell.execute_reply.started":"2024-11-12T10:28:58.662752Z","shell.execute_reply":"2024-11-12T10:28:58.699109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate data\nDATA = {}\n\nfor fp in tqdm(TRAIN_FILE_PATHS):\n    # Extract sample index\n    sample_idx = fp.split('/')[7]\n    # Acquire highest resolution data\n    tomogram = np.array(zarr.open(fp, mode='r')[0])\n    print(f'{sample_idx.ljust(7)} | tomogram shape: {tomogram.shape}')\n    print(f'min: {tomogram.min():.2E}, max: {tomogram.max():.2E}', end=', ')\n    print(f'µ: {tomogram.mean():.2E}, σ: {tomogram.std():.2E}')\n    # Read input image\n    DATA[sample_idx] = { 'tomogram': tomogram }\n    # Read labels\n    for fp_label in glob.glob(f'{CONFIG.ROOT}/train/*/*/{sample_idx}/Picks/*'):\n        # Read label JSON file\n        with open(fp_label, 'r') as f:\n            data = json.load(f)\n        # Assign points\n        points = [v['location'] for v in data['points']]\n        points = np.array([(v['x'], v['y'], v['z']) for v in points])\n        DATA[sample_idx][data['pickable_object_name']] = points","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:28:58.701602Z","iopub.execute_input":"2024-11-12T10:28:58.702026Z","iopub.status.idle":"2024-11-12T10:29:18.320868Z","shell.execute_reply.started":"2024-11-12T10:28:58.701984Z","shell.execute_reply":"2024-11-12T10:29:18.318185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalise Tomograms","metadata":{}},{"cell_type":"code","source":"tomograms_sum = 0\ntomograms_count = 0\ntomograms_std = 0\nfor k, v in DATA.items():\n    tomograms_sum += v['tomogram'].sum()\n    tomograms_count += v['tomogram'].size\n    tomograms_std += v['tomogram'].std() / len(DATA)\n    \nCONFIG.MEAN = tomograms_sum / tomograms_count\nCONFIG.STD = tomograms_std\n\nprint(f'tomograms µ: {CONFIG.MEAN:.2E}, σ: {CONFIG.STD:.2E}')","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:29:18.322671Z","iopub.execute_input":"2024-11-12T10:29:18.323219Z","iopub.status.idle":"2024-11-12T10:29:19.813084Z","shell.execute_reply.started":"2024-11-12T10:29:18.323157Z","shell.execute_reply":"2024-11-12T10:29:19.811963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalise Tomograms\nfor k, v in DATA.items():\n    DATA[k]['tomogram'] = (v['tomogram'] - CONFIG.MEAN) / CONFIG.STD\n    tomogram = v['tomogram']\n    print(f'{k} tomogram µ: {tomogram.mean():.2f}, σ: {tomogram.std():.2f}')","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:29:19.814662Z","iopub.execute_input":"2024-11-12T10:29:19.815149Z","iopub.status.idle":"2024-11-12T10:29:22.170039Z","shell.execute_reply.started":"2024-11-12T10:29:19.815094Z","shell.execute_reply":"2024-11-12T10:29:22.168968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Class Counts","metadata":{}},{"cell_type":"code","source":"# Get the color cycle from Matplotlib\nCLASS_COLORS = plt.rcParams['axes.prop_cycle'].by_key()['color']","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:29:22.17312Z","iopub.execute_input":"2024-11-12T10:29:22.173508Z","iopub.status.idle":"2024-11-12T10:29:22.178843Z","shell.execute_reply.started":"2024-11-12T10:29:22.173468Z","shell.execute_reply":"2024-11-12T10:29:22.177677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w = 0.10\n\nplt.figure(figsize=(15,8))\nplt.title('Class Count By Label Type')\n\nfor x, (k, v) in enumerate(DATA.items()):\n    for l_idx, (l, c) in enumerate(zip(CONFIG.LABELS, CLASS_COLORS)):\n        plt.bar(x - ((l_idx - 2) * w), v[l].shape[0], w, label=l if x == 0 else None, color=c)\n\nplt.xlabel('Sample')\nplt.ylabel('Class Count')\nxticks = np.array(list(DATA.keys()))\nplt.xticks(np.arange(xticks.size), xticks)\nplt.legend(fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:29:22.180785Z","iopub.execute_input":"2024-11-12T10:29:22.181266Z","iopub.status.idle":"2024-11-12T10:29:22.68062Z","shell.execute_reply.started":"2024-11-12T10:29:22.181214Z","shell.execute_reply":"2024-11-12T10:29:22.679397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,10))\nplt.title('Class Count Global')\n\n# Data\nsizes = dict([(l,0) for l in CONFIG.LABELS])\nfor k, v in DATA.items():\n    for l in CONFIG.LABELS:\n        sizes[l] += v[l].shape[0]\n\n# Create the pie chart\n_, texts, _ = plt.pie(\n    sizes.values(), labels=sizes.keys(),\n    startangle=90, counterclock=False,\n    colors=CLASS_COLORS, autopct='%1.1f%%',\n)\n\n# Manually set the labels\ns = np.sum(list(sizes.values()))\nfor (l, c), text in zip(sizes.items(), texts):\n    text.set_text(f'{l.upper()} | N=[{c}]')\n\nplt.axis('equal')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:39:33.114078Z","iopub.execute_input":"2024-11-12T10:39:33.114622Z","iopub.status.idle":"2024-11-12T10:39:33.402766Z","shell.execute_reply.started":"2024-11-12T10:39:33.114576Z","shell.execute_reply":"2024-11-12T10:39:33.401126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Point Distribution","metadata":{}},{"cell_type":"code","source":"# coordinate distribution\nfig, axes = plt.subplots(3,1, figsize=(15,20))\n# x coordinates\nnames = ('x','y','z')\nlims = (6300, 6300, 1840)\nfor ax_idx, (ax, name, lim) in enumerate(zip(axes, names, lims)):\n    ax.set_title(f'{name.upper()} Coordinates Distribution')\n    for l in CONFIG.LABELS:\n        coordinates = [v[l][:,ax_idx].flatten() for v in DATA.values()]\n        coordinates =  [e for l in coordinates for e in l]\n        ax.set_xlim(0, lim)\n        ax.hist(coordinates, bins=32, label=l, alpha=1.00)\n        ax.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:45:48.781859Z","iopub.execute_input":"2024-11-12T10:45:48.782467Z","iopub.status.idle":"2024-11-12T10:45:50.877231Z","shell.execute_reply.started":"2024-11-12T10:45:48.782405Z","shell.execute_reply":"2024-11-12T10:45:50.875927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Data","metadata":{}},{"cell_type":"code","source":"# Write data as LZMA compressed dictionary\nwith open('data.pkl', 'wb') as f:\n    pickle.dump(DATA, f)","metadata":{"execution":{"iopub.status.busy":"2024-11-12T10:37:21.327869Z","iopub.execute_input":"2024-11-12T10:37:21.329441Z","iopub.status.idle":"2024-11-12T10:37:27.395757Z","shell.execute_reply.started":"2024-11-12T10:37:21.329349Z","shell.execute_reply":"2024-11-12T10:37:27.390003Z"},"trusted":true},"execution_count":null,"outputs":[]}]}