{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":214207666,"sourceType":"kernelVersion"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-23T15:04:31.460550Z","iopub.execute_input":"2024-12-23T15:04:31.460777Z","iopub.status.idle":"2024-12-23T15:04:34.057734Z","shell.execute_reply.started":"2024-12-23T15:04:31.460753Z","shell.execute_reply":"2024-12-23T15:04:34.056640Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Initial Configuration","metadata":{}},{"cell_type":"code","source":"pip install zarr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:29:45.090131Z","iopub.execute_input":"2024-12-23T15:29:45.090852Z","iopub.status.idle":"2024-12-23T15:30:32.424494Z","shell.execute_reply.started":"2024-12-23T15:29:45.090816Z","shell.execute_reply":"2024-12-23T15:30:32.423445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport zarr\nimport json\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nfrom sklearn.metrics import fbeta_score\nfrom sklearn.model_selection import train_test_split\n\n# Paths to data\nTRAIN_BASE = '/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns'\nTEST_BASE = '/kaggle/input/czii-cryo-et-object-identification/test/static/ExperimentRuns'\nOVERLAY_BASE = '/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns'\n\n# Radio of particles\nPARTICLE_RADIUS = {\n'ribosome': 150.0,\n'virus-like-particle': 135.0,\n'apo-ferritin': 60.0,\n'beta-galactosidase': 90.0,\n'thyroglobulin': 130.0,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:30:41.070145Z","iopub.execute_input":"2024-12-23T15:30:41.070478Z","iopub.status.idle":"2024-12-23T15:30:52.659004Z","shell.execute_reply.started":"2024-12-23T15:30:41.070450Z","shell.execute_reply":"2024-12-23T15:30:52.658098Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"code","source":"import os\n\ndef get_training_experiments():\n    return os.listdir(TRAIN_BASE)\n    \ndef get_test_experiments():\n    return os.listdir(TEST_BASE)\n    \ndef get_experiment_path(experiment, type='denoised', test=False):\n    base = TEST_BASE if test else TRAIN_BASE\n    return base + '/' + experiment + '/VoxelSpacing10.000/' + type + '.zarr'\n    \ndef open_experiment(experiment, test=False):    \n    return zarr.open(get_experiment_path(experiment, test=test), mode='r')\n  \ndef get_label_path(experiment, particle_type):\n    return OVERLAY_BASE + '/' + experiment + '/Picks/' + particle_type + '.json'\n    \ndef open_labels_file(experiment, particle_type):\n    path = get_label_path(experiment, particle_type)\n    with open(path, 'r') as f:\n        data = json.load(f)\n    return data\n    \nimport os\n\ndef get_training_experiments():\n    return os.listdir(TRAIN_BASE)\n    \ndef get_test_experiments():\n    return os.listdir(TEST_BASE)\n    \ndef get_experiment_path(experiment, type='denoised', test=False):\n    base = TEST_BASE if test else TRAIN_BASE\n    return base + '/' + experiment + '/VoxelSpacing10.000/' + type + '.zarr'\n    \ndef open_experiment(experiment, test=False):    \n    return zarr.open(get_experiment_path(experiment, test=test), mode='r')\n    \ndef get_label_path(experiment, particle_type):\n    return OVERLAY_BASE + '/' + experiment + '/Picks/' + particle_type + '.json'\n    \ndef open_labels_file(experiment, particle_type):\n    path = get_label_path(experiment, particle_type)\n    with open(path, 'r') as f:\n        data = json.load(f)\n    return data\n    \ndef get_labels_with_radius(experiment, scaling_factor=10):\n    labels = {}\n    for particle_type in PARTICLE_RADIUS.keys():\n        raw_data = open_labels_file(experiment, particle_type)\n        labels[particle_type] = []\n        for item in raw_data['points']:\n            # Verificamos que los datos de cada punto tengan la estructura correcta\n            x = abs(item['location']['x'] // scaling_factor)\n            y = abs(item['location']['y'] // scaling_factor)\n            z = abs(item['location']['z'] // scaling_factor)\n            radius = abs(PARTICLE_RADIUS[particle_type] // scaling_factor)\n        # Añadimos las coordenadas (x, y, z) y el radio\n        labels[particle_type].append((x, y, z, radius))\n    return labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:30:57.589848Z","iopub.execute_input":"2024-12-23T15:30:57.591000Z","iopub.status.idle":"2024-12-23T15:30:57.601752Z","shell.execute_reply.started":"2024-12-23T15:30:57.590964Z","shell.execute_reply":"2024-12-23T15:30:57.601050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zarr\n\ndef inspect_zarr(experiment):\n    data_path = get_experiment_path(experiment)\n    z = zarr.open(data_path, mode='r')\n    print(z.tree())\n    return z\n\ninspect_zarr(\"TS_86_3\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:02.506359Z","iopub.execute_input":"2024-12-23T15:31:02.507017Z","iopub.status.idle":"2024-12-23T15:31:02.609476Z","shell.execute_reply.started":"2024-12-23T15:31:02.506971Z","shell.execute_reply":"2024-12-23T15:31:02.608727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\n\ndef preprocess_data(experiment):\n    # Open the tomogram data and labels for a specific experiment\n    tomogram = open_experiment(experiment)\n    tomogram = np.expand_dims(tomogram, axis=-1)  # Add an extra dimension for channel (grayscale)\n\n    labels = get_labels_with_radius(experiment)  # Get the particle labels\n    print(f\"Data shape for {experiment}: {tomogram.shape}\")\n    print(f\"Labels shape for {experiment}: {len(labels)}\")\n\n    return tomogram, labels\n\ndef create_training_dataset():\n    # Get the list of training experiments\n    training_experiments = get_training_experiments()\n\n    train_data = []\n    train_labels = []\n\n    for experiment in training_experiments:\n        print(f\"Processing the experiment: {experiment}\")\n\n        # Preprocess the data and labels\n        data, labels = preprocess_data(experiment)\n        print(f\"Shape of the data after preprocess: {data.shape}\")\n\n        # Add the processed data\n        train_data.append(data)\n\n        # Prepare the labels in proper format\n        label_dict = {}\n        for particle_type, coordinates in labels.items():\n            label_dict[particle_type] = coordinates\n        train_labels.append(label_dict)\n\n    # Stack the training data in an array\n    train_data = np.stack(train_data, axis=0)\n\n    # Ensure that the labels have a consistent shape (with a maximum of 250 particles)\n    train_labels = pad_labels(train_labels, max_particles=250)\n\n    print(f\"Final shape of train_data: {train_data.shape}\")\n    print(f\"Final shape of train_labels: {train_labels.shape}\")\n\n    return train_data, train_labels\n\ndef pad_labels(labels, max_particles=250):\n    # Create an array of zeros to fill the labels with a maximum of 250 particles\n    padded_labels = np.zeros((len(labels), max_particles, 3))  # Shape (num_samples, max_particles, 3)\n\n    for i, label_dict in enumerate(labels):\n        print(f\"Checking labels for experiment {i}...\")\n\n        for particle_type, particle_data in label_dict.items():\n            particles_coordinates = [(item[0], item[1], item[2]) for item in particle_data]\n\n            num_particles = len(particles_coordinates)\n            for j in range(min(num_particles, max_particles)):\n                # Fill particle coordinates\n                padded_labels[i, j, 0] = particles_coordinates[j][0]  # x\n                padded_labels[i, j, 1] = particles_coordinates[j][1]  # y\n                padded_labels[i, j, 2] = particles_coordinates[j][2]  # z\n\n    return padded_labels\n\ndef prepare_train_val_split(train_data, train_labels, test_size=0.2):\n    # Split the data into training and validation sets\n    return train_test_split(train_data, train_labels, test_size=test_size, random_state=42)\n\n# Create the training dataset\ntrain_data, train_labels = create_training_dataset()\nprint(f\"Train data before split: {train_data.shape}\")\nprint(f\"Train labels before split: {train_labels.shape}\")\n\n# Split into training and validation set\nX_train, X_val, y_train, y_val = prepare_train_val_split(train_data, train_labels)\n\nprint(f\"X_train: {X_train.shape}, y_train: {y_train.shape}\")\nprint(f\"X_val: {X_val.shape}, y_val: {y_val.shape}\")\n\n# Expand dimensions of X_train and X_val if needed\nprint(f\"Shape of X_train before expanding: {X_train.shape}\")\nX_train = np.expand_dims(X_train, axis=-1)  # Expand dimensions for channel\nprint(f\"Shape of X_train after expanding: {X_train.shape}\")\n\nX_val = np.expand_dims(X_val, axis=-1)  # Expand dimensions for channel\nprint(f\"Shape of X_val after expanding: {X_val.shape}\")\n\n# Final shape of data\nprint(f\"Final shape of X_train: {X_train.shape}\")\nprint(f\"Final shape of X_val: {X_val.shape}\")\nprint(f\"Final shape of y_train: {y_train.shape}\")\nprint(f\"Final shape of y_val: {y_val.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:05.424135Z","iopub.execute_input":"2024-12-23T15:31:05.424432Z","iopub.status.idle":"2024-12-23T15:31:05.709658Z","shell.execute_reply.started":"2024-12-23T15:31:05.424406Z","shell.execute_reply":"2024-12-23T15:31:05.708867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(f\"Size of X_train after expansion: {X_train.shape}\")\nprint(f\"Size of X_val after expansion: {X_val.shape}\")\nprint(f\"Size of y_train: {y_train.shape}\")\nprint(f\"Size of y_val: {y_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:13.428600Z","iopub.execute_input":"2024-12-23T15:31:13.428914Z","iopub.status.idle":"2024-12-23T15:31:13.433587Z","shell.execute_reply.started":"2024-12-23T15:31:13.428888Z","shell.execute_reply":"2024-12-23T15:31:13.432727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Size of X_train: {X_train.shape}\")\nprint(f\"Size of X_val: {X_val.shape}\")\nprint(f\"Size of y_train: {y_train.shape}\")\nprint(f\"Size of y_val: {y_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:15.947170Z","iopub.execute_input":"2024-12-23T15:31:15.947492Z","iopub.status.idle":"2024-12-23T15:31:15.952629Z","shell.execute_reply.started":"2024-12-23T15:31:15.947464Z","shell.execute_reply":"2024-12-23T15:31:15.951729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Development","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers, models\nimport numpy as np\n\ndef create_3d_cnn_model(input_shape, output_shape):\n    model = models.Sequential()\n    \n   \n    model.add(layers.Conv3D(16, (5, 5, 5), activation='relu', input_shape=input_shape, padding='same'))  \n    model.add(layers.BatchNormalization())  \n    model.add(layers.MaxPooling3D(pool_size=(2, 2, 2), padding='same'))  \n    model.add(layers.Dropout(0.3))  # Regularización (Dropout)\n\n   \n    model.add(layers.Conv3D(32, (5, 5, 5), activation='relu', padding='same')) \n    model.add(layers.BatchNormalization())  \n    model.add(layers.MaxPooling3D(pool_size=(2, 2, 2), padding='same')) \n    model.add(layers.Dropout(0.3)) \n\n   \n    model.add(layers.Conv3D(64, (5, 5, 5), activation='relu', padding='same'))  \n    model.add(layers.BatchNormalization()) \n    model.add(layers.Dropout(0.3))  \n\n    \n    model.add(layers.GlobalAveragePooling3D()) \n    model.add(layers.Dense(128, activation='relu'))  \n    model.add(layers.Dropout(0.5)) \n\n   \n    model.add(layers.Dense(np.prod(output_shape), activation='linear'))  \n    model.add(layers.Reshape(output_shape))  \n    \n    return model\n\n\ninput_shape = (46, 158, 158, 1)  \noutput_shape = (250, 3) \n\n# Crear el modelo\nmodel = create_3d_cnn_model(input_shape, output_shape)\n\n# Compilar el modelo\nmodel.compile(optimizer='adam', loss='mean_squared_error', metrics=['mae'])\n\n# Resumen del modelo\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:19.173104Z","iopub.execute_input":"2024-12-23T15:31:19.173986Z","iopub.status.idle":"2024-12-23T15:31:20.109785Z","shell.execute_reply.started":"2024-12-23T15:31:19.173948Z","shell.execute_reply":"2024-12-23T15:31:20.108980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"Shape of X_val: {X_val.shape}\")\nprint(f\"Shape of y_train: {y_train.shape}\")\nprint(f\"Shape of y_val: {y_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:28.640602Z","iopub.execute_input":"2024-12-23T15:31:28.640969Z","iopub.status.idle":"2024-12-23T15:31:28.646399Z","shell.execute_reply.started":"2024-12-23T15:31:28.640906Z","shell.execute_reply":"2024-12-23T15:31:28.645424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX_train = np.array(X_train, dtype='float32')\nX_val = np.array(X_val, dtype='float32')\ny_train = np.array(y_train, dtype='float32')\ny_val = np.array(y_val, dtype='float32')\n\nprint(X_train.dtype) \nprint(y_train.dtype) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:31.609392Z","iopub.execute_input":"2024-12-23T15:31:31.610198Z","iopub.status.idle":"2024-12-23T15:31:31.615346Z","shell.execute_reply.started":"2024-12-23T15:31:31.610165Z","shell.execute_reply":"2024-12-23T15:31:31.614392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TRAINING MODEL","metadata":{}},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=80, validation_data=(X_val, y_val),\nbatch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:31:38.101298Z","iopub.execute_input":"2024-12-23T15:31:38.101649Z","iopub.status.idle":"2024-12-23T15:31:51.223742Z","shell.execute_reply.started":"2024-12-23T15:31:38.101602Z","shell.execute_reply":"2024-12-23T15:31:51.223122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save the template in H5 format","metadata":{}},{"cell_type":"code","source":"model.save('/kaggle/working/CryoET-MLAnnotator.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T20:17:08.722602Z","iopub.execute_input":"2024-12-21T20:17:08.723235Z","iopub.status.idle":"2024-12-21T20:17:08.780930Z","shell.execute_reply.started":"2024-12-21T20:17:08.723198Z","shell.execute_reply":"2024-12-21T20:17:08.779967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.legend()\nplt.title('Loss vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.show()\n\nplt.plot(history.history['mae'], label='Train MAE')\nplt.plot(history.history['val_mae'], label='Validation MAE')\nplt.legend()\nplt.title('MAE vs Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('MAE')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:00.830296Z","iopub.execute_input":"2024-12-23T15:32:00.830635Z","iopub.status.idle":"2024-12-23T15:32:01.342892Z","shell.execute_reply.started":"2024-12-23T15:32:00.830604Z","shell.execute_reply":"2024-12-23T15:32:01.342060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nhistory_df = pd.DataFrame({\n'Epoch': range(1, len(history.history['loss']) + 1),\n'Train Loss': history.history['loss'],\n'Validation Loss': history.history['val_loss'],\n'Train MAE': history.history['mae'],\n'Validation MAE': history.history['val_mae']\n})\n\nprint(history_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:06.521307Z","iopub.execute_input":"2024-12-23T15:32:06.521788Z","iopub.status.idle":"2024-12-23T15:32:06.542293Z","shell.execute_reply.started":"2024-12-23T15:32:06.521739Z","shell.execute_reply":"2024-12-23T15:32:06.541296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"import os\ntest_experiments = os.listdir(TEST_BASE)\nprint(f\"Test experiments available: {test_experiments}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:10.203180Z","iopub.execute_input":"2024-12-23T15:32:10.203519Z","iopub.status.idle":"2024-12-23T15:32:10.208873Z","shell.execute_reply.started":"2024-12-23T15:32:10.203489Z","shell.execute_reply":"2024-12-23T15:32:10.208050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zarr\n\nexperiment_path = f\"{TEST_BASE}/TS_6_4/VoxelSpacing10.000/denoised.zarr\"\ntest_data = zarr.open(experiment_path, mode='r')\n\n\nprint(test_data.tree())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:12.919849Z","iopub.execute_input":"2024-12-23T15:32:12.920209Z","iopub.status.idle":"2024-12-23T15:32:12.953489Z","shell.execute_reply.started":"2024-12-23T15:32:12.920180Z","shell.execute_reply":"2024-12-23T15:32:12.952807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"experiment_name = \"TS_6_4\" \nexperiment_dir = f\"{TEST_BASE}/{experiment_name}\"\nprint(f\"Contents of {experiment_name}:\")\nprint(os.listdir(experiment_dir))\nvoxel_dir = f\"{experiment_dir}/VoxelSpacing10.000\"\nprint(f\"Contents of {voxel_dir}:\")\nprint(os.listdir(voxel_dir))\nzarr_path = f\"{voxel_dir}/denoised.zarr\"\ntry:\n    test_data = zarr.open(zarr_path, mode='r')\n    print(f\"Zarr file structure: ({experiment_name}):\")\n    print(test_data.tree())\nexcept Exception as e:\n    print(f\"Failed to open Zarr file: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:16.158114Z","iopub.execute_input":"2024-12-23T15:32:16.159045Z","iopub.status.idle":"2024-12-23T15:32:16.173858Z","shell.execute_reply.started":"2024-12-23T15:32:16.158997Z","shell.execute_reply":"2024-12-23T15:32:16.172898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"level = 2 # Change to 0 or 1 if you need other resolutions\ndata = test_data[str(level)]\n\ndata_array = np.array(data)\n\ndata_array = np.expand_dims(data_array, axis=-1)\nprint(f\"Final data shape: {data_array.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:19.382055Z","iopub.execute_input":"2024-12-23T15:32:19.382397Z","iopub.status.idle":"2024-12-23T15:32:19.458843Z","shell.execute_reply.started":"2024-12-23T15:32:19.382368Z","shell.execute_reply":"2024-12-23T15:32:19.457966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"experiment_name = \"TS_6_4\" # Please change to 'TS_5_4' or 'TS_69_2' for other experiments\nX_test, y_test = preprocess_data(experiment_name) \nX_test = np.expand_dims(X_test, axis=-1) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:22.472714Z","iopub.execute_input":"2024-12-23T15:32:22.473082Z","iopub.status.idle":"2024-12-23T15:32:22.491443Z","shell.execute_reply.started":"2024-12-23T15:32:22.473050Z","shell.execute_reply":"2024-12-23T15:32:22.490752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nif isinstance(X_test, list): \n    X_test = np.array(X_test, dtype=np.float32)\nelif X_test.dtype not in [np.float32, np.float64]: \n    X_test = X_test.astype(np.float32)\n    print(type(X_test)) \n    print(X_test.dtype) \n    print(X_test.shape) \n    if len(X_test.shape) == 4: \n        X_test = np.expand_dims(X_test, axis=-1)\n        print(f\"Form of X_test: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:25.142317Z","iopub.execute_input":"2024-12-23T15:32:25.142996Z","iopub.status.idle":"2024-12-23T15:32:25.148492Z","shell.execute_reply.started":"2024-12-23T15:32:25.142962Z","shell.execute_reply":"2024-12-23T15:32:25.147530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport zarr\n\ndata = zarr.open(f\"{TEST_BASE}/TS_6_4/VoxelSpacing10.000/denoised.zarr\",\nmode='r')['2']\n\ndata = np.array(data)\n\nX_test = np.expand_dims(data, axis=-1).astype(np.float32)\nprint(f\"Shape of X_test after correction: {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:28.608769Z","iopub.execute_input":"2024-12-23T15:32:28.609620Z","iopub.status.idle":"2024-12-23T15:32:28.661851Z","shell.execute_reply.started":"2024-12-23T15:32:28.609584Z","shell.execute_reply":"2024-12-23T15:32:28.661068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = []\nfor experiment in ['TS_5_4', 'TS_6_4', 'TS_69_2']:\n    data = zarr.open(f\"{TEST_BASE}/{experiment}/VoxelSpacing10.000/denoised.zarr\",mode='r')['2']\n    data = np.expand_dims(np.array(data), axis=-1).astype(np.float32) \n    X_test.append(data)\n\nX_test = np.stack(X_test)\nprint(f\"Combined X_test form: {X_test.shape}\")\ny_pred = model.predict(X_test)\nprint(f\"Form of predictions: {y_pred.shape}\")\nprint(f\"Predictions: {y_pred}\")\n\npredicted_classes = np.argmax(y_pred, axis=-1)\n\n# print(\"Predicted Classes (Final Labels):\")\n# print(predicted_classes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:15.290224Z","iopub.execute_input":"2024-12-23T15:33:15.290560Z","iopub.status.idle":"2024-12-23T15:33:15.540727Z","shell.execute_reply.started":"2024-12-23T15:33:15.290530Z","shell.execute_reply":"2024-12-23T15:33:15.539879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Predictions for the first tomogram:\")\nprint(y_pred[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:32:38.128260Z","iopub.execute_input":"2024-12-23T15:32:38.128917Z","iopub.status.idle":"2024-12-23T15:32:38.139008Z","shell.execute_reply.started":"2024-12-23T15:32:38.128885Z","shell.execute_reply":"2024-12-23T15:32:38.138060Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the real labels","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n# Assuming 'predicted_classes' is a 3x250 array containing your final class predictions\n\n# Now print the class distribution for each instance\nprint(\"Class distribution per instance:\")\nfor i in range(predicted_classes.shape[0]):\n    print(f\"Instance {i+1}:\")\n    # Get the unique class labels and their counts for the current instance\n    unique, counts = np.unique(predicted_classes[i], return_counts=True)\n    # Print the class counts for the current instance\n    for label, count in zip(unique, counts):\n        print(f\"Class {label}: {count} samples\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:23.419542Z","iopub.execute_input":"2024-12-23T15:33:23.420240Z","iopub.status.idle":"2024-12-23T15:33:23.426257Z","shell.execute_reply.started":"2024-12-23T15:33:23.420206Z","shell.execute_reply":"2024-12-23T15:33:23.425279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_true = [] \nfor experiment in ['TS_6_4', 'TS_5_4', 'TS_69_2']:\n    labels = get_labels_with_radius(experiment) \n    y_true.append(labels)\n    print(y_true)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:30.639304Z","iopub.execute_input":"2024-12-23T15:33:30.639701Z","iopub.status.idle":"2024-12-23T15:33:30.658589Z","shell.execute_reply.started":"2024-12-23T15:33:30.639655Z","shell.execute_reply":"2024-12-23T15:33:30.657757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Type of y_true: {type(y_true)}\")\nprint(f\"Type of y_pred: {type(y_pred)}\")\nprint(f\"First value of y_true: {y_true[0]}\")\nprint(f\"First value of y_pred: {y_pred[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:42.696375Z","iopub.execute_input":"2024-12-23T15:33:42.697277Z","iopub.status.idle":"2024-12-23T15:33:42.708781Z","shell.execute_reply.started":"2024-12-23T15:33:42.697241Z","shell.execute_reply":"2024-12-23T15:33:42.707903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Binarization of predictions and labels","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import fbeta_score\n\ndef binarize_predictions(y_true, y_pred, tolerance=0.5):\n    binary_true = []\n    binary_pred = []\n    \n    # Iterar sobre cada experimento\n    for true_data, pred_data in zip(y_true, y_pred):\n        # Inicializamos listas para las coordenadas binarizadas\n        correct_predictions_true = []\n        correct_predictions_pred = []\n        \n        # Iterar sobre cada tipo de partícula en `y_true`\n        for particle_type in true_data:\n            # Extraer las coordenadas (x, y, z) de las partículas en `y_true`\n            true_coords = np.array([coord[:3] for coord in true_data[particle_type]], dtype=np.float32)\n            \n            # Obtener las coordenadas predichas para la partícula (suponiendo que están en el mismo orden)\n            pred_coords = np.array(pred_data[:len(true_coords)], dtype=np.float32)\n            \n            # Inicializar las listas de resultados binarizados\n            correct_predictions_true = [0] * len(true_coords)\n            correct_predictions_pred = [0] * len(true_coords)\n            \n            # Comparar las coordenadas verdaderas y las predicciones\n            for i, true in enumerate(true_coords):\n                for j, pred in enumerate(pred_coords):\n                    # Calcular la distancia euclidiana\n                    distance = np.linalg.norm(true - pred)\n                    if distance <= tolerance:\n                        correct_predictions_true[i] = 1  # Predicción correcta para y_true\n                        correct_predictions_pred[i] = 1  # Predicción correcta para y_pred\n            \n            # Agregar las coordenadas binarizadas para esta partícula\n            binary_true.append(correct_predictions_true)\n            binary_pred.append(correct_predictions_pred)\n    \n    # Convertir a np.array, que tendrá estructuras de listas con longitudes variables\n    binary_true = np.array(binary_true, dtype=object)\n    binary_pred = np.array(binary_pred, dtype=object)\n    \n    return binary_true, binary_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:51.665983Z","iopub.execute_input":"2024-12-23T15:33:51.666876Z","iopub.status.idle":"2024-12-23T15:33:51.673974Z","shell.execute_reply.started":"2024-12-23T15:33:51.666840Z","shell.execute_reply":"2024-12-23T15:33:51.673256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Aplicar la función de binarización\ny_true_binary, y_pred_binary = binarize_predictions(y_true, y_pred)\n\n# Aplanar las matrices para que tengan la misma forma\ny_true_binary_flat = np.concatenate(y_true_binary)\ny_pred_binary_flat = np.concatenate(y_pred_binary)\n\n# Verificar las formas de los resultados binarizados\nprint(f\"Shape of y_true_binary_flat: {y_true_binary_flat.shape}\")\nprint(f\"Shape of y_pred_binary_flat: {y_pred_binary_flat.shape}\")\nprint(f\"y_true: {y_true}\")\nprint(f\"y_pred: {y_pred}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:33:58.092346Z","iopub.execute_input":"2024-12-23T15:33:58.093516Z","iopub.status.idle":"2024-12-23T15:33:58.100652Z","shell.execute_reply.started":"2024-12-23T15:33:58.093470Z","shell.execute_reply":"2024-12-23T15:33:58.099781Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calculate the F-beta metric","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import fbeta_score\nimport numpy as np\n\n# Asegúrate de que y_true_binary y y_pred_binary sean arrays de una sola dimensión\ny_true_binary_flat = np.array(y_true_binary, dtype=int)\ny_pred_binary_flat = np.array(y_pred_binary, dtype=int)\n\n# Calcular el F-beta score\nfbeta = fbeta_score(y_true_binary_flat, y_pred_binary_flat, beta=4, average=\"micro\")\nprint(f\"F-beta metric with beta=4: {fbeta}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T20:19:33.274311Z","iopub.execute_input":"2024-12-21T20:19:33.275350Z","iopub.status.idle":"2024-12-21T20:19:33.282059Z","shell.execute_reply.started":"2024-12-21T20:19:33.275315Z","shell.execute_reply":"2024-12-21T20:19:33.281211Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generating the Submission File","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\ny_pred = np.random.rand(3, 6, 6, 3)  \n\n\nexperiments = ['TS_6_4', 'TS_5_4', 'TS_69_2']\n\n\nparticle_types = [\n    'beta-amylase', 'virus-like-particle', 'apo-ferritin',\n    'thyroglobulin', 'beta-galactosidase', 'ribosome',\n]\n\n\npredictions_list = []\nid_counter = 0\n\n\nfor experiment_idx, experiment in enumerate(experiments):\n    for particle_idx, particle_type in enumerate(particle_types):\n      \n        coords = y_pred[experiment_idx, particle_idx]\n        \n       \n        for coord in coords:\n            x, y, z = coord\n            predictions_list.append([id_counter, experiment, particle_type, x, y, z])\n            id_counter += 1\n\npredictions_df = pd.DataFrame(predictions_list, columns=[\"id\", \"experiment\", \"particle_type\", \"x\", \"y\", \"z\"])\n\n\npredictions_df.to_csv('/kaggle/working/submission.csv', index=False)\n\n\nprint(predictions_df.head())\n ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:34:03.963523Z","iopub.execute_input":"2024-12-23T15:34:03.963857Z","iopub.status.idle":"2024-12-23T15:34:03.978061Z","shell.execute_reply.started":"2024-12-23T15:34:03.963827Z","shell.execute_reply":"2024-12-23T15:34:03.977144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('/kaggle/working/submission.csv')\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:34:09.842650Z","iopub.execute_input":"2024-12-23T15:34:09.843147Z","iopub.status.idle":"2024-12-23T15:34:09.864392Z","shell.execute_reply.started":"2024-12-23T15:34:09.843114Z","shell.execute_reply":"2024-12-23T15:34:09.863415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport plotly.graph_objects as go\n\n# Cargar el archivo CSV de las predicciones\ndf = pd.read_csv('/kaggle/working/submission.csv')\n\n# Filtrar las predicciones para un experimento específico\nexperiment = 'TS_6_4'\npredictions = df[df['experiment'] == experiment]\n\n# Crear una figura 3D\nfig = go.Figure()\n\n# Iterar sobre las predicciones y añadir puntos para cada partícula\nfor idx, row in predictions.iterrows():\n    fig.add_trace(go.Scatter3d(\n        x=[row['x']],\n        y=[row['y']],\n        z=[row['z']],\n        mode='markers',\n        marker=dict(size=5, color='red'),\n        name=row['particle_type']\n    ))\n\n# Mostrar la visualización\nfig.update_layout(\n    title=f'Predicciones de Partículas en el Tomograma ({experiment})',\n    scene=dict(\n        xaxis_title='X',\n        yaxis_title='Y',\n        zaxis_title='Z'\n    ),\n    showlegend=True\n)\n\nfig.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:34:13.566105Z","iopub.execute_input":"2024-12-23T15:34:13.566772Z","iopub.status.idle":"2024-12-23T15:34:14.168295Z","shell.execute_reply.started":"2024-12-23T15:34:13.566739Z","shell.execute_reply":"2024-12-23T15:34:14.167451Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Training is progressing well, as the loss and error metrics are decreasing. However, it is important to monitor these metrics as training progresses to ensure that the model is not overfitting. If the validation loss starts to increase while the training loss continues to decrease, that could be a sign of overfitting.","metadata":{}}]}