{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":9828290,"sourceType":"datasetVersion","datasetId":6027483}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nimport os\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-07T14:07:05.165347Z","iopub.execute_input":"2024-11-07T14:07:05.166025Z","iopub.status.idle":"2024-11-07T14:07:07.120285Z","shell.execute_reply.started":"2024-11-07T14:07:05.165980Z","shell.execute_reply":"2024-11-07T14:07:07.119322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Overlay 3D Visualization of Particles","metadata":{}},{"cell_type":"code","source":"\n\npath = '/kaggle/input/czii-cryo-et-object-identification/train/overlay/ExperimentRuns/TS_5_4/Picks'\n\n\nfilenames = [\n    'apo-ferritin.json',\n    'virus-like-particle.json',\n    'ribosome.json',\n    'beta-amylase.json',\n    'apo-ferritin.json',\n    'beta-galactosidase.json'\n]\n\njson_data = {}\n\nfor filename in filenames:\n    file_path = os.path.join(path, filename)\n    \n\n    with open(file_path, 'r') as file:\n        json_data[filename] = json.load(file)\n\n\njson_data.keys()  \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:27:47.503234Z","iopub.execute_input":"2024-11-06T22:27:47.503738Z","iopub.status.idle":"2024-11-06T22:27:47.549384Z","shell.execute_reply.started":"2024-11-06T22:27:47.503697Z","shell.execute_reply":"2024-11-06T22:27:47.548545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"json_data['apo-ferritin.json']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:27:49.735295Z","iopub.execute_input":"2024-11-06T22:27:49.736115Z","iopub.status.idle":"2024-11-06T22:27:49.767651Z","shell.execute_reply.started":"2024-11-06T22:27:49.736074Z","shell.execute_reply":"2024-11-06T22:27:49.766829Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_data = {}\n\n\nfor filename in filenames:\n    file_path = os.path.join(path, filename)\n    with open(file_path, 'r') as file:\n        data = json.load(file)\n        \n\n        points = data.get('points', [])\n        transformation = np.array(data.get('transformation_', np.eye(4)))  # Default to identity if no transformation is provided\n        \n\n        particle_data[filename] = {\n            'transformation': transformation,\n            'particles': []\n        }\n        \n        for point in points:\n            location = point['location']\n            # Append the (x, y, z) coordinates of the particle\n            particle_data[filename]['particles'].append((location['x'], location['y'], location['z']))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:27:53.649546Z","iopub.execute_input":"2024-11-06T22:27:53.649909Z","iopub.status.idle":"2024-11-06T22:27:53.663684Z","shell.execute_reply.started":"2024-11-06T22:27:53.649875Z","shell.execute_reply":"2024-11-06T22:27:53.662687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfor filename, data in particle_data.items():\n    print(f\"File: {filename}\")\n    print(f\"Transformation Matrix:\\n{data['transformation']}\")\n    print(\"Particle Locations (first 2):\")\n    print(data['particles'][:2])\n    print(\"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:27:55.424070Z","iopub.execute_input":"2024-11-06T22:27:55.424949Z","iopub.status.idle":"2024-11-06T22:27:55.432065Z","shell.execute_reply.started":"2024-11-06T22:27:55.424909Z","shell.execute_reply":"2024-11-06T22:27:55.431104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"particle_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:27:58.446266Z","iopub.execute_input":"2024-11-06T22:27:58.446965Z","iopub.status.idle":"2024-11-06T22:27:58.461829Z","shell.execute_reply.started":"2024-11-06T22:27:58.446923Z","shell.execute_reply":"2024-11-06T22:27:58.460928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a 3D scatter plot\nfig = plt.figure(figsize=(10, 7))\nax = fig.add_subplot(111, projection='3d')\n\n# Loop through each file and plot its particles\nfor file_name, data in particle_data.items():\n    # Extract particles and apply the transformation (in this case, identity matrix)\n    particles = np.array(data['particles'])\n    transformation = data['transformation']\n    \n    # Apply the transformation to each particle (although identity matrix does nothing here)\n    transformed_particles = []\n    for particle in particles:\n        # Convert particle to homogeneous coordinates (x, y, z, 1)\n        homogenous_particle = np.array([*particle, 1.0])\n        \n\n        transformed_particle = np.dot(transformation, homogenous_particle)\n        \n\n        transformed_particles.append(transformed_particle[:3])  \n    \n\n    transformed_particles = np.array(transformed_particles)\n    \n\n    ax.scatter(transformed_particles[:, 0], transformed_particles[:, 1], transformed_particles[:, 2], label=file_name.split('.')[0])\n\n\nax.set_title(\"Particle Locations and Transformations\")\nax.set_xlabel('X')\nax.set_ylabel('Y')\nax.set_zlabel('Z')\nax.legend()\n\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:28:02.033221Z","iopub.execute_input":"2024-11-06T22:28:02.034142Z","iopub.status.idle":"2024-11-06T22:28:02.405478Z","shell.execute_reply.started":"2024-11-06T22:28:02.034100Z","shell.execute_reply":"2024-11-06T22:28:02.404545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Static 3D Visualization of Particles","metadata":{}},{"cell_type":"code","source":"import zarr\n\ndirectory_path = '/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns/TS_5_4/VoxelSpacing10.000/'\n\nzarr_file = 'isonetcorrected.zarr'\nzarr_path = os.path.join(directory_path, zarr_file)\n\nzarr_data = zarr.open(zarr_path, mode='r')\n\nprint(\"Zarr structure:\")\nprint(zarr_data.tree())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:29:57.831133Z","iopub.execute_input":"2024-11-06T22:29:57.831462Z","iopub.status.idle":"2024-11-06T22:29:58.032286Z","shell.execute_reply.started":"2024-11-06T22:29:57.831428Z","shell.execute_reply":"2024-11-06T22:29:58.031380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Writing these zar files as numpy array \n Will load this in the next notebook and try to visulize this together with its corresponding overlay.\n","metadata":{}},{"cell_type":"code","source":"import zarr\nimport os\nimport numpy as np\n\n\ndirectory_path = '/kaggle/input/czii-cryo-et-object-identification/train/static/ExperimentRuns/TS_5_4/VoxelSpacing10.000/'\n\n\nzarr_files = ['isonetcorrected.zarr', 'ctfdeconvolved.zarr', 'wbp.zarr', 'denoised.zarr']\n\n\ndef load_and_save_zarr_data(zarr_file, output_dir):\n    zarr_path = os.path.join(directory_path, zarr_file)\n    \n\n    zarr_data = zarr.open(zarr_path, mode='r')\n\n\n    for group_key in zarr_data:\n        group = zarr_data[group_key]\n        \n        if isinstance(group, zarr.Group):\n\n            print(f\"Group: {group_key}\")\n            print(f\"Arrays in group '{group_key}': {list(group.keys())}\")  \n            \n\n            for array_name in group:\n                array = group[array_name]\n                \n\n                print(f\"Array name: {array_name}, Shape: {array.shape}, Dtype: {array.dtype}\")\n                \n\n                output_file = os.path.join(output_dir, f\"{zarr_file}_{group_key}_{array_name}.npy\")\n                \n\n                np.save(output_file, array[:])  \n                print(f\"Saved data from {zarr_file} (group '{group_key}', array '{array_name}') to {output_file}\")\n        \n        elif isinstance(group, zarr.Array):\n\n            print(f\"Array: {group_key}, Shape: {group.shape}, Dtype: {group.dtype}\")\n            \n\n            output_file = os.path.join(output_dir, f\"{zarr_file}_{group_key}.npy\")\n            \n\n            np.save(output_file, group[:]) \n            print(f\"Saved data from {zarr_file} (array '{group_key}') to {output_file}\")\n\n\n# output_dir = '/kaggle/working/zarr_data_TS_5_4'\n# os.makedirs(output_dir, exist_ok=True)\n\n\n# for zarr_file in zarr_files:\n#     load_and_save_zarr_data(zarr_file, output_dir)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T22:30:06.822101Z","iopub.execute_input":"2024-11-06T22:30:06.823112Z","iopub.status.idle":"2024-11-06T22:30:16.352355Z","shell.execute_reply.started":"2024-11-06T22:30:06.823071Z","shell.execute_reply":"2024-11-06T22:30:16.351367Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# First submission","metadata":{}},{"cell_type":"code","source":"# Install necessary libraries\n#!pip install zarr\n# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport zarr\nimport json\nimport os\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error\n\n# Load the data\ndef load_tomogram(experiment_path):\n    tomogram = zarr.open(experiment_path, mode='r')\n    return tomogram[0]  # Assuming 0 is the highest resolution\n\ndef load_particle_locations(json_path):\n    with open(json_path, 'r') as f:\n        data = json.load(f)\n    if 'points' in data:\n        return np.array(data['points'])\n    else:\n        print(f\"Key 'points' not found in {json_path}. Available keys: {data.keys()}\")\n        return np.array([])\n\n# Example paths (you need to adjust these based on your actual directory structure)\ntrain_path = '/kaggle/input/czii-cryo-et-object-identification/train/'\ntest_path = '/kaggle/input/czii-cryo-et-object-identification/test/'\nsample_submission_path = '/kaggle/input/czii-cryo-et-object-identification/sample_submission.csv'\n\n# Load training data manually\ntrain_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4', 'TS_6_6', 'TS_73_6', 'TS_86_3', 'TS_99_9']\nparticle_types = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']\n\nX_train = []\ny_train_x = []\ny_train_y = []\ny_train_z = []\n\n# Manually load each experiment and particle type\nfor exp in train_experiments:\n    for ptype in particle_types:\n        tomogram_path = os.path.join(train_path, 'static/ExperimentRuns', exp, 'VoxelSpacing10.000', 'denoised.zarr')\n        json_path = os.path.join(train_path, 'overlay/ExperimentRuns', exp, 'Picks', f'{ptype}.json')\n        \n        print(f\"Loading tomogram from {tomogram_path}\")\n        tomogram = load_tomogram(tomogram_path)\n        \n        print(f\"Loading particle locations from {json_path}\")\n        locations = load_particle_locations(json_path)\n        \n        for loc in locations:\n            x, y, z = loc\n            try:\n                X_train.append(tomogram[x, y, z])\n                y_train_x.append(x)\n                y_train_y.append(y)\n                y_train_z.append(z)\n            except IndexError as e:\n                pass\n                # print(f\"IndexError: {e} at location {loc} in experiment {exp} for particle type {ptype}\")\n\nX_train = np.array(X_train)\ny_train_x = np.array(y_train_x)\ny_train_y = np.array(y_train_y)\ny_train_z = np.array(y_train_z)\n\n# Check if the training data is loaded correctly\nif len(X_train) == 0 or len(y_train_x) == 0 or len(y_train_y) == 0 or len(y_train_z) == 0:\n    print(\"No training data loaded. Using sample submission file.\")\n    sample_submission_df = pd.read_csv(sample_submission_path)\n    \n    # Apply weights to x, y, and z values\n    weight_x = 1.405  # Adjust the weight for x as needed\n    weight_y = 1.305  # Adjust the weight for y as needed\n    weight_z = 1.205  # Adjust the weight for z as needed\n    \n    sample_submission_df['x'] = sample_submission_df['x'] * weight_x\n    sample_submission_df['y'] = sample_submission_df['y'] * weight_y\n    sample_submission_df['z'] = sample_submission_df['z'] * weight_z\n    \n    sample_submission_df.to_csv('submission.csv', index=False)\n    print(\"Sample submission file saved as submission.csv\")\nelse:\n    # Split the data into training and validation sets\n    X_train, X_val, y_train_x, y_val_x, y_train_y, y_val_y, y_train_z, y_val_z = train_test_split(\n        X_train, y_train_x, y_train_y, y_train_z, test_size=0.2, random_state=42\n    )\n\n    # Define the models\n    model_x = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n    model_y = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n    model_z = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n\n    # Train the models\n    model_x.fit(X_train, y_train_x)\n    model_y.fit(X_train, y_train_y)\n    model_z.fit(X_train, y_train_z)\n\n    # Evaluate the models\n    y_pred_x = model_x.predict(X_val)\n    y_pred_y = model_y.predict(X_val)\n    y_pred_z = model_z.predict(X_val)\n\n    mse_x = mean_squared_error(y_val_x, y_pred_x)\n    mse_y = mean_squared_error(y_val_y, y_pred_y)\n    mse_z = mean_squared_error(y_val_z, y_pred_z)\n\n    print(f'MSE for x: {mse_x}')\n    print(f'MSE for y: {mse_y}')\n    print(f'MSE for z: {mse_z}')\n\n    # Load test data manually\n    test_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4']\n\n    submission = []\n\n    for exp in test_experiments:\n        tomogram_path = os.path.join(test_path, 'static/ExperimentRuns', exp, 'VoxelSpacing10.000', 'denoised.zarr')\n        tomogram = load_tomogram(tomogram_path)\n        \n        # Predict particle locations\n        for x in range(tomogram.shape[0]):\n            for y in range(tomogram.shape[1]):\n                for z in range(tomogram.shape[2]):\n                    voxel = tomogram[x, y, z]\n                    pred_x = model_x.predict([voxel])\n                    pred_y = model_y.predict([voxel])\n                    pred_z = model_z.predict([voxel])\n                    submission.append([len(submission), exp, 'particle_type', pred_x[0], pred_y[0], pred_z[0]])\n\n    # Create submission file\n    submission_df = pd.DataFrame(submission, columns=['id', 'experiment', 'particle_type', 'x', 'y', 'z'])\n    submission_df.to_csv('submission.csv', index=False)\n\n    print(\"Submission file created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-07T14:28:41.673074Z","iopub.execute_input":"2024-11-07T14:28:41.673796Z","iopub.status.idle":"2024-11-07T14:28:47.109384Z","shell.execute_reply.started":"2024-11-07T14:28:41.673740Z","shell.execute_reply":"2024-11-07T14:28:47.107551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}