{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install necessary libraries\n#!pip install zarr","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport zarr\nimport json\nimport os\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error\n\n# Load the data\ndef load_tomogram(experiment_path):\n    tomogram = zarr.open(experiment_path, mode='r')\n    return tomogram[0]  # Assuming 0 is the highest resolution\n\ndef load_particle_locations(json_path):\n    with open(json_path, 'r') as f:\n        data = json.load(f)\n    if 'points' in data:\n        return np.array(data['points'])\n    else:\n        print(f\"Key 'points' not found in {json_path}. Available keys: {data.keys()}\")\n        return np.array([])\n\n# Example paths (you need to adjust these based on your actual directory structure)\ntrain_path = '/kaggle/input/czii-cryo-et-object-identification/train/'\ntest_path = '/kaggle/input/czii-cryo-et-object-identification/test/'\nsample_submission_path = '/kaggle/input/czii-cryo-et-object-identification/sample_submission.csv'\n\n# Load training data manually\ntrain_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4', 'TS_6_6', 'TS_73_6', 'TS_86_3', 'TS_99_9']\nparticle_types = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']\n\nX_train = []\ny_train_x = []\ny_train_y = []\ny_train_z = []\n\n# Manually load each experiment and particle type\nfor exp in train_experiments:\n    for ptype in particle_types:\n        tomogram_path = os.path.join(train_path, 'static/ExperimentRuns', exp, 'VoxelSpacing10.000', 'denoised.zarr')\n        json_path = os.path.join(train_path, 'overlay/ExperimentRuns', exp, 'Picks', f'{ptype}.json')\n        \n        print(f\"Loading tomogram from {tomogram_path}\")\n        tomogram = load_tomogram(tomogram_path)\n        \n        print(f\"Loading particle locations from {json_path}\")\n        locations = load_particle_locations(json_path)\n        \n        for loc in locations:\n            x, y, z = loc\n            try:\n                X_train.append(tomogram[x, y, z])\n                y_train_x.append(x)\n                y_train_y.append(y)\n                y_train_z.append(z)\n            except IndexError as e:\n                print(f\"IndexError: {e} at location {loc} in experiment {exp} for particle type {ptype}\")\n\nX_train = np.array(X_train)\ny_train_x = np.array(y_train_x)\ny_train_y = np.array(y_train_y)\ny_train_z = np.array(y_train_z)\n\n# Check if the training data is loaded correctly\nif len(X_train) == 0 or len(y_train_x) == 0 or len(y_train_y) == 0 or len(y_train_z) == 0:\n    print(\"No training data loaded. Using sample submission file.\")\n    sample_submission_df = pd.read_csv(sample_submission_path)\n    \n    # Apply weights to x, y, and z values\n    weight_x = 1.405  # Adjust the weight for x as needed\n    weight_y = 1.305  # Adjust the weight for y as needed\n    weight_z = 1.205  # Adjust the weight for z as needed\n    \n    sample_submission_df['x'] = sample_submission_df['x'] * weight_x\n    sample_submission_df['y'] = sample_submission_df['y'] * weight_y\n    sample_submission_df['z'] = sample_submission_df['z'] * weight_z\n    \n    sample_submission_df.to_csv('submission.csv', index=False)\n    print(\"Sample submission file saved as submission.csv\")\nelse:\n    # Split the data into training and validation sets\n    X_train, X_val, y_train_x, y_val_x, y_train_y, y_val_y, y_train_z, y_val_z = train_test_split(\n        X_train, y_train_x, y_train_y, y_train_z, test_size=0.2, random_state=42\n    )\n\n    # Define the models\n    model_x = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n    model_y = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n    model_z = lgb.LGBMRegressor(n_estimators=100, random_state=42)\n\n    # Train the models\n    model_x.fit(X_train, y_train_x)\n    model_y.fit(X_train, y_train_y)\n    model_z.fit(X_train, y_train_z)\n\n    # Evaluate the models\n    y_pred_x = model_x.predict(X_val)\n    y_pred_y = model_y.predict(X_val)\n    y_pred_z = model_z.predict(X_val)\n\n    mse_x = mean_squared_error(y_val_x, y_pred_x)\n    mse_y = mean_squared_error(y_val_y, y_pred_y)\n    mse_z = mean_squared_error(y_val_z, y_pred_z)\n\n    print(f'MSE for x: {mse_x}')\n    print(f'MSE for y: {mse_y}')\n    print(f'MSE for z: {mse_z}')\n\n    # Load test data manually\n    test_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4']\n\n    submission = []\n\n    for exp in test_experiments:\n        tomogram_path = os.path.join(test_path, 'static/ExperimentRuns', exp, 'VoxelSpacing10.000', 'denoised.zarr')\n        tomogram = load_tomogram(tomogram_path)\n        \n        # Predict particle locations\n        for x in range(tomogram.shape[0]):\n            for y in range(tomogram.shape[1]):\n                for z in range(tomogram.shape[2]):\n                    voxel = tomogram[x, y, z]\n                    pred_x = model_x.predict([voxel])\n                    pred_y = model_y.predict([voxel])\n                    pred_z = model_z.predict([voxel])\n                    submission.append([len(submission), exp, 'particle_type', pred_x[0], pred_y[0], pred_z[0]])\n\n    # Create submission file\n    submission_df = pd.DataFrame(submission, columns=['id', 'experiment', 'particle_type', 'x', 'y', 'z'])\n    submission_df.to_csv('submission.csv', index=False)\n\n    print(\"Submission file created successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}