{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84969,"databundleVersionId":10033515,"sourceType":"competition"},{"sourceId":11384,"sourceType":"modelInstanceVersion","modelInstanceId":6216,"modelId":3301}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install necessary libraries\n#!pip install zarr transformers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:14:46.557367Z","iopub.execute_input":"2024-11-08T21:14:46.557877Z","iopub.status.idle":"2024-11-08T21:15:04.616712Z","shell.execute_reply.started":"2024-11-08T21:14:46.557839Z","shell.execute_reply":"2024-11-08T21:15:04.615736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport zarr\nimport json\nimport os\nimport random\nfrom transformers import AutoModelForSequenceClassification, AutoTokenizer, AutoConfig","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:04.619092Z","iopub.execute_input":"2024-11-08T21:15:04.619530Z","iopub.status.idle":"2024-11-08T21:15:13.570226Z","shell.execute_reply.started":"2024-11-08T21:15:04.619472Z","shell.execute_reply":"2024-11-08T21:15:13.569267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the data\ndef load_tomogram(experiment_path):\n    tomogram = zarr.open(experiment_path, mode='r')\n    return tomogram[0]  # Assuming 0 is the highest resolution\n\ndef load_particle_locations(json_path):\n    with open(json_path, 'r') as f:\n        data = json.load(f)\n    if 'points' in data:\n        return np.array(data['points'])\n    else:\n        print(f\"Key 'points' not found in {json_path}. Available keys: {data.keys()}\")\n        return np.array([])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:13.571633Z","iopub.execute_input":"2024-11-08T21:15:13.572178Z","iopub.status.idle":"2024-11-08T21:15:13.578082Z","shell.execute_reply.started":"2024-11-08T21:15:13.572145Z","shell.execute_reply":"2024-11-08T21:15:13.577038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example paths (you need to adjust these based on your actual directory structure)\ntrain_path = '/kaggle/input/czii-cryo-et-object-identification/train/'\ntest_path = '/kaggle/input/czii-cryo-et-object-identification/test/'\nsample_submission_path = '/kaggle/input/czii-cryo-et-object-identification/sample_submission.csv'\n\ngemma_2b_model_path = '/kaggle/input/gemma/transformers/2b/2'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:13.580305Z","iopub.execute_input":"2024-11-08T21:15:13.580628Z","iopub.status.idle":"2024-11-08T21:15:13.590945Z","shell.execute_reply.started":"2024-11-08T21:15:13.580595Z","shell.execute_reply":"2024-11-08T21:15:13.590169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load training data manually\ntrain_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4', 'TS_6_6', 'TS_73_6', 'TS_86_3', 'TS_99_9']\nparticle_types = ['apo-ferritin', 'beta-amylase', 'beta-galactosidase', 'ribosome', 'thyroglobulin', 'virus-like-particle']\n\nX_train = []\ny_train_x = []\ny_train_y = []\ny_train_z = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:13.592089Z","iopub.execute_input":"2024-11-08T21:15:13.592713Z","iopub.status.idle":"2024-11-08T21:15:13.604235Z","shell.execute_reply.started":"2024-11-08T21:15:13.592665Z","shell.execute_reply":"2024-11-08T21:15:13.603511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Manually load each experiment and particle type\nfor exp in train_experiments:\n    for ptype in particle_types:\n        tomogram_path = os.path.join(train_path, 'static/ExperimentRuns', exp, 'VoxelSpacing10.000', 'denoised.zarr')\n        json_path = os.path.join(train_path, 'overlay/ExperimentRuns', exp, 'Picks', f'{ptype}.json')\n        \n        print(f\"Loading tomogram from {tomogram_path}\")\n        tomogram = load_tomogram(tomogram_path)\n        \n        print(f\"Loading particle locations from {json_path}\")\n        locations = load_particle_locations(json_path)\n        \n        for loc in locations:\n            x, y, z = loc\n            try:\n                X_train.append(str(tomogram[x, y, z]))\n                y_train_x.append(str(x))\n                y_train_y.append(str(y))\n                y_train_z.append(str(z))\n            except IndexError as e:\n                print(f\"IndexError: {e} at location {loc} in experiment {exp} for particle type {ptype}\")\n\n# Save the training data into a .txt file\nwith open('training_data.txt', 'w') as f:\n    for i in range(len(X_train)):\n        f.write(f\"{X_train[i]}\\t{y_train_x[i]}\\t{y_train_y[i]}\\t{y_train_z[i]}\\n\")\n\nprint(\"Training data saved to training_data.txt\")\n\n# Convert the .txt file into a .json file\ntraining_data = {\n    'X_train': [],\n    'y_train_x': [],\n    'y_train_y': [],\n    'y_train_z': []\n}\n\nwith open('training_data.txt', 'r') as f:\n    for line in f:\n        voxel, x, y, z = line.strip().split('\\t')\n        training_data['X_train'].append(voxel)\n        training_data['y_train_x'].append(x)\n        training_data['y_train_y'].append(y)\n        training_data['y_train_z'].append(z)\n\nwith open('training_data.json', 'w') as f:\n    json.dump(training_data, f)\n\nprint(\"Training data converted to training_data.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:13.605376Z","iopub.execute_input":"2024-11-08T21:15:13.605746Z","iopub.status.idle":"2024-11-08T21:15:14.333670Z","shell.execute_reply.started":"2024-11-08T21:15:13.605686Z","shell.execute_reply":"2024-11-08T21:15:14.332546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define a function to load the Gemma 2B LLM model\ndef load_gemma_2b_model(model_path):\n    # Load the configuration and set the hidden activation function\n    config = AutoConfig.from_pretrained(model_path)\n    config.hidden_activation = 'gelu_pytorch_tanh'\n    \n    # Load the Gemma 2B LLM model using Hugging Face's transformers library\n    model = AutoModelForSequenceClassification.from_pretrained(model_path, config=config)\n    tokenizer = AutoTokenizer.from_pretrained(model_path)\n    return model, tokenizer\n\n# Define a function to predict using the Gemma 2B LLM model\ndef predict_with_gemma_2b(model, tokenizer, X_train):\n    # Use the Gemma 2B LLM model to predict the test file\n    predictions = []\n    \n    for i, voxel in enumerate(X_train):\n        # Tokenize the input\n        inputs = tokenizer(voxel, return_tensors='pt', padding=True, truncation=True)\n        \n        # Get the model's output\n        outputs = model(**inputs)\n        \n        # Assuming the model outputs logits, convert them to predictions\n        pred_x = outputs.logits[0, 0].item()\n        pred_y = outputs.logits[0, 1].item()\n        pred_z = outputs.logits[0, 2].item()\n        \n        predictions.append([i, 'TS_5_4', 'particle_type', pred_x, pred_y, pred_z])\n    \n    return predictions\n\n# Load the Gemma 2B LLM model\ngemma_2b_model, gemma_2b_tokenizer = load_gemma_2b_model(gemma_2b_model_path)\n\n# Read the training data from the .json file\nwith open('training_data.json', 'r') as f:\n    training_data = json.load(f)\n\n# Convert the data back to lists\nX_train = training_data['X_train']\ny_train_x = training_data['y_train_x']\ny_train_y = training_data['y_train_y']\ny_train_z = training_data['y_train_z']\n\n# Predict the test file using Gemma 2B LLM\npredictions = predict_with_gemma_2b(gemma_2b_model, gemma_2b_tokenizer, X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:15:14.335051Z","iopub.execute_input":"2024-11-08T21:15:14.335421Z","iopub.status.idle":"2024-11-08T21:15:55.708744Z","shell.execute_reply.started":"2024-11-08T21:15:14.335378Z","shell.execute_reply":"2024-11-08T21:15:55.707911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a DataFrame for the predictions\nsubmission_df = pd.DataFrame(predictions, columns=['id', 'experiment', 'particle_type', 'x', 'y', 'z'])\n\n# Filter the DataFrame to include only the specified experiments\nspecified_experiments = ['TS_5_4', 'TS_69_2', 'TS_6_4']\nfiltered_submission_df = submission_df[submission_df['experiment'].isin(specified_experiments)]\n\n# Save the filtered DataFrame to a CSV file\nfiltered_submission_df.to_csv('submission.csv', index=False)\n\n# Check if the submission file is empty and use the sample submission file if necessary\nif filtered_submission_df.empty:\n    print(\"Submission file is empty. Using sample submission file.\")\n    sample_submission_df = pd.read_csv(sample_submission_path)\n    \n    # Filter the sample submission file to include only the specified experiments\n    filtered_sample_submission_df = sample_submission_df[sample_submission_df['experiment'].isin(specified_experiments)]\n    \n    # Randomly increase the x, y, z values to ensure they are higher than the sample submission file's values\n    def increase_values(row):\n        row['x'] = row['x'] + random.randint(1, 19999)\n        row['y'] = row['y'] + random.randint(1, 6999)\n        row['z'] = row['z'] + random.randint(1, 899)\n        return row\n    \n    filtered_sample_submission_df = filtered_sample_submission_df.apply(increase_values, axis=1)\n    \n    # Save the modified sample submission file to a CSV file\n    filtered_sample_submission_df.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:16:58.676544Z","iopub.execute_input":"2024-11-08T21:16:58.677189Z","iopub.status.idle":"2024-11-08T21:16:58.698110Z","shell.execute_reply.started":"2024-11-08T21:16:58.677149Z","shell.execute_reply":"2024-11-08T21:16:58.697137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_sample_submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T21:17:05.101213Z","iopub.execute_input":"2024-11-08T21:17:05.101934Z","iopub.status.idle":"2024-11-08T21:17:05.119457Z","shell.execute_reply.started":"2024-11-08T21:17:05.101893Z","shell.execute_reply":"2024-11-08T21:17:05.118526Z"}},"outputs":[],"execution_count":null}]}