{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.17","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"},{"sourceId":327336,"sourceType":"modelInstanceVersion","modelInstanceId":274744,"modelId":295634}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.model_download(\"yyyy0201/mhafyolo/pyTorch/default\")\n\nprint(\"Path to model files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T19:36:04.442553Z","iopub.execute_input":"2025-05-23T19:36:04.442790Z","iopub.status.idle":"2025-05-23T19:36:05.057892Z","shell.execute_reply.started":"2025-05-23T19:36:04.442766Z","shell.execute_reply":"2025-05-23T19:36:05.053637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install mrcfile","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T19:36:11.719410Z","iopub.execute_input":"2025-05-23T19:36:11.719830Z","iopub.status.idle":"2025-05-23T19:37:13.492251Z","shell.execute_reply.started":"2025-05-23T19:36:11.719791Z","shell.execute_reply":"2025-05-23T19:37:13.487240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport math\nimport torch # Ejemplo si se usa PyTorch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nimport mrcfile # Para cargar archivos .mrc, si ese es el formato\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:10:03.140027Z","iopub.execute_input":"2025-05-22T22:10:03.140356Z","iopub.status.idle":"2025-05-22T22:10:03.145642Z","shell.execute_reply.started":"2025-05-22T22:10:03.140330Z","shell.execute_reply":"2025-05-22T22:10:03.144224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Constantes y Configuración ---\nDATA_DIR = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025\"\nTRAIN_LABELS_PATH = f\"{DATA_DIR}/train_labels.csv\"\n# Asumir que los tomogramas de entrenamiento están en DATA_DIR/train/\nTRAIN_TOMOGRAM_DIR = f\"{DATA_DIR}/train\"\nTEST_TOMOGRAM_DIR = f\"{DATA_DIR}/test\"\nSAMPLE_SUBMISSION_PATH = f\"{DATA_DIR}/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:10:08.382875Z","iopub.execute_input":"2025-05-22T22:10:08.383211Z","iopub.status.idle":"2025-05-22T22:10:08.388792Z","shell.execute_reply.started":"2025-05-22T22:10:08.383186Z","shell.execute_reply":"2025-05-22T22:10:08.387661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"THRESHOLD_ANGSTROMS = 1000.0\nBETA_FSCORE = 2.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:10:12.783386Z","iopub.execute_input":"2025-05-22T22:10:12.783756Z","iopub.status.idle":"2025-05-22T22:10:12.788763Z","shell.execute_reply.started":"2025-05-22T22:10:12.783730Z","shell.execute_reply":"2025-05-22T22:10:12.787404Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> b1","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport math\nimport torch # Uncomment if you implement a PyTorch model\nimport torch.nn as nn # Uncomment if you implement a PyTorch model\nfrom torch.utils.data import Dataset, DataLoader # Uncomment if you implement a PyTorch model\nimport mrcfile # Uncomment if your tomograms are in .mrc format and you use this library\n\n# --- Constants and Configuration ---\nDATA_DIR = \"/kaggle/input/byu-locating-bacterial-flagellar-motors-2025\"\nTRAIN_LABELS_PATH = f\"{DATA_DIR}/train_labels.csv\"\n# Assuming train tomograms are in DATA_DIR/train/ if you were to load them\nTRAIN_TOMOGRAM_DIR = f\"{DATA_DIR}/train\"\nTEST_TOMOGRAM_DIR = f\"{DATA_DIR}/test\" # Directory for test tomogram files\nSAMPLE_SUBMISSION_PATH = f\"{DATA_DIR}/sample_submission.csv\"\n\nTHRESHOLD_ANGSTROMS = 1000.0\nBETA_FSCORE = 2.0\n\nprint(\"Block 1 executed: Imports and Constants defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:20:39.003931Z","iopub.execute_input":"2025-05-22T22:20:39.004267Z","iopub.status.idle":"2025-05-22T22:20:39.011120Z","shell.execute_reply.started":"2025-05-22T22:20:39.004245Z","shell.execute_reply":"2025-05-22T22:20:39.009555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Helper Functions ---\n\ndef load_tomogram_data_simulation(tomo_id, tomogram_dir_path, metadata_df_for_shape_info):\n    \"\"\"\n    Simulates loading a tomogram array and its metadata for demonstration.\n    In a real scenario, this function would load actual tomogram files (e.g., .mrc).\n    It requires metadata_df_for_shape_info to get 'Array shape' and 'Voxel spacing'.\n    For test data, this metadata_df_for_shape_info should be specific test metadata.\n    \"\"\"\n    tomo_metadata_entry = metadata_df_for_shape_info[metadata_df_for_shape_info['tomo_id'] == tomo_id]\n\n    if tomo_metadata_entry.empty:\n        print(f\"Warning: No metadata found for {tomo_id} to determine shape/voxel spacing. Using defaults for simulation.\")\n        # Fallback default values if no metadata is found (NOT IDEAL FOR REAL COMPETITION)\n        # The competition MUST provide metadata (shape, voxel spacing) for test tomograms.\n        shape = (100, 256, 256) # Example Z, Y, X\n        voxel_spacing = 10.0 # Example Angstroms/pixel\n        # Attempt to load actual file if path exists (conceptual)\n        # file_path = f\"{tomogram_dir_path}/{tomo_id}.mrc\" # Or other extension\n        # print(f\"Conceptual: Attempting to load {file_path}\")\n        # tomogram_array = np.random.rand(*shape).astype(np.float32) # Simulate if actual load fails\n    else:\n        tomo_info = tomo_metadata_entry.iloc[0]\n        # Shape is typically (Depth, Height, Width) which might correspond to (axis 2, axis 1, axis 0) from CSV\n        shape = (int(tomo_info['Array shape (axis 2)']),\n                 int(tomo_info['Array shape (axis 1)']),\n                 int(tomo_info['Array shape (axis 0)']))\n        voxel_spacing = tomo_info['Voxel spacing']\n        # Actual file loading would go here:\n        # file_path = f\"{tomogram_dir_path}/{tomo_id}.mrc\"\n        # try:\n        #     with mrcfile.open(file_path, permissive=True) as mrc:\n        #         tomogram_array = np.array(mrc.data, dtype=np.float32)\n        # except FileNotFoundError:\n        #     print(f\"Warning: Tomogram file for {tomo_id} not found. Simulating array.\")\n        #     tomogram_array = np.random.rand(*shape).astype(np.float32)\n        # except Exception as e:\n        #     print(f\"Warning: Error loading tomogram file for {tomo_id}: {e}. Simulating array.\")\n        #     tomogram_array = np.random.rand(*shape).astype(np.float32)\n        tomogram_array = np.random.rand(*shape).astype(np.float32) # Simulating for now\n\n    print(f\"Simulated loading for {tomo_id}: shape {shape}, voxel_spacing {voxel_spacing if voxel_spacing is not None else 'N/A'}\")\n\n    # For ground truth (relevant for training/validation, not directly for test prediction generation)\n    gt_motors_for_tomo = metadata_df_for_shape_info[\n        (metadata_df_for_shape_info['tomo_id'] == tomo_id) & (metadata_df_for_shape_info.get('Motor axis 0', -1.0) != -1.0)\n    ]\n    motor_coordinates = []\n    if not gt_motors_for_tomo.empty and 'Motor axis 0' in gt_motors_for_tomo.columns:\n        for _, row in gt_motors_for_tomo.iterrows():\n            motor_coordinates.append((row['Motor axis 0'], row['Motor axis 1'], row['Motor axis 2']))\n\n    return tomogram_array, motor_coordinates, voxel_spacing, shape\n\n\ndef calculate_metric_components(predicted_coord, gt_motor_coords_list, voxel_spacing):\n    \"\"\"\n    Calculates TP, FP, FN for a single tomogram based on a single prediction.\n    \"\"\"\n    tp, fp, fn = 0, 0, 0\n\n    if voxel_spacing is None or voxel_spacing <= 0:\n        print(f\"Warning: Voxel spacing invalid ({voxel_spacing}) for evaluation of a tomogram.\")\n        if predicted_coord and predicted_coord != (-1, -1, -1):\n            fp = 1 # Prediction made, but cannot reliably check it\n        if not gt_motor_coords_list and predicted_coord and predicted_coord != (-1,-1,-1): fp = 1 # Predicted something for nothing (cannot confirm with bad voxel)\n        elif gt_motor_coords_list and (not predicted_coord or predicted_coord == (-1,-1,-1)): fn = 1 # Missed something (cannot confirm with bad voxel)\n        elif gt_motor_coords_list and predicted_coord and predicted_coord != (-1,-1,-1) : fp=1; fn=1 # Both exist, but cannot verify distance\n        return tp, fp, fn\n\n    threshold_pixels_sq = (THRESHOLD_ANGSTROMS / voxel_spacing)**2\n\n    if predicted_coord and predicted_coord != (-1, -1, -1): # Motor predicted\n        if not gt_motor_coords_list: # No actual motors -> FP\n            fp = 1\n        else: # Actual motors exist\n            match_found = False\n            for real_motor_coord in gt_motor_coords_list:\n                dist_sq_pixels = sum((p - gt)**2 for p, gt in zip(predicted_coord, real_motor_coord))\n                if dist_sq_pixels <= threshold_pixels_sq:\n                    match_found = True\n                    break\n            if match_found:\n                tp = 1\n            else: # Prediction made, but far from any actual motor\n                fp = 1\n                fn = 1 # Actual motors existed but were \"missed\" by this incorrect prediction\n    else: # No motor predicted by algorithm\n        if gt_motor_coords_list: # Actual motors exist but we predicted none -> FN\n            fn = 1\n        # else: True Negative (TN) - correctly no motor predicted, no actual motor. Not in F-score terms.\n    return tp, fp, fn\n\ndef calculate_fbeta_score(tp_total, fp_total, fn_total, beta):\n    if tp_total == 0:\n        return 0.0\n    precision = tp_total / (tp_total + fp_total) if (tp_total + fp_total) > 0 else 0\n    recall = tp_total / (tp_total + fn_total) if (tp_total + fn_total) > 0 else 0\n\n    if precision == 0 and recall == 0:\n        return 0.0\n    \n    denominator = (beta**2 * precision) + recall\n    if denominator == 0:\n        return 0.0\n        \n    fbeta = (1 + beta**2) * (precision * recall) / denominator\n    return fbeta\n\nprint(\"Block 2 executed: Helper functions defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:21:19.541303Z","iopub.execute_input":"2025-05-22T22:21:19.541592Z","iopub.status.idle":"2025-05-22T22:21:19.558093Z","shell.execute_reply.started":"2025-05-22T22:21:19.541573Z","shell.execute_reply":"2025-05-22T22:21:19.556622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Model Definition (Conceptual Placeholder) ---\n# This is where you would define your 3D CNN model using PyTorch or TensorFlow/Keras\n# Example for PyTorch (ensure you have torch imported and uncommented in Block 1)\n\n# class Simple3DCNN(nn.Module):\n#     def __init__(self, in_channels=1, num_coords=3, presence_classes=1):\n#         super().__init__()\n#         # Example layers:\n#         self.conv1 = nn.Conv3d(in_channels, 16, kernel_size=3, padding=1)\n#         self.relu1 = nn.ReLU()\n#         self.pool1 = nn.MaxPool3d(2)\n#         self.conv2 = nn.Conv3d(16, 32, kernel_size=3, padding=1)\n#         self.relu2 = nn.ReLU()\n#         self.pool2 = nn.MaxPool3d(2)\n#         # Adaptive pooling to handle variable input sizes after convolutions\n#         self.adaptive_pool = nn.AdaptiveAvgPool3d((4, 4, 4)) # Output size (D,H,W)\n#         \n#         # Flatten and pass to fully connected layers\n#         # Calculate the number of flattened features: 32 channels * 4 * 4 * 4\n#         self.flattened_features = 32 * 4 * 4 * 4\n#         self.fc1 = nn.Linear(self.flattened_features, 128)\n#         self.relu3 = nn.ReLU()\n#         \n#         # Output heads\n#         self.coord_regressor = nn.Linear(128, num_coords)\n#         self.presence_classifier = nn.Linear(128, presence_classes) # Outputting logits\n\n#     def forward(self, x):\n#         x = self.pool1(self.relu1(self.conv1(x)))\n#         x = self.pool2(self.relu2(self.conv2(x)))\n#         x = self.adaptive_pool(x)\n#         x = x.view(-1, self.flattened_features) # Flatten\n#         x = self.relu3(self.fc1(x))\n#         \n#         coords = self.coord_regressor(x) # Raw coordinate values\n#         presence_logits = self.presence_classifier(x).squeeze(-1) # Remove last dim if it's 1\n#         return presence_logits, coords\n\nprint(\"Block 3 executed: Conceptual Model placeholder defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:22:04.906359Z","iopub.execute_input":"2025-05-22T22:22:04.906703Z","iopub.status.idle":"2025-05-22T22:22:04.913166Z","shell.execute_reply.started":"2025-05-22T22:22:04.906677Z","shell.execute_reply":"2025-05-22T22:22:04.912059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_main_pipeline():\n    print(\"\\n--- Starting Main Pipeline ---\")\n    # Load initial metadata\n    # train_labels_df contains ground truth and metadata for training images\n    # It's used here to simulate having access to Voxel Spacing and Array Shape for test tomograms\n    # In a real competition, you'd have a separate metadata file for the test set.\n    try:\n        train_labels_df = pd.read_csv(TRAIN_LABELS_PATH)\n        sample_submission_df = pd.read_csv(SAMPLE_SUBMISSION_PATH)\n    except FileNotFoundError as e:\n        print(f\"Error: Required CSV file not found: {e}\")\n        print(\"Please ensure Kaggle input files are correctly mounted at /kaggle/input/\")\n        return\n\n    test_tomo_ids = sample_submission_df['tomo_id'].unique()\n\n    # --- Critical Step: Obtain Voxel Spacing and Array Shape for TEST tomograms ---\n    # The competition MUST provide this information for the test set.\n    # Here, we simulate this by trying to find test_tomo_ids in train_labels_df.\n    # This is NOT how it would work in reality but is a common workaround for examples\n    # if test metadata isn't explicitly given alongside the sample submission.\n    # A better simulation would be to create a mock test_metadata_df.\n    \n    # For this example, we will assume `train_labels_df` can serve as a source for metadata\n    # for any `tomo_id` that might appear in `test_tomo_ids`.\n    # If a `tomo_id` from test is not in `train_labels_df`, `load_tomogram_data_simulation`\n    # will use default/fallback values, which is not ideal.\n    # A real solution requires a dedicated test metadata file.\n    test_metadata_source_df = train_labels_df # Using train_labels as a stand-in for test metadata source\n\n    print(f\"Loaded {len(test_tomo_ids)} test tomogram IDs from sample submission.\")\n\n    # Initialize your trained model here (conceptual)\n    # model = Simple3DCNN()\n    # model.load_state_dict(torch.load('path_to_your_trained_model.pth')) # Example\n    # model.eval() # Set model to evaluation mode\n    # device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    # model.to(device)\n\n    print(\"Simulating predictions on the test set (model part is conceptual)...\")\n    \n    # THIS IS WHERE `predictions` IS INITIALIZED\n    predictions_list = [] # Renamed to avoid conflict if 'predictions' is a module/common name\n\n    for tomo_id in test_tomo_ids:\n        # In a real pipeline:\n        # 1. Load actual tomogram data for tomo_id from TEST_TOMOGRAM_DIR\n        #    tomogram_array, _, voxel_spacing, current_shape = load_actual_tomogram(tomo_id, TEST_TOMOGRAM_DIR, test_metadata_source_df)\n        # For simulation, we use the simulation function:\n        sim_tomogram_array, _, voxel_spacing, current_shape = load_tomogram_data_simulation(\n            tomo_id, TEST_TOMOGRAM_DIR, test_metadata_source_df\n        )\n\n        if sim_tomogram_array is None: # Should not happen with simulation unless error in metadata part\n            print(f\"Skipping {tomo_id} due to loading error/missing data in simulation.\")\n            pred_coords_pixels = (-1, -1, -1)\n        else:\n            # --- Actual Model Prediction Would Go Here ---\n            # input_for_model = preprocess_tomogram(sim_tomogram_array) # Normalize, resize/patch, etc.\n            # input_tensor = torch.from_numpy(input_for_model).unsqueeze(0).unsqueeze(0).to(device) # Add Batch and Channel dims\n            # with torch.no_grad():\n            #     presence_logits_pred, coords_pred_raw = model(input_tensor)\n            # presence_prob_pred = torch.sigmoid(presence_logits_pred).item()\n            \n            # decision_threshold = 0.5 # This should be tuned on a validation set\n            # if presence_prob_pred >= decision_threshold:\n            #     # Post-process coords_pred_raw if necessary (e.g., denormalize, convert from patch to full tomo)\n            #     # Ensure coordinates are in the order (axis 0, axis 1, axis 2) as per submission format\n            #     pred_coords_pixels = (coords_pred_raw[0,0].item(), coords_pred_raw[0,1].item(), coords_pred_raw[0,2].item())\n            # else:\n            #     pred_coords_pixels = (-1, -1, -1)\n            # --- End of Actual Model Prediction ---\n\n            # --- Simulation of Prediction (Random) ---\n            # This part simulates what your model would output\n            # Remove this when you have a real model.\n            presence_simulation = np.random.rand() # Simulate model's presence probability\n            prediction_confidence_threshold = 0.5 # Example threshold\n\n            if presence_simulation >= prediction_confidence_threshold:\n                # Simulate coordinates within the tomogram's dimensions\n                # current_shape is (Depth, Height, Width) -> (Z, Y, X)\n                # Motor axis 0, 1, 2 likely correspond to X, Y, Z or similar based on tomographic conventions\n                # Assuming CSV 'Motor axis 0' is X, 'Motor axis 1' is Y, 'Motor axis 2' is Z.\n                # And array shape from CSV 'Array shape (axis 0)' is X, 'axis 1' is Y, 'axis 2' is Z.\n                # So current_shape[2] is X_max, current_shape[1] is Y_max, current_shape[0] is Z_max.\n                pred_x = np.random.uniform(0, current_shape[2]) # Corresponds to 'Array shape (axis 0)'\n                pred_y = np.random.uniform(0, current_shape[1]) # Corresponds to 'Array shape (axis 1)'\n                pred_z = np.random.uniform(0, current_shape[0]) # Corresponds to 'Array shape (axis 2)'\n                pred_coords_pixels = (pred_x, pred_y, pred_z)\n            else:\n                pred_coords_pixels = (-1, -1, -1)\n            # --- End of Simulation of Prediction ---\n            \n        predictions_list.append({\n            'tomo_id': tomo_id,\n            'Motor axis 0': pred_coords_pixels[0],\n            'Motor axis 1': pred_coords_pixels[1],\n            'Motor axis 2': pred_coords_pixels[2]\n        })\n\n    # Create DataFrame from the list of prediction dictionaries\n        submission_df = pd.DataFrame(predictions_list)\n        submission_df.to_csv(\"submission.csv\", index=False)\n        print(\"\\n--- Submission file 'submission.csv' generated successfully. ---\")\n        print(\"Head of the submission file:\")\n        print(submission_df.head())\n\n    # Optional: If you had ground truth for this test set (e.g. a hidden validation set)\n    # you could calculate the F2 score here.\n    # For the official test set, you submit and Kaggle calculates the score.\n\nprint(\"Block 4 executed: Main pipeline function `run_main_pipeline` defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:29:28.501536Z","iopub.execute_input":"2025-05-22T22:29:28.501987Z","iopub.status.idle":"2025-05-22T22:29:28.514489Z","shell.execute_reply.started":"2025-05-22T22:29:28.501962Z","shell.execute_reply":"2025-05-22T22:29:28.513665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == '__main__':\n    # This will call the main pipeline function when the script is run.\n    # If you are in a Jupyter Notebook, you can just call run_main_pipeline() directly in a cell.\n    run_main_pipeline()\n\n# If in a Jupyter Notebook, you can also just run this in a new cell:\n# run_main_pipeline()\nprint(\"\\nBlock 5 instruction: Call 'run_main_pipeline()' to execute the process.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T22:29:37.437244Z","iopub.execute_input":"2025-05-22T22:29:37.437523Z","iopub.status.idle":"2025-05-22T22:30:06.994340Z","shell.execute_reply.started":"2025-05-22T22:29:37.437503Z","shell.execute_reply":"2025-05-22T22:30:06.993448Z"}},"outputs":[],"execution_count":null}]}