{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":4192885,"sourceType":"datasetVersion","datasetId":2472763},{"sourceId":423914,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":345486,"modelId":366785},{"sourceId":424167,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":345689,"modelId":366985}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-06T06:46:22.207992Z","iopub.execute_input":"2025-06-06T06:46:22.208273Z","iopub.status.idle":"2025-06-06T06:46:24.313013Z","shell.execute_reply.started":"2025-06-06T06:46:22.208244Z","shell.execute_reply":"2025-06-06T06:46:24.312218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision\nfrom torch.utils.data import DataLoader\nimport torchvision.models as models\nfrom torchvision.models.detection import FasterRCNN\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection import fasterrcnn_resnet50_fpn\nfrom torchvision.datasets import CocoDetection\nfrom torchvision.transforms import functional as F\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nfrom sklearn.cluster import DBSCAN\nimport numpy as np\nimport math\nimport shutil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T06:46:24.314838Z","iopub.execute_input":"2025-06-06T06:46:24.315260Z","iopub.status.idle":"2025-06-06T06:46:33.173927Z","shell.execute_reply.started":"2025-06-06T06:46:24.315227Z","shell.execute_reply":"2025-06-06T06:46:33.173113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /root/.cache/torch/hub/checkpoints\n\nsrc = '/kaggle/input/fasterrcnn_resnet50_fpn/pytorch/default/1/fasterrcnn_resnet50_fpn.pth'   #'../input/offline-resnet50/resnet50'\ndst = '/root/.cache/torch/hub/checkpoints/fasterrcnn_resnet50_fpn_coco-258fb6c6.pth'\n\nshutil.copy(src, dst)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T06:46:33.174689Z","iopub.execute_input":"2025-06-06T06:46:33.175042Z","iopub.status.idle":"2025-06-06T06:46:35.951312Z","shell.execute_reply.started":"2025-06-06T06:46:33.175024Z","shell.execute_reply":"2025-06-06T06:46:35.950338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_folder_names(directory_path):\n    \"\"\"\n    Gets a list of folder names in the specified directory.\n\n    Args:\n        directory_path: The path to the directory.\n\n    Returns:\n        A list of folder names.\n    \"\"\"\n    folder_names = [entry.name for entry in os.scandir(directory_path) if entry.is_dir()]\n    return folder_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:19:55.272078Z","iopub.execute_input":"2025-06-05T23:19:55.272467Z","iopub.status.idle":"2025-06-05T23:19:55.277590Z","shell.execute_reply.started":"2025-06-05T23:19:55.272436Z","shell.execute_reply":"2025-06-05T23:19:55.276719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"INPUT_PATH = '/kaggle/input/'\nDATA_PATH = '/kaggle/input/byu-locating-bacterial-flagellar-motors-2025'\nTEST_PATH = '/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/test'\nOUTPUT_PATH = 'kaggle/working'\nCHECKPOINT_PATH = '/kaggle/input/bb50_fasterrcnn_resnet50_epoch_100/pytorch/default/1/fasterrcnn_resnet50_epoch_100.pth'\ntest_tomograms = get_folder_names('/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:19:57.693642Z","iopub.execute_input":"2025-06-05T23:19:57.693917Z","iopub.status.idle":"2025-06-05T23:19:57.699411Z","shell.execute_reply.started":"2025-06-05T23:19:57.693895Z","shell.execute_reply":"2025-06-05T23:19:57.698738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Faster R-CNN with ResNet-50 backbone\ndef get_model(num_classes):\n    # Load pre-trained Faster R-CNN\n    model = torchvision.models.detection.fasterrcnn_resnet50_fpn(pretrained=True)\n    #model = fasterrcnn_resnet50_fpn\n    \n\n    # Get the number of input features for the classifier\n    in_features = model.roi_heads.box_predictor.cls_score.in_features\n\n    # Replace the pre-trained head with a new one\n    model.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:19:59.897599Z","iopub.execute_input":"2025-06-05T23:19:59.898268Z","iopub.status.idle":"2025-06-05T23:19:59.903616Z","shell.execute_reply.started":"2025-06-05T23:19:59.898235Z","shell.execute_reply":"2025-06-05T23:19:59.902876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n# Assuming that we are on a CUDA machine, this should print a CUDA device:\nprint(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:20:03.397341Z","iopub.execute_input":"2025-06-05T23:20:03.398115Z","iopub.status.idle":"2025-06-05T23:20:03.477614Z","shell.execute_reply.started":"2025-06-05T23:20:03.398083Z","shell.execute_reply":"2025-06-05T23:20:03.476809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize the model\nnum_classes = 2 # Background + motor\n\n# Load the trained model\nmodel = get_model(num_classes)\n\n# Load the checkpoint dictionary\nif torch.cuda.is_available():\n    checkpoint = torch.load(CHECKPOINT_PATH, weights_only=False)\nelse:\n    checkpoint = torch.load(CHECKPOINT_PATH, map_location=torch.device('cpu'), weights_only=False)\n\n# Extract and load only the model's state_dict\nmodel.load_state_dict(checkpoint['model_state_dict'])\n\n\nmodel.to(device)\nmodel.eval()  # Set the model to evaluation mode","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:20:06.237011Z","iopub.execute_input":"2025-06-05T23:20:06.237364Z","iopub.status.idle":"2025-06-05T23:20:10.081510Z","shell.execute_reply.started":"2025-06-05T23:20:06.237341Z","shell.execute_reply":"2025-06-05T23:20:10.080808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_image(image_path, device):\n    \"\"\"Prepares a single image for model inference.\"\"\"\n    image = Image.open(image_path).convert(\"RGB\") # Ensure image is RGB if your model expects it\n    image_tensor = F.to_tensor(image).unsqueeze(0)\n    return image_tensor.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:20:10.082687Z","iopub.execute_input":"2025-06-05T23:20:10.082953Z","iopub.status.idle":"2025-06-05T23:20:10.087092Z","shell.execute_reply.started":"2025-06-05T23:20:10.082934Z","shell.execute_reply":"2025-06-05T23:20:10.086257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# `prediction` contains:\n# - boxes: predicted bounding boxes\n# - labels: predicted class labels\n# - scores: predicted scores for each box (confidence level)\nCOCO_CLASSES = {0: \"Background\", 1: \"Motor\"}\n\ndef get_class_name(class_id):\n    return COCO_CLASSES.get(class_id, \"Unknown\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:20:11.891222Z","iopub.execute_input":"2025-06-05T23:20:11.891931Z","iopub.status.idle":"2025-06-05T23:20:11.895464Z","shell.execute_reply.started":"2025-06-05T23:20:11.891909Z","shell.execute_reply":"2025-06-05T23:20:11.894611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_motor_detections_3d(image_series_path, model, device, score_threshold=0.5, decimation=2):\n    \"\"\"\n    Performs object detection on a series of images and extracts 3D locations and scores.\n\n    Args:\n        image_series_path (str): Path to the directory containing the image series.\n        model: The trained object detection model.\n        device: The device to run inference on (e.g., 'cuda' or 'cpu').\n        score_threshold (float): Confidence threshold for detected objects.\n\n    Returns:\n        pandas.DataFrame: DataFrame with 'x', 'y', 'z', 'score' columns for detected motor centers.\n    \"\"\"\n    detections_3d = []\n    image_filenames = sorted(os.listdir(image_series_path)) # Assuming filenames allow sorting by slice number\n\n    for i in range(0, len(image_filenames), decimation):\n        filename = image_filenames[i]\n        if filename.endswith(\".jpg\"): # Adjust file extension if needed\n            image_path = os.path.join(image_series_path, filename)\n            image_tensor = prepare_image(image_path, device)\n\n            with torch.no_grad():\n                prediction = model(image_tensor)\n\n            # Process predictions\n            boxes = prediction[0]['boxes'].cpu().numpy()\n            labels = prediction[0]['labels'].cpu().numpy()\n            scores = prediction[0]['scores'].cpu().numpy()\n\n            # Extract center (x, y) and set z as slice number, also include score\n            for box, label, score in zip(boxes, labels, scores):\n                if score > score_threshold and label == 1: # Assuming label 1 is 'motor'\n                    x_min, y_min, x_max, y_max = box\n                    center_x = (x_min + x_max) / 2\n                    center_y = (y_min + y_max) / 2\n                    # Use 'i' as the z-coordinate (slice number)\n                    detections_3d.append({'x': center_x, 'y': center_y, 'z': i, 'score': score})\n\n    return pd.DataFrame(detections_3d)\n\ndef cluster_and_average_3d_points_with_scores(detections_df, eps=20, min_samples=5):\n    \"\"\"\n    Clusters 3D points (including scores) and averages the locations within each cluster,\n    also keeping track of the maximum score in each cluster.\n\n    Args:\n        detections_df (pandas.DataFrame): DataFrame with 'x', 'y', 'z', 'score' columns.\n        eps (float): The maximum distance between two samples for one to be considered as in the neighborhood of the other.\n        min_samples (int): The number of samples in a neighborhood for a point to be considered as a core point.\n\n    Returns:\n        pandas.DataFrame: DataFrame with 'x', 'y', 'z', 'max_score' columns for averaged 3D motor locations\n                          and the maximum score within each cluster.\n    \"\"\"\n    if detections_df.empty:\n        return pd.DataFrame(columns=['x', 'y', 'z', 'max_score'])\n\n    X = detections_df[['x', 'y', 'z']].values\n    scores = detections_df['score'].values\n\n    # Apply DBSCAN\n    dbscan = DBSCAN(eps=eps, min_samples=min_samples)\n    clusters = dbscan.fit_predict(X)\n\n    averaged_locations_with_scores = []\n    # Iterate through unique cluster labels (excluding noise, labeled as -1)\n    for cluster_id in np.unique(clusters):\n        if cluster_id != -1:\n            cluster_points = X[clusters == cluster_id]\n            cluster_scores = scores[clusters == cluster_id]\n\n            avg_x = np.mean(cluster_points[:, 0])\n            avg_y = np.mean(cluster_points[:, 1])\n            avg_z = np.mean(cluster_points[:, 2])\n            max_score = np.max(cluster_scores) # Get the maximum score in the cluster\n\n            averaged_locations_with_scores.append({'x': avg_x, 'y': avg_y, 'z': avg_z, 'max_score': max_score})\n\n    return pd.DataFrame(averaged_locations_with_scores)\n\ndef process_tomograms_and_generate_csv_single_best(tomograms_path, model, device, output_csv_path):\n    \"\"\"\n    Cycles through tomogram folders, processes them, and generates a CSV with the single averaged\n    motor location with the highest associated confidence score for each tomogram.\n    If no motors are found in a tomogram, returns -1 for all 3 motor axis values.\n\n    Args:\n        tomograms_path (str): Path to the directory containing tomogram folders.\n        model: The trained object detection model.\n        device: The device to run inference on (e.g., 'cuda' or 'cpu').\n        output_csv_path (str): Path to save the output CSV file.\n    \"\"\"\n    all_best_averaged_locations = []\n    tomogram_folders = get_folder_names(tomograms_path)\n\n    for tomo_id in tomogram_folders:\n        image_series_path = os.path.join(tomograms_path, tomo_id)\n\n        # Get 3D detections with scores\n        motor_detections_3d_df = get_motor_detections_3d(image_series_path, model, device, score_threshold=0.5, decimation=1)\n\n        # Cluster and average detections, also getting max score per cluster\n        averaged_motor_locations_with_scores_df = cluster_and_average_3d_points_with_scores(motor_detections_3d_df, eps=25, min_samples=2)\n\n        if averaged_motor_locations_with_scores_df.empty:\n            # No motors found, add a row with -1\n            all_best_averaged_locations.append({\n                'tomo_id': tomo_id,\n                'Motor axis 0': -1,\n                'Motor axis 1': -1,\n                'Motor axis 2': -1\n            })\n        else:\n            # Motors found, find the row with the highest 'max_score'\n            best_row = averaged_motor_locations_with_scores_df.loc[averaged_motor_locations_with_scores_df['max_score'].idxmax()]\n\n            all_best_averaged_locations.append({\n                'tomo_id': tomo_id,\n                'Motor axis 0': best_row['z'],\n                'Motor axis 1': best_row['y'],\n                'Motor axis 2': best_row['x']\n            })\n\n    # Create a DataFrame from the results\n    results_df = pd.DataFrame(all_best_averaged_locations)\n\n    # Save the DataFrame to a CSV file\n    results_df.to_csv(output_csv_path, index=False)\n    print(f\"Single best averaged motor locations saved to {output_csv_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T06:51:06.478954Z","iopub.execute_input":"2025-06-06T06:51:06.479819Z","iopub.status.idle":"2025-06-06T06:51:06.493625Z","shell.execute_reply.started":"2025-06-06T06:51:06.479789Z","shell.execute_reply":"2025-06-06T06:51:06.492660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage:\noutput_csv_filename = 'submission.csv'\noutput_csv_path = os.path.join('/kaggle/working/', output_csv_filename)\n\nprocess_tomograms_and_generate_csv_single_best(TEST_PATH, model, device, output_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:49:36.380010Z","iopub.execute_input":"2025-06-05T23:49:36.380764Z","iopub.status.idle":"2025-06-05T23:50:32.219125Z","shell.execute_reply.started":"2025-06-05T23:49:36.380732Z","shell.execute_reply":"2025-06-05T23:50:32.218153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#df = pd.read_csv('/kaggle/working/submission.csv')\n\n#print(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T23:50:38.541371Z","iopub.execute_input":"2025-06-05T23:50:38.541672Z","iopub.status.idle":"2025-06-05T23:50:38.553820Z","shell.execute_reply.started":"2025-06-05T23:50:38.541650Z","shell.execute_reply":"2025-06-05T23:50:38.552983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}