{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":120286,"databundleVersionId":14384428,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nVAM Dataset Challenge Metric for Kaggle\n\nThis metric evaluates action detection predictions for the RetailAction dataset using:\n- Temporal IoU (Intersection over Union) for action timing\n- Spatial distance in meters for action localization\n- Average Precision (AP) computed across multiple thresholds\n\nThe metric returns a global mean Average Precision (mAPglobal) score that combines\nspatial and temporal evaluation criteria.\n\nExpected DataFrame format:\n- Row ID column: video_id (string)\n- 'prediction' column: JSON string containing prediction data\n- 'groundtruth' column: JSON string containing groundtruth data\n\nPrediction JSON format:\n{\n  \"actions\": [\n    {\n      \"score\": 0.8,  # Confidence score\n      \"start\": 0.0,  # Normalized time [0, 1]\n      \"end\": 0.3,    # Normalized time [0, 1]\n      \"spatial\": {\n        \"action_cam\": {\n          \"rank0\": {\"x\": 0.5, \"y\": 0.5, \"visible\": true}\n        }\n      }\n    }\n  ]\n}\n\nGroundtruth JSON format:\n{\n  \"content\": {\n    \"action_cam\": {\n      \"rank0\": {\n        \"poses\": [...]  # Pose data for computing meter/pixel conversion\n      }\n    },\n    \"m_px_factors\": {\"rank0\": 3.07, ...}  # Optional precomputed factors\n  },\n  \"labels\": {\n    \"action\": [\n      {\n        \"label\": \"take\",\n        \"start\": 0.0,\n        \"end\": 0.4,\n        \"spatial\": {\n          \"action_cam\": {\n            \"rank0\": {\"x\": 0.5, \"y\": 0.5}\n          }\n        }\n      }\n    ]\n  }\n}\n\"\"\"\n\nimport json\nimport logging\nfrom dataclasses import dataclass, field\nfrom enum import IntEnum\nfrom typing import Dict, List, NamedTuple, Optional, Union\n\nimport numpy as np\nimport pandas as pd\n\n# Configure logging only if not already configured\nif not logging.getLogger().hasHandlers():\n    logging.basicConfig(level=logging.WARNING)\nlogger = logging.getLogger(__name__)\n\n\nclass ParticipantVisibleError(Exception):\n    \"\"\"Error messages shown to competition participants.\"\"\"\n    pass\n\n\n# ============================================================================\n# Data Types (embedded to avoid external file dependencies)\n# ============================================================================\n\nclass ActionPoint(NamedTuple):\n    \"\"\"Represents a spatial point for an action.\"\"\"\n    x: float  # spatial dimension: normalized [0,1] wrt video size\n    y: float  # spatial dimension: normalized [0,1] wrt video size\n    visible: Union[float, bool] = True  # probability of the point being visible in the view\n    m_px_factor: Optional[float] = None  # meter / pixel conversion factor\n\n\nclass ActionEnum(IntEnum):\n    \"\"\"Enumeration of action types.\"\"\"\n    take = 0\n    put = 1\n    touch = 2\n    other = 3\n    none = 4  # Null class (means no action)\n\n    @classmethod\n    def from_raw_label(cls, label: str) -> \"ActionEnum\":\n        label = label.lower()\n        if \"take\" in label:\n            return ActionEnum.take\n        elif \"put\" in label:\n            return ActionEnum.put\n        elif \"touch\" in label:\n            return ActionEnum.touch\n        else:\n            return ActionEnum.other\n\n\nclass ActionLocationEnum(IntEnum):\n    \"\"\"Enumeration of action locations.\"\"\"\n    null = -1\n    other = 0\n    cart = 1\n\n\n@dataclass\nclass ActionPrediction:\n    \"\"\"Represents a predicted action.\"\"\"\n    probs: Union[List[float], float]  # List[float] for multi-class or float for single score\n    start: float  # temporal dimension: normalized [0,1] wrt video length\n    end: float  # temporal dimension: normalized [0,1] wrt video length\n    spatial: Dict[str, Dict[str, ActionPoint]] = field(\n        default_factory=lambda: {\"action_cam\": dict()}\n    )\n\n\n@dataclass\nclass Prediction:\n    \"\"\"Container for all predictions for a video.\"\"\"\n    interaction: Optional[float] = None\n    actions: List[ActionPrediction] = field(default_factory=list)\n\n\n@dataclass\nclass ActionLabel:\n    \"\"\"Represents a ground truth action label.\"\"\"\n    label: ActionEnum\n    start: float  # temporal dimension: normalized [0,1] wrt video length\n    end: float  # temporal dimension: normalized [0,1] wrt video length\n    spatial: Dict[str, Dict[str, ActionPoint]] = field(\n        default_factory=lambda: {\"action_cam\": dict()}\n    )\n    extends: bool = False  # if the action length spans outside the video time\n    location: ActionLocationEnum = ActionLocationEnum.null  # location of the action\n\n    @classmethod\n    def from_dict(cls, data: Dict) -> \"ActionLabel\":\n        spatial_raw = data[\"spatial\"]\n        spatial = {\n            \"action_cam\": dict(),\n        }\n        if spatial_raw is not None:\n            if \"action_cam\" in spatial_raw:\n                spatial[\"action_cam\"] = {\n                    rank: ActionPoint(**{**point, \"visible\": True})\n                    for rank, point in spatial_raw[\"action_cam\"].items()\n                }\n        return cls(\n            label=ActionEnum[data[\"label\"]],\n            start=data[\"start\"],\n            end=data[\"end\"],\n            spatial=spatial,\n            extends=data.get(\"extends\", None),\n            location=ActionLocationEnum[data.get(\"location\", \"null\")],\n        )\n\n\n@dataclass\nclass RawTarget:\n    \"\"\"Container for ground truth data.\"\"\"\n    classification_label: Optional[List] = None\n    action_labels: Optional[List[ActionLabel]] = field(default_factory=list)\n    keypoints: Dict = None\n    shelves: Dict = None\n    poses: Dict = None\n    m_px_factors: Optional[Dict[str, float]] = None  # Precomputed meters/pixel factors per rank\n\n# Average bone lengths in meters for pose-based meter/pixel conversion\nBONE_LENGTH_MEANS = {\n    (\"neck\", \"nose\"): 0.19354,\n    (\"left_shoulder\", \"left_elbow\"): 0.27096,\n    (\"left_elbow\", \"left_wrist\"): 0.21228,\n    (\"right_shoulder\", \"right_elbow\"): 0.27210,\n    (\"right_elbow\", \"right_wrist\"): 0.21316,\n    (\"left_hip\", \"left_knee\"): 0.39204,\n    (\"left_knee\", \"left_ankle\"): 0.39530,\n    (\"right_hip\", \"right_knee\"): 0.39266,\n    (\"right_knee\", \"right_ankle\"): 0.39322,\n    (\"left_shoulder\", \"right_shoulder\"): 0.35484,\n    (\"left_hip\", \"right_hip\"): 0.17150,\n    (\"neck\", \"left_shoulder\"): 0.18136,\n    (\"neck\", \"right_shoulder\"): 0.18081,\n    (\"left_shoulder\", \"left_hip\"): 0.51375,\n    (\"right_shoulder\", \"right_hip\"): 0.51226,\n}\n\nM_PX_FACTOR_AVG = 3.07\n\n\ndef compute_m_px_factor_from_bones(raw_target: RawTarget) -> Optional[Dict]:\n    \"\"\"Compute meters/pixel factors based on precomputed bone lengths.\"\"\"\n    m_px_factors = dict()\n    for rank_name, rank_poses in raw_target.poses.items():\n        m_px_factors[rank_name] = []\n        for pose in rank_poses:\n            if pose is None:\n                continue\n            for (first_bone, second_bone), bone_length_meters in BONE_LENGTH_MEANS.items():\n                first_joint = pose.get(first_bone)\n                second_joint = pose.get(second_bone)\n                \n                if first_joint is not None and second_joint is not None:\n                    bone_length_pixels = np.sqrt(\n                        np.sum((first_joint - second_joint) ** 2)\n                    )\n                    if bone_length_pixels > 0:\n                        m_px_factors[rank_name].append(bone_length_meters / bone_length_pixels)\n    \n    if len(m_px_factors) == 0:\n        return None\n    else:\n        output = {rank_name: np.nanmedian(factors) for rank_name, factors in m_px_factors.items()}\n        output = {\n            rank_name: M_PX_FACTOR_AVG if np.isnan(factor) else factor\n            for rank_name, factor in output.items()\n        }\n        return output\n\n\ndef parse_prediction_from_json(pred_json: str) -> Prediction:\n    \"\"\"Parse prediction from JSON string.\"\"\"\n    pred_data = json.loads(pred_json)\n    actions = []\n    \n    for action_dict in pred_data.get(\"actions\", []):\n        spatial = {\"action_cam\": {}}\n        spatial_raw = action_dict.get(\"spatial\", {})\n        if \"action_cam\" in spatial_raw:\n            spatial[\"action_cam\"] = {\n                rank: ActionPoint(**point)\n                for rank, point in spatial_raw[\"action_cam\"].items()\n            }\n        \n        # Require \"score\" field with single float value\n        if \"score\" not in action_dict:\n            raise ParticipantVisibleError(\"Action must have a 'score' field with a single float value\")\n        \n        score = action_dict[\"score\"]\n        if not isinstance(score, (int, float)):\n            raise ParticipantVisibleError(f\"Action 'score' must be a number, got {type(score).__name__}\")\n        \n        action_pred = ActionPrediction(\n            probs=float(score),  # Store as float in probs field for compatibility\n            start=action_dict[\"start\"],\n            end=action_dict[\"end\"],\n            spatial=spatial\n        )\n        actions.append(action_pred)\n    \n    return Prediction(actions=actions)\n\n\ndef parse_groundtruth_from_json(gt_json: str) -> RawTarget:\n    \"\"\"Parse groundtruth from JSON string.\"\"\"\n    gt_data = json.loads(gt_json)\n    \n    # Load poses\n    poses = {}\n    content = gt_data.get(\"content\", gt_data)\n    for rank, rank_dict in content.get(\"action_cam\", {}).items():\n        poses_data = rank_dict.get(\"poses\", [])\n        poses[rank] = []\n        for frame_content in poses_data:\n            if frame_content is not None and len(frame_content.get(\"pose\", {})) > 0:\n                pose_data = frame_content[\"pose\"]\n                joint_scores = frame_content.get(\"joint_scores\", {})\n                \n                pose_dict = {}\n                for joint_name, coords in pose_data.items():\n                    score = joint_scores.get(joint_name, 0.0)\n                    coords_array = np.array(coords)\n                    if score > 0 or np.any(coords_array != 0):\n                        pose_dict[joint_name] = coords_array\n                \n                poses[rank].append(pose_dict if pose_dict else None)\n            else:\n                poses[rank].append(None)\n    \n    # Load action labels\n    labels = content.get(\"labels\", {})\n    raw_action_labels = labels.get(\"action\", None)\n    \n    action_labels = []\n    if raw_action_labels is not None:\n        for action_label_dict in raw_action_labels:\n            action_labels.append(ActionLabel.from_dict(action_label_dict))\n    \n    # Load precomputed m_px_factors\n    m_px_factors = content.get(\"m_px_factors\", None)\n    \n    return RawTarget(\n        poses=poses,\n        action_labels=action_labels,\n        shelves={},\n        m_px_factors=m_px_factors\n    )\n\n\ndef compute_metric(\n    predictions: List[Prediction],\n    targets: List[RawTarget],\n    confidence_thresholds: np.ndarray = np.linspace(1, 0, 101),\n    tiou_thresholds: np.ndarray = np.linspace(0.05, 0.5, 10),\n    spatial_thresholds: np.ndarray = np.linspace(1, 0.05, 20),\n) -> float:\n    \"\"\"Compute mAPglobal metric.\n    \n    This is the core metric computation adapted from GlobalMetric.compute().\n    \"\"\"\n    eps = 1e-4\n    \n    # Initialize counters\n    tp = np.zeros((len(spatial_thresholds), len(tiou_thresholds), len(confidence_thresholds)))\n    fp = np.zeros((len(spatial_thresholds), len(tiou_thresholds), len(confidence_thresholds)))\n    fn = np.zeros((len(spatial_thresholds), len(tiou_thresholds), len(confidence_thresholds)))\n    \n    # Process each video\n    for pred, raw_target in zip(predictions, targets):\n        # Compute or use precomputed m_px_factors\n        if raw_target.m_px_factors is not None:\n            m_px_factors = raw_target.m_px_factors\n        else:\n            m_px_factors = compute_m_px_factor_from_bones(raw_target)\n        \n        if m_px_factors is None or any(np.isnan(factor) for factor in m_px_factors.values()):\n            continue\n        \n        pred_actions = pred.actions\n        target_actions = raw_target.action_labels\n        \n        if not target_actions:\n            continue\n        \n        # Helper function to get confidence score (always a single float)\n        def get_confidence(action_pred: ActionPrediction) -> float:\n            # probs field always contains a single float for this metric\n            return float(action_pred.probs)\n        \n        # Sort predictions by confidence\n        pred_actions = sorted(pred_actions, key=get_confidence, reverse=True)\n        \n        # Iterate through thresholds\n        for s_idx, spatial_thresh in enumerate(spatial_thresholds):\n            for t_idx, tiou_thresh in enumerate(tiou_thresholds):\n                for c_idx, conf_thresh in enumerate(confidence_thresholds):\n                    matched_targets = set()\n                    tp_count = fp_count = 0\n                    \n                    for pred_action in pred_actions:\n                        confidence_score = get_confidence(pred_action)\n                        if confidence_score < conf_thresh:\n                            continue\n                        \n                        best_iou = 0\n                        best_target_idx = -1\n                        \n                        for target_idx, target in enumerate(target_actions):\n                            if target_idx in matched_targets:\n                                continue\n                            \n                            # Compute temporal IoU\n                            t_intersection = min(pred_action.end, target.end) - max(\n                                pred_action.start, target.start\n                            )\n                            t_union = max(pred_action.end, target.end) - min(\n                                pred_action.start, target.start\n                            )\n                            tiou = max(0, t_intersection / (t_union + eps))\n                            \n                            # Check spatial match\n                            spatial_match = True\n                            for cam, ranks in pred_action.spatial.items():\n                                if cam not in target.spatial:\n                                    continue\n                                valid_distances = []\n                                for rank, pred_point in ranks.items():\n                                    if rank not in target.spatial[cam] or rank not in m_px_factors:\n                                        continue\n                                    target_point = target.spatial[cam][rank]\n                                    dist_pixels = (\n                                        (pred_point.x - target_point.x) ** 2\n                                        + (pred_point.y - target_point.y) ** 2\n                                    ) ** 0.5\n                                    dist_meters = dist_pixels * m_px_factors[rank]\n                                    valid_distances.append(dist_meters)\n                                if all(d > spatial_thresh for d in valid_distances):\n                                    spatial_match = False\n                                    break\n                                if not spatial_match:\n                                    break\n                            \n                            if tiou >= tiou_thresh and spatial_match:\n                                if tiou > best_iou:\n                                    best_iou = tiou\n                                    best_target_idx = target_idx\n                        \n                        if best_target_idx >= 0:\n                            tp_count += 1\n                            matched_targets.add(best_target_idx)\n                        else:\n                            fp_count += 1\n                    \n                    fn_count = len(target_actions) - len(matched_targets)\n                    \n                    tp[s_idx, t_idx, c_idx] += tp_count\n                    fp[s_idx, t_idx, c_idx] += fp_count\n                    fn[s_idx, t_idx, c_idx] += fn_count\n    \n    # Compute AP\n    precisions = tp / (tp + fp + eps)\n    recalls = tp / (tp + fn + eps)\n    \n    # Compute maximum precision for each recall level\n    for s_idx in range(len(spatial_thresholds)):\n        for t_idx in range(len(tiou_thresholds)):\n            precisions[s_idx, t_idx] = np.flip(\n                np.maximum.accumulate(np.flip(precisions[s_idx, t_idx]))\n            )\n    \n    # Calculate AP values\n    recall_diffs = recalls[..., 1:] - recalls[..., :-1]\n    ap_values = np.sum(recall_diffs * precisions[..., :-1], axis=-1)\n    \n    # Global mAP is mean of all values\n    mAP_global = np.mean(ap_values)\n    \n    return float(mAP_global)\n\n\ndef metric(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    row_id_column_name: str = \"id\",\n    confidence_thresholds: np.ndarray = None,\n    tiou_thresholds: np.ndarray = None,\n    spatial_thresholds: np.ndarray = None,\n) -> float:\n    \"\"\"\n    Compute the VAM Dataset Challenge metric (mAPglobal).\n    \n    This metric evaluates action detection in retail surveillance videos using:\n    - Temporal IoU for action timing accuracy\n    - Spatial distance (in meters) for action localization accuracy\n    - Average Precision computed across multiple threshold combinations\n    \n    Args:\n        solution: DataFrame with groundtruth data containing:\n            - Row ID column (default: \"id\")\n            - 'groundtruth' column with JSON strings\n        submission: DataFrame with prediction data containing:\n            - Row ID column (default: \"id\")\n            - 'prediction' column with JSON strings\n        row_id_column_name: Name of the row ID column (default: \"id\")\n        confidence_thresholds: Array of confidence thresholds (default: 101 values from 1 to 0)\n        tiou_thresholds: Array of temporal IoU thresholds (default: 10 values from 0.05 to 0.5)\n        spatial_thresholds: Array of spatial distance thresholds in meters (default: 20 values from 1 to 0.05)\n    \n    Returns:\n        mAPglobal: Single float value representing the global mean Average Precision\n    \n    Raises:\n        ParticipantVisibleError: For validation errors that participants should see\n    \n    Example:\n        >>> import pandas as pd\n        >>> import json\n        >>> # Create simple test data\n        >>> solution_data = [{\n        ...     'id': 'video_1',\n        ...     'groundtruth': json.dumps({\n        ...         \"content\": {\n        ...             \"action_cam\": {\"rank0\": {\"poses\": []}},\n        ...             \"m_px_factors\": {\"rank0\": 3.0},\n        ...             \"labels\": {\n        ...                 \"action\": [{\n        ...                     \"label\": \"take\",\n        ...                     \"start\": 0.0,\n        ...                     \"end\": 0.4,\n        ...                     \"spatial\": {\"action_cam\": {\"rank0\": {\"x\": 0.5, \"y\": 0.5}}}\n        ...                 }]\n        ...             }\n        ...         }\n        ...     })\n        ... }]\n        >>> submission_data = [{\n        ...     'id': 'video_1',\n        ...     'prediction': json.dumps({\n        ...         \"actions\": [{\n        ...             \"score\": 0.8,\n        ...             \"start\": 0.0,\n        ...             \"end\": 0.5,\n        ...             \"spatial\": {\"action_cam\": {\"rank0\": {\"x\": 0.5, \"y\": 0.5, \"visible\": True}}}\n        ...         }]\n        ...     })\n        ... }]\n        >>> solution_df = pd.DataFrame(solution_data)\n        >>> submission_df = pd.DataFrame(submission_data)\n        >>> score = metric(solution_df, submission_df)\n        >>> assert 0.0 <= score <= 1.0\n        >>> print(f\"Score: {score:.4f}\")  # doctest: +ELLIPSIS\n        Score: 0...\n    \"\"\"\n    # Set default thresholds if not provided\n    if confidence_thresholds is None:\n        confidence_thresholds = np.linspace(1, 0, 101)\n    if tiou_thresholds is None:\n        tiou_thresholds = np.linspace(0.05, 0.5, 10)\n    if spatial_thresholds is None:\n        spatial_thresholds = np.linspace(1, 0.05, 20)\n    \n    # Validate that required columns exist\n    if 'prediction' not in submission.columns:\n        raise ParticipantVisibleError(\"Submission must have a 'prediction' column\")\n    if 'groundtruth' not in solution.columns:\n        raise ParticipantVisibleError(\"Solution must have a 'groundtruth' column\")\n    if row_id_column_name not in submission.columns:\n        raise ParticipantVisibleError(f\"Submission must have a '{row_id_column_name}' column\")\n    if row_id_column_name not in solution.columns:\n        raise ParticipantVisibleError(f\"Solution must have a '{row_id_column_name}' column\")\n    \n    # Check that submission and solution have same video IDs\n    solution_ids = set(solution[row_id_column_name])\n    submission_ids = set(submission[row_id_column_name])\n    \n    if not submission_ids.issubset(solution_ids):\n        extra_ids = submission_ids - solution_ids\n        raise ParticipantVisibleError(\n            f\"Submission contains {len(extra_ids)} video IDs not in groundtruth\"\n        )\n    \n    if submission_ids != solution_ids:\n        missing_ids = solution_ids - submission_ids\n        raise ParticipantVisibleError(\n            f\"Submission is missing {len(missing_ids)} video IDs from groundtruth\"\n        )\n    \n    # Align dataframes by row_id\n    solution = solution.sort_values(row_id_column_name).reset_index(drop=True)\n    submission = submission.sort_values(row_id_column_name).reset_index(drop=True)\n    \n    # Parse predictions and groundtruth\n    predictions = []\n    targets = []\n    \n    for idx in range(len(submission)):\n        video_id = submission.loc[idx, row_id_column_name]\n        \n        try:\n            # Parse prediction\n            pred_json = submission.loc[idx, 'prediction']\n            pred = parse_prediction_from_json(pred_json)\n            predictions.append(pred)\n            \n            # Parse groundtruth\n            gt_json = solution.loc[idx, 'groundtruth']\n            gt = parse_groundtruth_from_json(gt_json)\n            targets.append(gt)\n            \n        except json.JSONDecodeError as e:\n            raise ParticipantVisibleError(f\"Invalid JSON format for video_id {video_id}: {e}\")\n        except KeyError as e:\n            raise ParticipantVisibleError(f\"Missing required field for video_id {video_id}: {e}\")\n        except Exception as e:\n            raise ParticipantVisibleError(f\"Error processing video_id {video_id}: {e}\")\n    \n    # Compute metric\n    try:\n        mAP_global = compute_metric(\n            predictions=predictions,\n            targets=targets,\n            confidence_thresholds=confidence_thresholds,\n            tiou_thresholds=tiou_thresholds,\n            spatial_thresholds=spatial_thresholds,\n        )\n    except Exception as e:\n        # Don't expose internal errors to participants\n        try:\n            logger.error(f\"Error computing metric: {e}\")\n        except:\n            pass  # Ignore logging errors\n        raise Exception(f\"Internal error computing metric\")\n    \n    # Validate output\n    if not np.isfinite(mAP_global):\n        raise Exception(\"Metric returned non-finite value\")\n    \n    return float(mAP_global)\n\n\n# Backward compatibility alias\ndef score(\n    solution: pd.DataFrame,\n    submission: pd.DataFrame,\n    row_id_column_name: str,\n    confidence_thresholds: Optional[np.ndarray] = None,\n    tiou_thresholds: Optional[np.ndarray] = None,\n    spatial_thresholds: Optional[np.ndarray] = None,\n) -> float:\n    \"\"\"\n    Backward compatibility alias for metric() function.\n    Use metric() instead for Kaggle compatibility.\n    \"\"\"\n    return metric(\n        solution=solution,\n        submission=submission,\n        row_id_column_name=row_id_column_name,\n        confidence_thresholds=confidence_thresholds,\n        tiou_thresholds=tiou_thresholds,\n        spatial_thresholds=spatial_thresholds,\n    )\n\n\n# Simple self-test to validate the metric works (Kaggle may run this)\nif __name__ == \"__main__\":\n    # Create minimal test case\n    test_prediction = {\n        \"actions\": [{\n            \"score\": 0.8,\n            \"start\": 0.0,\n            \"end\": 0.5,\n            \"spatial\": {\"action_cam\": {\"rank0\": {\"x\": 0.5, \"y\": 0.5, \"visible\": True}}}\n        }]\n    }\n    \n    test_groundtruth = {\n        \"content\": {\n            \"action_cam\": {\"rank0\": {\"poses\": []}},\n            \"m_px_factors\": {\"rank0\": 3.0},\n            \"labels\": {\n                \"action\": [{\n                    \"label\": \"take\",\n                    \"start\": 0.0,\n                    \"end\": 0.4,\n                    \"spatial\": {\"action_cam\": {\"rank0\": {\"x\": 0.5, \"y\": 0.5}}}\n                }]\n            }\n        }\n    }\n    \n    submission_df = pd.DataFrame([{\n        'id': 'test_1',\n        'prediction': json.dumps(test_prediction)\n    }])\n    \n    solution_df = pd.DataFrame([{\n        'id': 'test_1',\n        'groundtruth': json.dumps(test_groundtruth)\n    }])\n    \n    try:\n        # Test with default row_id_column_name\n        result = metric(solution_df, submission_df)\n        print(f\"Self-test passed. mAPglobal: {result:.6f}\")\n        assert isinstance(result, float), \"Result must be a float\"\n        assert np.isfinite(result), \"Result must be finite\"\n    except Exception as e:\n        print(f\"Self-test failed: {e}\")\n        raise","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}