{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":46105,"databundleVersionId":5087314}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"c58d9ac8-0308-45b6-9197-0d2a5ee095bb","cell_type":"markdown","source":"# 🦾 Sign Language Motion Dataset Builder\n### Raw Keypoints → Structured 3D Animation-Ready Dataset\n\n**Purpose:** Transform raw ASL landmark sequences into a clean, semantically-labelled\nmotion dataset suitable for driving a 3D character.\n\n**Input:** Parquet files — one per sign sample — containing MediaPipe landmarks  \n**Output:**\n- `joint_schema.json` — immutable joint contract  \n- `signs/<word>.json` — one structured file per sign  \n- `motion_dataset.csv` — flat CSV for quick inspection  \n- `metadata.json` — global index & statistics\n\n---\n*Pipeline: Load → Extract → Detect Active Region → Detect Handedness → Clean NaN → Normalise → Remap to Schema → Export*","metadata":{}},{"id":"2b74c496-b819-45f4-a344-0c87f19798c1","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 1 · SETUP & IMPORTS\n# ════════════════════════════════════════════════\n!pip install pandas numpy matplotlib pyarrow tqdm -q\n\nimport os, json, warnings\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom scipy import interpolate as sci_interp\nfrom tqdm.notebook import tqdm\n\nwarnings.filterwarnings('ignore')\nprint('✅ All libraries loaded.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:04.606607Z","iopub.execute_input":"2026-04-13T21:50:04.607505Z","iopub.status.idle":"2026-04-13T21:50:07.972692Z","shell.execute_reply.started":"2026-04-13T21:50:04.607470Z","shell.execute_reply":"2026-04-13T21:50:07.971858Z"}},"outputs":[],"execution_count":null},{"id":"1b8321d7-03dc-4c1e-a06a-2d19990a8ba3","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 2 · CONFIGURATION\n# ════════════════════════════════════════════════\n# ──────────────────────────────────────────────\n# 📂  Paths — adjust to your environment\n# ──────────────────────────────────────────────\nDATA_ROOT   = '/kaggle/input/competitions/asl-signs'        # root with train.csv + parquets\n# DATA_ROOT = '/content/asl-signs'             # Colab alternative\nTRAIN_CSV   = os.path.join(DATA_ROOT, 'train.csv')\nOUTPUT_DIR  = './motion_dataset'\n\nos.makedirs(OUTPUT_DIR,                      exist_ok=True)\nos.makedirs(os.path.join(OUTPUT_DIR,'signs'),exist_ok=True)\n\n# ──────────────────────────────────────────────\n# 🎛  Pipeline parameters\n# ──────────────────────────────────────────────\nFPS                    = 30\nMAX_FRAMES_HARD_CAP    = 384       # absolute ceiling for a single sign\nFIXED_LENGTH_OPTION    = 60        # resampling target (used on demand)\nPADDING_VALUE          = -100.0    # sentinel used by the original dataset\nACTIVITY_THRESHOLD     = 0.01      # velocity threshold for active-region detection\nHAND_ABSENCE_THRESHOLD = 0.60      # ≥ 60 % NaN across all frames → hand absent\n\n# ──────────────────────────────────────────────\n# 🎯  Which signs to process\n# ──────────────────────────────────────────────\n# None  →  process ALL 250 signs (one representative sample each)\n# list  →  process only these words (useful for debugging)\nTARGET_SIGNS = None\n# TARGET_SIGNS = ['book', 'sleep', 'drink', 'think', 'help']\n\n# ──────────────────────────────────────────────\n# 🦴  MediaPipe landmark indices used in the\n#     original notebook (DO NOT CHANGE)\n# ──────────────────────────────────────────────\nLIP_LANDMARKS  = [61,185,40,39,37,0,267,269,270,409,\n                  291,375,321,405,314,17,84,181,91,146,\n                  78,95,88,178,87,14,317,402,318,324,\n                  308,415,310,311,312,13,82,81,80,191]   # 40\n\nEYE_LANDMARKS  = [33,7,163,144,145,153,154,155,133,\n                  246,161,160,159,158,157,173]            # 16\n\nNOSE_LANDMARKS = [1,2,98,327]                             # 4\n\nHAND_LANDMARKS = list(range(21))                          # 0-20\n\nPOSE_LANDMARKS = [0,1,2,3,4,11,12,13,14,15,16]          # 11\n\nHAND_CONNECTIONS = [\n    (0,1),(1,2),(2,3),(3,4),\n    (0,5),(5,6),(6,7),(7,8),\n    (0,9),(9,10),(10,11),(11,12),\n    (0,13),(13,14),(14,15),(15,16),\n    (0,17),(17,18),(18,19),(19,20),\n    (5,9),(9,13),(13,17),\n]\n\nprint('✅ Configuration loaded.')\nprint(f'   Output directory : {os.path.abspath(OUTPUT_DIR)}')\nprint(f'   Dataset root     : {DATA_ROOT}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:07.974428Z","iopub.execute_input":"2026-04-13T21:50:07.974736Z","iopub.status.idle":"2026-04-13T21:50:07.985257Z","shell.execute_reply.started":"2026-04-13T21:50:07.974707Z","shell.execute_reply":"2026-04-13T21:50:07.984472Z"}},"outputs":[],"execution_count":null},{"id":"6a008ac9-9a2c-46d1-ad09-b643f37c8696","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 3 · JOINT SCHEMA DEFINITION\n# ════════════════════════════════════════════════\n\"\"\"\nThe joint schema is the immutable contract between this builder\nand any downstream consumer (animation engine, ML model, debugger).\n\nLayout in the output frame array (113 joints × 3 coords):\n  ┌──────────────┬────────┬──────────┐\n  │ Group        │ Offset │  Count   │\n  ├──────────────┼────────┼──────────┤\n  │ pose         │   0    │   11     │\n  │ face         │  11    │   60     │\n  │   lips_outer │  11    │   20     │\n  │   lips_inner │  31    │   20     │\n  │   left_eye   │  51    │    8     │\n  │   right_eye  │  59    │    8     │\n  │   nose       │  67    │    4     │\n  │ left_hand    │  71    │   21     │\n  │ right_hand   │  92    │   21     │\n  └──────────────┴────────┴──────────┘\n\"\"\"\n\n_POSE_JOINT_NAMES = [\n    'nose',\n    'left_eye_inner', 'left_eye', 'left_eye_outer',\n    'right_eye_inner',\n    'left_shoulder',  'right_shoulder',\n    'left_elbow',     'right_elbow',\n    'left_wrist',     'right_wrist',\n]  # 11\n\n_HAND_JOINT_NAMES = [\n    'wrist',\n    'thumb_cmc',  'thumb_mcp',  'thumb_ip',   'thumb_tip',\n    'index_mcp',  'index_pip',  'index_dip',  'index_tip',\n    'middle_mcp', 'middle_pip', 'middle_dip', 'middle_tip',\n    'ring_mcp',   'ring_pip',   'ring_dip',   'ring_tip',\n    'pinky_mcp',  'pinky_pip',  'pinky_dip',  'pinky_tip',\n]  # 21\n\n# ──  flat ordered list (same order as frame[\"joints\"])  ──────────────────\nJOINT_NAMES: list[str] = (\n    _POSE_JOINT_NAMES\n    + [f'lips_outer_{i}' for i in range(20)]\n    + [f'lips_inner_{i}' for i in range(20)]\n    + [f'left_eye_{i}'   for i in range(8)]\n    + [f'right_eye_{i}'  for i in range(8)]\n    + [f'nose_{i}'       for i in range(4)]\n    + [f'lh_{n}'         for n in _HAND_JOINT_NAMES]\n    + [f'rh_{n}'         for n in _HAND_JOINT_NAMES]\n)  # 113 names\n\nassert len(JOINT_NAMES) == 113, f\"Schema broken: {len(JOINT_NAMES)} joints\"\n\n# ──  group slice helpers  ─────────────────────────────────────────────────\nSCHEMA_GROUPS = {\n    'pose'       : slice( 0, 11),\n    'lips_outer' : slice(11, 31),\n    'lips_inner' : slice(31, 51),\n    'left_eye'   : slice(51, 59),\n    'right_eye'  : slice(59, 67),\n    'nose'       : slice(67, 71),\n    'left_hand'  : slice(71, 92),\n    'right_hand' : slice(92,113),\n}\n\n# ──  index look-up  ───────────────────────────────────────────────────────\nJOINT_INDEX = {name: i for i, name in enumerate(JOINT_NAMES)}\n\n# ──  serialise to disk  ───────────────────────────────────────────────────\n_schema_doc = {\n    \"version\"     : \"1.0\",\n    \"description\" : \"Immutable joint ordering contract for the sign-language motion dataset.\",\n    \"total_joints\": 113,\n    \"coordinate_axes\": [\"x\", \"y\", \"z\"],\n    \"coordinate_system\": {\n        \"origin\"    : \"nose (joint index 0)\",\n        \"scale\"     : \"body_stddev\",\n        \"normalized\": True,\n    },\n    \"groups\": {\n        \"pose\"       : {\"offset\":  0, \"count\": 11, \"joints\": _POSE_JOINT_NAMES},\n        \"face\": {\n            \"offset\": 11, \"count\": 60,\n            \"sub_groups\": {\n                \"lips_outer\": {\"offset\": 11, \"count\": 20},\n                \"lips_inner\": {\"offset\": 31, \"count\": 20},\n                \"left_eye\"  : {\"offset\": 51, \"count\":  8},\n                \"right_eye\" : {\"offset\": 59, \"count\":  8},\n                \"nose\"      : {\"offset\": 67, \"count\":  4},\n            },\n        },\n        \"left_hand\"  : {\"offset\": 71, \"count\": 21, \"joints\": _HAND_JOINT_NAMES},\n        \"right_hand\" : {\"offset\": 92, \"count\": 21, \"joints\": _HAND_JOINT_NAMES},\n    },\n    \"hand_connections\": HAND_CONNECTIONS,\n    \"joint_names_flat\": JOINT_NAMES,\n}\n\n_schema_path = os.path.join(OUTPUT_DIR, 'joint_schema.json')\nwith open(_schema_path, 'w') as _f:\n    json.dump(_schema_doc, _f, indent=2)\n\nprint(f'✅ Joint schema saved → {_schema_path}')\nprint(f'   Total joints : {len(JOINT_NAMES)}')\nprint(f'   Groups       : {list(SCHEMA_GROUPS.keys())}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:07.986492Z","iopub.execute_input":"2026-04-13T21:50:07.986685Z","iopub.status.idle":"2026-04-13T21:50:08.006566Z","shell.execute_reply.started":"2026-04-13T21:50:07.986668Z","shell.execute_reply":"2026-04-13T21:50:08.005807Z"}},"outputs":[],"execution_count":null},{"id":"bb0600c9-a408-4a33-934e-4fa592492613","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 4 · RAW LANDMARK EXTRACTION\n# ════════════════════════════════════════════════\n\"\"\"\nReads one parquet file and returns a (T, 113, 3) matrix\nin the RAW layout used by the original notebook:\n  cols 0:40   → lips\n  cols 40:56  → eyes\n  cols 56:60  → nose\n  cols 60:81  → left_hand\n  cols 81:102 → right_hand\n  cols 102:113→ pose\n\nThis raw matrix is later remapped to the schema order in § 8.\n\"\"\"\n\ndef _extract_type(df: pd.DataFrame, frame: int, lm_type: str,\n                  indices: list[int]) -> np.ndarray:\n    \"\"\"Return (len(indices), 3) array of xyz for given type+frame.\"\"\"\n    mask = (df['frame'] == frame) & (df['type'] == lm_type)\n    sub  = df[mask].sort_values('landmark_index')\n    sub  = sub[sub['landmark_index'].isin(indices)]\n    if len(sub) == 0:\n        return np.full((len(indices), 3), np.nan)\n    # reindex to requested order (preserves original landmark ordering)\n    idx_map = {r['landmark_index']: r[['x','y','z']].values\n               for _, r in sub.iterrows()}\n    return np.array([idx_map.get(i, [np.nan]*3) for i in indices],\n                    dtype=np.float32)\n\n\ndef build_raw_frame(df: pd.DataFrame, frame: int) -> np.ndarray:\n    \"\"\"\n    Return (113, 3) raw array for one frame.\n    NaN where the landmark was not detected.\n    \"\"\"\n    lips  = _extract_type(df, frame, 'face',        LIP_LANDMARKS)   # 40\n    eyes  = _extract_type(df, frame, 'face',        EYE_LANDMARKS)   # 16\n    nose  = _extract_type(df, frame, 'face',        NOSE_LANDMARKS)  #  4\n    lhand = _extract_type(df, frame, 'left_hand',   HAND_LANDMARKS)  # 21\n    rhand = _extract_type(df, frame, 'right_hand',  HAND_LANDMARKS)  # 21\n    pose  = _extract_type(df, frame, 'pose',        POSE_LANDMARKS)  # 11\n    return np.concatenate([lips, eyes, nose, lhand, rhand, pose], axis=0)\n\n\ndef load_raw_matrix(parquet_path: str) -> tuple[np.ndarray, list[int]]:\n    \"\"\"\n    Load a parquet file and return:\n      matrix : (T, 113, 3)  float32  — raw layout\n      frames : list of frame indices\n    \"\"\"\n    df     = pd.read_parquet(parquet_path)\n    frames = sorted(df['frame'].unique().tolist())\n    matrix = np.stack([build_raw_frame(df, f) for f in frames], axis=0)\n    return matrix.astype(np.float32), frames\n\n\ndef pick_best_sample(df_train: pd.DataFrame, sign: str,\n                     n_candidates: int = 10) -> str:\n    \"\"\"\n    Among the first `n_candidates` samples for `sign`,\n    return the parquet path with the lowest NaN percentage\n    (= cleanest motion capture data).\n    \"\"\"\n    rows     = df_train[df_train['sign'] == sign].head(n_candidates)\n    best_path, best_nan = None, 1.0\n    for _, row in rows.iterrows():\n        path = os.path.join(DATA_ROOT, row['path'])\n        try:\n            df_lm  = pd.read_parquet(path)\n            nan_pc = float(df_lm[['x','y','z']].isna().mean().mean())\n            if nan_pc < best_nan:\n                best_nan, best_path = nan_pc, path\n        except Exception:\n            continue\n    return best_path\n\n\nprint('✅ Extraction functions defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.008229Z","iopub.execute_input":"2026-04-13T21:50:08.008518Z","iopub.status.idle":"2026-04-13T21:50:08.022561Z","shell.execute_reply.started":"2026-04-13T21:50:08.008499Z","shell.execute_reply":"2026-04-13T21:50:08.021993Z"}},"outputs":[],"execution_count":null},{"id":"b4952efb-8838-467f-b2b9-e5b542197c5d","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 5 · ACTIVE REGION DETECTION\n# ════════════════════════════════════════════════\n\"\"\"\nSigns are recorded with idle padding at start and end.\nWe detect the \"active\" window where meaningful motion occurs\nby computing per-frame velocity and thresholding.\n\"\"\"\n\ndef detect_active_region(matrix: np.ndarray,\n                          threshold: float = ACTIVITY_THRESHOLD,\n                          padding_val: float = PADDING_VALUE\n                         ) -> tuple[int, int]:\n    \"\"\"\n    Returns (start_frame, end_frame) of the active motion window.\n    Falls back to full sequence if nothing passes the threshold.\n\n    Parameters\n    ----------\n    matrix    : (T, 113, 3)\n    threshold : minimum mean absolute velocity to be considered active\n    \"\"\"\n    # ignore hard-padding frames\n    not_padded = ~np.all(np.abs(matrix - padding_val) < 0.1, axis=(1, 2))\n\n    pos      = matrix[:, :, :2]                                # xy only\n    vel      = np.abs(np.diff(pos, axis=0, prepend=pos[:1]))   # (T,113,2)\n    activity = np.nanmean(vel, axis=(1, 2))                    # (T,)\n\n    active = (activity > threshold) & not_padded\n\n    if not active.any():\n        valid = np.where(not_padded)[0]\n        return (int(valid[0]), int(valid[-1])) if len(valid) else (0, len(matrix)-1)\n\n    start = int(np.argmax(active))\n    end   = int(len(active) - np.argmax(active[::-1]) - 1)\n    return start, end\n\n\ndef trim_to_active(matrix: np.ndarray) -> tuple[np.ndarray, int, int]:\n    \"\"\"\n    Trim matrix to its active window.\n    Returns trimmed matrix + original start/end indices.\n    \"\"\"\n    start, end  = detect_active_region(matrix)\n    trimmed     = matrix[start:end+1]\n    return trimmed, start, end\n\n\nprint('✅ Active-region detection defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.023423Z","iopub.execute_input":"2026-04-13T21:50:08.023675Z","iopub.status.idle":"2026-04-13T21:50:08.037936Z","shell.execute_reply.started":"2026-04-13T21:50:08.023646Z","shell.execute_reply":"2026-04-13T21:50:08.037443Z"}},"outputs":[],"execution_count":null},{"id":"46c73847-ef11-4d10-a43d-1baa4222a7dc","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 6 · HANDEDNESS DETECTION\n# ════════════════════════════════════════════════\n\"\"\"\nA hand is considered \"absent\" in a sign if its landmarks are NaN\nfor ≥ HAND_ABSENCE_THRESHOLD of all frames.\n\nRaw matrix columns:  lhand → 60:81,  rhand → 81:102\n\"\"\"\n\n_LHAND_SLICE = slice(60, 81)   # in raw matrix\n_RHAND_SLICE = slice(81, 102)  # in raw matrix\n\n\ndef _hand_nan_fraction(matrix: np.ndarray, hand_slice: slice) -> float:\n    \"\"\"Fraction of (frame, joint) cells that are NaN for the given hand.\"\"\"\n    chunk = matrix[:, hand_slice, :]\n    return float(np.isnan(chunk).mean())\n\n\ndef detect_handedness(matrix: np.ndarray,\n                      threshold: float = HAND_ABSENCE_THRESHOLD\n                     ) -> dict:\n    \"\"\"\n    Returns a dict with:\n      left_present  : bool\n      right_present : bool\n      label         : 'both' | 'left' | 'right' | 'none'\n      left_nan_pct  : float   (percentage of NaN, 0-100)\n      right_nan_pct : float\n    \"\"\"\n    l_nan = _hand_nan_fraction(matrix, _LHAND_SLICE)\n    r_nan = _hand_nan_fraction(matrix, _RHAND_SLICE)\n\n    l_present = l_nan < threshold\n    r_present = r_nan < threshold\n\n    if l_present and r_present:\n        label = 'both'\n    elif l_present:\n        label = 'left'\n    elif r_present:\n        label = 'right'\n    else:\n        label = 'none'\n\n    return {\n        'left_present' : l_present,\n        'right_present': r_present,\n        'label'        : label,\n        'left_nan_pct' : round(l_nan  * 100, 2),\n        'right_nan_pct': round(r_nan  * 100, 2),\n    }\n\n\nprint('✅ Handedness detection defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.038861Z","iopub.execute_input":"2026-04-13T21:50:08.039175Z","iopub.status.idle":"2026-04-13T21:50:08.053157Z","shell.execute_reply.started":"2026-04-13T21:50:08.039155Z","shell.execute_reply":"2026-04-13T21:50:08.052427Z"}},"outputs":[],"execution_count":null},{"id":"d3b346b7-fd96-4be3-993e-66ab3ce78ad3","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 7 · NaN HANDLING & REST POSE\n# ════════════════════════════════════════════════\n\"\"\"\nStrategy:\n  1. For joints that have SOME valid frames → linear interpolation,\n     then forward/backward fill for edges.\n  2. For hands that are absent in this sign (present=False):\n     → rest pose (wrist near hip, fingers naturally extended downward).\n  3. Any remaining NaN after step 1-2 → median of the sequence.\n\"\"\"\n\ndef _interpolate_joint(seq: np.ndarray) -> np.ndarray:\n    \"\"\"\n    seq : (T, 3)  — single joint over time, may contain NaN.\n    Returns interpolated array. If fully NaN, returns unchanged.\n    \"\"\"\n    T     = seq.shape[0]\n    times = np.arange(T, dtype=float)\n    out   = seq.copy()\n\n    for c in range(3):       # x, y, z independently\n        col   = seq[:, c]\n        valid = ~np.isnan(col)\n        if valid.sum() < 2:\n            continue         # not enough points — handled below\n        f   = sci_interp.interp1d(times[valid], col[valid],\n                                   kind='linear', fill_value='extrapolate')\n        out[:, c] = f(times)\n\n    return out\n\n\ndef interpolate_matrix(matrix: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Apply linear interpolation joint-wise across all 113 joints.\n    Returns (T, 113, 3) with far fewer NaNs.\n    \"\"\"\n    out = matrix.copy()\n    for j in range(matrix.shape[1]):\n        out[:, j, :] = _interpolate_joint(matrix[:, j, :])\n    return out\n\n\ndef _compute_rest_pose_hand(matrix: np.ndarray, hand_slice: slice,\n                              pose_wrist_idx: int) -> np.ndarray:\n    \"\"\"\n    Build a (T, 21, 3) rest-pose block for an absent hand.\n\n    Logic:\n      - Wrist is placed near the corresponding shoulder/hip:\n        we use the pose wrist landmark (col pose_wrist_idx in raw matrix)\n        and offset slightly outward + downward.\n      - Fingers hang naturally downward (small offsets in y).\n    \"\"\"\n    T   = matrix.shape[0]\n    out = np.zeros((T, 21, 3), dtype=np.float32)\n\n    # Try to use pose wrist as anchor\n    pose_block = matrix[:, 102:113, :]                      # pose in raw\n    wrist_raw  = pose_block[:, pose_wrist_idx, :]           # (T, 3)\n\n    # If pose wrist also NaN, fall back to zero\n    anchor = np.where(\n        np.isnan(wrist_raw),\n        np.zeros_like(wrist_raw),\n        wrist_raw\n    )  # (T, 3)\n\n    # Finger rest offsets (y = downward in normalised space)\n    _finger_y = [0.00, 0.04, 0.08, 0.12, 0.16,   # wrist + thumb\n                 0.05, 0.10, 0.15, 0.20,           # index\n                 0.05, 0.10, 0.15, 0.20,           # middle\n                 0.05, 0.10, 0.15, 0.20,           # ring\n                 0.05, 0.10, 0.15, 0.20]           # pinky\n\n    for j, dy in enumerate(_finger_y):\n        out[:, j, :] = anchor\n        out[:, j, 1] += dy      # shift each joint downward\n\n    return out\n\n\ndef clean_matrix(matrix: np.ndarray,\n                 handedness: dict) -> np.ndarray:\n    \"\"\"\n    Full NaN-cleaning pass:\n      1. Interpolate all joints.\n      2. Set absent-hand blocks to anatomical rest pose.\n      3. Fill any residual NaN with per-joint median.\n\n    Returns clean (T, 113, 3) — no NaN.\n    \"\"\"\n    # Step 1 · interpolation\n    m = interpolate_matrix(matrix)\n\n    # Step 2 · rest pose for absent hands\n    # pose_wrist_idx inside pose block (102:113):\n    #   POSE_LANDMARKS = [0,1,2,3,4,11,12,13,14,15,16]\n    #   index 9 = left_wrist (landmark 15), index 10 = right_wrist (landmark 16)\n    if not handedness['left_present']:\n        m[:, 60:81, :] = _compute_rest_pose_hand(m, slice(60, 81),\n                                                   pose_wrist_idx=9)\n\n    if not handedness['right_present']:\n        m[:, 81:102, :] = _compute_rest_pose_hand(m, slice(81, 102),\n                                                    pose_wrist_idx=10)\n\n    # Step 3 · residual NaN → per-joint median\n    for j in range(m.shape[1]):\n        for c in range(3):\n            col = m[:, j, c]\n            bad = np.isnan(col)\n            if bad.any():\n                med = float(np.nanmedian(col)) if not np.all(bad) else 0.0\n                m[:, j, c] = np.where(bad, med, col)\n\n    return m\n\n\nprint('✅ NaN handling & rest-pose functions defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.054058Z","iopub.execute_input":"2026-04-13T21:50:08.054649Z","iopub.status.idle":"2026-04-13T21:50:08.069930Z","shell.execute_reply.started":"2026-04-13T21:50:08.054629Z","shell.execute_reply":"2026-04-13T21:50:08.069206Z"}},"outputs":[],"execution_count":null},{"id":"65c08cd3-881a-4e7e-8a7c-429ba27563ec","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 8 · NORMALISATION & SCHEMA REMAP\n# ════════════════════════════════════════════════\n\"\"\"\nTwo operations performed together:\n\n(A) Normalise coordinates\n    · Translate: origin → nose (raw index 56, first of NOSE_LANDMARKS)\n    · Scale    : divide by std-dev of all xy joints (body scale)\n\n(B) Remap raw layout → schema layout\n    Raw order : lips(40) eyes(16) nose(4) lhand(21) rhand(21) pose(11)\n    Schema    : pose(11) lips_outer(20) lips_inner(20) left_eye(8)\n                right_eye(8) nose(4) lhand(21) rhand(21)\n\"\"\"\n\n# ──  Nose reference index in the RAW matrix  ─────────────────────────────\n# After lips(40)+eyes(16), nose starts at index 56.\n# NOSE_LANDMARKS = [1,2,98,327] — we use index 0 of that sub-array as ref.\n_RAW_NOSE_IDX = 56   # first nose point in raw (113,3) array\n\n# ──  Remap: raw → schema  ─────────────────────────────────────────────────\n# Schema layout indices mapped FROM raw matrix indices\n_SCHEMA_FROM_RAW = (\n    list(range(102, 113))          # pose      (11) ← raw 102:113\n    + list(range(0,   20))         # lips_outer(20) ← raw 0:20\n    + list(range(20,  40))         # lips_inner(20) ← raw 20:40\n    + list(range(40,  48))         # left_eye  ( 8) ← raw 40:48\n    + list(range(48,  56))         # right_eye ( 8) ← raw 48:56\n    + list(range(56,  60))         # nose      ( 4) ← raw 56:60\n    + list(range(60,  81))         # left_hand (21) ← raw 60:81\n    + list(range(81, 102))         # right_hand(21) ← raw 81:102\n)\n\nassert len(_SCHEMA_FROM_RAW) == 113, \"Remap index list length mismatch\"\n\n\ndef normalise_and_remap(matrix: np.ndarray) -> np.ndarray:\n    \"\"\"\n    (T, 113, 3) raw  →  (T, 113, 3) schema-ordered, normalised.\n\n    Coordinate system:\n        Origin  = nose tip (raw index 56)\n        Scale   = std-dev of all (x, y) values across the sequence\n        z       = kept but not used in scale calculation\n    \"\"\"\n    m   = matrix.copy().astype(np.float64)\n\n    # ── (A) Normalise ──────────────────────────────────────────────────\n    nose_ref = m[:, _RAW_NOSE_IDX:_RAW_NOSE_IDX+1, :2]   # (T, 1, 2)\n    m[:, :, :2] -= nose_ref                                # translate\n\n    std = np.nanstd(m[:, :, :2])\n    if std > 1e-8:\n        m[:, :, :2] /= std                                 # scale\n\n    # ── (B) Remap to schema order ──────────────────────────────────────\n    return m[:, _SCHEMA_FROM_RAW, :].astype(np.float32)\n\n\nprint('✅ Normalisation & schema-remap defined.')\nprint(f'   Schema remap list has {len(_SCHEMA_FROM_RAW)} entries.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.071012Z","iopub.execute_input":"2026-04-13T21:50:08.071307Z","iopub.status.idle":"2026-04-13T21:50:08.086594Z","shell.execute_reply.started":"2026-04-13T21:50:08.071279Z","shell.execute_reply":"2026-04-13T21:50:08.086041Z"}},"outputs":[],"execution_count":null},{"id":"116508b8-d4f6-49c9-bac1-d60684def97c","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 9 · RESAMPLING (OPTIONAL FIXED LENGTH)\n# ════════════════════════════════════════════════\n\"\"\"\nThe dataset stores VARIABLE length (active frames only).\nA utility is provided to resample any sign to a fixed number\nof frames via linear interpolation — useful for batching.\n\"\"\"\n\ndef resample_sequence(matrix: np.ndarray, target_len: int) -> np.ndarray:\n    \"\"\"\n    Resample (T, 113, 3) to (target_len, 113, 3) using linear interpolation.\n    Works on both x,y,z simultaneously.\n    \"\"\"\n    T = matrix.shape[0]\n    if T == target_len:\n        return matrix.copy()\n\n    t_src = np.linspace(0, 1, T)\n    t_dst = np.linspace(0, 1, target_len)\n    out   = np.empty((target_len, 113, 3), dtype=np.float32)\n\n    for j in range(113):\n        for c in range(3):\n            out[:, j, c] = np.interp(t_dst, t_src, matrix[:, j, c])\n\n    return out\n\n\nprint('✅ Resampling utility defined.')\nprint(f'   Default fixed length = {FIXED_LENGTH_OPTION} frames ({FIXED_LENGTH_OPTION/FPS:.1f}s @ {FPS}fps)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.087505Z","iopub.execute_input":"2026-04-13T21:50:08.087799Z","iopub.status.idle":"2026-04-13T21:50:08.101817Z","shell.execute_reply.started":"2026-04-13T21:50:08.087778Z","shell.execute_reply":"2026-04-13T21:50:08.101315Z"}},"outputs":[],"execution_count":null},{"id":"af783737-24ff-4d78-a146-58e23622cc9a","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 10 · STRUCTURED JSON FRAME BUILDER\n# ════════════════════════════════════════════════\n\"\"\"\nConverts the clean (T, 113, 3) schema-ordered matrix into the\nhuman-readable JSON structure for one sign.\n\nFrame layout (schema order, 113 joints):\n  joints[0:11]   → pose joints (named via _POSE_JOINT_NAMES)\n  joints[11:71]  → face (lips_outer, lips_inner, eyes, nose)\n  joints[71:92]  → left_hand  (wrist + 5 fingers)\n  joints[92:113] → right_hand (wrist + 5 fingers)\n\"\"\"\n\n\ndef _hand_block(joints_21: np.ndarray, present: bool) -> dict:\n    \"\"\"\n    Build a semantically-labelled hand dict from 21 xyz joints.\n    joints_21 : (21, 3)\n    \"\"\"\n    j = joints_21\n    return {\n        \"present\" : present,\n        \"wrist\"   : j[0].tolist(),\n        \"thumb\"   : {\"cmc\": j[1].tolist(), \"mcp\": j[2].tolist(),\n                     \"ip\" : j[3].tolist(), \"tip\": j[4].tolist()},\n        \"index\"   : {\"mcp\": j[5].tolist(), \"pip\": j[6].tolist(),\n                     \"dip\": j[7].tolist(), \"tip\": j[8].tolist()},\n        \"middle\"  : {\"mcp\": j[9].tolist(), \"pip\": j[10].tolist(),\n                     \"dip\": j[11].tolist(),\"tip\": j[12].tolist()},\n        \"ring\"    : {\"mcp\": j[13].tolist(),\"pip\": j[14].tolist(),\n                     \"dip\": j[15].tolist(),\"tip\": j[16].tolist()},\n        \"pinky\"   : {\"mcp\": j[17].tolist(),\"pip\": j[18].tolist(),\n                     \"dip\": j[19].tolist(),\"tip\": j[20].tolist()},\n    }\n\n\ndef _face_block(joints_60: np.ndarray) -> dict:\n    \"\"\"\n    Build face dict from 60 xyz joints.\n    joints_60 : (60, 3)\n    \"\"\"\n    return {\n        \"lips_outer\": joints_60[0:20].tolist(),\n        \"lips_inner\": joints_60[20:40].tolist(),\n        \"left_eye\"  : joints_60[40:48].tolist(),\n        \"right_eye\" : joints_60[48:56].tolist(),\n        \"nose\"      : joints_60[56:60].tolist(),\n    }\n\n\ndef _pose_block(joints_11: np.ndarray) -> dict:\n    \"\"\"Named pose joints from 11 xyz joints.\"\"\"\n    return {name: joints_11[i].tolist()\n            for i, name in enumerate(_POSE_JOINT_NAMES)}\n\n\ndef build_frame(t: int, joints_113: np.ndarray,\n                l_present: bool, r_present: bool) -> dict:\n    \"\"\"\n    Build one animation frame dict.\n\n    Parameters\n    ----------\n    t            : frame index (0-based within active window)\n    joints_113   : (113, 3)  schema-ordered, cleaned, normalised\n    l_present    : left hand present in this sign\n    r_present    : right hand present in this sign\n    \"\"\"\n    return {\n        \"t\"          : t,\n        \"pose\"       : _pose_block(joints_113[0:11]),\n        \"face\"       : _face_block(joints_113[11:71]),\n        \"left_hand\"  : _hand_block(joints_113[71:92],  l_present),\n        \"right_hand\" : _hand_block(joints_113[92:113], r_present),\n    }\n\n\ndef build_sign_doc(sign: str, matrix_schema: np.ndarray,\n                   handedness: dict, active_start: int,\n                   active_end: int, raw_num_frames: int,\n                   nan_stats: dict) -> dict:\n    \"\"\"\n    Build the complete JSON document for one sign.\n\n    Parameters\n    ----------\n    sign           : word label\n    matrix_schema  : (T, 113, 3) — clean, normalised, schema-ordered\n    handedness     : output of detect_handedness()\n    active_start   : start frame in the ORIGINAL sequence\n    active_end     : end frame in the ORIGINAL sequence\n    raw_num_frames : total frames before trimming\n    nan_stats      : NaN percentages before cleaning\n    \"\"\"\n    T = matrix_schema.shape[0]\n    l = handedness['left_present']\n    r = handedness['right_present']\n\n    frames = [build_frame(t, matrix_schema[t], l, r) for t in range(T)]\n\n    return {\n        \"sign\"              : sign,\n        \"handedness\"        : handedness['label'],\n        \"num_frames\"        : T,\n        \"duration_sec\"      : round(T / FPS, 3),\n        \"original_num_frames\": raw_num_frames,\n        \"active_range\"      : {\"start\": active_start, \"end\": active_end},\n        \"fixed_length_option\": FIXED_LENGTH_OPTION,\n        \"coordinate_system\" : {\n            \"origin\"    : \"nose\",\n            \"scale\"     : \"body_stddev\",\n            \"normalized\": True,\n        },\n        \"nan_stats_before_cleaning\": nan_stats,\n        \"frames\"            : frames,\n    }\n\n\nprint('✅ JSON frame-builder defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.103933Z","iopub.execute_input":"2026-04-13T21:50:08.104301Z","iopub.status.idle":"2026-04-13T21:50:08.120772Z","shell.execute_reply.started":"2026-04-13T21:50:08.104279Z","shell.execute_reply":"2026-04-13T21:50:08.120039Z"}},"outputs":[],"execution_count":null},{"id":"368bee2b-b59f-46b6-9214-2053e4fce2f7","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 11 · FULL PROCESSING PIPELINE (PER SIGN)\n# ════════════════════════════════════════════════\n\"\"\"\nOrchestrates §4-§10 into a single function that:\n  1. Picks the cleanest sample for the sign\n  2. Loads raw parquet\n  3. Detects active region\n  4. Detects handedness\n  5. Cleans NaN\n  6. Normalises & remaps to schema\n  7. Builds and returns the sign document\n\"\"\"\n\ndef process_sign(sign: str,\n                 df_train: pd.DataFrame,\n                 n_candidates: int = 10) -> dict | None:\n    \"\"\"\n    Full pipeline for one sign word.\n    Returns the sign document dict, or None on failure.\n    \"\"\"\n    # ── 1 · pick best sample ─────────────────────────────────────────\n    path = pick_best_sample(df_train, sign, n_candidates)\n    if path is None:\n        print(f'  ⚠️  {sign}: no valid parquet found — skipped.')\n        return None\n\n    # ── 2 · load raw ─────────────────────────────────────────────────\n    try:\n        raw, _ = load_raw_matrix(path)\n    except Exception as e:\n        print(f'  ⚠️  {sign}: load error — {e}')\n        return None\n\n    raw_num_frames = raw.shape[0]\n\n    # ── 3 · detect active region ─────────────────────────────────────\n    trimmed, active_start, active_end = trim_to_active(raw)\n\n    # ── 4 · detect handedness (on trimmed, before cleaning) ──────────\n    handedness = detect_handedness(trimmed)\n\n    # ── 5 · record NaN stats BEFORE cleaning ─────────────────────────\n    nan_stats = {\n        \"left_hand_nan_pct\" : handedness['left_nan_pct'],\n        \"right_hand_nan_pct\": handedness['right_nan_pct'],\n        \"face_nan_pct\"      : round(float(\n            np.isnan(trimmed[:, 0:60, :]).mean()) * 100, 2),\n    }\n\n    # ── 6 · clean NaN ────────────────────────────────────────────────\n    cleaned = clean_matrix(trimmed, handedness)\n\n    # ── 7 · normalise + remap to schema ──────────────────────────────\n    schema_mat = normalise_and_remap(cleaned)\n\n    # ── 8 · build sign document ──────────────────────────────────────\n    doc = build_sign_doc(\n        sign        = sign,\n        matrix_schema   = schema_mat,\n        handedness  = handedness,\n        active_start    = active_start,\n        active_end  = active_end,\n        raw_num_frames  = raw_num_frames,\n        nan_stats   = nan_stats,\n    )\n    return doc\n\n\nprint('✅ Full pipeline function defined.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.121728Z","iopub.execute_input":"2026-04-13T21:50:08.122085Z","iopub.status.idle":"2026-04-13T21:50:08.138199Z","shell.execute_reply.started":"2026-04-13T21:50:08.122066Z","shell.execute_reply":"2026-04-13T21:50:08.137551Z"}},"outputs":[],"execution_count":null},{"id":"9628899c-d4ee-439a-95fc-0cb03b345fb0","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 12 · RUN — PROCESS ALL SIGNS\n# ════════════════════════════════════════════════\n\ndf_train = pd.read_csv(TRAIN_CSV)\nall_signs = sorted(df_train['sign'].unique().tolist())\n\nif TARGET_SIGNS is not None:\n    signs_to_process = [s for s in TARGET_SIGNS if s in all_signs]\n    print(f'🎯 Targeted mode: {len(signs_to_process)} signs')\nelse:\n    signs_to_process = all_signs\n    print(f'🌐 Full mode: {len(signs_to_process)} signs')\n\n# ──  Process ─────────────────────────────────────────────────────────────\nresults = {}     # sign → doc\nfailures = []\n\nfor sign in tqdm(signs_to_process, desc='Processing signs'):\n    doc = process_sign(sign, df_train)\n    if doc is None:\n        failures.append(sign)\n        continue\n\n    # save individual sign file\n    out_path = os.path.join(OUTPUT_DIR, 'signs', f'{sign}.json')\n    with open(out_path, 'w') as f:\n        json.dump(doc, f, separators=(',', ':'))   # compact — no spaces\n\n    results[sign] = doc\n\nprint(f'\\n✅ Done.')\nprint(f'   Processed : {len(results):>4} signs')\nprint(f'   Failed    : {len(failures):>4} signs')\nif failures:\n    print(f'   Failed list: {failures}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:50:08.139127Z","iopub.execute_input":"2026-04-13T21:50:08.139420Z","iopub.status.idle":"2026-04-13T21:57:44.858086Z","shell.execute_reply.started":"2026-04-13T21:50:08.139400Z","shell.execute_reply":"2026-04-13T21:57:44.857161Z"}},"outputs":[],"execution_count":null},{"id":"6669a835-5008-4153-ac07-17bb828ab485","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 13 · METADATA INDEX\n# ════════════════════════════════════════════════\n\"\"\"\nmetadata.json lets any consumer quickly look up a sign\nwithout parsing the full sign file.\n\"\"\"\n\nindex_entries = []\nfor sign, doc in results.items():\n    index_entries.append({\n        \"sign\"       : sign,\n        \"handedness\" : doc[\"handedness\"],\n        \"num_frames\" : doc[\"num_frames\"],\n        \"duration_sec\": doc[\"duration_sec\"],\n        \"file\"       : f\"signs/{sign}.json\",\n    })\n\nmetadata = {\n    \"dataset_version\"       : \"1.0\",\n    \"description\"           : \"Structured sign-language motion dataset for 3D animation.\",\n    \"schema_file\"           : \"joint_schema.json\",\n    \"fps\"                   : FPS,\n    \"fixed_length_option\"   : FIXED_LENGTH_OPTION,\n    \"hand_absence_threshold\": HAND_ABSENCE_THRESHOLD,\n    \"coordinate_system\"     : {\n        \"origin\"    : \"nose\",\n        \"scale\"     : \"body_stddev\",\n        \"normalized\": True,\n    },\n    \"num_signs\"    : len(results),\n    \"handedness_counts\": {\n        k: sum(1 for e in index_entries if e[\"handedness\"] == k)\n        for k in [\"both\", \"right\", \"left\", \"none\"]\n    },\n    \"frame_stats\" : {\n        \"min\" : int(min(e[\"num_frames\"] for e in index_entries)),\n        \"max\" : int(max(e[\"num_frames\"] for e in index_entries)),\n        \"mean\": round(float(np.mean([e[\"num_frames\"] for e in index_entries])), 1),\n    },\n    \"signs\": index_entries,\n}\n\nmeta_path = os.path.join(OUTPUT_DIR, 'metadata.json')\nwith open(meta_path, 'w') as f:\n    json.dump(metadata, f, indent=2)\n\nprint(f'✅ metadata.json saved → {meta_path}')\nprint()\nprint(f'   Signs processed  : {metadata[\"num_signs\"]}')\nprint(f'   Handedness dist  : {metadata[\"handedness_counts\"]}')\nprint(f'   Frame range      : {metadata[\"frame_stats\"][\"min\"]} – '\n      f'{metadata[\"frame_stats\"][\"max\"]} '\n      f'(mean {metadata[\"frame_stats\"][\"mean\"]})')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:57:44.859175Z","iopub.execute_input":"2026-04-13T21:57:44.859437Z","iopub.status.idle":"2026-04-13T21:57:44.871304Z","shell.execute_reply.started":"2026-04-13T21:57:44.859414Z","shell.execute_reply":"2026-04-13T21:57:44.870644Z"}},"outputs":[],"execution_count":null},{"id":"adbf7bb5-834b-4a16-a864-1573bbc171db","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 14 · CSV EXPORT (flat inspection format)\n# ════════════════════════════════════════════════\n\"\"\"\nFlat CSV — one row per (sign, frame, joint).\nColumns: sign | handedness | frame_t | joint_name | x | y | z\n\nUseful for:\n  • Quick pandas queries\n  • Sanity-checking individual joint trajectories\n  • Prototype experiments before reading full JSON\n\"\"\"\n\nrows = []\nfor sign, doc in results.items():\n    h = doc[\"handedness\"]\n    for frame in doc[\"frames\"]:\n        t = frame[\"t\"]\n        # Rebuild flat joints list in schema order from the frame dict\n        pose_joints = list(frame[\"pose\"].values())   # 11 joints as [x,y,z]\n\n        face = frame[\"face\"]\n        face_joints = (face[\"lips_outer\"] + face[\"lips_inner\"]\n                       + face[\"left_eye\"] + face[\"right_eye\"]\n                       + face[\"nose\"])               # 60 joints\n\n        lh = frame[\"left_hand\"]\n        lh_joints = (\n            [lh[\"wrist\"]]\n            + list(lh[\"thumb\"].values())\n            + list(lh[\"index\"].values())\n            + list(lh[\"middle\"].values())\n            + list(lh[\"ring\"].values())\n            + list(lh[\"pinky\"].values())\n        )  # 21 joints\n\n        rh = frame[\"right_hand\"]\n        rh_joints = (\n            [rh[\"wrist\"]]\n            + list(rh[\"thumb\"].values())\n            + list(rh[\"index\"].values())\n            + list(rh[\"middle\"].values())\n            + list(rh[\"ring\"].values())\n            + list(rh[\"pinky\"].values())\n        )  # 21 joints\n\n        all_joints = pose_joints + face_joints + lh_joints + rh_joints\n        for name, xyz in zip(JOINT_NAMES, all_joints):\n            rows.append((sign, h, t, name, xyz[0], xyz[1], xyz[2]))\n\ndf_flat = pd.DataFrame(rows,\n    columns=[\"sign\",\"handedness\",\"frame_t\",\"joint_name\",\"x\",\"y\",\"z\"])\n\ncsv_path = os.path.join(OUTPUT_DIR, 'motion_dataset.csv')\ndf_flat.to_csv(csv_path, index=False)\n\nprint(f'✅ CSV saved → {csv_path}')\nprint(f'   Shape : {df_flat.shape}')\nprint(f'   Memory: {df_flat.memory_usage(deep=True).sum()/1e6:.1f} MB')\nprint()\nprint(df_flat.head(10).to_string(index=False))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:57:44.872455Z","iopub.execute_input":"2026-04-13T21:57:44.872749Z","iopub.status.idle":"2026-04-13T21:57:48.145403Z","shell.execute_reply.started":"2026-04-13T21:57:44.872729Z","shell.execute_reply":"2026-04-13T21:57:48.144482Z"}},"outputs":[],"execution_count":null},{"id":"5b1545cd-0c61-45fd-9749-fa01b1163312","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 15 · VALIDATION & STATISTICS\n# ════════════════════════════════════════════════\n\"\"\"\nThree validation checks:\n  A. NaN audit   — no NaN should remain in any sign file\n  B. Range audit — all values within expected normalised range\n  C. Schema audit— joint count and name consistency\n\"\"\"\n\nprint(\"=\" * 60)\nprint(\"  VALIDATION REPORT\")\nprint(\"=\" * 60)\n\nnan_violations   = []\nrange_violations = []\nschema_violations = []\n\n_EXPECTED_RANGE = (-15.0, 15.0)    # normalised xyz should stay within this\n\nfor sign, doc in results.items():\n    T = doc[\"num_frames\"]\n    for frame in doc[\"frames\"]:\n        # Rebuild flat array\n        all_xyz = []\n        all_xyz += list(frame[\"pose\"].values())\n        face = frame[\"face\"]\n        all_xyz += face[\"lips_outer\"] + face[\"lips_inner\"]\n        all_xyz += face[\"left_eye\"]   + face[\"right_eye\"] + face[\"nose\"]\n        lh = frame[\"left_hand\"]\n        rh = frame[\"right_hand\"]\n        for hand in [lh, rh]:\n            all_xyz += [hand[\"wrist\"]]\n            for finger in [\"thumb\",\"index\",\"middle\",\"ring\",\"pinky\"]:\n                all_xyz += list(hand[finger].values())\n\n        arr = np.array(all_xyz, dtype=np.float32)   # (113, 3)\n\n        # A · NaN check\n        if np.isnan(arr).any():\n            nan_violations.append(f\"{sign} frame {frame['t']}\")\n\n        # B · Range check\n        lo, hi = _EXPECTED_RANGE\n        if (arr < lo).any() or (arr > hi).any():\n            range_violations.append(f\"{sign} frame {frame['t']}\")\n\n        # C · Schema check\n        if len(all_xyz) != 113:\n            schema_violations.append(f\"{sign} frame {frame['t']} \"\n                                      f\"({len(all_xyz)} joints)\")\n\nprint(f\"  Signs validated    : {len(results)}\")\nprint(f\"  NaN violations     : {len(nan_violations)}\"\n      + (\" ✅\" if not nan_violations else f\" ❌  e.g. {nan_violations[:3]}\"))\nprint(f\"  Range violations   : {len(range_violations)}\"\n      + (\" ✅\" if not range_violations else f\" ⚠️  e.g. {range_violations[:3]}\"))\nprint(f\"  Schema violations  : {len(schema_violations)}\"\n      + (\" ✅\" if not schema_violations else f\" ❌  {schema_violations[:3]}\"))\nprint(\"=\" * 60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:57:48.146652Z","iopub.execute_input":"2026-04-13T21:57:48.147147Z","iopub.status.idle":"2026-04-13T21:57:48.394180Z","shell.execute_reply.started":"2026-04-13T21:57:48.147122Z","shell.execute_reply":"2026-04-13T21:57:48.393465Z"}},"outputs":[],"execution_count":null},{"id":"7b83d2ff-3360-42ee-9cf1-4a89209451e1","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 16 · STATISTICAL ANALYSIS PLOTS\n# ════════════════════════════════════════════════\n\nfig, axes = plt.subplots(2, 2, figsize=(14, 9))\nfig.patch.set_facecolor('#0f0f1a')\nfor ax in axes.flat:\n    ax.set_facecolor('#1a1a2e')\n    ax.tick_params(colors='white')\n    for spine in ax.spines.values():\n        spine.set_edgecolor('#2d3436')\n\nframe_counts = [doc[\"num_frames\"] for doc in results.values()]\ndurations    = [doc[\"duration_sec\"] for doc in results.values()]\nhandedness   = [doc[\"handedness\"] for doc in results.values()]\nl_nan        = [doc[\"nan_stats_before_cleaning\"][\"left_hand_nan_pct\"]\n                for doc in results.values()]\nr_nan        = [doc[\"nan_stats_before_cleaning\"][\"right_hand_nan_pct\"]\n                for doc in results.values()]\n\n# ── Plot A: Frame count distribution ────────────────────────────────────\nax = axes[0, 0]\nax.hist(frame_counts, bins=30, color='#74b9ff', edgecolor='#0984e3', alpha=0.85)\nax.axvline(np.mean(frame_counts), color='#fdcb6e', lw=2,\n           label=f'Mean: {np.mean(frame_counts):.0f}')\nax.axvline(FIXED_LENGTH_OPTION, color='#00b894', lw=2, ls='--',\n           label=f'Fixed option: {FIXED_LENGTH_OPTION}')\nax.set_title('Frame Count Distribution', color='white')\nax.set_xlabel('Frames', color='white')\nax.legend(facecolor='#2d3436', labelcolor='white', fontsize=8)\n\n# ── Plot B: Duration distribution ────────────────────────────────────────\nax = axes[0, 1]\nax.hist(durations, bins=30, color='#a29bfe', edgecolor='#6c5ce7', alpha=0.85)\nax.axvline(np.mean(durations), color='#fdcb6e', lw=2,\n           label=f'Mean: {np.mean(durations):.2f}s')\nax.set_title('Duration Distribution (seconds)', color='white')\nax.set_xlabel('Seconds', color='white')\nax.legend(facecolor='#2d3436', labelcolor='white', fontsize=8)\n\n# ── Plot C: Handedness pie ────────────────────────────────────────────────\nax = axes[1, 0]\nlabels  = ['both', 'right', 'left', 'none']\ncounts  = [handedness.count(l) for l in labels]\ncolors  = ['#00b894', '#e17055', '#74b9ff', '#636e72']\npatches = [mpatches.Patch(color=c, label=f'{l} ({n})')\n           for c, l, n in zip(colors, labels, counts) if n > 0]\nax.pie([c for c in counts if c > 0],\n       colors=[c for c, n in zip(colors, counts) if n > 0],\n       autopct='%1.0f%%', textprops={'color': 'white'})\nax.legend(handles=patches, facecolor='#2d3436', labelcolor='white',\n          fontsize=8, loc='lower right')\nax.set_title('Handedness Distribution', color='white')\n\n# ── Plot D: NaN % before cleaning ────────────────────────────────────────\nax = axes[1, 1]\nax.scatter(l_nan, r_nan, alpha=0.6, color='#fd79a8', s=20)\nax.axhline(HAND_ABSENCE_THRESHOLD * 100, color='#e17055', lw=1.5, ls='--',\n           label=f'Absence threshold ({HAND_ABSENCE_THRESHOLD*100:.0f}%)')\nax.axvline(HAND_ABSENCE_THRESHOLD * 100, color='#e17055', lw=1.5, ls='--')\nax.set_xlabel('Left hand NaN %', color='white')\nax.set_ylabel('Right hand NaN %', color='white')\nax.set_title('Hand NaN % Before Cleaning', color='white')\nax.legend(facecolor='#2d3436', labelcolor='white', fontsize=8)\n\nplt.suptitle('Motion Dataset — Statistical Overview',\n             color='white', fontsize=13, y=1.01)\nplt.tight_layout()\nplt.savefig(os.path.join(OUTPUT_DIR, 'dataset_stats.png'),\n            dpi=150, bbox_inches='tight', facecolor='#0f0f1a')\nplt.show()\nprint('✅ Stats plot saved.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:57:48.395188Z","iopub.execute_input":"2026-04-13T21:57:48.395505Z","iopub.status.idle":"2026-04-13T21:57:49.684449Z","shell.execute_reply.started":"2026-04-13T21:57:48.395469Z","shell.execute_reply":"2026-04-13T21:57:49.683773Z"}},"outputs":[],"execution_count":null},{"id":"37a3fcbb-be9e-4342-a969-705bc0c0dbd1","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 17 · SIGN PREVIEW — skeleton animation\n# ════════════════════════════════════════════════\n\"\"\"\nRenders a quick 2D skeleton preview for any sign in the dataset.\nReads directly from the JSON output to verify the full round-trip.\n\"\"\"\nimport matplotlib.animation as animation\n\nPREVIEW_SIGN = list(results.keys())[0]    # ← change to any sign name\n\ndef load_sign_json(sign: str) -> dict:\n    path = os.path.join(OUTPUT_DIR, 'signs', f'{sign}.json')\n    with open(path) as f:\n        return json.load(f)\n\n\ndef frame_to_array(frame: dict) -> np.ndarray:\n    \"\"\"Rebuild (113, 3) from a loaded JSON frame (schema order).\"\"\"\n    pose  = list(frame['pose'].values())\n    face  = frame['face']\n    lh    = frame['left_hand']\n    rh    = frame['right_hand']\n\n    face_pts = (face['lips_outer'] + face['lips_inner']\n                + face['left_eye'] + face['right_eye'] + face['nose'])\n\n    def hand_pts(h):\n        out = [h['wrist']]\n        for f_ in ['thumb','index','middle','ring','pinky']:\n            out += list(h[f_].values())\n        return out\n\n    all_pts = pose + face_pts + hand_pts(lh) + hand_pts(rh)\n    return np.array(all_pts, dtype=np.float32)\n\n\ndef draw_frame(ax, joints: np.ndarray, title: str = ''):\n    ax.clear()\n    ax.set_facecolor('#1a1a2e')\n    ax.set_xlim(-3, 3); ax.set_ylim(3, -3)\n    ax.set_aspect('equal'); ax.axis('off')\n    if title: ax.set_title(title, color='white', fontsize=9)\n\n    pose  = joints[0:11,  :2]\n    lhand = joints[71:92, :2]\n    rhand = joints[92:,   :2]\n    lips  = joints[11:31, :2]\n\n    def scatter(pts, col, s=18):\n        ax.scatter(pts[:,0], pts[:,1], c=col, s=s, alpha=0.85, zorder=4)\n\n    def lines(pts, conns, col, lw=1.8):\n        for i,j in conns:\n            if i<len(pts) and j<len(pts):\n                ax.plot([pts[i,0],pts[j,0]],[pts[i,1],pts[j,1]],\n                        color=col, lw=lw, alpha=0.75, zorder=3)\n\n    # pose skeleton\n    pose_conns = [(5,6),(5,7),(7,9),(6,8),(8,10)]\n    lines(pose, pose_conns, '#00b894', lw=3)\n    scatter(pose[5:], '#55efc4', 35)\n\n    # hands\n    lines(lhand, HAND_CONNECTIONS, '#74b9ff', lw=2)\n    scatter(lhand, '#0984e3', 25)\n    lines(rhand, HAND_CONNECTIONS, '#fd79a8', lw=2)\n    scatter(rhand, '#e84393', 25)\n\n    # lips outline\n    valid = lips[~np.isnan(lips).any(axis=1)]\n    if len(valid) > 3:\n        from matplotlib.patches import Polygon\n        ax.add_patch(Polygon(valid[:20], closed=True,\n                             fc='#e74c3c', alpha=0.45, ec='#c0392b'))\n\n\ndoc    = load_sign_json(PREVIEW_SIGN)\nframes = [frame_to_array(f) for f in doc['frames']]\nT      = len(frames)\n\nfig  = plt.figure(figsize=(5, 6), facecolor='#0f0f1a')\nax   = fig.add_axes([0.05, 0.05, 0.9, 0.9])\n\ndef update(i):\n    draw_frame(ax, frames[i],\n               title=f'🤟 {PREVIEW_SIGN}  [{i+1}/{T}]  '\n                     f'{doc[\"handedness\"]} hand')\n\nani = animation.FuncAnimation(fig, update, frames=T, interval=1000//FPS)\n\ngif_path = os.path.join(OUTPUT_DIR, f'{PREVIEW_SIGN}_preview.gif')\nani.save(gif_path, writer='pillow', fps=FPS)\nprint(f'✅ Preview saved → {gif_path}')\n\nfrom IPython.display import Image, display\ndisplay(Image(gif_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:57:49.685391Z","iopub.execute_input":"2026-04-13T21:57:49.686133Z","iopub.status.idle":"2026-04-13T21:58:03.506651Z","shell.execute_reply.started":"2026-04-13T21:57:49.686107Z","shell.execute_reply":"2026-04-13T21:58:03.506035Z"}},"outputs":[],"execution_count":null},{"id":"a9b7b865-9c69-4720-a405-dc14f6fd6a5f","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 18 · UTILITY FUNCTIONS FOR CONSUMERS\n# ════════════════════════════════════════════════\n\"\"\"\nHelper functions that downstream developers can copy/import\nto work with this dataset without re-reading this notebook.\n\"\"\"\n\ndef load_sign(sign: str, dataset_dir: str = OUTPUT_DIR) -> dict:\n    \"\"\"Load a sign JSON document from the dataset.\"\"\"\n    with open(os.path.join(dataset_dir, 'signs', f'{sign}.json')) as f:\n        return json.load(f)\n\n\ndef sign_to_matrix(sign_doc: dict) -> np.ndarray:\n    \"\"\"\n    Convert a sign document to a (T, 113, 3) numpy array.\n    Schema order is preserved.\n    \"\"\"\n    T = sign_doc['num_frames']\n    M = np.zeros((T, 113, 3), dtype=np.float32)\n    for frame in sign_doc['frames']:\n        M[frame['t']] = frame_to_array(frame)\n    return M\n\n\ndef get_joint(matrix: np.ndarray, joint_name: str) -> np.ndarray:\n    \"\"\"\n    Extract the (T, 3) trajectory of a single named joint.\n\n    Examples\n    --------\n    >>> m = sign_to_matrix(load_sign('book'))\n    >>> get_joint(m, 'rh_index_tip')   # right index fingertip path\n    >>> get_joint(m, 'nose')            # nose trajectory\n    \"\"\"\n    idx = JOINT_INDEX[joint_name]\n    return matrix[:, idx, :]\n\n\ndef get_group(matrix: np.ndarray, group: str) -> np.ndarray:\n    \"\"\"\n    Extract a whole body-part group.\n\n    group options: 'pose', 'lips_outer', 'lips_inner',\n                   'left_eye', 'right_eye', 'nose',\n                   'left_hand', 'right_hand'\n    \"\"\"\n    sl = SCHEMA_GROUPS[group]\n    return matrix[:, sl, :]\n\n\ndef resample_sign(sign_doc: dict, target_len: int = FIXED_LENGTH_OPTION) -> np.ndarray:\n    \"\"\"\n    Return (target_len, 113, 3) array resampled from the sign doc.\n    Useful for batching multiple signs together.\n    \"\"\"\n    M = sign_to_matrix(sign_doc)\n    return resample_sequence(M, target_len)\n\n\n# ──  Quick demo  ──────────────────────────────────────────────────────────\ndemo_sign = list(results.keys())[0]\ndemo_doc  = load_sign(demo_sign)\ndemo_mat  = sign_to_matrix(demo_doc)\n\nprint(f'✅ Utility functions defined.')\nprint()\nprint(f'  Demo: \"{demo_sign}\"')\nprint(f'  Matrix shape              : {demo_mat.shape}')\nprint(f'  Right index tip (first 3 frames):')\nprint(f'  {get_joint(demo_mat, \"rh_index_tip\")[:3]}')\nprint()\nprint(f'  Resampled to {FIXED_LENGTH_OPTION} frames shape: ')\nprint(f'  {resample_sign(demo_doc).shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:58:03.507762Z","iopub.execute_input":"2026-04-13T21:58:03.508117Z","iopub.status.idle":"2026-04-13T21:58:03.551518Z","shell.execute_reply.started":"2026-04-13T21:58:03.508093Z","shell.execute_reply":"2026-04-13T21:58:03.550988Z"}},"outputs":[],"execution_count":null},{"id":"6884488f-8953-44bf-8589-317b481cea10","cell_type":"code","source":"# ════════════════════════════════════════════════\n# § 19 · FINAL SUMMARY\n# ════════════════════════════════════════════════\n\nprint(\"╔══════════════════════════════════════════════╗\")\nprint(\"║    MOTION DATASET BUILD — COMPLETE           ║\")\nprint(\"╠══════════════════════════════════════════════╣\")\nprint(f\"║  Signs processed     : {len(results):<5}                ║\")\nprint(f\"║  Output directory    : {os.path.basename(OUTPUT_DIR):<22} ║\")\nprint(\"╠══════════════════════════════════════════════╣\")\nprint(\"║  Files generated:                            ║\")\nprint(\"║    joint_schema.json  — immutable contract   ║\")\nprint(\"║    metadata.json      — global index         ║\")\nprint(\"║    signs/<word>.json  — one file per sign    ║\")\nprint(\"║    motion_dataset.csv — flat inspection CSV  ║\")\nprint(\"║    dataset_stats.png  — visual report        ║\")\nprint(\"╠══════════════════════════════════════════════╣\")\nprint(\"║  JSON structure per sign:                    ║\")\nprint(\"║    sign, handedness, num_frames              ║\")\nprint(\"║    frames[t].pose          (11 joints)       ║\")\nprint(\"║    frames[t].face          (60 joints)       ║\")\nprint(\"║    frames[t].left_hand     (21 joints)       ║\")\nprint(\"║    frames[t].right_hand    (21 joints)       ║\")\nprint(\"╠══════════════════════════════════════════════╣\")\nprint(\"║  Consumer API (§18):                         ║\")\nprint(\"║    load_sign(word)                           ║\")\nprint(\"║    sign_to_matrix(doc)  → (T,113,3)          ║\")\nprint(\"║    get_joint(mat, name) → (T,3)              ║\")\nprint(\"║    get_group(mat, grp)  → (T,n,3)            ║\")\nprint(\"║    resample_sign(doc, 60) → (60,113,3)       ║\")\nprint(\"╚══════════════════════════════════════════════╝\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-13T21:58:03.552469Z","iopub.execute_input":"2026-04-13T21:58:03.552744Z","iopub.status.idle":"2026-04-13T21:58:03.559721Z","shell.execute_reply.started":"2026-04-13T21:58:03.552714Z","shell.execute_reply":"2026-04-13T21:58:03.559041Z"}},"outputs":[],"execution_count":null}]}