{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.cluster import KMeans\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:21.337404Z","iopub.execute_input":"2026-02-07T03:18:21.337705Z","iopub.status.idle":"2026-02-07T03:18:22.669386Z","shell.execute_reply.started":"2026-02-07T03:18:21.337677Z","shell.execute_reply":"2026-02-07T03:18:22.668817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_hand_landmarks(parquet_path):\n    df = pd.read_parquet(parquet_path)\n\n    # Keep only hands\n    df = df[df[\"type\"].isin([\"left_hand\", \"right_hand\"])]\n\n    frames = sorted(df[\"frame\"].unique())\n    T = len(frames)\n\n    if T < 2:\n        return None\n\n    hand_xyz = np.zeros((T, 42, 3), dtype=np.float32)\n\n    for t, frame in enumerate(frames):\n        fdf = df[df[\"frame\"] == frame]\n\n        left = fdf[fdf[\"type\"] == \"left_hand\"].sort_values(\"landmark_index\")\n        right = fdf[fdf[\"type\"] == \"right_hand\"].sort_values(\"landmark_index\")\n\n        if len(left) == 21:\n            hand_xyz[t, :21] = left[[\"x\", \"y\", \"z\"]].values\n\n        if len(right) == 21:\n            hand_xyz[t, 21:] = right[[\"x\", \"y\", \"z\"]].values\n\n    return hand_xyz\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.670800Z","iopub.execute_input":"2026-02-07T03:18:22.671172Z","iopub.status.idle":"2026-02-07T03:18:22.677171Z","shell.execute_reply.started":"2026-02-07T03:18:22.671148Z","shell.execute_reply":"2026-02-07T03:18:22.676437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_motion_score(hand_xyz):\n    if hand_xyz is None or hand_xyz.shape[0] < 2:\n        return 0.0\n\n    diffs = hand_xyz[1:] - hand_xyz[:-1]\n    dist = np.linalg.norm(diffs, axis=-1)\n\n    valid = dist[~np.isnan(dist)]\n\n    if len(valid) == 0:\n        return 0.0\n\n    return valid.mean()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.678068Z","iopub.execute_input":"2026-02-07T03:18:22.678345Z","iopub.status.idle":"2026-02-07T03:18:22.693654Z","shell.execute_reply.started":"2026-02-07T03:18:22.678318Z","shell.execute_reply":"2026-02-07T03:18:22.692870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def analyze_dataset(train_csv_path, base_path, sample_limit=None):\n\n    train_df = pd.read_csv(train_csv_path)\n\n    if sample_limit:\n        train_df = train_df.head(sample_limit)\n\n    motion_scores = []\n\n    for _, row in tqdm(train_df.iterrows(), total=len(train_df)):\n        parquet_path = os.path.join(base_path, row[\"path\"])\n\n        try:\n            hand_xyz = load_hand_landmarks(parquet_path)\n\n            if hand_xyz is None:\n                motion_scores.append(0.0)\n                continue\n\n            score = compute_motion_score(hand_xyz)\n            motion_scores.append(score)\n\n        except:\n            motion_scores.append(0.0)\n\n    train_df[\"motion_score\"] = motion_scores\n\n    return train_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.694725Z","iopub.execute_input":"2026-02-07T03:18:22.695248Z","iopub.status.idle":"2026-02-07T03:18:22.705005Z","shell.execute_reply.started":"2026-02-07T03:18:22.695226Z","shell.execute_reply":"2026-02-07T03:18:22.704306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_class_clusters(df, n_clusters=4):\n\n    # Average motion per class\n    class_motion = (\n        df.groupby(\"sign\")[\"motion_score\"]\n        .mean()\n        .reset_index()\n        .rename(columns={\"motion_score\": \"avg_motion_score\"})\n    )\n\n    scores = class_motion[\"avg_motion_score\"].values.reshape(-1, 1)\n\n    kmeans = KMeans(n_clusters=n_clusters, random_state=42)\n    class_motion[\"motion_cluster\"] = kmeans.fit_predict(scores)\n\n    # Sort clusters from static → dynamic\n    cluster_order = (\n        class_motion.groupby(\"motion_cluster\")[\"avg_motion_score\"]\n        .mean()\n        .sort_values()\n        .index\n    )\n\n    mapping = {old: new for new, old in enumerate(cluster_order)}\n    class_motion[\"motion_cluster\"] = class_motion[\"motion_cluster\"].map(mapping)\n\n    print(\"\\nCluster meaning:\")\n    print(class_motion.groupby(\"motion_cluster\")[\"avg_motion_score\"].mean())\n\n    return class_motion\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.707241Z","iopub.execute_input":"2026-02-07T03:18:22.707716Z","iopub.status.idle":"2026-02-07T03:18:22.716936Z","shell.execute_reply.started":"2026-02-07T03:18:22.707692Z","shell.execute_reply":"2026-02-07T03:18:22.716383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def assign_clusters_to_samples(df, class_motion):\n\n    df = df.merge(\n        class_motion[[\"sign\", \"motion_cluster\"]],\n        on=\"sign\",\n        how=\"left\"\n    )\n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.717779Z","iopub.execute_input":"2026-02-07T03:18:22.718057Z","iopub.status.idle":"2026-02-07T03:18:22.729833Z","shell.execute_reply.started":"2026-02-07T03:18:22.718036Z","shell.execute_reply":"2026-02-07T03:18:22.729308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_subsets(df):\n\n    subsets = {}\n\n    for cluster_id in sorted(df[\"motion_cluster\"].unique()):\n        subset_df = df[df[\"motion_cluster\"] == cluster_id].reset_index(drop=True)\n        subsets[cluster_id] = subset_df\n\n        print(f\"Cluster {cluster_id}:\")\n        print(\"  Samples:\", len(subset_df))\n        print(\"  Classes:\", subset_df[\"sign\"].nunique())\n\n    return subsets\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.730774Z","iopub.execute_input":"2026-02-07T03:18:22.731136Z","iopub.status.idle":"2026-02-07T03:18:22.741452Z","shell.execute_reply.started":"2026-02-07T03:18:22.731114Z","shell.execute_reply":"2026-02-07T03:18:22.740735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n\n    TRAIN_CSV = \"/kaggle/input/asl-signs/train.csv\"\n    BASE_PATH = \"/kaggle/input/asl-signs\"\n\n    # FULL dataset (no sampling)\n    df = analyze_dataset(TRAIN_CSV, BASE_PATH)\n\n    print(\"Finished computing motion scores.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T03:18:22.742151Z","iopub.execute_input":"2026-02-07T03:18:22.742421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_static_dynamic_split(df):\n    \n    # Average motion per class\n    class_motion = (\n        df.groupby(\"sign\")[\"motion_score\"]\n        .mean()\n        .reset_index()\n        .rename(columns={\"motion_score\": \"avg_motion_score\"})\n    )\n\n    # Median threshold (balanced split)\n    threshold = class_motion[\"avg_motion_score\"].median()\n\n    class_motion[\"motion_category\"] = class_motion[\"avg_motion_score\"].apply(\n        lambda x: \"static\" if x <= threshold else \"dynamic\"\n    )\n\n    return class_motion\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def assign_category_to_samples(df, class_motion):\n\n    df = df.merge(\n        class_motion[[\"sign\", \"motion_category\"]],\n        on=\"sign\",\n        how=\"left\"\n    )\n\n    return df\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_motion = create_static_dynamic_split(df)\n\ndf = assign_category_to_samples(df, class_motion)\n\nprint(\"Samples per category:\")\nprint(df[\"motion_category\"].value_counts())\n\nprint(\"\\nClasses per category:\")\nprint(df.groupby(\"motion_category\")[\"sign\"].nunique())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"static_df = df[df[\"motion_category\"] == \"static\"].reset_index(drop=True)\ndynamic_df = df[df[\"motion_category\"] == \"dynamic\"].reset_index(drop=True)\n\nprint(\"Static dataset:\", static_df.shape)\nprint(\"Dynamic dataset:\", dynamic_df.shape)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"static_df.to_csv(\"train_static.csv\", index=False)\ndynamic_df.to_csv(\"train_dynamic.csv\", index=False)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}