{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture \nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:05:42.912518Z","iopub.execute_input":"2026-02-26T04:05:42.913169Z","iopub.status.idle":"2026-02-26T04:07:19.458578Z","shell.execute_reply.started":"2026-02-26T04:05:42.913139Z","shell.execute_reply":"2026-02-26T04:07:19.457735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:13:24.328084Z","iopub.execute_input":"2026-02-26T04:13:24.328604Z","iopub.status.idle":"2026-02-26T04:13:24.332746Z","shell.execute_reply.started":"2026-02-26T04:13:24.328570Z","shell.execute_reply":"2026-02-26T04:13:24.331983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/motion-s-hierarchical-text-to-motion-generation-for-sign-language\"\n\ntrain_df = pd.read_csv(f\"{DATA_PATH}/train.csv\")\ntest_df = pd.read_csv(f\"{DATA_PATH}/test.csv\")\nmotion_root = DATA_PATH+ \"/Motion-Features\"\n\nprint(\"Train:\", train_df.shape)\nprint(\"Test:\", test_df.shape)\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:16:29.924659Z","iopub.execute_input":"2026-02-26T04:16:29.924984Z","iopub.status.idle":"2026-02-26T04:16:30.316293Z","shell.execute_reply.started":"2026-02-26T04:16:29.924956Z","shell.execute_reply":"2026-02-26T04:16:30.315697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom transformers import AutoTokenizer, AutoModel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:07:21.209241Z","iopub.execute_input":"2026-02-26T04:07:21.209495Z","iopub.status.idle":"2026-02-26T04:07:44.657984Z","shell.execute_reply.started":"2026-02-26T04:07:21.209472Z","shell.execute_reply":"2026-02-26T04:07:44.657198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Helper function\ndef load_metadata(sample_dir: Path) -> dict:\n    metadata_path = sample_dir / \"metadata.txt\"\n    if not metadata_path.exists():\n        return {}\n    result = {}\n\n    with open(metadata_path, 'r', encoding='utf-8') as f:\n        for line in f:\n            line = line.strip()\n\n            if line.startswith(\"SENTENCE:\"):\n                result[\"sentence\"] =line.split(\":\",1)[1].strip()\n            elif line.startswith(\"GLOSS:\"):\n                result[\"gloss\"] = line.split(\":\",1)[1].strip()\n\n    return result\n\nfrom pathlib import Path\n\ndef count_bvh_files(sample_dir: Path) -> int:\n    return len(list(sample_dir.glob(\"*.bvh\")))\n\ndef parse_glosses(gloss_str: str) -> list:\n    cleaned = gloss_str.replace(\"//\", \"\").strip()\n    return [g.strip() for g in cleaned.split() if g.strip()]\n\ndef is_fingerspelling(gloss: str) -> bool:\n    \"\"\"Check if a gloss is a fingerspelled letter (single uppercase letter).\"\"\"\n    return len(gloss) == 1 and gloss.isupper()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:13:27.767826Z","iopub.execute_input":"2026-02-26T04:13:27.768132Z","iopub.status.idle":"2026-02-26T04:13:27.774911Z","shell.execute_reply.started":"2026-02-26T04:13:27.768105Z","shell.execute_reply":"2026-02-26T04:13:27.774163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.nn.utils.rnn import pad_sequence \n\n#Collate function\n\"\"\"Collate functions will pad shorter sequences (for variable motion length)\"\"\"\ndef collate_fn(batch):\n    sentences = [b[0] for b in batch]\n\n    motions= [torch.tensor(b[1], dtype=torch.float32) for b in batch ]\n    lengths= torch.tensor([m.shape[0] for m in motions])\n\n    motions = pad_sequence(motions, batch_first =True)\n\n    return sentences,motions,lenghts\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:13:57.200132Z","iopub.execute_input":"2026-02-26T04:13:57.200554Z","iopub.status.idle":"2026-02-26T04:13:57.205442Z","shell.execute_reply.started":"2026-02-26T04:13:57.200525Z","shell.execute_reply":"2026-02-26T04:13:57.204794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''DATASET CLASS'''\nfrom pathlib import Path\nimport numpy as np\nfrom torch.utils.data import Dataset\n\nclass MotionDataset(Dataset):\n    def __init__(self, df, motion_root):\n        self.df = df.reset_index(drop=True)\n        self.motion_root = Path(motion_root)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n\n        sid = str(row[\"id\"])\n        sentence = row[\"sentence\"]\n\n        npy_path = self.motion_root / f\"{sid}.npy\"\n        motion = np.load(npy_path)\n\n        return sentence, motion\n\n\ndataset = MotionDataset(train_df, motion_root)\nprint(len(dataset))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:23:57.705007Z","iopub.execute_input":"2026-02-26T04:23:57.705796Z","iopub.status.idle":"2026-02-26T04:23:57.714232Z","shell.execute_reply.started":"2026-02-26T04:23:57.705764Z","shell.execute_reply":"2026-02-26T04:23:57.713695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"Dataset sample\"\"\"\ns, m = dataset[0]\n\nprint(type(s))\nprint(m.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-26T04:24:00.180709Z","iopub.execute_input":"2026-02-26T04:24:00.181007Z","iopub.status.idle":"2026-02-26T04:24:00.211983Z","shell.execute_reply.started":"2026-02-26T04:24:00.180981Z","shell.execute_reply":"2026-02-26T04:24:00.211409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}