{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:38:29.590854Z","iopub.execute_input":"2026-04-26T18:38:29.59103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install transformers -q\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom transformers import BertTokenizer, BertModel\nfrom tqdm import tqdm\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:43:59.704374Z","iopub.execute_input":"2026-04-26T18:43:59.70476Z","iopub.status.idle":"2026-04-26T18:44:02.861158Z","shell.execute_reply.started":"2026-04-26T18:43:59.704726Z","shell.execute_reply":"2026-04-26T18:44:02.860049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/motion-s-hierarchical-text-to-motion-generation-for-sign-language\"\n\ntrain_df = pd.read_csv(\"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:45:38.535258Z","iopub.execute_input":"2026-04-26T18:45:38.53558Z","iopub.status.idle":"2026-04-26T18:45:39.16079Z","shell.execute_reply.started":"2026-04-26T18:45:38.53555Z","shell.execute_reply":"2026-04-26T18:45:39.159639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = BertTokenizer.from_pretrained(\"bert-base-uncased\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:48:07.302956Z","iopub.execute_input":"2026-04-26T18:48:07.303263Z","iopub.status.idle":"2026-04-26T18:48:08.501033Z","shell.execute_reply.started":"2026-04-26T18:48:07.303236Z","shell.execute_reply":"2026-04-26T18:48:08.500251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MotionDataset(Dataset):\n    def __init__(self, df):\n        self.df = df\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n\n        text = f\"GLOSS: {row['gloss']} TEXT: {row['sentence']}\"\n\n        enc = tokenizer(\n            text,\n            padding='max_length',\n            truncation=True,\n            max_length=128,\n            return_tensors=\"pt\"\n        )\n\n        item = {\n            \"input_ids\": enc[\"input_ids\"].squeeze(),\n            \"attention_mask\": enc[\"attention_mask\"].squeeze()\n        }\n\n        # Training only\n        if \"base_tokens\" in row:\n            tokens = []\n            for col in [\"base_tokens\",\"residual_1\",\"residual_2\",\"residual_3\",\"residual_4\",\"residual_5\"]:\n                tokens.append(list(map(int, row[col].split())))\n            tokens = np.array(tokens).T  # shape (L, 6)\n\n            item[\"tokens\"] = torch.tensor(tokens, dtype=torch.long)\n\n        return item","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:48:15.961601Z","iopub.execute_input":"2026-04-26T18:48:15.961893Z","iopub.status.idle":"2026-04-26T18:48:15.968687Z","shell.execute_reply.started":"2026-04-26T18:48:15.96187Z","shell.execute_reply":"2026-04-26T18:48:15.967756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MotionModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        \n        self.bert = BertModel.from_pretrained(\"bert-base-uncased\")\n\n        self.decoder_layer = nn.TransformerDecoderLayer(\n            d_model=768,\n            nhead=8\n        )\n        self.decoder = nn.TransformerDecoder(self.decoder_layer, num_layers=3)\n\n        self.token_embed = nn.Embedding(512, 768)\n\n        # 6 output heads\n        self.heads = nn.ModuleList([nn.Linear(768, 512) for _ in range(6)])\n\n    def forward(self, input_ids, attention_mask, tgt_tokens):\n        \n        enc_out = self.bert(input_ids=input_ids, attention_mask=attention_mask)\n        memory = enc_out.last_hidden_state.transpose(0,1)\n\n        tgt = self.token_embed(tgt_tokens[:,:,0])  # base token embedding\n        tgt = tgt.transpose(0,1)\n\n        out = self.decoder(tgt, memory)\n        out = out.transpose(0,1)\n\n        outputs = []\n        for i in range(6):\n            outputs.append(self.heads[i](out))\n\n        return outputs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:48:21.738974Z","iopub.execute_input":"2026-04-26T18:48:21.739357Z","iopub.status.idle":"2026-04-26T18:48:21.748712Z","shell.execute_reply.started":"2026-04-26T18:48:21.739318Z","shell.execute_reply":"2026-04-26T18:48:21.747503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collate_fn(batch):\n    input_ids = torch.stack([x[\"input_ids\"] for x in batch])\n    attention_mask = torch.stack([x[\"attention_mask\"] for x in batch])\n\n    # get max length in batch\n    max_len = max(x[\"tokens\"].shape[0] for x in batch)\n\n    padded_tokens = []\n\n    for x in batch:\n        tokens = x[\"tokens\"]\n        pad_len = max_len - tokens.shape[0]\n\n        if pad_len > 0:\n            pad = torch.zeros((pad_len, 6), dtype=torch.long)\n            tokens = torch.cat([tokens, pad], dim=0)\n\n        padded_tokens.append(tokens)\n\n    padded_tokens = torch.stack(padded_tokens)\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_mask\": attention_mask,\n        \"tokens\": padded_tokens\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:57:29.774497Z","iopub.execute_input":"2026-04-26T18:57:29.77486Z","iopub.status.idle":"2026-04-26T18:57:29.782688Z","shell.execute_reply.started":"2026-04-26T18:57:29.774829Z","shell.execute_reply":"2026-04-26T18:57:29.781416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# length_model = torch.load(f\"{DATA_PATH}/length_estimator.pth\", map_location=device)\n# length_model.eval()\n\ndef predict_length(text):\n    words = text.split()\n    \n    # simple heuristic\n    length = int(len(words) * 8)\n    \n    # clamp to valid range\n    return max(40, min(800, length))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:57:37.662473Z","iopub.execute_input":"2026-04-26T18:57:37.663054Z","iopub.status.idle":"2026-04-26T18:57:37.667767Z","shell.execute_reply.started":"2026-04-26T18:57:37.663025Z","shell.execute_reply":"2026-04-26T18:57:37.667128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = MotionDataset(train_df.head(2000))  # use subset first\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=8,\n    shuffle=True,\n    collate_fn=collate_fn\n)\nmodel = MotionModel().to(device)\noptimizer = torch.optim.AdamW(model.parameters(), lr=2e-5)\n\ncriterion = nn.CrossEntropyLoss()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:57:42.049452Z","iopub.execute_input":"2026-04-26T18:57:42.049741Z","iopub.status.idle":"2026-04-26T18:57:43.20361Z","shell.execute_reply.started":"2026-04-26T18:57:42.049719Z","shell.execute_reply":"2026-04-26T18:57:43.202415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EPOCHS = 2\n\nfor epoch in range(EPOCHS):\n    model.train()\n    \n    for batch in tqdm(train_loader):\n        input_ids = batch[\"input_ids\"].to(device)\n        attention_mask = batch[\"attention_mask\"].to(device)\n        tokens = batch[\"tokens\"].to(device)\n\n        # shift for teacher forcing\n        tgt_input = tokens[:,:-1,:]\n        tgt_output = tokens[:,1:,:]\n\n        outputs = model(input_ids, attention_mask, tgt_input)\n\n        loss = 0\n        for i in range(6):\n            loss += criterion(\n                outputs[i].reshape(-1,512),\n                tgt_output[:,:,i].reshape(-1)\n            )\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n    print(f\"Epoch {epoch+1} Loss: {loss.item()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T18:57:54.596241Z","iopub.execute_input":"2026-04-26T18:57:54.596535Z","iopub.status.idle":"2026-04-26T18:59:06.704254Z","shell.execute_reply.started":"2026-04-26T18:57:54.596512Z","shell.execute_reply":"2026-04-26T18:59:06.702692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_length(text):\n    # dummy: fallback (length model integration is complex)\n    return 120","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate(model, text):\n    model.eval()\n\n    enc = tokenizer(\n        text,\n        return_tensors=\"pt\",\n        truncation=True,\n        padding=True,\n        max_length=128\n    )\n\n    input_ids = enc[\"input_ids\"].to(device)\n    attention_mask = enc[\"attention_mask\"].to(device)\n\n    L = predict_length(text)\n\n    tokens = torch.zeros((1, L, 6), dtype=torch.long).to(device)\n\n    for t in range(1, L):\n        outputs = model(input_ids, attention_mask, tokens[:,:t,:])\n\n        for i in range(6):\n            tokens[0,t,i] = torch.argmax(outputs[i][0,-1])\n\n    return tokens[0].cpu().numpy()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = []\n\nfor _, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    text = f\"GLOSS: {row['gloss']} TEXT: {row['sentence']}\"\n    \n    tokens = generate(model, text)\n\n    entry = {\n        \"id\": row[\"id\"]\n    }\n\n    for i, col in enumerate([\"base_tokens\",\"residual_1\",\"residual_2\",\"residual_3\",\"residual_4\",\"residual_5\"]):\n        entry[col] = \" \".join(map(str, tokens[:,i]))\n\n    submission.append(entry)\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}