{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":14.766368,"end_time":"2026-02-25T17:08:59.125516","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-25T17:08:44.359148","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, re, random\nfrom pathlib import Path\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom transformers import AutoTokenizer, AutoModel\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# ============================================================\n# CONFIG\n# ============================================================\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nHIDDEN_DIM = 512\nNUM_LAYERS = 3\nDROPOUT = 0.2\nBATCH_SIZE = 32\nEPOCHS = 15\nLEARNING_RATE = 1e-3\nMAX_SEQ_LEN = 800\nMIN_SEQ_LEN = 40\nNUM_TOKENS = 512\nNUM_LAYERS_TO_GEN = 6\n\nINPUT_DIR = Path(\"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language\")\n\ntrain_df = pd.read_csv(INPUT_DIR / \"train.csv\")\ntest_df = pd.read_csv(INPUT_DIR / \"test.csv\")\n\nprint(f\"Train: {train_df.shape}, Test: {test_df.shape}\")\n\n# ============================================================\n# TEXT PROCESSING\n# ============================================================\n_ws_re = re.compile(r\"\\s+\")\n\ndef norm_text(x):\n    if not isinstance(x, str):\n        return \"\"\n    return _ws_re.sub(\" \", x.strip().lower())\n\ndef build_combined_text(df):\n    gloss = df[\"gloss\"].map(norm_text)\n    sent = df[\"sentence\"].map(norm_text)\n    return (gloss + \" \" + gloss + \" \" + gloss + \" \" + sent).fillna(\"\")\n\ntrain_text = build_combined_text(train_df)\ntest_text = build_combined_text(test_df)\n\n# ============================================================\n# TOKEN UTILS\n# ============================================================\nTOKEN_COLS = [\"base_tokens\", \"residual_1\", \"residual_2\",\n              \"residual_3\", \"residual_4\", \"residual_5\"]\n\ndef parse_tokens(x):\n    if not isinstance(x, str) or x.strip() == \"\":\n        return []\n    return [int(t) for t in x.split()]\n\ndef tokens_to_str(t):\n    return \" \".join(map(str, t))\n\n# ============================================================\n# DATA CLEANING\n# ============================================================\ngood_mask = np.ones(len(train_df), dtype=bool)\n\nfor i, row in train_df.iterrows():\n    lens = []\n    ok = True\n    for c in TOKEN_COLS:\n        t = parse_tokens(row[c])\n        if any((x < 0 or x > 511) for x in t):\n            ok = False\n            break\n        lens.append(len(t))\n    if (not ok) or (len(set(lens)) != 1) or (lens[0] < MIN_SEQ_LEN):\n        good_mask[i] = False\n\ntrain_clean = train_df.loc[good_mask].reset_index(drop=True)\ntrain_text_clean = train_text.loc[good_mask].reset_index(drop=True)\n\nprint(f\"Usable train samples: {len(train_clean)}\")\n\ntrain_tokens_cache = {}\nfor c in TOKEN_COLS:\n    train_tokens_cache[c] = [parse_tokens(x) for x in train_clean[c].astype(str)]\n\n# ============================================================\n# BERT TEXT ENCODER\n# ============================================================\nENCODER_MODEL = \"bert-base-uncased\"\nbert_tokenizer = AutoTokenizer.from_pretrained(ENCODER_MODEL)\nbert_model = AutoModel.from_pretrained(ENCODER_MODEL).to(DEVICE)\nbert_model.eval()\nBERT_DIM = 768\n\nprint(\"BERT encoder loaded\")\n\ndef encode_text_batch(texts):\n    inputs = bert_tokenizer(texts, padding=True, truncation=True, \n                           max_length=128, return_tensors=\"pt\").to(DEVICE)\n    with torch.no_grad():\n        outputs = bert_model(**inputs)\n        embeddings = outputs.last_hidden_state[:, 0, :]\n    return embeddings\n\n# ============================================================\n# CUSTOM COLLATE FUNCTION (FIXES PADDING ISSUE)\n# ============================================================\ndef collate_motion_batch(batch):\n    \"\"\"Pad sequences to max length in batch\"\"\"\n    texts = [item['text'] for item in batch]\n    token_sequences = [item['tokens'] for item in batch]\n    lengths = [item['length'] for item in batch]\n    \n    # Find max length in this batch\n    max_len = max([len(seq) for seq in token_sequences])\n    \n    # Pad all sequences to max_len (use 0 as padding token)\n    padded_sequences = []\n    masks = []\n    \n    for seq in token_sequences:\n        pad_amount = max_len - len(seq)\n        if pad_amount > 0:\n            padded_seq = torch.cat([seq, torch.zeros(pad_amount, dtype=torch.long)])\n            mask = torch.cat([torch.ones(len(seq)), torch.zeros(pad_amount)])\n        else:\n            padded_seq = seq\n            mask = torch.ones(len(seq))\n        \n        padded_sequences.append(padded_seq)\n        masks.append(mask)\n    \n    padded_sequences = torch.stack(padded_sequences)\n    masks = torch.stack(masks)\n    \n    return {\n        'text': texts,\n        'tokens': padded_sequences,\n        'masks': masks,\n        'lengths': torch.tensor(lengths, dtype=torch.long)\n    }\n\n# ============================================================\n# DATASET\n# ============================================================\nclass MotionDataset(Dataset):\n    def __init__(self, texts, token_sequences):\n        self.texts = texts\n        self.token_sequences = token_sequences\n        self.lengths = [len(seq) for seq in token_sequences]\n        \n    def __len__(self):\n        return len(self.texts)\n    \n    def __getitem__(self, idx):\n        text = self.texts.iloc[idx]\n        tokens = self.token_sequences[idx]\n        \n        # Clip to MAX_SEQ_LEN\n        if len(tokens) > MAX_SEQ_LEN:\n            tokens = tokens[:MAX_SEQ_LEN]\n        \n        return {\n            'text': text,\n            'tokens': torch.tensor(tokens, dtype=torch.long),\n            'length': len(tokens)\n        }\n\n# Prepare token sequences\nall_token_sequences = []\nfor idx in range(len(train_clean)):\n    seq = []\n    for c in TOKEN_COLS:\n        seq.extend(train_tokens_cache[c][idx])\n    all_token_sequences.append(seq)\n\ndataset = MotionDataset(train_text_clean, all_token_sequences)\ndataloader = DataLoader(dataset, batch_size=BATCH_SIZE, shuffle=True, collate_fn=collate_motion_batch)\n\nprint(f\"Dataset created: {len(dataset)} samples\")\n\n# ============================================================\n# GENERATOR MODEL\n# ============================================================\nclass MotionGenerator(nn.Module):\n    def __init__(self, input_dim, hidden_dim, num_layers, dropout=0.2):\n        super().__init__()\n        \n        self.text_encoder_proj = nn.Linear(input_dim, hidden_dim)\n        \n        self.length_predictor = nn.Sequential(\n            nn.Linear(hidden_dim, hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, 1)\n        )\n        \n        self.token_embedding = nn.Embedding(NUM_TOKENS, hidden_dim)\n        \n        self.decoder = nn.LSTM(\n            input_size=hidden_dim * 2,\n            hidden_size=hidden_dim,\n            num_layers=num_layers,\n            dropout=dropout if num_layers > 1 else 0,\n            batch_first=True\n        )\n        \n        self.output_proj = nn.Linear(hidden_dim, NUM_TOKENS)\n        \n        self.hidden_dim = hidden_dim\n        \n    def forward(self, text_embeddings, token_sequence=None, mask=None, max_len=None):\n        batch_size = text_embeddings.shape[0]\n        \n        text_context = self.text_encoder_proj(text_embeddings)\n        \n        predicted_length = torch.sigmoid(self.length_predictor(text_context))\n        predicted_length = (predicted_length * (MAX_SEQ_LEN - MIN_SEQ_LEN) + MIN_SEQ_LEN).squeeze(-1)\n        \n        if token_sequence is not None:\n            seq_len = token_sequence.shape[1]\n            \n            # Start token\n            start_tokens = torch.zeros(batch_size, 1, dtype=torch.long, device=text_embeddings.device)\n            prev_tokens = torch.cat([start_tokens, token_sequence[:, :-1]], dim=1)\n            \n            # Embed tokens\n            token_embeds = self.token_embedding(prev_tokens)\n            \n            # Expand text context\n            text_context_expanded = text_context.unsqueeze(1).expand(batch_size, seq_len, -1)\n            \n            # Concatenate\n            decoder_input = torch.cat([token_embeds, text_context_expanded], dim=-1)\n            \n            # Decode\n            decoder_output, _ = self.decoder(decoder_input)\n            logits = self.output_proj(decoder_output)\n            \n            return logits, predicted_length, mask\n        \n        else:\n            # Inference\n            if max_len is None:\n                max_len = int(predicted_length.max().item())\n                max_len = min(max_len, MAX_SEQ_LEN)\n                max_len = max(max_len, MIN_SEQ_LEN)\n            \n            generated_tokens = []\n            current_token = torch.zeros(batch_size, 1, dtype=torch.long, device=text_embeddings.device)\n            h = None\n            \n            for t in range(max_len):\n                token_embed = self.token_embedding(current_token)\n                text_context_t = text_context.unsqueeze(1)\n                decoder_input = torch.cat([token_embed, text_context_t], dim=-1)\n                \n                if h is None:\n                    decoder_output, h = self.decoder(decoder_input)\n                else:\n                    decoder_output, h = self.decoder(decoder_input, h)\n                \n                logits = self.output_proj(decoder_output)\n                next_token = torch.argmax(logits, dim=-1)\n                \n                generated_tokens.append(next_token)\n                current_token = next_token\n            \n            generated_tokens = torch.cat(generated_tokens, dim=1)\n            \n            return generated_tokens, predicted_length, None\n\n# ============================================================\n# TRAINING\n# ============================================================\nmodel = MotionGenerator(BERT_DIM, HIDDEN_DIM, NUM_LAYERS, DROPOUT).to(DEVICE)\noptimizer = optim.AdamW(model.parameters(), lr=LEARNING_RATE)\ncriterion = nn.CrossEntropyLoss(reduction='none')\n\nprint(\"\\nStarting training...\")\n\nfor epoch in range(EPOCHS):\n    model.train()\n    total_loss = 0\n    num_tokens_processed = 0\n    \n    for batch in tqdm(dataloader, desc=f\"Epoch {epoch+1}/{EPOCHS}\"):\n        # Encode text\n        text_embeddings = encode_text_batch(batch['text'])\n        \n        tokens_tensor = batch['tokens'].to(DEVICE)\n        mask = batch['masks'].to(DEVICE)\n        \n        # Forward pass\n        logits, _, batch_mask = model(text_embeddings, tokens_tensor, mask)\n        \n        # Loss with masking (ignore padding)\n        loss = criterion(logits.reshape(-1, NUM_TOKENS), tokens_tensor.reshape(-1))\n        loss = loss * mask.reshape(-1)\n        loss = loss.sum() / mask.sum()\n        \n        # Backward\n        optimizer.zero_grad()\n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n        optimizer.step()\n        \n        total_loss += loss.item()\n        num_tokens_processed += mask.sum().item()\n    \n    avg_loss = total_loss / len(dataloader)\n    print(f\"Epoch {epoch+1}/{EPOCHS} - Loss: {avg_loss:.4f}\")\n\nprint(\"Training complete!\")\n\n# ============================================================\n# INFERENCE\n# ============================================================\nmodel.eval()\npredictions = pd.DataFrame({\"id\": test_df[\"id\"].values})\n\nfor c in TOKEN_COLS:\n    predictions[c] = \"\"\n\nprint(\"\\nGenerating predictions...\")\n\nfor i in tqdm(range(len(test_df))):\n    test_text_sample = test_text.iloc[i:i+1]\n    \n    with torch.no_grad():\n        text_embedding = encode_text_batch(test_text_sample)\n        generated_tokens, pred_length, _ = model(text_embedding, token_sequence=None, max_len=None)\n    \n    tokens_list = generated_tokens[0].cpu().numpy().tolist()\n    \n    # Ensure valid length\n    if len(tokens_list) < MIN_SEQ_LEN:\n        tokens_list.extend([np.random.randint(0, NUM_TOKENS) for _ in range(MIN_SEQ_LEN - len(tokens_list))])\n    elif len(tokens_list) > MAX_SEQ_LEN:\n        tokens_list = tokens_list[:MAX_SEQ_LEN]\n    \n    # Split into 6 equal-length layers\n    layer_size = len(tokens_list) // NUM_LAYERS_TO_GEN\n    remainder = len(tokens_list) % NUM_LAYERS_TO_GEN\n    \n    layers = []\n    start_idx = 0\n    \n    for layer_idx in range(NUM_LAYERS_TO_GEN):\n        # Distribute remainder across first layers\n        current_layer_size = layer_size + (1 if layer_idx < remainder else 0)\n        end_idx = start_idx + current_layer_size\n        \n        layer_tokens = tokens_list[start_idx:end_idx]\n        layers.append(tokens_to_str(layer_tokens))\n        start_idx = end_idx\n    \n    for j, col in enumerate(TOKEN_COLS):\n        predictions.at[i, col] = layers[j]\n\n# ============================================================\n# SAVE\n# ============================================================\nOUT_PATH = Path(\"/kaggle/working/submission_generator_v1.csv\")\npredictions.to_csv(OUT_PATH, index=False)\n\nprint(f\"Saved: {OUT_PATH}\")\nprint(predictions.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T16:42:51.083507Z","iopub.execute_input":"2026-04-26T16:42:51.083817Z","iopub.status.idle":"2026-04-26T17:51:43.577184Z","shell.execute_reply.started":"2026-04-26T16:42:51.083788Z","shell.execute_reply":"2026-04-26T17:51:43.575942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# INFERENCE\n# ============================================================\nmodel.eval()\npredictions = pd.DataFrame({\"id\": test_df[\"id\"].values})\n\nfor c in TOKEN_COLS:\n    predictions[c] = \"\"\n\nprint(\"\\nGenerating predictions...\")\n\nfor i in tqdm(range(len(test_df))):\n    test_text_sample = test_text.iloc[i]  # Get single string, not Series\n    \n    with torch.no_grad():\n        # Convert to list for batch processing\n        text_embedding = encode_text_batch([test_text_sample])\n        generated_tokens, pred_length, _ = model(text_embedding, token_sequence=None, max_len=None)\n    \n    tokens_list = generated_tokens[0].cpu().numpy().tolist()\n    \n    # Ensure valid length\n    if len(tokens_list) < MIN_SEQ_LEN:\n        tokens_list.extend([np.random.randint(0, NUM_TOKENS) for _ in range(MIN_SEQ_LEN - len(tokens_list))])\n    elif len(tokens_list) > MAX_SEQ_LEN:\n        tokens_list = tokens_list[:MAX_SEQ_LEN]\n    \n    # Split into 6 equal-length layers\n    layer_size = len(tokens_list) // NUM_LAYERS_TO_GEN\n    remainder = len(tokens_list) % NUM_LAYERS_TO_GEN\n    \n    layers = []\n    start_idx = 0\n    \n    for layer_idx in range(NUM_LAYERS_TO_GEN):\n        # Distribute remainder across first layers\n        current_layer_size = layer_size + (1 if layer_idx < remainder else 0)\n        end_idx = start_idx + current_layer_size\n        \n        layer_tokens = tokens_list[start_idx:end_idx]\n        layers.append(tokens_to_str(layer_tokens))\n        start_idx = end_idx\n    \n    for j, col in enumerate(TOKEN_COLS):\n        predictions.at[i, col] = layers[j]\n\n# ============================================================\n# SAVE\n# ============================================================\nOUT_PATH = Path(\"/kaggle/working/submission_generator_v1.csv\")\npredictions.to_csv(OUT_PATH, index=False)\n\nprint(f\"Saved: {OUT_PATH}\")\nprint(predictions.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T17:52:42.798080Z","iopub.execute_input":"2026-04-26T17:52:42.798814Z","iopub.status.idle":"2026-04-26T18:01:56.862988Z","shell.execute_reply.started":"2026-04-26T17:52:42.798779Z","shell.execute_reply":"2026-04-26T18:01:56.862243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}