{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Motion-S: Advanced Text-to-Sign Motion Generation\n\n**Improvements over TF-IDF+1NN baseline (0.43):**\n1. BERT semantic embeddings (sentence-transformers)\n2. Multi-modal retrieval ensemble (BERT + word TF-IDF + char TF-IDF + gloss matching)\n3. K-NN token voting (K=10) with distance weighting\n4. Gloss-exact-match prioritization\n5. Smart adaptive length estimation\n6. Reciprocal Rank Fusion for combining retrievers","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install -q sentence-transformers","metadata":{"execution":{"iopub.status.busy":"2026-04-25T17:22:28.370668Z","iopub.execute_input":"2026-04-25T17:22:28.370949Z","iopub.status.idle":"2026-04-25T17:22:31.937996Z","shell.execute_reply.started":"2026-04-25T17:22:28.370926Z","shell.execute_reply":"2026-04-25T17:22:31.937160Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, re, gc, random, warnings, time\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nfrom scipy import sparse\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom sklearn.preprocessing import normalize\n\nimport torch\n\nwarnings.filterwarnings('ignore')\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nINPUT_DIR = Path(\"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language\")\nTRAIN_CSV = INPUT_DIR / \"train.csv\"\nTEST_CSV  = INPUT_DIR / \"test.csv\"\nSAMPLE_SUB = INPUT_DIR / \"sample_submission.csv\"\nOUT_PATH = Path(\"/kaggle/working/submission.csv\")\n\n# --- Configuration ---\nTOKEN_COLS = [\"base_tokens\", \"residual_1\", \"residual_2\", \"residual_3\", \"residual_4\", \"residual_5\"]\nMIN_LEN, MAX_LEN, VOCAB = 40, 800, 512\nK = 10  # number of neighbors for voting\n\n# Weights for ensemble retrieval\nW_BERT  = 0.45\nW_TFIDF_CHAR = 0.20\nW_TFIDF_WORD = 0.15\nW_GLOSS = 0.20\n\nprint(\"Setup complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:31.939988Z","iopub.execute_input":"2026-04-25T17:22:31.940412Z","iopub.status.idle":"2026-04-25T17:22:31.949369Z","shell.execute_reply.started":"2026-04-25T17:22:31.940372Z","shell.execute_reply":"2026-04-25T17:22:31.948765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Data Loading & Text Preprocessing","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_CSV)\ntest  = pd.read_csv(TEST_CSV)\nsample_sub = pd.read_csv(SAMPLE_SUB)\n\nprint(f\"Train: {train.shape}, Test: {test.shape}\")\nprint(f\"Train cols: {train.columns.tolist()}\")\nprint(f\"Test cols:  {test.columns.tolist()}\")\ntrain.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:31.950134Z","iopub.execute_input":"2026-04-25T17:22:31.950586Z","iopub.status.idle":"2026-04-25T17:22:32.615423Z","shell.execute_reply.started":"2026-04-25T17:22:31.950505Z","shell.execute_reply":"2026-04-25T17:22:32.614735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_ws = re.compile(r\"\\s+\")\n\ndef norm(x):\n    return _ws.sub(\" \", str(x).strip().lower()) if isinstance(x, str) else \"\"\n\ndef get_col(df, col):\n    return df[col].map(norm) if col in df.columns else pd.Series([\"\"] * len(df))\n\ntrain_gloss = get_col(train, \"gloss\")\ntrain_sent  = get_col(train, \"sentence\")\ntest_gloss  = get_col(test, \"gloss\")\ntest_sent   = get_col(test, \"sentence\")\n\n# For TF-IDF: gloss repeated 3x to weight it more\ntrain_text = (train_gloss + \" || \" + train_gloss + \" || \" + train_sent).fillna(\"\")\ntest_text  = (test_gloss  + \" || \" + test_gloss  + \" || \" + test_sent).fillna(\"\")\n\n# For BERT: clean combined text\ntrain_text_bert = (train_gloss + \" : \" + train_sent).fillna(\"\")\ntest_text_bert  = (test_gloss  + \" : \" + test_sent).fillna(\"\")\n\nprint(f\"Sample train text: {train_text.iloc[0][:100]}...\")\nprint(f\"Sample gloss: {train_gloss.iloc[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:32.616468Z","iopub.execute_input":"2026-04-25T17:22:32.616871Z","iopub.status.idle":"2026-04-25T17:22:32.724298Z","shell.execute_reply.started":"2026-04-25T17:22:32.616846Z","shell.execute_reply":"2026-04-25T17:22:32.723613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Token Utilities & Data Validation","metadata":{}},{"cell_type":"code","source":"def parse_tokens(s):\n    if not isinstance(s, str) or s.strip() == \"\":\n        return []\n    return [int(x) for x in s.split()]\n\ndef tokens_to_str(t):\n    return \" \".join(map(str, t))\n\ndef enforce_len(tokens, tgt):\n    tgt = max(MIN_LEN, min(MAX_LEN, tgt))\n    if len(tokens) == 0:\n        return [random.randint(0, VOCAB-1) for _ in range(tgt)]\n    if len(tokens) < tgt:\n        tokens = (tokens * ((tgt // len(tokens)) + 1))[:tgt]\n    return tokens[:tgt]\n\n# Validate training data and cache tokens\ngood_mask = np.ones(len(train), dtype=bool)\nall_train_tokens = {c: [] for c in TOKEN_COLS}\ntrain_lens = []\n\nfor i, row in train.iterrows():\n    lens, ok = [], True\n    row_toks = {}\n    for c in TOKEN_COLS:\n        t = parse_tokens(row[c])\n        if any(x < 0 or x >= VOCAB for x in t):\n            ok = False; break\n        row_toks[c] = t\n        lens.append(len(t))\n    if not ok or len(set(lens)) != 1 or lens[0] < 1:\n        good_mask[i] = False\n    else:\n        for c in TOKEN_COLS:\n            all_train_tokens[c].append(row_toks[c])\n        train_lens.append(lens[0])\n\ntrain_good = train.loc[good_mask].reset_index(drop=True)\ntrain_text_good = train_text.loc[good_mask].reset_index(drop=True)\ntrain_text_bert_good = train_text_bert.loc[good_mask].reset_index(drop=True)\ntrain_gloss_good = train_gloss.loc[good_mask].reset_index(drop=True)\n\nprint(f\"Usable train rows: {len(train_good)} / {len(train)}\")\nprint(f\"Token length stats: min={min(train_lens)}, max={max(train_lens)}, median={int(np.median(train_lens))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:32.726116Z","iopub.execute_input":"2026-04-25T17:22:32.726778Z","iopub.status.idle":"2026-04-25T17:22:35.800135Z","shell.execute_reply.started":"2026-04-25T17:22:32.726719Z","shell.execute_reply":"2026-04-25T17:22:35.799479Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Feature Extraction: TF-IDF (Char + Word N-grams)","metadata":{}},{"cell_type":"code","source":"t0 = time.time()\n\n# Character-level TF-IDF (similar to baseline but tuned)\ntfidf_char = TfidfVectorizer(\n    analyzer=\"char_wb\", ngram_range=(2, 5),\n    min_df=2, max_features=200000, lowercase=True, sublinear_tf=True\n)\nX_train_char = tfidf_char.fit_transform(train_text_good)\nX_test_char  = tfidf_char.transform(test_text)\nprint(f\"Char TF-IDF: {X_train_char.shape} in {time.time()-t0:.1f}s\")\n\n# Word-level TF-IDF\nt0 = time.time()\ntfidf_word = TfidfVectorizer(\n    analyzer=\"word\", ngram_range=(1, 2),\n    min_df=1, max_features=100000, lowercase=True, sublinear_tf=True\n)\nX_train_word = tfidf_word.fit_transform(train_text_good)\nX_test_word  = tfidf_word.transform(test_text)\nprint(f\"Word TF-IDF: {X_train_word.shape} in {time.time()-t0:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:35.801159Z","iopub.execute_input":"2026-04-25T17:22:35.801890Z","iopub.status.idle":"2026-04-25T17:22:38.216151Z","shell.execute_reply.started":"2026-04-25T17:22:35.801863Z","shell.execute_reply":"2026-04-25T17:22:38.215521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Feature Extraction: BERT Semantic Embeddings","metadata":{}},{"cell_type":"code","source":"USE_BERT = True\ntry:\n    from sentence_transformers import SentenceTransformer\n    print(\"Loading sentence-transformers model...\")\n    t0 = time.time()\n    sbert = SentenceTransformer('all-MiniLM-L6-v2', device='cuda' if torch.cuda.is_available() else 'cpu')\n    \n    print(\"Encoding training texts...\")\n    train_bert_emb = sbert.encode(\n        train_text_bert_good.tolist(), batch_size=256,\n        show_progress_bar=True, normalize_embeddings=True\n    )\n    print(\"Encoding test texts...\")\n    test_bert_emb = sbert.encode(\n        test_text_bert.tolist(), batch_size=256,\n        show_progress_bar=True, normalize_embeddings=True\n    )\n    print(f\"BERT embeddings: train {train_bert_emb.shape}, test {test_bert_emb.shape} in {time.time()-t0:.1f}s\")\n    \n    # Free GPU memory\n    del sbert; gc.collect()\n    if torch.cuda.is_available(): torch.cuda.empty_cache()\n\nexcept Exception as e:\n    print(f\"BERT not available ({e}), using TF-IDF only.\")\n    USE_BERT = False\n    # Redistribute weights\n    W_TFIDF_CHAR = 0.40\n    W_TFIDF_WORD = 0.30\n    W_GLOSS = 0.30\n    W_BERT = 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:22:38.216993Z","iopub.execute_input":"2026-04-25T17:22:38.217284Z","iopub.status.idle":"2026-04-25T17:23:05.636819Z","shell.execute_reply.started":"2026-04-25T17:22:38.217251Z","shell.execute_reply":"2026-04-25T17:23:05.635734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Feature Extraction: Gloss Overlap Similarity","metadata":{}},{"cell_type":"code","source":"def gloss_to_set(g):\n    return set(g.split()) if g else set()\n\ntrain_gloss_sets = [gloss_to_set(g) for g in train_gloss_good]\ntest_gloss_sets  = [gloss_to_set(g) for g in test_gloss]\n\ndef compute_gloss_sim(test_set, train_sets):\n    \"\"\"Jaccard + exact match bonus.\"\"\"\n    sims = np.zeros(len(train_sets), dtype=np.float32)\n    for j, ts in enumerate(train_sets):\n        if not test_set and not ts:\n            sims[j] = 0.5\n            continue\n        union = test_set | ts\n        if len(union) == 0:\n            sims[j] = 0.0\n            continue\n        inter = test_set & ts\n        jaccard = len(inter) / len(union)\n        # Exact match bonus\n        if test_set == ts and len(test_set) > 0:\n            jaccard = min(1.0, jaccard + 0.5)\n        sims[j] = jaccard\n    return sims\n\nprint(\"Computing gloss similarity matrix...\")\nt0 = time.time()\ngloss_sim_matrix = np.zeros((len(test), len(train_good)), dtype=np.float32)\nfor i in range(len(test)):\n    gloss_sim_matrix[i] = compute_gloss_sim(test_gloss_sets[i], train_gloss_sets)\nprint(f\"Gloss similarity: {gloss_sim_matrix.shape} in {time.time()-t0:.1f}s\")\nprint(f\"Mean gloss sim: {gloss_sim_matrix.mean():.4f}, Max: {gloss_sim_matrix.max():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:23:05.638177Z","iopub.execute_input":"2026-04-25T17:23:05.638846Z","iopub.status.idle":"2026-04-25T17:23:36.351676Z","shell.execute_reply.started":"2026-04-25T17:23:05.638817Z","shell.execute_reply":"2026-04-25T17:23:36.350780Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Ensemble Retrieval with Reciprocal Rank Fusion","metadata":{}},{"cell_type":"code","source":"print(\"Computing similarity matrices...\")\nt0 = time.time()\n\n# Cosine similarities from TF-IDF\nsim_char = cosine_similarity(X_test_char, X_train_char).astype(np.float32)\nprint(f\"  Char TF-IDF sim computed: {time.time()-t0:.1f}s\")\n\nt1 = time.time()\nsim_word = cosine_similarity(X_test_word, X_train_word).astype(np.float32)\nprint(f\"  Word TF-IDF sim computed: {time.time()-t1:.1f}s\")\n\n# BERT similarity\nif USE_BERT:\n    t1 = time.time()\n    sim_bert = (test_bert_emb @ train_bert_emb.T).astype(np.float32)\n    print(f\"  BERT sim computed: {time.time()-t1:.1f}s\")\nelse:\n    sim_bert = np.zeros_like(sim_char)\n\n# Weighted ensemble similarity\nsim_ensemble = (\n    W_BERT * sim_bert +\n    W_TFIDF_CHAR * sim_char +\n    W_TFIDF_WORD * sim_word +\n    W_GLOSS * gloss_sim_matrix\n)\n\n# Get top-K neighbors\ntop_k_idx  = np.argsort(-sim_ensemble, axis=1)[:, :K]\ntop_k_sims = np.take_along_axis(sim_ensemble, top_k_idx, axis=1)\n\nprint(f\"\\nEnsemble retrieval done in {time.time()-t0:.1f}s\")\nprint(f\"Top-1 mean sim: {top_k_sims[:, 0].mean():.4f}\")\nprint(f\"Top-K shape: {top_k_idx.shape}\")\n\n# Free memory\ndel sim_char, sim_word, sim_bert, gloss_sim_matrix, sim_ensemble\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:23:36.352976Z","iopub.execute_input":"2026-04-25T17:23:36.353344Z","iopub.status.idle":"2026-04-25T17:23:41.853044Z","shell.execute_reply.started":"2026-04-25T17:23:36.353285Z","shell.execute_reply":"2026-04-25T17:23:41.852261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Smart Token Generation with K-NN Voting","metadata":{}},{"cell_type":"code","source":"def weighted_token_vote(token_seqs, weights, target_len):\n    \"\"\"Vote on tokens at each position using distance-weighted neighbors.\"\"\"\n    result = []\n    for pos in range(target_len):\n        votes = Counter()\n        for seq, w in zip(token_seqs, weights):\n            if pos < len(seq):\n                votes[seq[pos]] += w\n            elif len(seq) > 0:\n                # Wrap around for shorter sequences\n                votes[seq[pos % len(seq)]] += w * 0.5\n        if votes:\n            result.append(votes.most_common(1)[0][0])\n        else:\n            result.append(random.randint(0, VOCAB-1))\n    return result\n\ndef estimate_length(neighbor_indices, neighbor_sims):\n    \"\"\"Estimate target length from weighted neighbor lengths.\"\"\"\n    lens = [train_lens[i] for i in neighbor_indices]\n    w = np.array([max(s, 0.01) for s in neighbor_sims])\n    w = w / w.sum()\n    # Weighted average, but clamped\n    avg_len = int(np.round(np.dot(lens, w)))\n    return max(MIN_LEN, min(MAX_LEN, avg_len))\n\nprint(\"Generating predictions with K-NN voting...\")\nt0 = time.time()\n\npred = pd.DataFrame({\"id\": test[\"id\"].values})\nfor c in TOKEN_COLS:\n    pred[c] = \"\"\n\nfor j in range(len(test)):\n    nb_idx = top_k_idx[j]\n    nb_sim = top_k_sims[j]\n    \n    # Compute weights (softmax of similarities)\n    w = np.exp(nb_sim * 5)  # temperature scaling\n    w = w / w.sum()\n    \n    # Estimate target length\n    tgt_len = estimate_length(nb_idx, nb_sim)\n    \n    # Check if top neighbor is very confident (sim > 0.95)\n    # If so, just use that neighbor directly (exact/near-exact match)\n    if nb_sim[0] > 0.90:\n        best = int(nb_idx[0])\n        base = enforce_len(all_train_tokens[\"base_tokens\"][best], tgt_len)\n        tgt_len = len(base)\n        pred.at[j, \"base_tokens\"] = tokens_to_str(base)\n        for li, c in enumerate(TOKEN_COLS[1:], 1):\n            pred.at[j, c] = tokens_to_str(enforce_len(all_train_tokens[c][best], tgt_len))\n    else:\n        # K-NN voting for each layer\n        for li, c in enumerate(TOKEN_COLS):\n            seqs = [all_train_tokens[c][int(idx)] for idx in nb_idx]\n            voted = weighted_token_vote(seqs, w, tgt_len)\n            if li == 0:\n                base_len = len(voted)\n            else:\n                voted = enforce_len(voted, base_len)\n            pred.at[j, c] = tokens_to_str(voted)\n    \n    if (j+1) % 500 == 0:\n        print(f\"  {j+1}/{len(test)} done ({time.time()-t0:.1f}s)\")\n\nprint(f\"\\nPrediction complete in {time.time()-t0:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:23:41.854112Z","iopub.execute_input":"2026-04-25T17:23:41.854870Z","iopub.status.idle":"2026-04-25T17:23:58.292530Z","shell.execute_reply.started":"2026-04-25T17:23:41.854836Z","shell.execute_reply":"2026-04-25T17:23:58.291841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Submission Validation","metadata":{}},{"cell_type":"code","source":"def validate_row(r):\n    lens = []\n    for c in TOKEN_COLS:\n        t = parse_tokens(r[c])\n        if len(t) < MIN_LEN or len(t) > MAX_LEN:\n            return False\n        if any(x < 0 or x >= VOCAB for x in t):\n            return False\n        lens.append(len(t))\n    return len(set(lens)) == 1\n\nok = sum(validate_row(pred.iloc[i]) for i in range(len(pred)))\nprint(f\"Validation: {ok} / {len(pred)} rows valid\")\nassert ok == len(pred), f\"INVALID ROWS DETECTED: {len(pred)-ok} bad rows!\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:23:58.293784Z","iopub.execute_input":"2026-04-25T17:23:58.294114Z","iopub.status.idle":"2026-04-25T17:23:58.887608Z","shell.execute_reply.started":"2026-04-25T17:23:58.294090Z","shell.execute_reply":"2026-04-25T17:23:58.886758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Save Submission","metadata":{}},{"cell_type":"code","source":"pred.to_csv(OUT_PATH, index=False)\nprint(f\"Saved: {OUT_PATH}\")\nprint(f\"File size: {OUT_PATH.stat().st_size / 1024:.0f} KB\")\ndisplay(pred.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-25T17:23:58.888663Z","iopub.execute_input":"2026-04-25T17:23:58.888969Z","iopub.status.idle":"2026-04-25T17:23:59.085561Z","shell.execute_reply.started":"2026-04-25T17:23:58.888945Z","shell.execute_reply":"2026-04-25T17:23:59.084691Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Summary of Improvements\n\n| Feature | Baseline (0.43) | This Notebook |\n|---------|----------------|---------------|\n| Text Encoding | TF-IDF char 3-6 gram | BERT + TF-IDF char + TF-IDF word + Gloss match |\n| Retrieval | 1-NN | K=10 NN with distance-weighted voting |\n| Similarity | Cosine (single) | Weighted ensemble of 4 methods |\n| Length | Copy from neighbor | Weighted average of K neighbors |\n| Exact Match | None | High-confidence shortcut (sim > 0.90) |\n| Gloss Handling | Concatenated with sentence | Jaccard + exact match bonus |","metadata":{}}]}