{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Motion-S: Length-Aware Retrieval Baseline\n\nThis notebook presents a retrieval-based baseline for the Motion-S competition.\n\n**What the pipeline does**\n- Cleans and combines gloss and sentence text\n- Converts text into numerical features\n- Finds similar training samples using Top-K nearest neighbors\n- Predicts sequence length and selects motions with similar duration\n- Copies and adjusts token sequences to match the required format\n\n**Key properties**\n- Produces valid submissions\n- Runs within standard Kaggle limits\n- Simple and easy to modify\n- Can be used as a starting point for further experiments","metadata":{}},{"cell_type":"markdown","source":"## Navigation\n- [Setup and Data Loading](#setup-and-data-loading)\n- [Text Preprocessing](#text-preprocessing)\n- [Token Parsing and Train Filtering](#token-parsing-and-train-filtering)\n- [Length Features and Length Model](#length-features-and-length-model)\n- [Embeddings Backend](#embeddings-backend)\n- [Top-K Retrieval](#top-k-retrieval)\n- [Length-Aware Sampling](#length-aware-sampling)\n- [Build Predictions](#build-predictions)\n- [Validate Submission](#validate-submission)\n- [Save Submission](#save-submission)","metadata":{}},{"cell_type":"markdown","source":"## Setup and Data Loading","metadata":{}},{"cell_type":"code","source":"import os, re, math, random, gc\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\n\nINPUT_DIR = Path(\"/kaggle/input/motion-s-hierarchical-text-to-motion-generation-for-sign-language\")\nTRAIN_CSV  = INPUT_DIR / \"train.csv\"\nTEST_CSV   = INPUT_DIR / \"test.csv\"\nOUT_PATH = Path(\"/kaggle/working/submission.csv\")\n\ntrain = pd.read_csv(TRAIN_CSV)\ntest  = pd.read_csv(TEST_CSV)\n\nprint(\"train:\", train.shape)\nprint(\"test :\", test.shape)\ndisplay(train.head(2))\ndisplay(test.head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:04.330334Z","iopub.execute_input":"2026-02-25T09:57:04.330524Z","iopub.status.idle":"2026-02-25T09:57:06.144421Z","shell.execute_reply.started":"2026-02-25T09:57:04.330504Z","shell.execute_reply":"2026-02-25T09:57:06.143703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Text Preprocessing","metadata":{}},{"cell_type":"code","source":"_ws_re = re.compile(r\"\\s+\")\n_punct_re = re.compile(r\"[^A-Za-z0-9_'\\- ]+\")\n_multi_bar = re.compile(r\"\\|+\")\n\ndef norm_text(x):\n    if not isinstance(x, str):\n        return \"\"\n    x = x.strip()\n    x = x.replace(\"\\n\", \" \").replace(\"\\t\", \" \")\n    x = _multi_bar.sub(\"|\", x)\n    x = _ws_re.sub(\" \", x)\n    return x\n\ndef clean_gloss(x):\n    x = norm_text(x)\n    x = x.replace(\"//\", \" \")\n    x = _ws_re.sub(\" \", x).strip()\n    return x\n\ndef clean_sentence(x):\n    x = norm_text(x).lower()\n    x = _punct_re.sub(\" \", x)\n    x = _ws_re.sub(\" \", x).strip()\n    return x\n\ndef build_text(df, gloss_w=2):\n    g = df[\"gloss\"].map(clean_gloss) if \"gloss\" in df.columns else \"\"\n    s = df[\"sentence\"].map(clean_sentence) if \"sentence\" in df.columns else \"\"\n    if isinstance(g, str):\n        return (g + \" || \" + s).strip()\n    return ((g + \" || \") * gloss_w + s).fillna(\"\").map(norm_text)\n\ntrain_text = build_text(train, gloss_w=2)\ntest_text  = build_text(test, gloss_w=2)\n\nprint(train_text.iloc[0])\nprint(test_text.iloc[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:06.146361Z","iopub.execute_input":"2026-02-25T09:57:06.146622Z","iopub.status.idle":"2026-02-25T09:57:06.565371Z","shell.execute_reply.started":"2026-02-25T09:57:06.146600Z","shell.execute_reply":"2026-02-25T09:57:06.564624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Token Parsing and Train Filtering","metadata":{}},{"cell_type":"code","source":"TOKEN_COLS = [\"base_tokens\", \"residual_1\", \"residual_2\", \"residual_3\", \"residual_4\", \"residual_5\"]\n\ndef parse_tokens(s):\n    if not isinstance(s, str):\n        return []\n    s = s.strip()\n    if s == \"\":\n        return []\n    return [int(x) for x in s.split()]\n\ndef tokens_to_str(tokens):\n    return \" \".join(map(str, tokens))\n\ndef enforce_exact_len(tokens, L):\n    L = int(L)\n    if L <= 0:\n        L = 40\n    if len(tokens) == 0:\n        return [random.randint(0, 511) for _ in range(L)]\n    if len(tokens) < L:\n        reps = (L + len(tokens) - 1) // len(tokens)\n        tokens = (tokens * reps)[:L]\n    if len(tokens) > L:\n        tokens = tokens[:L]\n    return tokens\n\ngood = np.ones(len(train), dtype=bool)\nlen_good = np.zeros(len(train), dtype=np.int32)\n\nfor i, row in train.iterrows():\n    lens = []\n    ok = True\n    for c in TOKEN_COLS:\n        t = parse_tokens(row[c])\n        if len(t) == 0:\n            ok = False\n            break\n        if any((x < 0 or x > 511) for x in t):\n            ok = False\n            break\n        lens.append(len(t))\n    if (not ok) or (len(set(lens)) != 1) or (lens[0] < 40) or (lens[0] > 800):\n        good[i] = False\n    else:\n        len_good[i] = lens[0]\n\ntrain_good = train.loc[good].reset_index(drop=True)\ntrain_text_good = train_text.loc[good].reset_index(drop=True)\nlen_good = len_good[good]\n\nprint(\"usable train:\", len(train_good), \"/\", len(train))\nprint(\"length stats:\", float(np.mean(len_good)), int(np.min(len_good)), int(np.max(len_good)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:06.569141Z","iopub.execute_input":"2026-02-25T09:57:06.569447Z","iopub.status.idle":"2026-02-25T09:57:09.146981Z","shell.execute_reply.started":"2026-02-25T09:57:06.569423Z","shell.execute_reply":"2026-02-25T09:57:09.146088Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Length Features and Length Model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.model_selection import KFold\n\ndef length_features(text_series):\n    feats = np.zeros((len(text_series), 8), dtype=np.float32)\n    for i, t in enumerate(text_series):\n        t = \"\" if not isinstance(t, str) else t\n        n_char = len(t)\n        n_words = len(t.split())\n        n_bar = t.count(\"|\")\n        n_digits = sum(ch.isdigit() for ch in t)\n        n_upper = sum(ch.isupper() for ch in t)\n        n_dash = t.count(\"-\")\n        avg_word = (sum(len(w) for w in t.split()) / max(1, n_words)) if n_words > 0 else 0.0\n        feats[i] = [n_char, n_words, n_bar, n_digits, n_upper, n_dash, avg_word, 1.0]\n    return feats\n\nX_len = length_features(train_text_good)\ny_len = len_good.astype(np.float32)\n\nkf = KFold(n_splits=5, shuffle=True, random_state=SEED)\noof = np.zeros(len(train_good), dtype=np.float32)\n\nfor tr_idx, va_idx in kf.split(X_len):\n    m = Ridge(alpha=2.0, random_state=SEED)\n    m.fit(X_len[tr_idx], y_len[tr_idx])\n    oof[va_idx] = m.predict(X_len[va_idx])\n\nrmse = float(np.sqrt(np.mean((oof - y_len) ** 2)))\nprint(\"Length RMSE:\", rmse)\n\nlen_model = Ridge(alpha=2.0, random_state=SEED)\nlen_model.fit(X_len, y_len)\n\nX_len_test = length_features(test_text)\npred_len_test = len_model.predict(X_len_test)\n\npred_len_test = np.clip(pred_len_test, 40, 800)\npred_len_test = np.round(pred_len_test).astype(np.int32)\n\nprint(\"Pred length stats:\", float(np.mean(pred_len_test)), int(np.min(pred_len_test)), int(np.max(pred_len_test)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:09.148129Z","iopub.execute_input":"2026-02-25T09:57:09.148473Z","iopub.status.idle":"2026-02-25T09:57:10.256241Z","shell.execute_reply.started":"2026-02-25T09:57:09.148441Z","shell.execute_reply":"2026-02-25T09:57:10.255304Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Embeddings Backend","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ndef l2_normalize(a, eps=1e-12):\n    n = np.linalg.norm(a, axis=1, keepdims=True)\n    return a / np.maximum(n, eps)\n\ndef try_gpu_embeddings(train_text, test_text, batch_size=256):\n    try:\n        import torch\n        from transformers import AutoTokenizer, AutoModel\n    except Exception:\n        return None, None, \"none\"\n\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    if device.type != \"cuda\":\n        return None, None, \"none\"\n\n    model_name = \"sentence-transformers/all-MiniLM-L6-v2\"\n    try:\n        tok = AutoTokenizer.from_pretrained(model_name, local_files_only=True)\n        mdl = AutoModel.from_pretrained(model_name, local_files_only=True).to(device)\n    except Exception:\n        return None, None, \"none\"\n\n    def embed(texts):\n        out = []\n        mdl.eval()\n        with torch.no_grad():\n            for i in range(0, len(texts), batch_size):\n                batch = texts[i:i+batch_size].tolist()\n                enc = tok(batch, padding=True, truncation=True, max_length=256, return_tensors=\"pt\").to(device)\n                h = mdl(**enc).last_hidden_state\n                m = enc[\"attention_mask\"].unsqueeze(-1).float()\n                pooled = (h * m).sum(dim=1) / torch.clamp(m.sum(dim=1), min=1e-6)\n                out.append(pooled.detach().cpu().numpy().astype(np.float32))\n        out = np.vstack(out)\n        return l2_normalize(out)\n\n    Xtr = embed(train_text)\n    Xte = embed(test_text)\n    return Xtr, Xte, \"gpu-st\"\n\nX_train_emb, X_test_emb, backend = try_gpu_embeddings(train_text_good, test_text, batch_size=256)\n\nif backend == \"none\":\n    vectorizer = TfidfVectorizer(\n        lowercase=True,\n        analyzer=\"char_wb\",\n        ngram_range=(3, 6),\n        min_df=2,\n        max_features=350000,\n    )\n    X_train_emb = vectorizer.fit_transform(train_text_good)\n    X_test_emb  = vectorizer.transform(test_text)\n    backend = \"tfidf\"\n\nprint(\"Embedding backend:\", backend)\nprint(\"Train emb shape:\", X_train_emb.shape)\nprint(\"Test  emb shape:\", X_test_emb.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:10.257449Z","iopub.execute_input":"2026-02-25T09:57:10.258794Z","iopub.status.idle":"2026-02-25T09:57:28.124352Z","shell.execute_reply.started":"2026-02-25T09:57:10.258768Z","shell.execute_reply":"2026-02-25T09:57:28.123636Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Top-K Retrieval","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import NearestNeighbors\n\nK = 20\n\nnn = NearestNeighbors(n_neighbors=K, metric=\"cosine\", algorithm=\"brute\")\nnn.fit(X_train_emb)\n\ndist, idx = nn.kneighbors(X_test_emb, return_distance=True)\n\nprint(\"idx:\", idx.shape, \"dist:\", dist.shape)\nprint(\"example neighbors:\", idx[0, :5], dist[0, :5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:28.126376Z","iopub.execute_input":"2026-02-25T09:57:28.126817Z","iopub.status.idle":"2026-02-25T09:57:30.201431Z","shell.execute_reply.started":"2026-02-25T09:57:28.126792Z","shell.execute_reply":"2026-02-25T09:57:30.200660Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Length-Aware Sampling","metadata":{}},{"cell_type":"code","source":"def choose_neighbor(nei_idx, nei_dist, target_len, train_lens, tau=0.08, lam=0.015):\n    Ls = train_lens[nei_idx].astype(np.float32)\n    dlen = np.abs(Ls - float(target_len))\n    score = (nei_dist.astype(np.float32) / max(tau, 1e-6)) + (dlen.astype(np.float32) * float(lam))\n    score = score - score.min()\n    w = np.exp(-score)\n    w = w / max(w.sum(), 1e-12)\n    pick = np.random.choice(len(nei_idx), p=w)\n    return int(nei_idx[pick]), w\n\ntrain_lens_arr = len_good.astype(np.int32)\n\npicked_train = np.zeros(len(test), dtype=np.int32)\npicked_wmax = np.zeros(len(test), dtype=np.float32)\n\nfor i in range(len(test)):\n    k, w = choose_neighbor(idx[i], dist[i], pred_len_test[i], train_lens_arr, tau=0.08, lam=0.015)\n    picked_train[i] = k\n    picked_wmax[i] = float(w.max())\n\nprint(\"picked_wmax mean:\", float(picked_wmax.mean()))\nprint(\"unique picked train:\", int(len(np.unique(picked_train))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:30.202432Z","iopub.execute_input":"2026-02-25T09:57:30.202943Z","iopub.status.idle":"2026-02-25T09:57:30.353322Z","shell.execute_reply.started":"2026-02-25T09:57:30.202919Z","shell.execute_reply":"2026-02-25T09:57:30.352489Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build Predictions","metadata":{}},{"cell_type":"code","source":"train_tokens_cache = {}\nfor c in TOKEN_COLS:\n    train_tokens_cache[c] = [parse_tokens(x) for x in train_good[c].astype(str).tolist()]\n\npred = pd.DataFrame({\"id\": test[\"id\"].values})\nfor c in TOKEN_COLS:\n    pred[c] = \"\"\n\nfor i in range(len(test)):\n    k = int(picked_train[i])\n    target_L = int(pred_len_test[i])\n\n    layers = [train_tokens_cache[c][k] for c in TOKEN_COLS]\n    L0 = min(len(t) for t in layers)\n    L0 = int(max(40, min(800, L0)))\n    base = enforce_exact_len(layers[0][:L0], target_L)\n    Lf = len(base)\n\n    fixed = [base]\n    for li in range(1, 6):\n        fixed.append(enforce_exact_len(layers[li][:L0], Lf))\n\n    pred.at[i, \"base_tokens\"] = tokens_to_str(fixed[0])\n    pred.at[i, \"residual_1\"]  = tokens_to_str(fixed[1])\n    pred.at[i, \"residual_2\"]  = tokens_to_str(fixed[2])\n    pred.at[i, \"residual_3\"]  = tokens_to_str(fixed[3])\n    pred.at[i, \"residual_4\"]  = tokens_to_str(fixed[4])\n    pred.at[i, \"residual_5\"]  = tokens_to_str(fixed[5])\n\ndisplay(pred.head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:30.354347Z","iopub.execute_input":"2026-02-25T09:57:30.354660Z","iopub.status.idle":"2026-02-25T09:57:32.605041Z","shell.execute_reply.started":"2026-02-25T09:57:30.354635Z","shell.execute_reply":"2026-02-25T09:57:32.604240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Validate Submission","metadata":{}},{"cell_type":"code","source":"def validate_row(r):\n    lens = []\n    for c in TOKEN_COLS:\n        t = parse_tokens(r[c])\n        if len(t) < 40 or len(t) > 800:\n            return False\n        if any((x < 0 or x > 511) for x in t):\n            return False\n        lens.append(len(t))\n    return len(set(lens)) == 1\n\nn = len(pred)\nok = 0\nfor i in range(n):\n    ok += int(validate_row(pred.iloc[i]))\n\nprint(\"valid rows:\", ok, \"/\", n)\nassert ok == n\n\nassert list(pred.columns) == [\"id\"] + TOKEN_COLS\nassert len(pred) == len(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:32.606126Z","iopub.execute_input":"2026-02-25T09:57:32.606564Z","iopub.status.idle":"2026-02-25T09:57:33.245986Z","shell.execute_reply.started":"2026-02-25T09:57:32.606525Z","shell.execute_reply":"2026-02-25T09:57:33.245150Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Submission","metadata":{}},{"cell_type":"code","source":"pred.to_csv(OUT_PATH, index=False)\nprint(\"Saved:\", OUT_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:57:33.247026Z","iopub.execute_input":"2026-02-25T09:57:33.247444Z","iopub.status.idle":"2026-02-25T09:57:33.438234Z","shell.execute_reply.started":"2026-02-25T09:57:33.247418Z","shell.execute_reply":"2026-02-25T09:57:33.437483Z"}},"outputs":[],"execution_count":null}]}