{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Motion-S Starter: Text-to-Motion Retrieval Baseline\n\nThis notebook provides a simple and reliable baseline for the **Motion-S: Hierarchical Text-to-Motion Generation for Sign Language** competition.\n\nThe approach uses:\n- Text preprocessing (gloss + sentence)\n- TF-IDF feature extraction\n- Nearest Neighbor retrieval\n- Direct token transfer from the most similar training sample\n\nThe goal of this notebook is to:\n- Produce a **valid submission**\n- Provide a **fast and reproducible baseline**\n- Serve as a starting point for further improvements","metadata":{}},{"cell_type":"markdown","source":"## Navigation\n\n- [Setup and Data Loading](#setup-and-data-loading)\n- [Token Utilities](#token-utilities)\n- [Training Data Validation](#training-data-validation)\n- [TF-IDF Feature Extraction](#tf-idf-feature-extraction)\n- [Nearest Neighbor Retrieval](#nearest-neighbor-retrieval)\n- [Token Prediction](#token-prediction)\n- [Submission Validation](#submission-validation)\n- [Save Submission](#save-submission)","metadata":{}},{"cell_type":"markdown","source":"## Setup and Data Loading","metadata":{}},{"cell_type":"code","source":"import os, re, gc, random\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.neighbors import NearestNeighbors\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\n\nINPUT_DIR = Path(\"/kaggle/input/motion-s-hierarchical-text-to-motion-generation-for-sign-language\")\n\nTRAIN_CSV  = INPUT_DIR / \"train.csv\"\nTEST_CSV   = INPUT_DIR / \"test.csv\"\nSAMPLE_SUB = INPUT_DIR / \"sample_submission.csv\"\n\nOUT_PATH = Path(\"/kaggle/working/submission.csv\")\n\ntrain = pd.read_csv(TRAIN_CSV)\ntest  = pd.read_csv(TEST_CSV)\nsample_sub = pd.read_csv(SAMPLE_SUB)\n\nprint(\"Train:\", train.shape)\nprint(\"Test:\", test.shape)\n\n_ws_re = re.compile(r\"\\s+\")\n\ndef norm_text(x):\n    if not isinstance(x, str):\n        return \"\"\n    x = x.strip()\n    x = _ws_re.sub(\" \", x)\n    return x\n\ndef build_text(df):\n    g = df[\"gloss\"].map(norm_text) if \"gloss\" in df.columns else \"\"\n    s = df[\"sentence\"].map(norm_text) if \"sentence\" in df.columns else \"\"\n    return (g + \" || \" + g + \" || \" + s).fillna(\"\")\n\ntrain_text = build_text(train)\ntest_text  = build_text(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:27.525870Z","iopub.execute_input":"2026-02-25T09:43:27.526620Z","iopub.status.idle":"2026-02-25T09:43:27.990712Z","shell.execute_reply.started":"2026-02-25T09:43:27.526589Z","shell.execute_reply":"2026-02-25T09:43:27.990029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Token Utilities","metadata":{}},{"cell_type":"code","source":"TOKEN_COLS = [\"base_tokens\", \"residual_1\", \"residual_2\", \"residual_3\", \"residual_4\", \"residual_5\"]\n\ndef parse_tokens(tok_str):\n    if not isinstance(tok_str, str):\n        return []\n    tok_str = tok_str.strip()\n    if tok_str == \"\":\n        return []\n    return [int(x) for x in tok_str.split()]\n\ndef tokens_to_str(tokens):\n    return \" \".join(map(str, tokens))\n\ndef enforce_len(tokens, min_len=40, max_len=800):\n    if len(tokens) == 0:\n        tokens = [random.randint(0, 511) for _ in range(min_len)]\n        return tokens\n    if len(tokens) < min_len:\n        reps = (min_len + len(tokens) - 1) // len(tokens)\n        tokens = (tokens * reps)[:min_len]\n    if len(tokens) > max_len:\n        tokens = tokens[:max_len]\n    return tokens","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:27.991924Z","iopub.execute_input":"2026-02-25T09:43:27.992208Z","iopub.status.idle":"2026-02-25T09:43:27.997904Z","shell.execute_reply.started":"2026-02-25T09:43:27.992184Z","shell.execute_reply":"2026-02-25T09:43:27.997374Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training Data Validation","metadata":{}},{"cell_type":"code","source":"good_mask = np.ones(len(train), dtype=bool)\ntrain_lens = []\n\nfor i, row in train.iterrows():\n    lens = []\n    ok = True\n    for c in TOKEN_COLS:\n        t = parse_tokens(row[c])\n        if any((x < 0 or x > 511) for x in t):\n            ok = False\n            break\n        lens.append(len(t))\n    if (not ok) or (len(set(lens)) != 1) or (lens[0] < 1):\n        good_mask[i] = False\n    else:\n        train_lens.append(lens[0])\n\ntrain_good = train.loc[good_mask].reset_index(drop=True)\ntrain_text_good = train_text.loc[good_mask].reset_index(drop=True)\n\nprint(\"Usable train rows:\", len(train_good))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:27.998653Z","iopub.execute_input":"2026-02-25T09:43:27.998908Z","iopub.status.idle":"2026-02-25T09:43:30.491014Z","shell.execute_reply.started":"2026-02-25T09:43:27.998884Z","shell.execute_reply":"2026-02-25T09:43:30.490332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TF-IDF Feature Extraction","metadata":{}},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(\n    lowercase=True,\n    analyzer=\"char_wb\",\n    ngram_range=(3, 6),\n    min_df=2,\n    max_features=250000,\n)\n\nX_train = vectorizer.fit_transform(train_text_good)\nX_test  = vectorizer.transform(test_text)\n\nprint(\"Feature shapes:\", X_train.shape, X_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:30.491999Z","iopub.execute_input":"2026-02-25T09:43:30.492302Z","iopub.status.idle":"2026-02-25T09:43:32.037871Z","shell.execute_reply.started":"2026-02-25T09:43:30.492273Z","shell.execute_reply":"2026-02-25T09:43:32.037249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Nearest Neighbor Retrieval","metadata":{}},{"cell_type":"code","source":"nn = NearestNeighbors(n_neighbors=1, metric=\"cosine\", algorithm=\"brute\")\nnn.fit(X_train)\n\ndist, idx = nn.kneighbors(X_test, return_distance=True)\n\nidx = idx.reshape(-1)\ndist = dist.reshape(-1)\n\nprint(\"Retrieval completed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:32.039502Z","iopub.execute_input":"2026-02-25T09:43:32.039709Z","iopub.status.idle":"2026-02-25T09:43:34.182668Z","shell.execute_reply.started":"2026-02-25T09:43:32.039690Z","shell.execute_reply":"2026-02-25T09:43:34.181874Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Token Prediction","metadata":{}},{"cell_type":"code","source":"pred = pd.DataFrame({\"id\": test[\"id\"].values})\n\nfor c in TOKEN_COLS:\n    pred[c] = \"\"\n\ntrain_tokens_cache = {}\nfor c in TOKEN_COLS:\n    train_tokens_cache[c] = [parse_tokens(x) for x in train_good[c].astype(str).tolist()]\n\nMIN_LEN, MAX_LEN = 40, 800\n\nfor j in range(len(test)):\n    k = int(idx[j])\n\n    layers = []\n    for c in TOKEN_COLS:\n        layers.append(train_tokens_cache[c][k])\n\n    L = min(len(t) for t in layers)\n    if L <= 0:\n        L = MIN_LEN\n\n    base = enforce_len(layers[0][:L], MIN_LEN, MAX_LEN)\n    L2 = len(base)\n\n    fixed_layers = [base]\n    for li in range(1, 6):\n        t = layers[li][:L]\n        t = enforce_len(t, L2, L2)\n        fixed_layers.append(t)\n\n    pred.at[j, \"base_tokens\"] = tokens_to_str(fixed_layers[0])\n    pred.at[j, \"residual_1\"]  = tokens_to_str(fixed_layers[1])\n    pred.at[j, \"residual_2\"]  = tokens_to_str(fixed_layers[2])\n    pred.at[j, \"residual_3\"]  = tokens_to_str(fixed_layers[3])\n    pred.at[j, \"residual_4\"]  = tokens_to_str(fixed_layers[4])\n    pred.at[j, \"residual_5\"]  = tokens_to_str(fixed_layers[5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:34.183599Z","iopub.execute_input":"2026-02-25T09:43:34.183862Z","iopub.status.idle":"2026-02-25T09:43:36.343399Z","shell.execute_reply.started":"2026-02-25T09:43:34.183842Z","shell.execute_reply":"2026-02-25T09:43:36.342376Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission Validation","metadata":{}},{"cell_type":"code","source":"def validate_row(r):\n    lens = []\n    for c in TOKEN_COLS:\n        t = parse_tokens(r[c])\n        if len(t) < 40 or len(t) > 800:\n            return False\n        if any((x < 0 or x > 511) for x in t):\n            return False\n        lens.append(len(t))\n    return len(set(lens)) == 1\n\ncheck_n = min(200, len(pred))\nok = 0\nfor i in range(check_n):\n    ok += int(validate_row(pred.iloc[i]))\n\nprint(\"Validation:\", ok, \"/\", check_n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:36.344402Z","iopub.execute_input":"2026-02-25T09:43:36.344699Z","iopub.status.idle":"2026-02-25T09:43:36.392039Z","shell.execute_reply.started":"2026-02-25T09:43:36.344669Z","shell.execute_reply":"2026-02-25T09:43:36.391523Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Submission","metadata":{}},{"cell_type":"code","source":"pred.to_csv(OUT_PATH, index=False)\nprint(\"Saved:\", OUT_PATH)\ndisplay(pred.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-25T09:43:36.392905Z","iopub.execute_input":"2026-02-25T09:43:36.393161Z","iopub.status.idle":"2026-02-25T09:43:36.583739Z","shell.execute_reply.started":"2026-02-25T09:43:36.393117Z","shell.execute_reply":"2026-02-25T09:43:36.583151Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Summary\n\n**What this notebook does**\n- Uses a simple nearest-neighbor approach to generate motion tokens\n- Runs fast and creates a valid submission without any model training\n\n**What you can try next**\n- Predict motion length more accurately\n- Use several nearest neighbors instead of one\n- Replace TF-IDF with better text embeddings","metadata":{}}]}