{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"CSV_PATH = \"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language/train.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T05:33:58.91614Z","iopub.execute_input":"2026-05-13T05:33:58.916423Z","iopub.status.idle":"2026-05-13T05:33:58.925416Z","shell.execute_reply.started":"2026-05-13T05:33:58.916367Z","shell.execute_reply":"2026-05-13T05:33:58.924663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q transformers datasets sentencepiece accelerate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T05:41:10.207064Z","iopub.execute_input":"2026-05-13T05:41:10.207873Z","iopub.status.idle":"2026-05-13T05:41:14.621303Z","shell.execute_reply.started":"2026-05-13T05:41:10.207842Z","shell.execute_reply":"2026-05-13T05:41:14.620568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from huggingface_hub import login\nlogin(token=\"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\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T05:44:35.406121Z","iopub.execute_input":"2026-05-13T05:44:35.407091Z","iopub.status.idle":"2026-05-13T05:44:35.670615Z","shell.execute_reply.started":"2026-05-13T05:44:35.407045Z","shell.execute_reply":"2026-05-13T05:44:35.669793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datasets import Dataset\nfrom transformers import (\n    T5Tokenizer, \n    T5ForConditionalGeneration, \n    Seq2SeqTrainingArguments, \n    Seq2SeqTrainer, \n    DataCollatorForSeq2Seq\n)\n\n# ---------------------------------------------------------\n# 1. Load and Prepare the Data\n# ---------------------------------------------------------\n# Load your dataset\ndf = pd.read_csv(CSV_PATH)\n\n# Drop missing values in crucial columns\ndf = df.dropna(subset=['sentence', 'gloss'])\n\n# Convert to HuggingFace Dataset\n# We will do a 90/10 train/val split\ndataset = Dataset.from_pandas(df[['sentence', 'gloss']])\ndataset = dataset.train_test_split(test_size=0.1, seed=42)\n\n# ---------------------------------------------------------\n# 2. Initialize Model & Tokenizer\n# ---------------------------------------------------------\nmodel_checkpoint = \"t5-small\"\n\n# T5 uses a SentencePiece tokenizer\ntokenizer = T5Tokenizer.from_pretrained(model_checkpoint)\nmodel = T5ForConditionalGeneration.from_pretrained(model_checkpoint)\n\n# ---------------------------------------------------------\n# 3. Tokenization Logic\n# ---------------------------------------------------------\n# T5 was trained on multiple tasks, so we give it a task prefix\nPREFIX = \"translate English to Kenyan Sign Language gloss: \"\nMAX_INPUT_LENGTH = 128\nMAX_TARGET_LENGTH = 128\n\ndef preprocess_function(examples):\n    # Prefix the English sentences\n    inputs = [PREFIX + text for text in examples[\"sentence\"]]\n    targets = examples[\"gloss\"]\n    \n    # Tokenize the inputs\n    model_inputs = tokenizer(\n        inputs, \n        max_length=MAX_INPUT_LENGTH, \n        truncation=True\n    )\n    \n    # Tokenize the targets (labels)\n    # T5 requires the labels to be tokenized separately\n    labels = tokenizer(\n        text_target=targets, \n        max_length=MAX_TARGET_LENGTH, \n        truncation=True\n    )\n    \n    model_inputs[\"labels\"] = labels[\"input_ids\"]\n    return model_inputs\n\n# Apply the tokenization to the dataset\ntokenized_datasets = dataset.map(\n    preprocess_function, \n    batched=True, \n    remove_columns=dataset[\"train\"].column_names\n)\n\n# ---------------------------------------------------------\n# 4. Training Setup\n# ---------------------------------------------------------\n# The data collator handles dynamic padding for the batches\ndata_collator = DataCollatorForSeq2Seq(\n    tokenizer=tokenizer, \n    model=model, \n    label_pad_token_id=-100 # -100 tells PyTorch to ignore these tokens in loss calculation\n)\n\n# Define training hyperparameters\nargs = Seq2SeqTrainingArguments(\n    output_dir=\"./t5-ksl-gloss-model\",\n    eval_strategy=\"epoch\",\n    learning_rate=5e-5,       # T5 requires a slightly higher LR than BERT\n    per_device_train_batch_size=32,\n    per_device_eval_batch_size=32,\n    weight_decay=0.01,\n    save_total_limit=2,\n    num_train_epochs=100,       # Adjust based on dataset size and convergence\n    predict_with_generate=True, # Required for Seq2Seq generation metrics\n    fp16=True,                # Mixed precision for faster training on Kaggle GPUs\n    logging_steps=50,\n)\n\ntrainer = Seq2SeqTrainer(\n    model=model,\n    args=args,\n    train_dataset=tokenized_datasets[\"train\"],\n    eval_dataset=tokenized_datasets[\"test\"],\n    processing_class=tokenizer,\n    data_collator=data_collator,\n)\n\n# ---------------------------------------------------------\n# 5. Execute Training\n# ---------------------------------------------------------\nprint(\"Starting training...\")\ntrainer.train()\n\n# Save the final model weights and tokenizer\ntrainer.save_model(\"./t5-ksl-gloss-final\")\nprint(\"Model saved to ./t5-ksl-gloss-final\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T05:44:45.142611Z","iopub.execute_input":"2026-05-13T05:44:45.143169Z","iopub.status.idle":"2026-05-13T05:46:04.268858Z","shell.execute_reply.started":"2026-05-13T05:44:45.143136Z","shell.execute_reply":"2026-05-13T05:46:04.267929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inference Test\ntest_sentence = \"But, remember, that I cannot use a four, until I first have a six, and then a five and than a four.\"\ninput_text = PREFIX + test_sentence\n\ninputs = tokenizer(input_text, return_tensors=\"pt\").to(model.device)\n\n# Generate output\noutputs = model.generate(\n    **inputs, \n    max_length=50, \n    num_beams=4, # Beam search for better translation\n    early_stopping=True\n)\n\npredicted_gloss = tokenizer.decode(outputs[0], skip_special_tokens=True)\nprint(f\"English: {test_sentence}\")\nprint(f\"Predicted Gloss: {predicted_gloss}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T16:57:16.245805Z","iopub.execute_input":"2026-04-28T16:57:16.246504Z","iopub.status.idle":"2026-04-28T16:57:16.67212Z","shell.execute_reply.started":"2026-04-28T16:57:16.246474Z","shell.execute_reply":"2026-04-28T16:57:16.671299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Force strict uppercase for pipeline consistency\nclean_gloss = predicted_gloss.upper() \n\nprint(f\"Cleaned Pipeline Output: {clean_gloss}\")\n# Output: ME WORK ABOUT MY STORY//","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-28T16:58:10.227235Z","iopub.execute_input":"2026-04-28T16:58:10.227934Z","iopub.status.idle":"2026-04-28T16:58:10.231924Z","shell.execute_reply.started":"2026-04-28T16:58:10.227903Z","shell.execute_reply":"2026-04-28T16:58:10.231091Z"}},"outputs":[],"execution_count":null}]}