{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Bengali to English Translation\n","metadata":{}},{"cell_type":"markdown","source":"### facebook/nllb-200-distilled-600M\nThe facebook/nllb-200-distilled-600M model is a machine translation model that was trained on the Flores-200 dataset, which consists of parallel text data for 200 languages. It is a distilled model, which means that it was trained on a smaller version of the original model, the facebook/nllb-200-1.3B model. This makes it smaller and faster than the original model, while still maintaining good translation quality.\n\nThe model is primarily intended for research in machine translation, especially for low-resource languages. It can be used to translate single sentences between any of the 200 languages in the Flores-200 dataset.","metadata":{}},{"cell_type":"code","source":"!pip download transformers==4.25.1\n!pip install /kaggle/working/transformers-4.25.1-py3-none-any.whl","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-20T13:39:34.276008Z","iopub.execute_input":"2023-08-20T13:39:34.277008Z","iopub.status.idle":"2023-08-20T13:40:04.080199Z","shell.execute_reply.started":"2023-08-20T13:39:34.276921Z","shell.execute_reply":"2023-08-20T13:40:04.078925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport torch\nimport pandas as pd\nfrom transformers import AutoModelForSeq2SeqLM, AutoTokenizer\nfrom tqdm.auto import tqdm\n%env TOKENIZERS_PARALLELISM=true\ndevice = torch.device('cuda')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-20T13:40:04.083202Z","iopub.execute_input":"2023-08-20T13:40:04.083588Z","iopub.status.idle":"2023-08-20T13:40:11.669721Z","shell.execute_reply.started":"2023-08-20T13:40:04.083555Z","shell.execute_reply":"2023-08-20T13:40:11.668606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    INPUT = '/kaggle/input/bengaliai-speech'\n    TRANS_MODEL = 'facebook/nllb-200-distilled-600M'","metadata":{"execution":{"iopub.status.busy":"2023-08-20T13:40:11.671196Z","iopub.execute_input":"2023-08-20T13:40:11.67212Z","iopub.status.idle":"2023-08-20T13:40:11.679247Z","shell.execute_reply.started":"2023-08-20T13:40:11.672059Z","shell.execute_reply":"2023-08-20T13:40:11.677008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f'{CFG.INPUT}/train.csv')\ndisplay(df)","metadata":{"execution":{"iopub.status.busy":"2023-08-20T13:44:03.414931Z","iopub.execute_input":"2023-08-20T13:44:03.415324Z","iopub.status.idle":"2023-08-20T13:44:04.113115Z","shell.execute_reply.started":"2023-08-20T13:44:03.415292Z","shell.execute_reply":"2023-08-20T13:44:04.112142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSeq2SeqLM.from_pretrained(CFG.TRANS_MODEL)\ntorch.save(model.state_dict(), './nllb-200-distilled-600M.pth')\nmodel.eval()\nmodel.to(device)\ntokenizer = AutoTokenizer.from_pretrained(CFG.TRANS_MODEL)\ntokenizer.save_pretrained('./tokenizer/')","metadata":{"execution":{"iopub.status.busy":"2023-08-20T13:40:12.350023Z","iopub.execute_input":"2023-08-20T13:40:12.351001Z","iopub.status.idle":"2023-08-20T13:42:15.875495Z","shell.execute_reply.started":"2023-08-20T13:40:12.350959Z","shell.execute_reply":"2023-08-20T13:42:15.874133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def translate_by_nllb(text, tokenizer, model, device):\n    inputs = tokenizer(text, return_tensors=\"pt\")    \n    for k, v in inputs.items():\n        inputs[k] = v.to(device)\n    translated_tokens = model.generate(\n        **inputs, forced_bos_token_id=tokenizer.lang_code_to_id[\"eng_Latn\"], max_length=64\n    )\n    return tokenizer.batch_decode(translated_tokens, skip_special_tokens=True)[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-20T13:42:15.877377Z","iopub.execute_input":"2023-08-20T13:42:15.87815Z","iopub.status.idle":"2023-08-20T13:42:15.886207Z","shell.execute_reply.started":"2023-08-20T13:42:15.878082Z","shell.execute_reply":"2023-08-20T13:42:15.884826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, row in tqdm(df.iterrows(), total=len(df)):\n    if i<5000:\n        df.loc[i,'eng'] = translate_by_nllb(row['sentence'], tokenizer, model, device)        \ndisplay(df[0:5000])","metadata":{"execution":{"iopub.status.busy":"2023-08-20T13:42:20.67119Z","iopub.execute_input":"2023-08-20T13:42:20.671584Z","iopub.status.idle":"2023-08-20T13:43:45.043469Z","shell.execute_reply.started":"2023-08-20T13:42:20.671552Z","shell.execute_reply":"2023-08-20T13:43:45.042094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}