{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"While large language models based on the Transformer architecture have become the standard in NLP, it is still very common to use an n-gram LM to boost speech recognition systems. For more information on how n-grams function and why they are (still) so useful for speech recognition, the reader is advised to take a look at this [excellent summary](https://web.stanford.edu/~jurafsky/slp3/3.pdf) from Stanford.\n\nIn this notebook, we train a 6 gram model with the texts from train and validation datasets using KenLM.","metadata":{}},{"cell_type":"code","source":"!pip -q install https://github.com/kpu/kenlm/archive/master.zip pyctcdecode","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:47:40.666285Z","iopub.execute_input":"2022-07-05T01:47:40.667244Z","iopub.status.idle":"2022-07-05T01:48:42.412446Z","shell.execute_reply.started":"2022-07-05T01:47:40.667113Z","shell.execute_reply":"2022-07-05T01:48:42.411361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T01:52:00.376863Z","iopub.execute_input":"2022-07-05T01:52:00.377300Z","iopub.status.idle":"2022-07-05T01:52:00.383150Z","shell.execute_reply.started":"2022-07-05T01:52:00.377259Z","shell.execute_reply":"2022-07-05T01:52:00.382155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chars_to_ignore_regex = '[\\,\\?\\.\\!\\-\\;\\:\\\"\\“\\%\\‘\\”\\�\\।\\’]'\n\ntrain_df = pd.read_csv('../input/dlsprint/train.csv')\nvalid_df = pd.read_csv('../input/dlsprint/validation.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:52:15.612136Z","iopub.execute_input":"2022-07-05T01:52:15.612483Z","iopub.status.idle":"2022-07-05T01:52:17.673797Z","shell.execute_reply.started":"2022-07-05T01:52:15.612439Z","shell.execute_reply":"2022-07-05T01:52:17.672872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('text.txt', 'w') as f:\n    for sentence in train_df['sentence']:\n        f.write(re.sub(chars_to_ignore_regex, '', sentence))\n        f.write(' ')\n        \n    for sentence in valid_df['sentence']:\n        f.write(re.sub(chars_to_ignore_regex, '', sentence))\n        f.write(' ')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:52:20.941770Z","iopub.execute_input":"2022-07-05T01:52:20.942452Z","iopub.status.idle":"2022-07-05T01:52:21.633029Z","shell.execute_reply.started":"2022-07-05T01:52:20.942411Z","shell.execute_reply":"2022-07-05T01:52:21.631831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! sudo apt -y install build-essential cmake libboost-system-dev libboost-thread-dev libboost-program-options-dev libboost-test-dev libeigen3-dev zlib1g-dev libbz2-dev liblzma-dev\n! wget -O - https://kheafield.com/code/kenlm.tar.gz | tar xz\n! mkdir kenlm/build && cd kenlm/build && cmake .. && make -j2\n! ls kenlm/build/bin\n! kenlm/build/bin/lmplz -o 6 < \"text.txt\" > \"6gram.arpa\"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:52:25.103957Z","iopub.execute_input":"2022-07-05T01:52:25.104493Z","iopub.status.idle":"2022-07-05T01:54:38.808635Z","shell.execute_reply.started":"2022-07-05T01:52:25.104418Z","shell.execute_reply":"2022-07-05T01:54:38.807239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"6gram.arpa\", \"r\") as read_file, open(\"6gram_correct.arpa\", \"w\") as write_file:\n  has_added_eos = False\n  for line in read_file:\n    if not has_added_eos and \"ngram 1=\" in line:\n      count=line.strip().split(\"=\")[-1]\n      write_file.write(line.replace(f\"{count}\", f\"{int(count)+1}\"))\n    elif not has_added_eos and \"<s>\" in line:\n      write_file.write(line)\n      write_file.write(line.replace(\"<s>\", \"</s>\"))\n      has_added_eos = True\n    else:\n      write_file.write(line)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:56:48.885320Z","iopub.execute_input":"2022-07-05T01:56:48.886742Z","iopub.status.idle":"2022-07-05T01:56:57.249160Z","shell.execute_reply.started":"2022-07-05T01:56:48.886684Z","shell.execute_reply":"2022-07-05T01:56:57.248078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import kenlm\nmodel = kenlm.LanguageModel('./6gram_correct.arpa')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:56:58.882723Z","iopub.execute_input":"2022-07-05T01:56:58.883244Z","iopub.status.idle":"2022-07-05T01:57:04.138508Z","shell.execute_reply.started":"2022-07-05T01:56:58.883210Z","shell.execute_reply":"2022-07-05T01:57:04.137444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model.score('প্রধান নায়ক', bos=True, eos=True))\nprint(model.score('প্রাধান নয়ক', bos=True, eos=True))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:57:26.141249Z","iopub.execute_input":"2022-07-05T01:57:26.141662Z","iopub.status.idle":"2022-07-05T01:57:26.147417Z","shell.execute_reply.started":"2022-07-05T01:57:26.141626Z","shell.execute_reply":"2022-07-05T01:57:26.146521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see that \"প্রধান নায়ক\" is a valid sentence. That's why it scored higher than \"প্রাধান নয়ক\"","metadata":{}},{"cell_type":"markdown","source":"Incorporating this language model with transformers library is easy!","metadata":{}},{"cell_type":"code","source":"from transformers import AutoProcessor\n\nprocessor = AutoProcessor.from_pretrained(\"Tahsin-Mayeesha/wav2vec2-bn-300m\")\nvocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k.lower(): v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}","metadata":{"execution":{"iopub.status.busy":"2022-07-05T01:59:35.641183Z","iopub.execute_input":"2022-07-05T01:59:35.641618Z","iopub.status.idle":"2022-07-05T01:59:43.303420Z","shell.execute_reply.started":"2022-07-05T01:59:35.641584Z","shell.execute_reply":"2022-07-05T01:59:43.301504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyctcdecode import build_ctcdecoder\n\ndecoder = build_ctcdecoder(\n    labels=list(sorted_vocab_dict.keys()),\n    kenlm_model_path=\"6gram_correct.arpa\",\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Hope you learnt something new!","metadata":{}}]}