{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Install necessary libraries\n## Make sure you have the accelerator set to: GPU P100","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"from distutils.dir_util import copy_tree\n\norig = \"/kaggle/input/emoji170/types-emoji-1.7.0\"\ndest = \"/kaggle/working/types-emoji-1.7.0\"\ncopy_tree(orig, dest)\nprint(\"Done\")\n\n#orig = \"/kaggle/input/bnlp003\"\n#dest = \"/kaggle/working/bnlp\"\n#copy_tree(orig, dest)\n#print(\"Done\")\n\n\n!pip uninstall --yes emoji\n!pip install /kaggle/working/types-emoji-1.7.0\n\nimport sys\nsys.path.append('/kaggle/input/bnlp003/bnlp01')\n!ls /kaggle/input/bnlp003\n\nimport os, shutil\n#shutil.rmtree(\"/kaggle/working/bnlp\")\nshutil.rmtree(\"/kaggle/working/types-emoji-1.7.0\")","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:33:45.977200Z","iopub.execute_input":"2023-10-17T23:33:45.977496Z","iopub.status.idle":"2023-10-17T23:34:22.927157Z","shell.execute_reply.started":"2023-10-17T23:33:45.977472Z","shell.execute_reply":"2023-10-17T23:34:22.926309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:34:22.928814Z","iopub.execute_input":"2023-10-17T23:34:22.929271Z","iopub.status.idle":"2023-10-17T23:35:23.148250Z","shell.execute_reply.started":"2023-10-17T23:34:22.929242Z","shell.execute_reply":"2023-10-17T23:35:23.147234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:35:23.149466Z","iopub.execute_input":"2023-10-17T23:35:23.149806Z","iopub.status.idle":"2023-10-17T23:35:24.109316Z","shell.execute_reply.started":"2023-10-17T23:35:23.149782Z","shell.execute_reply":"2023-10-17T23:35:24.108373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nfrom functools import partial","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:35:24.111437Z","iopub.execute_input":"2023-10-17T23:35:24.111670Z","iopub.status.idle":"2023-10-17T23:35:36.331441Z","shell.execute_reply.started":"2023-10-17T23:35:24.111650Z","shell.execute_reply":"2023-10-17T23:35:36.330775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define model y process based in wave-2-vec-2-bengali-ai","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', 100)\nmodel = Wav2Vec2ForCTC.from_pretrained(\"/kaggle/input/wave-2-vec-2-bengali-ai\")\nprocessor = Wav2Vec2Processor.from_pretrained(\"/kaggle/input/wave-2-vec-2-bengali-ai\")","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:35:36.332434Z","iopub.execute_input":"2023-10-17T23:35:36.332650Z","iopub.status.idle":"2023-10-17T23:35:48.203158Z","shell.execute_reply.started":"2023-10-17T23:35:36.332631Z","shell.execute_reply":"2023-10-17T23:35:48.202325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We get the vocab and sort them in a list from wave-2-vec-2-bengali-ai. We created a decoder with wav2vec2-xls-r-300m-bengali and a processor with Wav2Vec2ProcessorWithLM. We obtain the cpu or cuda device regarding the equipment where the training will be done to pass it as a parameter to Wav2Vec2ForCTC.\n## In this case we will use cuda directly.","metadata":{}},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(\"/kaggle/input/arijitx-full-model/wav2vec2-xls-r-300m-bengali/language_model/5gram.bin\"),\n)\nprocessor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)\ndevice = torch.device(\"cuda\")\nmodel = model.to(device)\nmodel = model.eval()\nmodel = model.half()","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:35:48.204246Z","iopub.execute_input":"2023-10-17T23:35:48.204475Z","iopub.status.idle":"2023-10-17T23:36:35.289742Z","shell.execute_reply.started":"2023-10-17T23:35:48.204456Z","shell.execute_reply":"2023-10-17T23:36:35.289051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We use absolute paths for each file and have 3 different lists to test. We process and obtain a float32 array for each mp3 or WAV file to process the audio and decode the sentences.","metadata":{}},{"cell_type":"code","source":"#mp3= [\"/kaggle/input/bengaliai-speech/examples/Audiobook.wav\", \"/kaggle/input/bengaliai-speech/examples/Bangladeshi TV Drama.wav\", \"/kaggle/input/bengaliai-speech/examples/Bengali Advertisement.wav\"]\n#mp3= [\"/kaggle/input/bengaliai-speech/examples/Telemedicine.mp3\", \"/kaggle/input/bengaliai-speech/examples/Slang Profanity.mp3\", \"/kaggle/input/bengaliai-speech/train_mp3s/0000c6d85d91.mp3\"]\n#mp3 = [\"/kaggle/input/bengaliai-speech/test_mp3s/0f3dac00655e.mp3\", \"/kaggle/input/bengaliai-speech/test_mp3s/a9395e01ad21.mp3\", \"/kaggle/input/bengaliai-speech/test_mp3s/bf36ea8b718d.mp3\"]\ntest = pd.read_csv(\"/kaggle/input/bengaliai-speech/sample_submission.csv\", dtype={\"id\": str})\nmp3 = [str(\"/kaggle/input/bengaliai-speech/test_mp3s/\" + f\"{aid}.mp3\") for aid in test[\"id\"].values]\n\ntest_dataset = []\nfor m in mp3:\n    test_dataset.append(librosa.load(m, sr=16_000, mono=False)[0])\n\nprint(test_dataset)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=16_000,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=16, shuffle=False,\n    num_workers=2, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)\n\npred_sentence_list = []\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=512).text\n            pred_sentence_list.append(sentence)\nprint(pred_sentence_list)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:36:35.290772Z","iopub.execute_input":"2023-10-17T23:36:35.291186Z","iopub.status.idle":"2023-10-17T23:36:48.189104Z","shell.execute_reply.started":"2023-10-17T23:36:35.291164Z","shell.execute_reply":"2023-10-17T23:36:48.188297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Library BNLP\n## We use the blnp library, it is similar to spacy for the Bengali language. We also use word normalization for Bengali.","metadata":{}},{"cell_type":"code","source":"#https://github.com/sagorbrur/bnlp/tree/main/docs\nfrom bnlp import BasicTokenizer\nfrom bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\ntokenizer = BasicTokenizer()\n\ndef normalize_word(word):            \n    result = bnorm(word)        \n    if (result['normalized']==None):\n        return \"।\"\n    return result['normalized']\n\ntokens_norm = []    \ndef tokenizer_sentence(sentence):\n    raw_text = sentence\n    tokens = tokenizer(raw_text)        \n    tokens_norm = [normalize_word(word) for word in tokens]            \n    tokens_sort = [token for token in tokens_norm]\n    new_sentence = \" \".join(tokens_sort)\n    final_sentence.append(new_sentence)    \nfinal_sentence = []    \nfor s in pred_sentence_list:\n    tokenizer_sentence(s)\n    \nprint(final_sentence)","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:36:48.192585Z","iopub.execute_input":"2023-10-17T23:36:48.193238Z","iopub.status.idle":"2023-10-17T23:37:05.003285Z","shell.execute_reply.started":"2023-10-17T23:36:48.193211Z","shell.execute_reply":"2023-10-17T23:37:05.002481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare the submission file","metadata":{}},{"cell_type":"code","source":"test_csv = pd.read_csv(\"/kaggle/input/bengaliai-speech/sample_submission.csv\", dtype={\"id\": str})\ntest_csv[\"sentence\"] = final_sentence\ntest_csv.to_csv(\"submission.csv\", index=False)\ndisplay(test_csv.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-17T23:37:05.004206Z","iopub.execute_input":"2023-10-17T23:37:05.004429Z","iopub.status.idle":"2023-10-17T23:37:05.023156Z","shell.execute_reply.started":"2023-10-17T23:37:05.004410Z","shell.execute_reply":"2023-10-17T23:37:05.022489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusions\n## It is important to try other libraries such as DeepSpeech, Kaldi and others similar to see possible improvements. The bnlp library works well, its functions need to be explored further to integrate them into future projects.","metadata":{}}]}