{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n# !pwd\n# !conda env create -f /kaggle/input/scv-bengali/kaggle_submission/asr_bengali.yml\n# !activate asr_env1\n# !pip download pyctcdecode -d ./pyctcdecode/\n# !pip download https://github.com/kpu/kenlm/archive/master.zip -d ./kenlm/","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-22T02:58:10.717837Z","iopub.execute_input":"2023-09-22T02:58:10.718250Z","iopub.status.idle":"2023-09-22T02:58:10.726418Z","shell.execute_reply.started":"2023-09-22T02:58:10.718212Z","shell.execute_reply":"2023-09-22T02:58:10.724912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index","metadata":{"execution":{"iopub.status.busy":"2023-09-22T02:58:10.728382Z","iopub.execute_input":"2023-09-22T02:58:10.729032Z","iopub.status.idle":"2023-09-22T02:59:23.951812Z","shell.execute_reply.started":"2023-09-22T02:58:10.728994Z","shell.execute_reply":"2023-09-22T02:59:23.950601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport os, re, torch, torchaudio\nfrom datasets import Dataset, load_metric\nimport pandas as pd\nfrom transformers import Wav2Vec2ForCTC, Wav2Vec2Processor\nfrom pyctcdecode import build_ctcdecoder\nfrom multiprocessing import Pool\nfrom torchinfo import summary\nimport argparse\nfrom glob import glob\nimport kenlm\nimport librosa\nimport numpy as np\n","metadata":{"execution":{"iopub.status.busy":"2023-09-22T02:59:23.953316Z","iopub.execute_input":"2023-09-22T02:59:23.954194Z","iopub.status.idle":"2023-09-22T02:59:36.992227Z","shell.execute_reply.started":"2023-09-22T02:59:23.954163Z","shell.execute_reply":"2023-09-22T02:59:36.989911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_DIRECTORY = '/kaggle/input/bengaliai-speech/test_mp3s'\npaths = glob(os.path.join(TEST_DIRECTORY,'*.mp3'))\nprint(paths)","metadata":{"execution":{"iopub.status.busy":"2023-09-22T02:59:36.998246Z","iopub.execute_input":"2023-09-22T02:59:36.998985Z","iopub.status.idle":"2023-09-22T02:59:37.011544Z","shell.execute_reply.started":"2023-09-22T02:59:36.998948Z","shell.execute_reply":"2023-09-22T02:59:37.010328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parser = argparse.ArgumentParser()\n# parser.add_argument('--checkpoint_path', type=str, required=True,\n#                     help=\"path of checkpoint\")\n\ncheckpoint_path = '/kaggle/input/scv-bengali/kaggle_submission/best_checkpoint_folder'\n# parser.add_argument('--use_LM', default=True,\n#                     help=\"Want to use LM during inference ? if yes, provide LM_path\")\nuse_LM = False\n# parser.add_argument('--LM_path', type=str, default='4gram.bin',\n#                     help=\"If it's passed the inference will be done using language model.\")\nLM_path = '/kaggle/input/scv-bengali/kaggle_submission/4gram.bin'\n\n# parser.add_argument('--dataset_path', type=str, required=True,default='dataset/dataset/test.csv',\n#                     help=\"If it's passed the inference will be done in all audio files in this path and the dataset present in the config json will be ignored\")\n\n# parser.add_argument('--calculate_wer', type=str, default='True',\n#                     help=\"Want to calculate Word Error Rate ?\")\n# args = parser.parse_args()\n\n\n# audio_df = pd.read_csv(args.dataset_path,sep='\\t')#,header=None)\naudio_df = pd.DataFrame(paths, columns=['path'])\n\nmr_test_dataset = Dataset.from_pandas(audio_df)\n\n# wer = load_metric(\"wer\")\n# cer = load_metric(\"cer\")\nmodel_dir=checkpoint_path\nif use_LM:\n  print(\"Utilizing LM...\")\n  LM_path=args.LM_path\n  \n  vocab_dict = processor.tokenizer.get_vocab()\n  sorted_dict = {k.lower(): v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n  decoder = build_ctcdecoder(list(sorted_dict.keys())[:87],LM_path)\nelse:\n  print(\"Not utilizing LM...\")\n\nprocessor = Wav2Vec2Processor.from_pretrained(model_dir)\nmodel = Wav2Vec2ForCTC.from_pretrained(model_dir)\n\nif torch.cuda.is_available():\n\tmodel.to(\"cuda\")\n\nsummary(model,(1,16000))\n\n    \n#chars_to_ignore_regex = '[\\,\\?\\.\\!\\-\\;\\:\\\"\\“]' \nchars_to_ignore_regex = '[\\,\\?]'\n\nresampler = torchaudio.transforms.Resample(16_000, 16_000)\n# Preprocessing the datasets. We need to read the aduio files as arrays\ndef speech_file_to_array_fn(batch):\n  # batch[\"text\"] = re.sub(chars_to_ignore_regex, '', batch[\"text\"]).lower()\n#   speech_array, sampling_rate = torchaudio.load(batch[\"path\"])\n  speech_array, sampling_rate = librosa.load(batch[\"path\"])\n#   batch[\"speech\"] = resampler(speech_array).squeeze().numpy()\n  batch[\"speech\"] = resampler(speech_array).squeeze()\n  return batch\n\n\nmr_test_dataset = mr_test_dataset.map(speech_file_to_array_fn)\n\ndef evaluate(batch):\n  inputs = processor(batch[\"speech\"], sampling_rate=16_000, return_tensors=\"pt\", padding=True)\n  with torch.no_grad():\n#     logits = model(inputs.input_values.to(\"cuda\"), attention_mask=inputs.attention_mask.to(\"cuda\")).logits\n    if torch.cuda.is_available():\n        logits = model(inputs.input_values.to(\"cuda\")).logits\n    else:\n        logits = model(inputs.input_values).logits\n#     logits = model(inputs.input_values).logits\n    if use_LM:\n        logits1=list(logits.cpu().numpy())\n        with Pool() as pool:\n            batch[\"pred_strings\"] = decoder.decode_batch(pool,logits1)\n        #batch[\"pred_strings\"] = decoder.decode(logits)\n        #batch[\"text\"] = batch[\"text\"]\n        #print(batch[\"pred_strings\"])\n    else:\n        pred_ids = torch.argmax(logits, dim=-1)\n        batch[\"pred_strings\"] = processor.batch_decode(pred_ids)\n  return batch\nresult = mr_test_dataset.map(evaluate, batch_size=1,batched=True)\nprint(\"/*/#\"*100)\nprint(\"Results Generated!\")\nprint(\"/*/#\"*100)\n# print(\"WER: {:2f}\".format(100 * wer.compute(predictions=result[\"pred_strings\"], references=result[\"text\"])))\n# print(\"CER: {:2f}\".format(100 * cer.compute(predictions=result[\"pred_strings\"], references=result[\"text\"])))\n\n# for i in range(len(result)): \n\t# print(\"ref = %s\"%result[\"text\"][i])\n# \tprint(\"hyp = %s\"%result[\"pred_strings\"][i])\n\t# print(\"wer is : \",wer.compute(predictions=[result[\"pred_strings\"][i]], references=[result[\"text\"][i]]) )\n\t","metadata":{"execution":{"iopub.status.busy":"2023-09-22T02:59:37.014472Z","iopub.execute_input":"2023-09-22T02:59:37.014825Z","iopub.status.idle":"2023-09-22T03:00:08.102595Z","shell.execute_reply.started":"2023-09-22T02:59:37.014793Z","shell.execute_reply":"2023-09-22T03:00:08.101640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ids = [x.split(\"/\")[-1][:-4] for x in result['path']]\n\n# submission_df = pd.DataFrame(zip(ids,result['pred_strings']),columns=['id','sentence']).set_index('id')\n# submission_df.to_csv(\"sumission_1.csv\")\n\nfrom bnunicodenormalizer import Normalizer \nbnorm = Normalizer()\n\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence\n\ndf= pd.DataFrame(\n    {\n        \"id\":[p.split(os.sep)[-1].replace('.mp3','') for p in paths],\n        \"sentence\":[p for p in result['pred_strings']]\n    }\n)\ndf.sentence= df.sentence.apply(lambda x:normalize(x))\ndf.sentence= df.sentence.apply(lambda x:dari(x))\nprint(\"#*\"*100)\nprint(\"Done!\")\nprint(\"*#\"*100)\ndf.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-22T03:00:08.103951Z","iopub.execute_input":"2023-09-22T03:00:08.105272Z","iopub.status.idle":"2023-09-22T03:00:08.148374Z","shell.execute_reply.started":"2023-09-22T03:00:08.105233Z","shell.execute_reply":"2023-09-22T03:00:08.147252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}