{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## What is SCTK ? \n\n- The NIST Scoring Toolkit or SCTK for short, is a command line tool that offers ways to score and compare text transcripts.\n\n- It can generate several types of reports that can be used to identify systematic errors and other common mistakes made by an ASR system.\n\n- Go-SCTK is a wrapper around the original SCTK toolkit that offers a simplified interface, and formats the reports better for languages that don't use an ASCII script.\n\n### Example SCTK Reports\n\n- Word / Character Confusion Pairs\n\n| Count | Conufusion Pair |\n|:-----:|:---------------:|\n| 17 | তার ==> তাঁর |\n| 12 | হিসাবে ==> হিসেবে |\n| 11 | তাঁর ==> তার |\n| 10 | করেন। ==> করে। |\n| 8 | কোন ==> কোনো |\n| 5 | এছাড়া ==> ছাড়া |\n| 5 | ঐ ==> ওই |\n| 4 | উপর ==> ওপর |\n| 4 | এছাড়াও ==> ছাড়াও |\n| 4 | ডানহাতি ==> হাতি |\n\n- Reference / Predicted Transcript Alignment\n\n| common_voice_bn_31621610.mp3 |  | | | | |\n|--- | :---: | :---: | :---: | :---: | :---: |\n|REF | তিনি | এথেন্সে | এসে | সক্রেটিসের | শিষ্য | হন।|\n|HYP1 | তিনি |  | অ্যাথেলসেএসে | সক্রেটির | শীর্ষ | হন।|\n|EVAL |  | D | S | S | S | |\n","metadata":{}},{"cell_type":"markdown","source":"## Download Go-SCTK tool","metadata":{}},{"cell_type":"code","source":"# Getting Go SCTK CLI tool and giving it executable permissions.\n!version=v0.3.0 && wget -q -O sctk https://github.com/shahruk10/go-sctk/releases/download/${version}/sctk && chmod +x ./sctk\n\n# Printing usage docs.\n!./sctk score --help","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T11:33:34.004966Z","iopub.execute_input":"2022-07-23T11:33:34.006508Z","iopub.status.idle":"2022-07-23T11:33:36.436323Z","shell.execute_reply.started":"2022-07-23T11:33:34.006364Z","shell.execute_reply":"2022-07-23T11:33:36.434904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transcribe validation set using pretrained model\n\n- First let's generate some results using a pretrained off-the-shelf model","metadata":{}},{"cell_type":"code","source":"# Installing extra deps.\n!pip install pyctcdecode pypi-kenlm jiwer bnunicodenormalizer","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:36:18.150897Z","iopub.execute_input":"2022-07-23T11:36:18.151445Z","iopub.status.idle":"2022-07-23T11:36:32.151002Z","shell.execute_reply.started":"2022-07-23T11:36:18.151401Z","shell.execute_reply":"2022-07-23T11:36:32.149648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Dict, List, Tuple, Any, Union\n\nimport os\n\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torchaudio\nimport torchaudio.functional as F\nimport torchaudio.transforms as T\n\nimport transformers\nfrom transformers import Wav2Vec2ForCTC, pipeline\nfrom datasets import load_metric\n\nfrom bnunicodenormalizer import Normalizer \n\nfrom tqdm.auto import tqdm\nfrom IPython.display import display, Audio, HTML\n\nbnorm = Normalizer()\ncer = load_metric(\"cer\")\nwer = load_metric(\"wer\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:53:25.115177Z","iopub.execute_input":"2022-07-23T11:53:25.115944Z","iopub.status.idle":"2022-07-23T11:53:26.672410Z","shell.execute_reply.started":"2022-07-23T11:53:25.115883Z","shell.execute_reply":"2022-07-23T11:53:26.671376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Path to validation csv file and MP3 audio directory.\nvalidMetaFile = \"../input/dlsprint/validation.csv\"\nvalidAudioDir = \"../input/dlsprint/validation_files\"\n\n# Loading metadata, and adding full path to audio files.\nvalidMeta = pd.read_csv(validMetaFile)\nvalidMeta[\"filepath\"] = [ os.path.join(validAudioDir, f) for f in validMeta[\"path\"] ]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:36:41.523463Z","iopub.execute_input":"2022-07-23T11:36:41.524070Z","iopub.status.idle":"2022-07-23T11:36:41.691013Z","shell.execute_reply.started":"2022-07-23T11:36:41.524025Z","shell.execute_reply":"2022-07-23T11:36:41.689786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hugging face model name to transcribe with.\nmodelName = \"arijitx/wav2vec2-xls-r-300m-bengali\"\n\n# Loading model and creating inference pipeline.\nasrPipeline = pipeline(\"automatic-speech-recognition\", modelName, device=-1) # Set to 0 for GPU\nsampleRate = 16000","metadata":{"execution":{"iopub.status.busy":"2022-07-23T11:36:43.564911Z","iopub.execute_input":"2022-07-23T11:36:43.565334Z","iopub.status.idle":"2022-07-23T11:42:03.282594Z","shell.execute_reply.started":"2022-07-23T11:36:43.565293Z","shell.execute_reply":"2022-07-23T11:42:03.281443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DLSprintDataset(torch.utils.data.Dataset):\n        \n    def __init__(self, audioPaths, sampleRate: int):\n        self.paths =  audioPaths\n        self.len = len(audioPaths)\n        self.sampleRate = sampleRate\n\n    def loadAudio(self, audioPath: str) -> np.ndarray:\n        x, sr = torchaudio.load(audioPath)\n        if self.sampleRate is not None and sr != self.sampleRate:\n            resampleParams = {\n                \"lowpass_filter_width\": 16, \"rolloff\": 0.85,\n                \"resampling_method\": \"kaiser_window\", \"beta\": 8.555504641634386,\n            }\n            x = F.resample(x, sr, self.sampleRate, **resampleParams)\n\n        return x[0].numpy()\n\n    def __len__(self):\n        return self.len\n    \n    def __getitem__(self, idx): \n        if idx >= self.len:\n            raise IndexError(\"index out of range\")\n        return self.loadAudio(self.paths[idx])\n\ndef normalizeUnicode(text):\n    \"\"\"\n    Normalizes Bengali unicode character representations.\n    \"\"\"\n    words = [ bnorm(w)['normalized'] for w in text.split() ]\n    return \" \".join([w for w in words if w is not None]) \n\ndef transcribe(\n    dataset: torch.utils.data.Dataset, asrPipeline: transformers.Pipeline,\n) -> List[str]:\n    \"\"\"\n    Transcribes the given dataset with the given ASR pipeline and returns the\n    text transcripts.\n    \"\"\"\n    iterator = asrPipeline(dataset)\n    text = [ normalizeUnicode(pred[\"text\"]) for pred in tqdm(iterator) ]\n    text = [ t if t.endswith(\"।\") else t + \"।\" for t in text ]\n   \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:15:07.035955Z","iopub.execute_input":"2022-07-23T12:15:07.036580Z","iopub.status.idle":"2022-07-23T12:15:07.055625Z","shell.execute_reply.started":"2022-07-23T12:15:07.036537Z","shell.execute_reply":"2022-07-23T12:15:07.054180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Taking only the first N files for demonstration purposes.\n# Wav2Vec2 is a chonky model and takes a while to decode, even on the GPU.\nnumFiles = 1000\n\n# Creating dataset.    \nvalDS = DLSprintDataset(validMeta[\"filepath\"][:numFiles], sampleRate)\n\n# Running inference.\nvalidPreds = validMeta[[\"path\", \"sentence\"]][:numFiles]\nvalidPreds[\"sentence\"] = transcribe(valDS, asrPipeline)\nvalidPreds.to_csv(\"./valid_pred.csv\", index=False)\n\n# Saving only path and normalized reference transcript from metadata file.\nvalidRef = validMeta[[\"path\", \"sentence\"]][:numFiles]\nvalidRef[\"sentence\"] = [ normalizeUnicode(t) for t in validRef[\"sentence\"] ]\nvalidRef.to_csv(\"./valid_ref.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:37:11.260506Z","iopub.execute_input":"2022-07-23T12:37:11.261548Z","iopub.status.idle":"2022-07-23T12:40:12.703364Z","shell.execute_reply.started":"2022-07-23T12:37:11.261504Z","shell.execute_reply":"2022-07-23T12:40:12.702180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate predictions using SCTK\n\n- We can now provide our predictions and reference transcripts to SCTK for analysis.\n- We could also use outputs from other models / previous iterations of models as the reference as well. This will highlight the things that changed between models.","metadata":{}},{"cell_type":"code","source":"# Running sctk on model output for the validation set.\n# \n# Since the first row in the file contains headers, we set --ignore-first=true\n#\n# The file ID is the first column (index = 0) and the transcripts\n# are in the second column (index = 1); so setting --col-id and\n# --col-trn to 0 and 1 respectively.\n#\n# By setting --cer=false, we are evaluating at the word level.\n!./sctk score \\\n  --ignore-first=true \\\n  --delimiter=\",\" \\\n  --col-id=0 \\\n  --col-trn=1 \\\n  --cer=false \\\n  --out=./report \\\n  --ref=./valid_ref.csv \\\n  --hyp=./valid_pred.csv","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-23T12:40:18.834936Z","iopub.execute_input":"2022-07-23T12:40:18.835330Z","iopub.status.idle":"2022-07-23T12:40:20.025482Z","shell.execute_reply.started":"2022-07-23T12:40:18.835296Z","shell.execute_reply":"2022-07-23T12:40:20.024017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- SCTK generates a bunch of files. Let's take a look at some.\n\n- First, the `*.dtl` file contains a detailed breakdown of the different errors; there are three sections: `Confusion Pairs` or `Substitutions`, `Deletions` and `Insertions`.\n\n- Looking at the `Confusion Pairs` section, we can see that the most frequent confused words between the reference and hypothesis are mostly differences in spelling, whitespace position and homonyms.\n\n| Count | Conufusion Pair |\n|:-----:|:---------------:|\n| 17 | তার ==> তাঁর |\n| 12 | হিসাবে ==> হিসেবে |\n| 11 | তাঁর ==> তার |\n| 10 | করেন। ==> করে। |\n| 8 | কোন ==> কোনো |\n| 5 | এছাড়া ==> ছাড়া |\n| 5 | ঐ ==> ওই |\n| 4 | উপর ==> ওপর |\n| 4 | এছাড়াও ==> ছাড়াও |\n| 4 | ডানহাতি ==> হাতি |\n\n- There are also quite a few errors counted due to absence of punctuation (commas etc.) in the model output.","metadata":{}},{"cell_type":"code","source":"display(HTML(open(\"./report/hyp1.trn.dtl\", 'r').read().replace(\"\\n\", \"<br>\")))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:40:23.968408Z","iopub.execute_input":"2022-07-23T12:40:23.969446Z","iopub.status.idle":"2022-07-23T12:40:23.983636Z","shell.execute_reply.started":"2022-07-23T12:40:23.969396Z","shell.execute_reply":"2022-07-23T12:40:23.982414Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Next, let's take a look at the `*.pra` file. This contains alignments between the reference and model output. This is provided in a variety of formats. HTML is probably the easiest to view in a notebook.\n\n - Sidenote, these alignments can also be used to combine the outputs from different ASR models (See [ROVER](https://ieeexplore.ieee.org/document/659110) algorithm).\n\n\n- One interesing example is the following (common_voice_bn_31675568.mp3). The transcription is mostly right, but has some issues with compound words and spelling. These could be improved by using a better / adapted language model, or even dictionary lookup.\n\n| Cor=55.6%\tSub=27.8%\tDel=0.0%\tIns=16.7% |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  |  | |\n| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | ---|\n|REF | তিনি | দারুল | উলুম | দেওবন্দের | প্রাক্তন | ছাত্র | এবং | মাহমুদ | হাসান |  |  |  | দেওবন্দির | থামরাতুত-তারবিয়াতের | প্রতিষ্ঠাতা | সদস্যদের | মধ্যে | ছিলেন। | |\n|HYP1 | তিনি | দারুল | উলম | দেওবন্দের | প্রাক্তন | ছাত্র | এবং | মাহমুদ | হাসান | দেহ | বন্দির | থাম | রাতু | তারবিয়াতের | প্রতি | সদস্যদের | মধ্য | ছিলেন। | |\n|EVAL |  |  | S |  |  |  |  |  |  | I | I | I | S | S | S |  | S |  | |\n","metadata":{}},{"cell_type":"code","source":"display(Audio(\"../input/dlsprint/validation_files/common_voice_bn_31675568.mp3\", rate=32000))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:56:15.837593Z","iopub.execute_input":"2022-07-23T12:56:15.838773Z","iopub.status.idle":"2022-07-23T12:56:15.851742Z","shell.execute_reply.started":"2022-07-23T12:56:15.838720Z","shell.execute_reply":"2022-07-23T12:56:15.850158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(HTML(open(\"./report/hyp1.trn.pra.html\", 'r').read()))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T12:48:40.419127Z","iopub.execute_input":"2022-07-23T12:48:40.419518Z","iopub.status.idle":"2022-07-23T12:48:40.461098Z","shell.execute_reply.started":"2022-07-23T12:48:40.419487Z","shell.execute_reply":"2022-07-23T12:48:40.460061Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluating at the character level.\n\n- We can run the same analysis as above at the character level instead of words.","metadata":{}},{"cell_type":"code","source":"# Running sctk on model output for the validation set.\n# \n# Since the first row in the file contains headers, we set --ignore-first=true\n#\n# The file ID is the first column (index = 0) and the transcripts\n# are in the second column (index = 1); so setting --col-id and\n# --col-trn to 0 and 1 respectively.\n!./sctk score \\\n  --ignore-first=true \\\n  --delimiter=\",\" \\\n  --col-id=0 \\\n  --col-trn=1 \\\n  --cer=true \\\n  --out=./report \\\n  --ref=./valid_ref.csv \\\n  --hyp=./valid_pred.csv","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:05:30.138889Z","iopub.execute_input":"2022-07-23T13:05:30.139284Z","iopub.status.idle":"2022-07-23T13:05:33.453303Z","shell.execute_reply.started":"2022-07-23T13:05:30.139253Z","shell.execute_reply":"2022-07-23T13:05:33.451720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Looking at the `*.dtl` file again, this time we can see the confused character pairs. It seems that like most humans, the ASR model is not sure when to use `ী` vs.`ি`\n\n| Count | Conufusion Pair |\n|:-----:|:---------------:|\n| 70 |  ী ==>  ি |\n| 28 |  ণ ==>  ন |\n| 26 |  া ==>  ে |\n| 24 |  ে ==>  া |\n| 22 |  ি ==>  ী |\n| 22 |  ে ==>  ি |\n| 22 |  ড় ==>  র |\n| 19 |  শ ==>  স |\n| 19 |  ি ==>  ে |\n| 16 |  এ ==>  ে |","metadata":{}},{"cell_type":"code","source":"display(HTML(open(\"./report/hyp1.trn.dtl\", 'r').read().replace(\"\\n\", \"<br>\")))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T13:05:36.664549Z","iopub.execute_input":"2022-07-23T13:05:36.665456Z","iopub.status.idle":"2022-07-23T13:05:36.676395Z","shell.execute_reply.started":"2022-07-23T13:05:36.665411Z","shell.execute_reply":"2022-07-23T13:05:36.675111Z"},"trusted":true},"execution_count":null,"outputs":[]}]}