{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center><span style=\"font-family:cursive;\"><b>DL Sprint - BUET CSE Fest 2022 </b> <br><font size=\"+2\">Bengali Automatic Speech Recognition Competition</font></span></center>\n\n<center><img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/37174/logos/header.png?t=2022-06-30-07-00-00\"></center>\n","metadata":{"id":"V7YOT2mnUiea"}},{"cell_type":"code","source":"gpu_info = !nvidia-smi\ngpu_info = '\\n'.join(gpu_info)\nif gpu_info.find('failed') >= 0:\n    print('Not connected to a GPU')\nelse:\n    print(gpu_info)","metadata":{"id":"YELVqGxMxnbG","outputId":"1ab7eb67-409d-4371-b99e-7eb1171cb5fb","execution":{"iopub.status.busy":"2022-08-25T13:24:55.948763Z","iopub.execute_input":"2022-08-25T13:24:55.949302Z","iopub.status.idle":"2022-08-25T13:24:56.003324Z","shell.execute_reply.started":"2022-08-25T13:24:55.94919Z","shell.execute_reply":"2022-08-25T13:24:56.001549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n!pip install datasets\n!pip install s3fs==2021.5.0\n!pip install fsspec==2021.5.0\n!pip install jiwer\n!pip install transformers\n!pip install torchaudio\n!pip install https://github.com/kpu/kenlm/archive/master.zip \n!pip install https://github.com/kpu/kenlm/archive/master.zip pyctcdecode","metadata":{"id":"c8eh87Hoee5d","execution":{"iopub.status.busy":"2022-08-25T13:25:02.210249Z","iopub.execute_input":"2022-08-25T13:25:02.210667Z","iopub.status.idle":"2022-08-25T13:27:33.322333Z","shell.execute_reply.started":"2022-08-25T13:25:02.210632Z","shell.execute_reply":"2022-08-25T13:27:33.3206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport torch\nimport librosa\nfrom sklearn.model_selection import train_test_split\nfrom datasets import load_dataset, load_metric,Dataset,concatenate_datasets,set_caching_enabled, ClassLabel\nfrom datasets import load_from_disk, Audio\nimport pandas as pd\n\nimport random\nfrom IPython.display import display, HTML\n\nimport json\nfrom transformers import Wav2Vec2CTCTokenizer,Wav2Vec2ForCTC,Wav2Vec2Processor,Trainer,TrainingArguments,Wav2Vec2FeatureExtractor,set_seed\nfrom transformers import Wav2Vec2Processor, HubertForCTC\nimport re\nset_caching_enabled(False)\n\nimport soundfile as sf\nimport torchaudio\n\n\nimport IPython.display as ipd","metadata":{"execution":{"iopub.status.busy":"2022-08-25T13:27:33.325788Z","iopub.execute_input":"2022-08-25T13:27:33.326305Z","iopub.status.idle":"2022-08-25T13:27:47.712091Z","shell.execute_reply.started":"2022-08-25T13:27:33.326253Z","shell.execute_reply":"2022-08-25T13:27:47.71075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nimport transformers\ntransformers.logging.get_verbosity = lambda: logging.NOTSET\ntransformers.logging.get_verbosity()\nimport datasets\ndatasets.logging.get_verbosity = lambda: logging.NOTSET","metadata":{"execution":{"iopub.status.busy":"2022-08-25T13:27:47.714017Z","iopub.execute_input":"2022-08-25T13:27:47.714947Z","iopub.status.idle":"2022-08-25T13:27:47.726458Z","shell.execute_reply.started":"2022-08-25T13:27:47.714894Z","shell.execute_reply":"2022-08-25T13:27:47.724909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Data, Tokenizer, Feature Extractor","metadata":{"id":"0mW-C1Nt-j7k"}},{"cell_type":"code","source":"dataset_loaded= load_from_disk(\"../input/dl-sprint-data\")","metadata":{"id":"2MMXcWFFgCXU","outputId":"00961862-9e79-4e0e-c887-db62adafa553","execution":{"iopub.status.busy":"2022-08-25T13:27:47.72967Z","iopub.execute_input":"2022-08-25T13:27:47.731918Z","iopub.status.idle":"2022-08-25T13:27:49.379204Z","shell.execute_reply.started":"2022-08-25T13:27:47.731858Z","shell.execute_reply":"2022-08-25T13:27:49.377879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_voice_train=dataset_loaded[\"train\"]\ncommon_voice_train=common_voice_train.filter(lambda x: (x[\"up_votes\"]-x[\"down_votes\"]>0) or (x[\"up_votes\"]!=0))\ncommon_voice_train=common_voice_train.filter(lambda x: (x[\"down_votes\"]==0))\ncommon_voice_test=dataset_loaded[\"validation\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-25T13:27:49.380685Z","iopub.execute_input":"2022-08-25T13:27:49.381031Z","iopub.status.idle":"2022-08-25T14:13:32.369521Z","shell.execute_reply.started":"2022-08-25T13:27:49.380998Z","shell.execute_reply":"2022-08-25T14:13:32.367342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_voice_train","metadata":{"execution":{"iopub.status.busy":"2022-08-25T14:13:40.768543Z","iopub.execute_input":"2022-08-25T14:13:40.769853Z","iopub.status.idle":"2022-08-25T14:13:40.783574Z","shell.execute_reply.started":"2022-08-25T14:13:40.769802Z","shell.execute_reply":"2022-08-25T14:13:40.782332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Many ASR datasets only provide the target text, `'sentence'` for each audio array `'audio'` and file `'path'`. Common Voice actually provides much more information about each audio file, such as the `'accent'`, etc. Keeping the notebook as general as possible, we only consider the transcribed text for fine-tuning.\n","metadata":{}},{"cell_type":"code","source":"common_voice_train = common_voice_train.remove_columns([\"accent\", \"age\", \"client_id\", \"down_votes\", \"gender\", \"locale\", \"segment\", \"up_votes\"])\ncommon_voice_test = common_voice_test.remove_columns([\"accent\", \"age\", \"client_id\", \"down_votes\", \"gender\", \"locale\", \"segment\", \"up_votes\"])","metadata":{"id":"kbyq6lDgQc2a","execution":{"iopub.status.busy":"2022-08-25T14:14:24.736187Z","iopub.execute_input":"2022-08-25T14:14:24.736562Z","iopub.status.idle":"2022-08-25T14:14:24.741869Z","shell.execute_reply.started":"2022-08-25T14:14:24.736532Z","shell.execute_reply":"2022-08-25T14:14:24.740829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's write a short function to display some random samples of the dataset and run it a couple of times to get a feeling for the transcriptions.","metadata":{"id":"Go9Hq4e4NDT9"}},{"cell_type":"code","source":"from datasets import ClassLabel\nimport random\nimport pandas as pd\nfrom IPython.display import display, HTML\n\ndef show_random_elements(dataset, num_examples=10):\n    assert num_examples <= len(dataset), \"Can't pick more elements than there are in the dataset.\"\n    picks = []\n    for _ in range(num_examples):\n        pick = random.randint(0, len(dataset)-1)\n        while pick in picks:\n            pick = random.randint(0, len(dataset)-1)\n        picks.append(pick)\n    \n    df = pd.DataFrame(dataset[picks])\n    display(HTML(df.to_html()))","metadata":{"id":"72737oog2F6U","execution":{"iopub.status.busy":"2022-08-23T18:29:44.46826Z","iopub.execute_input":"2022-08-23T18:29:44.468656Z","iopub.status.idle":"2022-08-23T18:29:44.476152Z","shell.execute_reply.started":"2022-08-23T18:29:44.468617Z","shell.execute_reply":"2022-08-23T18:29:44.475113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_random_elements(common_voice_train.remove_columns([\"path\", \"audio\"]), num_examples=10)","metadata":{"id":"K_JUmf3G3b9S","outputId":"e3a0d9c8-4b68-4255-fd29-8dba65632a24","execution":{"iopub.status.busy":"2022-08-23T18:29:44.477874Z","iopub.execute_input":"2022-08-23T18:29:44.478534Z","iopub.status.idle":"2022-08-23T18:29:44.500821Z","shell.execute_reply.started":"2022-08-23T18:29:44.478497Z","shell.execute_reply":"2022-08-23T18:29:44.499975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove special characters","metadata":{}},{"cell_type":"code","source":"import re\nchars_to_remove_regex = '[\\,\\?\\.\\!\\-\\;\\:\\\"\\“\\%\\‘\\”\\�\\']'\n\ndef remove_special_characters(batch):\n    batch[\"sentence\"] = re.sub(chars_to_remove_regex, '', batch[\"sentence\"]).lower()\n    return batch","metadata":{"id":"svKzVJ_hQGK6","execution":{"iopub.status.busy":"2022-08-23T18:29:44.502022Z","iopub.execute_input":"2022-08-23T18:29:44.502359Z","iopub.status.idle":"2022-08-23T18:29:44.508157Z","shell.execute_reply.started":"2022-08-23T18:29:44.502325Z","shell.execute_reply":"2022-08-23T18:29:44.506948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_voice_train = common_voice_train.map(remove_special_characters)\ncommon_voice_test = common_voice_test.map(remove_special_characters)","metadata":{"id":"XIHocAuTQbBR","outputId":"32b09453-6f60-42c4-e417-4e87edad58dd","execution":{"iopub.status.busy":"2022-08-23T18:29:44.509627Z","iopub.execute_input":"2022-08-23T18:29:44.510662Z","iopub.status.idle":"2022-08-23T18:30:00.70093Z","shell.execute_reply.started":"2022-08-23T18:29:44.510628Z","shell.execute_reply":"2022-08-23T18:30:00.699699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare `Wav2Vec2CTCTokenizer`","metadata":{}},{"cell_type":"code","source":"from transformers import Wav2Vec2CTCTokenizer\n\ntokenizer = Wav2Vec2CTCTokenizer.from_pretrained(\"arijitx/wav2vec2-xls-r-300m-bengali\", unk_token=\"[UNK]\", pad_token=\"[PAD]\", word_delimiter_token=\"|\")","metadata":{"id":"xriFGEWQkO4M","outputId":"95e57a04-c48e-4748-8d77-bac71dd2750e","execution":{"iopub.status.busy":"2022-08-23T18:30:02.956509Z","iopub.execute_input":"2022-08-23T18:30:02.957029Z","iopub.status.idle":"2022-08-23T18:30:06.223755Z","shell.execute_reply.started":"2022-08-23T18:30:02.956993Z","shell.execute_reply":"2022-08-23T18:30:06.222806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare `Wav2Vec2FeatureExtractor`","metadata":{"id":"mYcIiR2FQ96i"}},{"cell_type":"code","source":"from transformers import Wav2Vec2FeatureExtractor\n\nfeature_extractor = Wav2Vec2FeatureExtractor(feature_size=1, sampling_rate=16000, padding_value=0.0, do_normalize=True, return_attention_mask=True)","metadata":{"id":"kAR0-2KLkopp","execution":{"iopub.status.busy":"2022-08-23T18:30:06.225207Z","iopub.execute_input":"2022-08-23T18:30:06.226188Z","iopub.status.idle":"2022-08-23T18:30:06.231583Z","shell.execute_reply.started":"2022-08-23T18:30:06.226149Z","shell.execute_reply":"2022-08-23T18:30:06.230447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2Processor\n\nprocessor = Wav2Vec2Processor(feature_extractor=feature_extractor, tokenizer=tokenizer)","metadata":{"id":"KYZtoW-tlZgl","execution":{"iopub.status.busy":"2022-08-23T18:30:06.233285Z","iopub.execute_input":"2022-08-23T18:30:06.233909Z","iopub.status.idle":"2022-08-23T18:30:06.243161Z","shell.execute_reply.started":"2022-08-23T18:30:06.233872Z","shell.execute_reply":"2022-08-23T18:30:06.242155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, we can prepare the dataset.","metadata":{"id":"DrKnYuvDIoOO"}},{"cell_type":"markdown","source":"### Preprocess Data","metadata":{"id":"YFmShnl7RE35"}},{"cell_type":"markdown","source":"Sampling rate 16kHz are expected by the model. We can set the audio feature to the correct sampling rate by making use of [`cast_column`](https://huggingface.co/docs/datasets/package_reference/main_classes.html?highlight=cast_column#datasets.DatasetDict.cast_column):","metadata":{"id":"WUUTgI1bGHW-"}},{"cell_type":"code","source":"common_voice_train = common_voice_train.cast_column(\"audio\", Audio(sampling_rate=16_000))\ncommon_voice_test = common_voice_test.cast_column(\"audio\", Audio(sampling_rate=16_000))","metadata":{"id":"rrv65aj7G95i","execution":{"iopub.status.busy":"2022-08-23T18:30:06.290017Z","iopub.execute_input":"2022-08-23T18:30:06.290789Z","iopub.status.idle":"2022-08-23T18:30:06.301659Z","shell.execute_reply.started":"2022-08-23T18:30:06.290745Z","shell.execute_reply":"2022-08-23T18:30:06.300696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's take a look at `changed sampling rate` .","metadata":{"id":"PcnO4x-NGBEi"}},{"cell_type":"code","source":"rand_int = random.randint(0, len(common_voice_train)-1)\n\nprint(\"Target text:\", common_voice_train[rand_int][\"sentence\"])\nprint(\"Input array shape:\", common_voice_train[rand_int][\"audio\"][\"array\"].shape)\nprint(\"Sampling rate:\", common_voice_train[rand_int][\"audio\"][\"sampling_rate\"])","metadata":{"id":"1Po2g7YPuRTx","outputId":"63478ae9-2927-4ec1-c13c-41fb3754e18e","execution":{"iopub.status.busy":"2022-08-23T18:30:06.377147Z","iopub.execute_input":"2022-08-23T18:30:06.377493Z","iopub.status.idle":"2022-08-23T18:30:06.429253Z","shell.execute_reply.started":"2022-08-23T18:30:06.377457Z","shell.execute_reply":"2022-08-23T18:30:06.428258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(batch):\n    audio = batch[\"audio\"]\n\n    # batched output is \"un-batched\"\n    batch[\"input_values\"] = processor(audio[\"array\"], sampling_rate=audio[\"sampling_rate\"]).input_values[0]\n    batch[\"input_length\"] = len(batch[\"input_values\"])\n    \n    with processor.as_target_processor():\n        batch[\"labels\"] = processor(batch[\"sentence\"]).input_ids\n    return batch","metadata":{"id":"eJY7I0XAwe9p","execution":{"iopub.status.busy":"2022-08-23T18:30:06.430717Z","iopub.execute_input":"2022-08-23T18:30:06.431274Z","iopub.status.idle":"2022-08-23T18:30:06.438118Z","shell.execute_reply.started":"2022-08-23T18:30:06.431238Z","shell.execute_reply":"2022-08-23T18:30:06.436965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's apply the data preparation function to all examples.","metadata":{"id":"q6Pg_WR3OGAP"}},{"cell_type":"markdown","source":"### Sample Training Example\ntrain samples=34001\nval samples=3000","metadata":{}},{"cell_type":"code","source":"#common_voice_train=common_voice_train.select(range(5000))\ncommon_voice_test=common_voice_test.select(range(3000))","metadata":{"execution":{"iopub.status.busy":"2022-08-23T18:30:06.439916Z","iopub.execute_input":"2022-08-23T18:30:06.44034Z","iopub.status.idle":"2022-08-23T18:30:06.450751Z","shell.execute_reply.started":"2022-08-23T18:30:06.440302Z","shell.execute_reply":"2022-08-23T18:30:06.449694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_voice_train = common_voice_train.map(prepare_dataset, remove_columns=common_voice_train.column_names)\ncommon_voice_test = common_voice_test.map(prepare_dataset, remove_columns=common_voice_test.column_names)","metadata":{"id":"-np9xYK-wl8q","outputId":"00d6940a-a7bf-4128-896b-76bc289e5b7f","execution":{"iopub.status.busy":"2022-08-23T18:30:06.45265Z","iopub.execute_input":"2022-08-23T18:30:06.453001Z","iopub.status.idle":"2022-08-23T18:37:21.277782Z","shell.execute_reply.started":"2022-08-23T18:30:06.452966Z","shell.execute_reply":"2022-08-23T18:37:21.276109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{"id":"gYlQkKVoRUos"}},{"cell_type":"markdown","source":"### Set-up Trainer","metadata":{"id":"Slk403unUS91"}},{"cell_type":"code","source":"import torch\n\nfrom dataclasses import dataclass, field\nfrom typing import Any, Dict, List, Optional, Union\n\n@dataclass\nclass DataCollatorCTCWithPadding:\n    \"\"\"\n    Data collator that will dynamically pad the inputs received.\n    Args:\n        processor (:class:`~transformers.Wav2Vec2Processor`)\n            The processor used for proccessing the data.\n        padding (:obj:`bool`, :obj:`str` or :class:`~transformers.tokenization_utils_base.PaddingStrategy`, `optional`, defaults to :obj:`True`):\n            Select a strategy to pad the returned sequences (according to the model's padding side and padding index)\n            among:\n            * :obj:`True` or :obj:`'longest'`: Pad to the longest sequence in the batch (or no padding if only a single\n              sequence if provided).\n            * :obj:`'max_length'`: Pad to a maximum length specified with the argument :obj:`max_length` or to the\n              maximum acceptable input length for the model if that argument is not provided.\n            * :obj:`False` or :obj:`'do_not_pad'` (default): No padding (i.e., can output a batch with sequences of\n              different lengths).\n    \"\"\"\n\n    processor: Wav2Vec2Processor\n    padding: Union[bool, str] = True\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lenghts and need\n        # different padding methods\n        input_features = [{\"input_values\": feature[\"input_values\"]} for feature in features]\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n\n        batch = self.processor.pad(\n            input_features,\n            padding=self.padding,\n            return_tensors=\"pt\",\n        )\n        with self.processor.as_target_processor():\n            labels_batch = self.processor.pad(\n                label_features,\n                padding=self.padding,\n                return_tensors=\"pt\",\n            )\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        batch[\"labels\"] = labels\n\n        return batch","metadata":{"id":"tborvC9hx88e","execution":{"iopub.status.busy":"2022-08-23T18:37:21.291224Z","iopub.execute_input":"2022-08-23T18:37:21.291893Z","iopub.status.idle":"2022-08-23T18:37:23.393273Z","shell.execute_reply.started":"2022-08-23T18:37:21.291857Z","shell.execute_reply":"2022-08-23T18:37:23.392121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_collator = DataCollatorCTCWithPadding(processor=processor, padding=True)","metadata":{"id":"lbQf5GuZyQ4_","execution":{"iopub.status.busy":"2022-08-23T18:37:23.395217Z","iopub.execute_input":"2022-08-23T18:37:23.395606Z","iopub.status.idle":"2022-08-23T18:37:23.406767Z","shell.execute_reply.started":"2022-08-23T18:37:23.395571Z","shell.execute_reply":"2022-08-23T18:37:23.405712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, the evaluation metric is defined. As mentioned earlier, the \npredominant metric in ASR is the word error rate (WER), hence we will use it in this notebook as well.","metadata":{"id":"xO-Zdj-5cxXp"}},{"cell_type":"code","source":"wer_metric = load_metric(\"wer\")","metadata":{"id":"9Xsux2gmyXso","outputId":"18ceeb9e-1a0d-4ee8-f511-a12ad3608bf1","execution":{"iopub.status.busy":"2022-08-23T18:37:23.408689Z","iopub.execute_input":"2022-08-23T18:37:23.409098Z","iopub.status.idle":"2022-08-23T18:37:24.390167Z","shell.execute_reply.started":"2022-08-23T18:37:23.409018Z","shell.execute_reply":"2022-08-23T18:37:24.389188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_metrics(pred):\n    pred_logits = pred.predictions\n    pred_ids = np.argmax(pred_logits, axis=-1)\n\n    pred.label_ids[pred.label_ids == -100] = processor.tokenizer.pad_token_id\n\n    pred_str = processor.batch_decode(pred_ids)\n    # we do not want to group tokens when computing the metrics\n    label_str = processor.batch_decode(pred.label_ids, group_tokens=False)\n\n    wer = wer_metric.compute(predictions=pred_str, references=label_str)\n\n    return {\"wer\": wer}","metadata":{"id":"1XZ-kjweyTy_","execution":{"iopub.status.busy":"2022-08-23T18:37:24.391499Z","iopub.execute_input":"2022-08-23T18:37:24.391954Z","iopub.status.idle":"2022-08-23T18:37:24.399451Z","shell.execute_reply.started":"2022-08-23T18:37:24.391916Z","shell.execute_reply":"2022-08-23T18:37:24.398397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> <p style='color:Black'> 📌 We used a publicly available model from <b>huggingface </b>for further finetuning on provided training data. This <a href=\"https://huggingface.co/arijitx/wav2vec2-xls-r-300m-bengali\"><b>model</b></a> is a fine-tuned version of <a href=\"https://huggingface.co/facebook/wav2vec2-xls-r-300m\"><b>facebook/wav2vec2-xls-r-300m </b></a>on the OPENSLR_SLR53 - bengali dataset.</p> </div>","metadata":{}},{"cell_type":"code","source":"from transformers import Wav2Vec2ForCTC\n\nmodel = Wav2Vec2ForCTC.from_pretrained(\n    \"arijitx/wav2vec2-xls-r-300m-bengali\", \n    attention_dropout=0.0,\n    activation_dropout=0.1,\n    hidden_dropout=0.0,\n    feat_proj_dropout=0.0,\n    mask_time_prob=0.75,\n    mask_time_length=10,\n    mask_feature_prob=0.25,\n    mask_feature_length=64,\n    layerdrop=0.0,\n    ctc_loss_reduction=\"mean\", \n    pad_token_id=processor.tokenizer.pad_token_id,\n    vocab_size=len(processor.tokenizer),\n)","metadata":{"id":"e7cqAWIayn6w","outputId":"3e6cebea-78ef-45df-87b0-f63b78ba9644","execution":{"iopub.status.busy":"2022-08-23T18:37:24.401061Z","iopub.execute_input":"2022-08-23T18:37:24.401709Z","iopub.status.idle":"2022-08-23T18:38:23.514622Z","shell.execute_reply.started":"2022-08-23T18:37:24.401672Z","shell.execute_reply":"2022-08-23T18:38:23.513615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> <p style='color:Black'> ⏩ The first component of XLS-R consists of a stack of CNN layers that are used to extract acoustically meaningful - but contextually independent - features from the raw speech signal. This part of the model has already been sufficiently trained during pretraining and as stated in the <a href=\"https://arxiv.org/pdf/2006.13979.pdf\"><b>paper</b></a> does not need to be fine-tuned anymore. \nThus, we can set the <i>requires_grad</i> to <i>False</i> for all parameters of the *feature extraction* part. <b>Additionally we freezed all the encoder layers except for the last two layers for further finetuning.</b><i> (From our experiments it gives the best result.)</i></p> </div>","metadata":{"id":"1DwR3XLSzGDD"}},{"cell_type":"code","source":"model.freeze_feature_extractor()\nmodel.freeze_feature_encoder()\nfor param in model.wav2vec2.encoder.parameters():\n    param.requires_grad = False\nlayer_no=len(model.wav2vec2.encoder.layers)\nfor l in range(layer_no-2,layer_no):\n    for param in model.wav2vec2.encoder.layers[l].parameters():\n        param.requires_grad = True","metadata":{"id":"oGI8zObtZ3V0","execution":{"iopub.status.busy":"2022-08-23T18:38:23.516161Z","iopub.execute_input":"2022-08-23T18:38:23.516535Z","iopub.status.idle":"2022-08-23T18:38:23.524834Z","shell.execute_reply.started":"2022-08-23T18:38:23.516484Z","shell.execute_reply":"2022-08-23T18:38:23.523848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(42)\nfrom transformers import TrainingArguments\n\nstep_n = 500 # Change it to 400 while training.\n\ntraining_args = TrainingArguments(\n  output_dir=\"./\",\n  group_by_length=True,\n  per_device_train_batch_size=8,\n  gradient_accumulation_steps=2,\n  evaluation_strategy=\"steps\",\n  num_train_epochs=9,\n  gradient_checkpointing=True,\n  fp16=True,\n  save_steps=step_n,\n  eval_steps=step_n,\n  logging_steps=step_n,\n  learning_rate=7.5e-5,\n  warmup_steps=500,\n  save_total_limit=4,\n  push_to_hub=False,\n)","metadata":{"id":"KbeKSV7uzGPP","execution":{"iopub.status.busy":"2022-08-23T18:43:27.859519Z","iopub.execute_input":"2022-08-23T18:43:27.860196Z","iopub.status.idle":"2022-08-23T18:43:27.955744Z","shell.execute_reply.started":"2022-08-23T18:43:27.860163Z","shell.execute_reply":"2022-08-23T18:43:27.954782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, all instances can be passed to Trainer and we are ready to start training!","metadata":{"id":"OsW-WZcL1ZtN"}},{"cell_type":"code","source":"from transformers import Trainer\n\ntrainer = Trainer(\n    model=model,\n    data_collator=data_collator,\n    args=training_args,\n    compute_metrics=compute_metrics,\n    train_dataset=common_voice_train,\n    eval_dataset=common_voice_test,\n    tokenizer=processor.feature_extractor,\n)","metadata":{"id":"rY7vBmFCPFgC","outputId":"c47ecc78-5259-44db-c121-5bd7945defb8","execution":{"iopub.status.busy":"2022-08-23T18:43:36.011676Z","iopub.execute_input":"2022-08-23T18:43:36.012042Z","iopub.status.idle":"2022-08-23T18:43:42.161364Z","shell.execute_reply.started":"2022-08-23T18:43:36.012008Z","shell.execute_reply":"2022-08-23T18:43:42.160344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{"id":"rpvZHM1xReIW"}},{"cell_type":"code","source":"import os\nos.environ[\"WANDB_DISABLED\"] = \"true\"","metadata":{"execution":{"iopub.status.busy":"2022-08-23T18:43:45.169635Z","iopub.execute_input":"2022-08-23T18:43:45.170022Z","iopub.status.idle":"2022-08-23T18:43:45.17556Z","shell.execute_reply.started":"2022-08-23T18:43:45.169986Z","shell.execute_reply":"2022-08-23T18:43:45.174409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()","metadata":{"id":"9fRr9TG5pGBl","outputId":"122ea040-7b24-452a-c7d4-7e72e1c46973","execution":{"iopub.status.busy":"2022-08-25T08:16:49.747336Z","iopub.execute_input":"2022-08-25T08:16:49.748385Z","iopub.status.idle":"2022-08-25T08:16:49.770233Z","shell.execute_reply.started":"2022-08-25T08:16:49.748238Z","shell.execute_reply":"2022-08-25T08:16:49.769061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"./checkpoint-19000\")\n#processor.save_pretrained(\"./checkpoint-14500\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:35:17.14425Z","iopub.execute_input":"2022-08-22T06:35:17.145311Z","iopub.status.idle":"2022-08-22T06:35:17.176065Z","shell.execute_reply.started":"2022-08-22T06:35:17.145265Z","shell.execute_reply":"2022-08-22T06:35:17.175052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Copying the best checkpoint to model directory","metadata":{}},{"cell_type":"code","source":"from distutils.dir_util import copy_tree\ncopy_tree('./checkpoint-19000/', './wav2vec2_large_xlsr_bangla_afia/')\nprocessor.save_pretrained(\"./wav2vec2_large_xlsr_bangla_afia/\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Integrating Language model","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> <p style='color:Black'>📎 Wav2Vec2 can be significantly improved when used with a language model. A language model should assist the acoustic model, such as Wav2Vec2, in anticipating the subsequent word in order to be helpful for a speech recognition system. Regardless of the audio input provided to the speech recognition system, the language model need to be capable of accurately predicting the following word given all previously transcribed words. Using the well-known <a href=\"https://github.com/kpu/kenlm\"><b>KenLM library</b></a>, we created an n-gram model for decoding the asr transcriptions.</p></div>. ","metadata":{}},{"cell_type":"code","source":"%%capture\n!sudo apt install -y build-essential cmake libboost-system-dev libboost-thread-dev libboost-program-options-dev libboost-test-dev libeigen3-dev zlib1g-dev libbz2-dev liblzma-dev\n!wget -O - https://kheafield.com/code/kenlm.tar.gz | tar xz\n!mkdir kenlm/build && cd kenlm/build && cmake .. && make -j2\n!ls kenlm/build/bin","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset=load_from_disk(\"../input/dl-sprint-data/train\")\nvalid_dataset=load_from_disk(\"../input/dl-sprint-data/validation\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:45:28.653677Z","iopub.execute_input":"2022-08-22T06:45:28.654592Z","iopub.status.idle":"2022-08-22T06:45:28.991298Z","shell.execute_reply.started":"2022-08-22T06:45:28.654554Z","shell.execute_reply":"2022-08-22T06:45:28.990237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chars_to_ignore_regex = '[,?.!\\-\\;\\:\\\"“%‘”�—’…–।]'\ndef extract_text(batch):\n    text = batch[\"sentence\"]\n    batch[\"text\"] = re.sub(chars_to_ignore_regex, \"\",text)\n    return batch\ntrain_dataset = train_dataset.map(extract_text, remove_columns=train_dataset.column_names)\nvalid_dataset = valid_dataset.map(extract_text, remove_columns=valid_dataset.column_names)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:45:35.949692Z","iopub.execute_input":"2022-08-22T06:45:35.950087Z","iopub.status.idle":"2022-08-22T06:45:35.982594Z","shell.execute_reply.started":"2022-08-22T06:45:35.950056Z","shell.execute_reply":"2022-08-22T06:45:35.981575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We used the sentences in training and validation dataset for building the 5-gram","metadata":{}},{"cell_type":"code","source":"dataset=train_dataset[\"text\"]+valid_dataset[\"text\"]\nwith open(\"./text.txt\", \"w\") as file:\n    file.write(\" \".join(set(dataset)))\n!kenlm/build/bin/lmplz -o 5 <\"./text.txt\" > \"5gram.arpa\"","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:46:49.418682Z","iopub.execute_input":"2022-08-22T06:46:49.419311Z","iopub.status.idle":"2022-08-22T06:46:50.387301Z","shell.execute_reply.started":"2022-08-22T06:46:49.41927Z","shell.execute_reply":"2022-08-22T06:46:50.385335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"5gram.arpa\", \"r\") as read_file, open(\"5gram_correct.arpa\", \"w\") as write_file:\n    has_added_eos = False\n    for line in read_file:\n        if not has_added_eos and \"ngram 1=\" in line:\n            count=line.strip().split(\"=\")[-1]\n            write_file.write(line.replace(f\"{count}\", f\"{int(count)+1}\"))\n        elif not has_added_eos and \"<s>\" in line:\n            write_file.write(line)\n            write_file.write(line.replace(\"<s>\", \"</s>\"))\n            has_added_eos = True\n        else:\n            write_file.write(line)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:46:52.736933Z","iopub.execute_input":"2022-08-22T06:46:52.738618Z","iopub.status.idle":"2022-08-22T06:46:52.825383Z","shell.execute_reply.started":"2022-08-22T06:46:52.73858Z","shell.execute_reply":"2022-08-22T06:46:52.824287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -20 5gram_correct.arpa","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:46:52.827964Z","iopub.execute_input":"2022-08-22T06:46:52.829167Z","iopub.status.idle":"2022-08-22T06:46:54.057525Z","shell.execute_reply.started":"2022-08-22T06:46:52.829129Z","shell.execute_reply":"2022-08-22T06:46:54.055396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2Processor\nprocessor = Wav2Vec2Processor.from_pretrained(\"arijitx/wav2vec2-xls-r-300m-bengali\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:46:54.060191Z","iopub.execute_input":"2022-08-22T06:46:54.061124Z","iopub.status.idle":"2022-08-22T06:47:00.260269Z","shell.execute_reply.started":"2022-08-22T06:46:54.061037Z","shell.execute_reply":"2022-08-22T06:47:00.259231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\nsorted_vocab_dict","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:00.261994Z","iopub.execute_input":"2022-08-22T06:47:00.262707Z","iopub.status.idle":"2022-08-22T06:47:00.299979Z","shell.execute_reply.started":"2022-08-22T06:47:00.262661Z","shell.execute_reply":"2022-08-22T06:47:00.298983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyctcdecode import build_ctcdecoder\n\ndecoder = build_ctcdecoder(\n    labels=list(sorted_vocab_dict.keys()),\n    kenlm_model_path=\"5gram_correct.arpa\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:13.322055Z","iopub.execute_input":"2022-08-22T06:47:13.322768Z","iopub.status.idle":"2022-08-22T06:47:13.452087Z","shell.execute_reply.started":"2022-08-22T06:47:13.322719Z","shell.execute_reply":"2022-08-22T06:47:13.450515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:36.689249Z","iopub.execute_input":"2022-08-22T06:47:36.690202Z","iopub.status.idle":"2022-08-22T06:47:36.721829Z","shell.execute_reply.started":"2022-08-22T06:47:36.690165Z","shell.execute_reply":"2022-08-22T06:47:36.720772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Saving the built LM in the model directory","metadata":{}},{"cell_type":"code","source":"processor_with_lm.save_pretrained(\"./wav2vec2_large_xlsr_bangla_afia/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:43.713933Z","iopub.execute_input":"2022-08-22T06:47:43.714608Z","iopub.status.idle":"2022-08-22T06:47:43.791868Z","shell.execute_reply.started":"2022-08-22T06:47:43.714571Z","shell.execute_reply":"2022-08-22T06:47:43.790344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kenlm/build/bin/build_binary ./wav2vec2_large_xlsr_bangla_afia/language_model/5gram_correct.arpa ./wav2vec2_large_xlsr_bangla_afia/language_model/5gram.bin","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:45.499226Z","iopub.execute_input":"2022-08-22T06:47:45.500159Z","iopub.status.idle":"2022-08-22T06:47:46.671578Z","shell.execute_reply.started":"2022-08-22T06:47:45.500096Z","shell.execute_reply":"2022-08-22T06:47:46.670383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm ./wav2vec2_large_xlsr_bangla_afia/language_model/5gram_correct.arpa && tree -h ./wav2vec2_large_xlsr_bangla_afia/","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:47:51.123159Z","iopub.execute_input":"2022-08-22T06:47:51.12363Z","iopub.status.idle":"2022-08-22T06:47:52.249138Z","shell.execute_reply.started":"2022-08-22T06:47:51.123596Z","shell.execute_reply":"2022-08-22T06:47:52.247877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm  -r ./kenlm ","metadata":{"execution":{"iopub.status.busy":"2022-08-22T06:49:20.664559Z","iopub.execute_input":"2022-08-22T06:49:20.665432Z","iopub.status.idle":"2022-08-22T06:49:21.802744Z","shell.execute_reply.started":"2022-08-22T06:49:20.665395Z","shell.execute_reply":"2022-08-22T06:49:21.801606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Resources:\n* blog: https://huggingface.co/blog/fine-tune-xlsr-wav2vec2\n* blog: https://huggingface.co/blog/wav2vec2-with-ngram","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}