{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Reference\n\nFirstly, Please upvote/refer to [@tawara's](https://www.kaggle.com/ttahara) discussions and inference [notebook](https://www.kaggle.com/code/ttahara/bengali-sr-public-wav2vec2-0-w-lm-baseline).\n\n","metadata":{}},{"cell_type":"markdown","source":"## What this notebook features??\n- I wanted to showcase the impact of finetuning the models on competition dataset.\n- Current version comprises of finetuned model only with 10% of competition training data.\n- I will publish the training code in upcoming days. You can refer to this [dataset]()\n\n\n\nPublic models from hugging faces:\n* `https://huggingface.co/ai4bharat/indicwav2vec_v1_bengali` for Wav2vec2CTC Model only\n* `https://huggingface.co/arijitx/wav2vec2-xls-r-300m-bengali` for Language Model\n\nI didn't trained these models using the competitaion data at all. I just want to know public models score as baseline.  \n\nSo we may get higher and higher score by fine-tuning on competition data.\n\n### Note: I only finetuned the indicwav2vec_v1_bengali which is a CTC model. I am still using the public LM model mentioned above.","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-28T08:47:22.968153Z","iopub.execute_input":"2023-08-28T08:47:22.968510Z","iopub.status.idle":"2023-08-28T08:48:38.624874Z","shell.execute_reply.started":"2023-08-28T08:47:22.968477Z","shell.execute_reply":"2023-08-28T08:48:38.623628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:48:38.628510Z","iopub.execute_input":"2023-08-28T08:48:38.628851Z","iopub.status.idle":"2023-08-28T08:48:39.599297Z","shell.execute_reply.started":"2023-08-28T08:48:38.628823Z","shell.execute_reply":"2023-08-28T08:48:39.598043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\n\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-28T08:48:39.601261Z","iopub.execute_input":"2023-08-28T08:48:39.601760Z","iopub.status.idle":"2023-08-28T08:48:52.058677Z","shell.execute_reply.started":"2023-08-28T08:48:39.601720Z","shell.execute_reply":"2023-08-28T08:48:52.057713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = DATA / \"test_mp3s\"\n\nSAMPLING_RATE = 16_000\nMODEL_PATH = INPUT / \"bengali-wav2vec2-finetuned/\"\nLM_PATH = INPUT / \"bengali-sr-download-public-trained-models/wav2vec2-xls-r-300m-bengali/language_model/\"","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:48:52.061444Z","iopub.execute_input":"2023-08-28T08:48:52.061806Z","iopub.status.idle":"2023-08-28T08:48:52.069208Z","shell.execute_reply.started":"2023-08-28T08:48:52.061771Z","shell.execute_reply":"2023-08-28T08:48:52.067319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### load model, processor, decoder","metadata":{}},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)\nprocessor = Wav2Vec2Processor.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:48:52.070498Z","iopub.execute_input":"2023-08-28T08:48:52.071304Z","iopub.status.idle":"2023-08-28T08:49:08.146806Z","shell.execute_reply.started":"2023-08-28T08:48:52.071269Z","shell.execute_reply":"2023-08-28T08:49:08.145796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(LM_PATH / \"5gram.bin\"),\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:08.148539Z","iopub.execute_input":"2023-08-28T08:49:08.148964Z","iopub.status.idle":"2023-08-28T08:49:51.240751Z","shell.execute_reply.started":"2023-08-28T08:49:08.148928Z","shell.execute_reply":"2023-08-28T08:49:51.239626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.242285Z","iopub.execute_input":"2023-08-28T08:49:51.244845Z","iopub.status.idle":"2023-08-28T08:49:51.251590Z","shell.execute_reply.started":"2023-08-28T08:49:51.244804Z","shell.execute_reply":"2023-08-28T08:49:51.250610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prepare dataloader","metadata":{}},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr, mono=False)[0]\n        \n        return w","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.252873Z","iopub.execute_input":"2023-08-28T08:49:51.253310Z","iopub.status.idle":"2023-08-28T08:49:51.263145Z","shell.execute_reply.started":"2023-08-28T08:49:51.253268Z","shell.execute_reply":"2023-08-28T08:49:51.262202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(DATA / \"sample_submission.csv\", dtype={\"id\": str})\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.264734Z","iopub.execute_input":"2023-08-28T08:49:51.265271Z","iopub.status.idle":"2023-08-28T08:49:51.293174Z","shell.execute_reply.started":"2023-08-28T08:49:51.265235Z","shell.execute_reply":"2023-08-28T08:49:51.292115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = [str(TEST / f\"{aid}.mp3\") for aid in test[\"id\"].values]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.298086Z","iopub.execute_input":"2023-08-28T08:49:51.298378Z","iopub.status.idle":"2023-08-28T08:49:51.304256Z","shell.execute_reply.started":"2023-08-28T08:49:51.298353Z","shell.execute_reply":"2023-08-28T08:49:51.302638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=8, shuffle=False,\n    num_workers=2, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.305631Z","iopub.execute_input":"2023-08-28T08:49:51.306317Z","iopub.status.idle":"2023-08-28T08:49:51.315081Z","shell.execute_reply.started":"2023-08-28T08:49:51.306160Z","shell.execute_reply":"2023-08-28T08:49:51.313696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.316866Z","iopub.execute_input":"2023-08-28T08:49:51.317413Z","iopub.status.idle":"2023-08-28T08:49:51.349854Z","shell.execute_reply.started":"2023-08-28T08:49:51.317377Z","shell.execute_reply":"2023-08-28T08:49:51.348046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.to(device)\nmodel = model.eval()\nmodel = model.half()","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:51.351491Z","iopub.execute_input":"2023-08-28T08:49:51.352138Z","iopub.status.idle":"2023-08-28T08:49:56.513682Z","shell.execute_reply.started":"2023-08-28T08:49:51.352104Z","shell.execute_reply":"2023-08-28T08:49:56.512720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=512).text\n            pred_sentence_list.append(sentence)","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:49:56.515349Z","iopub.execute_input":"2023-08-28T08:49:56.515750Z","iopub.status.idle":"2023-08-28T08:50:11.346800Z","shell.execute_reply.started":"2023-08-28T08:49:56.515712Z","shell.execute_reply":"2023-08-28T08:50:11.345594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Submission","metadata":{}},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:50:11.348953Z","iopub.execute_input":"2023-08-28T08:50:11.349339Z","iopub.status.idle":"2023-08-28T08:50:11.358170Z","shell.execute_reply.started":"2023-08-28T08:50:11.349299Z","shell.execute_reply":"2023-08-28T08:50:11.355688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [\n    postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:50:11.359976Z","iopub.execute_input":"2023-08-28T08:50:11.360488Z","iopub.status.idle":"2023-08-28T08:50:11.408357Z","shell.execute_reply.started":"2023-08-28T08:50:11.360452Z","shell.execute_reply":"2023-08-28T08:50:11.407461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\n\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-28T08:50:11.409801Z","iopub.execute_input":"2023-08-28T08:50:11.410626Z","iopub.status.idle":"2023-08-28T08:50:11.424863Z","shell.execute_reply.started":"2023-08-28T08:50:11.410589Z","shell.execute_reply":"2023-08-28T08:50:11.423908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EOF","metadata":{}}]}