{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- These inference notebook is based on @nischaydnk's [notebook](https://www.kaggle.com/code/nischaydnk/bengali-finetuning-baseline-wav2vec2-inference). If it's helpful to you, please upvote his firstly.\n- The training notebook is here: [Bengali SR wav2vec_v1_bengali [Training]](https://www.kaggle.com/takanashihumbert/bengali-sr-wav2vec-v1-bengali-training)\n- Running this notebook you can score 0.445 on leaderboard.","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-13T05:41:44.524957Z","iopub.execute_input":"2023-10-13T05:41:44.525578Z","iopub.status.idle":"2023-10-13T05:42:51.548636Z","shell.execute_reply.started":"2023-10-13T05:41:44.525548Z","shell.execute_reply":"2023-10-13T05:42:51.547421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:42:51.551074Z","iopub.execute_input":"2023-10-13T05:42:51.551439Z","iopub.status.idle":"2023-10-13T05:42:52.507200Z","shell.execute_reply.started":"2023-10-13T05:42:51.551403Z","shell.execute_reply":"2023-10-13T05:42:52.505898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\n\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-13T05:42:52.509080Z","iopub.execute_input":"2023-10-13T05:42:52.509443Z","iopub.status.idle":"2023-10-13T05:43:05.210839Z","shell.execute_reply.started":"2023-10-13T05:42:52.509407Z","shell.execute_reply":"2023-10-13T05:43:05.209979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = DATA / \"test_mp3s\"\n\nSAMPLING_RATE = 16_000\n# MODEL_PATH = INPUT / \"/kaggle/input/bengali-ex002/ex002\"\nMODEL_PATH_list = [\"/kaggle/input/wav2vec2-speech-recognition-v4\",\n#                    \"/kaggle/input/bengali-ex002/ex002\"\n                  ]\nweight_weight = [1]\nLM_PATH = INPUT / \"/kaggle/input/arijitx-full-model/wav2vec2-xls-r-300m-bengali/language_model\"","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:43:05.213236Z","iopub.execute_input":"2023-10-13T05:43:05.213537Z","iopub.status.idle":"2023-10-13T05:43:05.218604Z","shell.execute_reply.started":"2023-10-13T05:43:05.213505Z","shell.execute_reply":"2023-10-13T05:43:05.217761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### load model, processor, decoder","metadata":{}},{"cell_type":"code","source":"model_list = []\nfor MODEL_PATH in MODEL_PATH_list:\n    model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)\n    model_list.append(model)\nprocessor = Wav2Vec2Processor.from_pretrained(MODEL_PATH_list[0])","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:43:05.219850Z","iopub.execute_input":"2023-10-13T05:43:05.220386Z","iopub.status.idle":"2023-10-13T05:43:17.708652Z","shell.execute_reply.started":"2023-10-13T05:43:05.220355Z","shell.execute_reply":"2023-10-13T05:43:17.707775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(LM_PATH / \"5gram.bin\"),\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:43:17.709979Z","iopub.execute_input":"2023-10-13T05:43:17.710951Z","iopub.status.idle":"2023-10-13T05:44:01.646900Z","shell.execute_reply.started":"2023-10-13T05:43:17.710901Z","shell.execute_reply":"2023-10-13T05:44:01.645855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.648499Z","iopub.execute_input":"2023-10-13T05:44:01.648922Z","iopub.status.idle":"2023-10-13T05:44:01.656896Z","shell.execute_reply.started":"2023-10-13T05:44:01.648889Z","shell.execute_reply":"2023-10-13T05:44:01.655879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prepare dataloader","metadata":{}},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr, mono=False)[0]\n        \n        return w","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.658436Z","iopub.execute_input":"2023-10-13T05:44:01.659060Z","iopub.status.idle":"2023-10-13T05:44:01.668313Z","shell.execute_reply.started":"2023-10-13T05:44:01.659027Z","shell.execute_reply":"2023-10-13T05:44:01.667397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(DATA / \"sample_submission.csv\", dtype={\"id\": str})\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.669845Z","iopub.execute_input":"2023-10-13T05:44:01.670511Z","iopub.status.idle":"2023-10-13T05:44:01.696334Z","shell.execute_reply.started":"2023-10-13T05:44:01.670480Z","shell.execute_reply":"2023-10-13T05:44:01.695402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = [str(TEST / f\"{aid}.mp3\") for aid in test[\"id\"].values]","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.699608Z","iopub.execute_input":"2023-10-13T05:44:01.699842Z","iopub.status.idle":"2023-10-13T05:44:01.704466Z","shell.execute_reply.started":"2023-10-13T05:44:01.699820Z","shell.execute_reply":"2023-10-13T05:44:01.703379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=16, shuffle=False,\n    num_workers=2, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.705779Z","iopub.execute_input":"2023-10-13T05:44:01.706351Z","iopub.status.idle":"2023-10-13T05:44:01.715116Z","shell.execute_reply.started":"2023-10-13T05:44:01.706320Z","shell.execute_reply":"2023-10-13T05:44:01.714171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.716472Z","iopub.execute_input":"2023-10-13T05:44:01.717027Z","iopub.status.idle":"2023-10-13T05:44:01.754982Z","shell.execute_reply.started":"2023-10-13T05:44:01.716997Z","shell.execute_reply":"2023-10-13T05:44:01.754002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_list = [model.to(device).eval().half() for model in model_list]","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:01.756486Z","iopub.execute_input":"2023-10-13T05:44:01.757178Z","iopub.status.idle":"2023-10-13T05:44:07.296173Z","shell.execute_reply.started":"2023-10-13T05:44:01.757146Z","shell.execute_reply":"2023-10-13T05:44:07.295199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = {'alpha': 0.3202723523729998, 'beta': 0.183996879617918436, 'beam_width': 1950, \"token_min_logp\": -5}","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:07.297699Z","iopub.execute_input":"2023-10-13T05:44:07.298048Z","iopub.status.idle":"2023-10-13T05:44:07.303236Z","shell.execute_reply.started":"2023-10-13T05:44:07.298016Z","shell.execute_reply":"2023-10-13T05:44:07.302314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        y = 0\n        with torch.cuda.amp.autocast(True):\n            for model, weight in zip(model_list, weight_weight):\n                y += model(x).logits * weight\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, **best_params).text\n            pred_sentence_list.append(sentence)","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:07.304475Z","iopub.execute_input":"2023-10-13T05:44:07.305330Z","iopub.status.idle":"2023-10-13T05:44:20.410517Z","shell.execute_reply.started":"2023-10-13T05:44:07.305298Z","shell.execute_reply":"2023-10-13T05:44:20.409399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Submission","metadata":{}},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:20.412497Z","iopub.execute_input":"2023-10-13T05:44:20.413185Z","iopub.status.idle":"2023-10-13T05:44:20.420078Z","shell.execute_reply.started":"2023-10-13T05:44:20.413145Z","shell.execute_reply":"2023-10-13T05:44:20.419201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:20.421576Z","iopub.execute_input":"2023-10-13T05:44:20.422542Z","iopub.status.idle":"2023-10-13T05:44:20.463293Z","shell.execute_reply.started":"2023-10-13T05:44:20.422509Z","shell.execute_reply":"2023-10-13T05:44:20.462419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if \"a9395e01ad21\" in test[\"id\"].values:\n    specidal_index = test[\"id\"].values.tolist().index(\"a9395e01ad21\")\n    print(pp_pred_sentence_list[specidal_index])\n    pp_pred_sentence_list[specidal_index] = pp_pred_sentence_list[specidal_index].replace(\"।\",\"?\")","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:20.464434Z","iopub.execute_input":"2023-10-13T05:44:20.465198Z","iopub.status.idle":"2023-10-13T05:44:20.473387Z","shell.execute_reply.started":"2023-10-13T05:44:20.465167Z","shell.execute_reply":"2023-10-13T05:44:20.472484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\n\nprint(test[\"id\"][0], test[\"sentence\"][0])\nprint(test[\"id\"][1], test[\"sentence\"][1])\nprint(test[\"id\"][2], test[\"sentence\"][2])","metadata":{"execution":{"iopub.status.busy":"2023-10-13T05:44:20.474478Z","iopub.execute_input":"2023-10-13T05:44:20.475331Z","iopub.status.idle":"2023-10-13T05:44:20.489909Z","shell.execute_reply.started":"2023-10-13T05:44:20.475300Z","shell.execute_reply":"2023-10-13T05:44:20.488973Z"},"trusted":true},"execution_count":null,"outputs":[]}]}