{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:48:21.396686Z","iopub.execute_input":"2023-10-07T04:48:21.397771Z","iopub.status.idle":"2023-10-07T04:49:36.284430Z","shell.execute_reply.started":"2023-10-07T04:48:21.397727Z","shell.execute_reply":"2023-10-07T04:49:36.283292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\n\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:49:36.286922Z","iopub.execute_input":"2023-10-07T04:49:36.287291Z","iopub.status.idle":"2023-10-07T04:49:57.135936Z","shell.execute_reply.started":"2023-10-07T04:49:36.287254Z","shell.execute_reply":"2023-10-07T04:49:57.134872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST = \"/kaggle/input/bengaliai-speech/test_mp3s/\"\n\nSAMPLING_RATE = 16_000\nMODEL_PATH = \"/kaggle/input/bengali-wav2vec/wav2vec-ver5\"\nLM_PATH = \"/kaggle/input/bengali-sr-download-public-trained-models/wav2vec2-xls-r-300m-bengali/language_model/\"","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:49:57.137795Z","iopub.execute_input":"2023-10-07T04:49:57.138619Z","iopub.status.idle":"2023-10-07T04:49:57.143692Z","shell.execute_reply.started":"2023-10-07T04:49:57.138578Z","shell.execute_reply":"2023-10-07T04:49:57.142670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)\nprocessor = Wav2Vec2Processor.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:49:57.147176Z","iopub.execute_input":"2023-10-07T04:49:57.147885Z","iopub.status.idle":"2023-10-07T04:50:09.862050Z","shell.execute_reply.started":"2023-10-07T04:49:57.147857Z","shell.execute_reply":"2023-10-07T04:50:09.861072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(Path(LM_PATH + \"5gram.bin\")),\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:09.866678Z","iopub.execute_input":"2023-10-07T04:50:09.868948Z","iopub.status.idle":"2023-10-07T04:50:54.894063Z","shell.execute_reply.started":"2023-10-07T04:50:09.868909Z","shell.execute_reply":"2023-10-07T04:50:54.893045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.895440Z","iopub.execute_input":"2023-10-07T04:50:54.897337Z","iopub.status.idle":"2023-10-07T04:50:54.909818Z","shell.execute_reply.started":"2023-10-07T04:50:54.897301Z","shell.execute_reply":"2023-10-07T04:50:54.908860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr, mono=False)[0]\n        \n        return w","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.911138Z","iopub.execute_input":"2023-10-07T04:50:54.911955Z","iopub.status.idle":"2023-10-07T04:50:54.925324Z","shell.execute_reply.started":"2023-10-07T04:50:54.911923Z","shell.execute_reply":"2023-10-07T04:50:54.924451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/bengaliai-speech/sample_submission.csv\", dtype={\"id\": str})\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.926738Z","iopub.execute_input":"2023-10-07T04:50:54.927694Z","iopub.status.idle":"2023-10-07T04:50:54.960370Z","shell.execute_reply.started":"2023-10-07T04:50:54.927581Z","shell.execute_reply":"2023-10-07T04:50:54.959370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = [str(Path(TEST + f\"{aid}.mp3\")) for aid in test[\"id\"].values]","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.961564Z","iopub.execute_input":"2023-10-07T04:50:54.963988Z","iopub.status.idle":"2023-10-07T04:50:54.968472Z","shell.execute_reply.started":"2023-10-07T04:50:54.963966Z","shell.execute_reply":"2023-10-07T04:50:54.967371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=4, shuffle=False,\n    num_workers=1, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.970272Z","iopub.execute_input":"2023-10-07T04:50:54.971131Z","iopub.status.idle":"2023-10-07T04:50:54.981353Z","shell.execute_reply.started":"2023-10-07T04:50:54.971096Z","shell.execute_reply":"2023-10-07T04:50:54.980435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:54.984986Z","iopub.execute_input":"2023-10-07T04:50:54.985244Z","iopub.status.idle":"2023-10-07T04:50:55.060244Z","shell.execute_reply.started":"2023-10-07T04:50:54.985225Z","shell.execute_reply":"2023-10-07T04:50:55.059157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.to(device)\nmodel = model.eval()\nmodel = model.half()","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:50:55.061590Z","iopub.execute_input":"2023-10-07T04:50:55.061906Z","iopub.status.idle":"2023-10-07T04:51:03.582909Z","shell.execute_reply.started":"2023-10-07T04:50:55.061875Z","shell.execute_reply":"2023-10-07T04:51:03.581843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=512).text\n            pred_sentence_list.append(sentence)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:51:03.584354Z","iopub.execute_input":"2023-10-07T04:51:03.585372Z","iopub.status.idle":"2023-10-07T04:51:23.439330Z","shell.execute_reply.started":"2023-10-07T04:51:03.585338Z","shell.execute_reply":"2023-10-07T04:51:23.437777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred_sentence_list = []\n\n# with torch.no_grad():\n#     for batch in tqdm(test_loader):\n#         x = batch[\"input_values\"]\n#         x = x.to(device, non_blocking=True)\n#         with torch.cuda.amp.autocast(True):\n#             y = model(x).logits\n#         y = y.detach().cpu().numpy()\n        \n#         for l in y:  \n#             sentence = processor_with_lm.decode(l, beam_width=512).text\n#             pred_sentence_list.append(sentence)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:51:23.441218Z","iopub.execute_input":"2023-10-07T04:51:23.442348Z","iopub.status.idle":"2023-10-07T04:51:23.448879Z","shell.execute_reply.started":"2023-10-07T04:51:23.442300Z","shell.execute_reply":"2023-10-07T04:51:23.447533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:51:23.450372Z","iopub.execute_input":"2023-10-07T04:51:23.451384Z","iopub.status.idle":"2023-10-07T04:51:23.469756Z","shell.execute_reply.started":"2023-10-07T04:51:23.451343Z","shell.execute_reply":"2023-10-07T04:51:23.468530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [\n    postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:51:23.471436Z","iopub.execute_input":"2023-10-07T04:51:23.472594Z","iopub.status.idle":"2023-10-07T04:51:23.523120Z","shell.execute_reply.started":"2023-10-07T04:51:23.472552Z","shell.execute_reply":"2023-10-07T04:51:23.521995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\n\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-07T04:51:23.524720Z","iopub.execute_input":"2023-10-07T04:51:23.525348Z","iopub.status.idle":"2023-10-07T04:51:23.545477Z","shell.execute_reply.started":"2023-10-07T04:51:23.525310Z","shell.execute_reply":"2023-10-07T04:51:23.544136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}