{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n# !tar xvfz ./python-packages2/jiwer.tgz\n# !pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-10-01T09:46:01.126021Z","iopub.execute_input":"2023-10-01T09:46:01.126387Z","iopub.status.idle":"2023-10-01T09:47:02.434751Z","shell.execute_reply.started":"2023-10-01T09:46:01.126358Z","shell.execute_reply":"2023-10-01T09:47:02.433589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ../input/jiwer-3-0-3/jiwer-3.0.3-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-10-01T09:47:02.437271Z","iopub.execute_input":"2023-10-01T09:47:02.437982Z","iopub.status.idle":"2023-10-01T09:47:34.622008Z","shell.execute_reply.started":"2023-10-01T09:47:02.437948Z","shell.execute_reply":"2023-10-01T09:47:34.620818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2023-10-01T09:47:34.623769Z","iopub.execute_input":"2023-10-01T09:47:34.624158Z","iopub.status.idle":"2023-10-01T09:47:35.608710Z","shell.execute_reply.started":"2023-10-01T09:47:34.624124Z","shell.execute_reply":"2023-10-01T09:47:35.607364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\n\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl\nimport math\nimport jiwer\nfrom torchmetrics.text import WordErrorRate\nimport optuna\nfrom optuna.trial import TrialState\nimport kenlm\nfrom jiwer import wer","metadata":{"execution":{"iopub.status.busy":"2023-10-01T09:47:35.611615Z","iopub.execute_input":"2023-10-01T09:47:35.612390Z","iopub.status.idle":"2023-10-01T09:47:48.761898Z","shell.execute_reply.started":"2023-10-01T09:47:35.612355Z","shell.execute_reply":"2023-10-01T09:47:48.760954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = DATA / \"test_mp3s\"\n\nSAMPLING_RATE = 16_000\n\n# MODEL_PATH = INPUT / \"/kaggle/input/bengali-dataset-asr-finetuned/umong_asr_model/\"\n# LM_PATH = INPUT / \"/kaggle/input/bengali-asr-5gram-lm-decoder-model/\"\n\n# MODEL_PATH = INPUT / \"wav2vec2-bengali-finetuned-shahrukh10/wav2vec2_bengali/\"\n# LM_PATH = INPUT / \"huggingface-model-shahruk10wav2vec2/wav2vec2-xls-r-300m-bengali-commonvoice/language_model/\"\n\n# NEW_MODEL_PATH = INPUT / \"wav2vec2-demo/train_demo/\"\n# NEW_PROCESSOR_PATH = INPUT / \"yellowking-dlsprint-model/YellowKing_processor/\"\n\n# MODEL_PATH = INPUT / \"bengali-ex006/checkpoint-95000/\"\n# LM_PATH = INPUT / \"bengali-asr-5gram-lm-decoder-model/\"\n\nMODEL_PATH = INPUT / \"wav2vec2-bengali-ex007/\"\nLM_PATH = INPUT / \"bengali-asr-5gram-lm-decoder-model/\"","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:03.280196Z","iopub.execute_input":"2023-10-01T10:07:03.280534Z","iopub.status.idle":"2023-10-01T10:07:03.286949Z","shell.execute_reply.started":"2023-10-01T10:07:03.280509Z","shell.execute_reply":"2023-10-01T10:07:03.285569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)\nprocessor = Wav2Vec2Processor.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:04.663716Z","iopub.execute_input":"2023-10-01T10:07:04.664086Z","iopub.status.idle":"2023-10-01T10:07:16.653373Z","shell.execute_reply.started":"2023-10-01T10:07:04.664031Z","shell.execute_reply":"2023-10-01T10:07:16.652316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor.tokenizer.get_vocab()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-10-01T10:07:27.314948Z","iopub.execute_input":"2023-10-01T10:07:27.315648Z","iopub.status.idle":"2023-10-01T10:07:27.323700Z","shell.execute_reply.started":"2023-10-01T10:07:27.315615Z","shell.execute_reply":"2023-10-01T10:07:27.322630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\n# vocab_dict = vocab_dict[\"ben\"]\n# vocab_dict['<s>'] = 64\n# vocab_dict['</s>'] =  65\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:28.925426Z","iopub.execute_input":"2023-10-01T10:07:28.925761Z","iopub.status.idle":"2023-10-01T10:07:28.930797Z","shell.execute_reply.started":"2023-10-01T10:07:28.925733Z","shell.execute_reply":"2023-10-01T10:07:28.929570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_vocab_dict","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-10-01T10:07:29.419368Z","iopub.execute_input":"2023-10-01T10:07:29.419709Z","iopub.status.idle":"2023-10-01T10:07:29.428080Z","shell.execute_reply.started":"2023-10-01T10:07:29.419680Z","shell.execute_reply":"2023-10-01T10:07:29.426376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kenlm_model_path = str(LM_PATH / \"5gram_correct.arpa\")\nkenlm_model_path\n\n# kenlm_model_path = str(LM_PATH / \"commonvoice-bn.5.arpa\")\n\n# unigrams_path = str(LM_PATH / \"unigrams.txt\")","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:31.731497Z","iopub.execute_input":"2023-10-01T10:07:31.732370Z","iopub.status.idle":"2023-10-01T10:07:31.739436Z","shell.execute_reply.started":"2023-10-01T10:07:31.732325Z","shell.execute_reply":"2023-10-01T10:07:31.738408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 0.1913582915441812\n\n# decoder_params = {'alpha': 0.6573403891137685,\n#                  'beta': 0.3177199293629181,\n#                  'beam_width': 1024,\n#                  'unk_score_offset': -8.716724230562996,\n#                  'lm_score_boundary': True,\n#                  'beam_prune_logp': -10.85719040831011,\n#                  'token_min_logp': -6.872247007624853,\n#                  'hotword_weight': 11.833543535497373,\n#                  'prune_history': True}\n\n\n# # 0.19289296213038215\n# decoder_params = {'alpha': 0.6413676501883724,\n#                  'beta': 0.4372542266063575,\n#                  'beam_width': 768,\n#                  'unk_score_offset': -11.163546177593888,\n#                  'lm_score_boundary': True,\n#                  'beam_prune_logp': -10.011891181912848,\n#                  'token_min_logp': -6.970828948239548,\n#                  'hotword_weight': 11.998447431011007,\n#                  'prune_history': False}\n\n\n# 0.20547293792149404\n# decoder_params = {'alpha': 0.6816385361581904, 'beta': 0.4497783009207039, 'beam_width': 512}\n\n# decoder_params = {'alpha': 0.6504972257686036, 'beta': 0.41167847045653994, 'beam_width': 256}","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:38.315439Z","iopub.execute_input":"2023-10-01T10:07:38.315776Z","iopub.status.idle":"2023-10-01T10:07:38.320496Z","shell.execute_reply.started":"2023-10-01T10:07:38.315750Z","shell.execute_reply":"2023-10-01T10:07:38.319569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decoder = pyctcdecode.build_ctcdecoder(\n#     labels=list(sorted_vocab_dict.keys()),\n#     kenlm_model_path=str(LM_PATH / \"5gram_correct.arpa\"),\n#     alpha=decoder_params['alpha'],\n#     beta = decoder_params['beta'],\n#     unk_score_offset = decoder_params['unk_score_offset'],\n#     lm_score_boundary = decoder_params['lm_score_boundary']\n# )\n\n\n# decoder = pyctcdecode.build_ctcdecoder(\n#     labels=list(sorted_vocab_dict.keys()),\n#     kenlm_model_path=str(LM_PATH / \"5gram_correct.arpa\"),\n#     alpha=decoder_params['alpha'],\n#     beta = decoder_params['beta'],\n# )\n\n\n# decoder = pyctcdecode.build_ctcdecoder(\n#      labels=list(sorted_vocab_dict.keys()),\n#      kenlm_model_path=kenlm_model_path,\n#      unigrams=\"/kaggle/input/arijitx-full-model/wav2vec2-xls-r-300m-bengali/language_model/unigrams.txt\",\n# )\n\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    labels=list(sorted_vocab_dict.keys()),\n    kenlm_model_path=kenlm_model_path\n)\n\n\n# decoder = pyctcdecode.build_ctcdecoder(\n#     labels=list(sorted_vocab_dict.keys()),\n#     kenlm_model_path=str(LM_PATH / \"5gram_correct.arpa\"),\n#     alpha=decoder_params['alpha'],\n#     beta = decoder_params['beta']\n# )","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:07:47.962182Z","iopub.execute_input":"2023-10-01T10:07:47.963162Z","iopub.status.idle":"2023-10-01T10:08:11.916932Z","shell.execute_reply.started":"2023-10-01T10:07:47.963123Z","shell.execute_reply":"2023-10-01T10:08:11.915953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:11.918786Z","iopub.execute_input":"2023-10-01T10:08:11.919367Z","iopub.status.idle":"2023-10-01T10:08:11.923724Z","shell.execute_reply.started":"2023-10-01T10:08:11.919334Z","shell.execute_reply":"2023-10-01T10:08:11.922659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr, mono=False)[0]\n        \n        return w","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:11.925007Z","iopub.execute_input":"2023-10-01T10:08:11.925635Z","iopub.status.idle":"2023-10-01T10:08:11.940784Z","shell.execute_reply.started":"2023-10-01T10:08:11.925597Z","shell.execute_reply":"2023-10-01T10:08:11.939664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\n\nmodel = model.to(device)\nmodel = model.eval()\nmodel = model.half()","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:11.943370Z","iopub.execute_input":"2023-10-01T10:08:11.943965Z","iopub.status.idle":"2023-10-01T10:08:16.764928Z","shell.execute_reply.started":"2023-10-01T10:08:11.943932Z","shell.execute_reply":"2023-10-01T10:08:16.763895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:16.766555Z","iopub.execute_input":"2023-10-01T10:08:16.767465Z","iopub.status.idle":"2023-10-01T10:08:16.776924Z","shell.execute_reply.started":"2023-10-01T10:08:16.767422Z","shell.execute_reply":"2023-10-01T10:08:16.772888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(DATA / \"sample_submission.csv\", dtype={\"id\": str})\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:16.778392Z","iopub.execute_input":"2023-10-01T10:08:16.779014Z","iopub.status.idle":"2023-10-01T10:08:16.823316Z","shell.execute_reply.started":"2023-10-01T10:08:16.778984Z","shell.execute_reply":"2023-10-01T10:08:16.822195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = [str(TEST / f\"{aid}.mp3\") for aid in test[\"id\"].values]","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:16.825044Z","iopub.execute_input":"2023-10-01T10:08:16.825770Z","iopub.status.idle":"2023-10-01T10:08:16.833309Z","shell.execute_reply.started":"2023-10-01T10:08:16.825734Z","shell.execute_reply":"2023-10-01T10:08:16.832260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n#     processor.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=8, shuffle=False,\n    num_workers=2, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:16.834901Z","iopub.execute_input":"2023-10-01T10:08:16.835291Z","iopub.status.idle":"2023-10-01T10:08:16.845947Z","shell.execute_reply.started":"2023-10-01T10:08:16.835257Z","shell.execute_reply":"2023-10-01T10:08:16.844855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=512).text\n#             sentence = processor_with_lm.decode(l, beam_width=decoder_params['beam_width']).text\n            pred_sentence_list.append(sentence)\n\n#         for l in y:\n#             beam = decoder.decode_beams(l, \n#                                         beam_width=decoder_params['beam_width'], \n#                                         beam_prune_logp=decoder_params['beam_prune_logp'], \n#                                         token_min_logp=decoder_params['token_min_logp'], \n#                                         hotword_weight=decoder_params['hotword_weight'], \n#                                         prune_history=decoder_params['prune_history'])\n\n#         for l in y:\n#             beam = decoder.decode_beams(l, beam_width=decoder_params['beam_width'])\n#             s = beam[0][0]\n#             pred_sentence_list.append(s)","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:16.847840Z","iopub.execute_input":"2023-10-01T10:08:16.848528Z","iopub.status.idle":"2023-10-01T10:08:31.039615Z","shell.execute_reply.started":"2023-10-01T10:08:16.848493Z","shell.execute_reply":"2023-10-01T10:08:31.038548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:31.043173Z","iopub.execute_input":"2023-10-01T10:08:31.043716Z","iopub.status.idle":"2023-10-01T10:08:31.050235Z","shell.execute_reply.started":"2023-10-01T10:08:31.043678Z","shell.execute_reply":"2023-10-01T10:08:31.049358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [\n    postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:31.051630Z","iopub.execute_input":"2023-10-01T10:08:31.052249Z","iopub.status.idle":"2023-10-01T10:08:31.097273Z","shell.execute_reply.started":"2023-10-01T10:08:31.052218Z","shell.execute_reply":"2023-10-01T10:08:31.096405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\n\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-10-01T10:08:31.098407Z","iopub.execute_input":"2023-10-01T10:08:31.099194Z","iopub.status.idle":"2023-10-01T10:08:31.113613Z","shell.execute_reply.started":"2023-10-01T10:08:31.099162Z","shell.execute_reply":"2023-10-01T10:08:31.112449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}