{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About\n\nAfter seeing [this post](https://www.kaggle.com/competitions/bengaliai-speech/discussion/425496)、I've been curious what the Public LB Score the combination of public trained models would get.\n\nIn this notebook I used two public models from hugging faces:\n* `https://huggingface.co/ai4bharat/indicwav2vec_v1_bengali` for Wav2vec2CTC Model only\n* `https://huggingface.co/arijitx/wav2vec2-xls-r-300m-bengali` for Language Model\n\nI didn't trained these models using the competitaion data at all. I just want to know public models score as baseline.  \n\nSo we may get higher and higher score by fine-tuning on competition data.","metadata":{}},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-18T22:49:25.977465Z","iopub.execute_input":"2023-08-18T22:49:25.977851Z","iopub.status.idle":"2023-08-18T22:50:46.732186Z","shell.execute_reply.started":"2023-08-18T22:49:25.977808Z","shell.execute_reply":"2023-08-18T22:50:46.730789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:50:46.735020Z","iopub.execute_input":"2023-08-18T22:50:46.735784Z","iopub.status.idle":"2023-08-18T22:50:47.788173Z","shell.execute_reply.started":"2023-08-18T22:50:46.735745Z","shell.execute_reply":"2023-08-18T22:50:47.786652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import typing as tp\nfrom pathlib import Path\nfrom functools import partial\nfrom dataclasses import dataclass, field\n\nimport pandas as pd\nimport pyctcdecode\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport librosa\n\nimport pyctcdecode\nimport kenlm\nimport torch\nfrom transformers import Wav2Vec2Processor, Wav2Vec2ProcessorWithLM, Wav2Vec2ForCTC\nfrom bnunicodenormalizer import Normalizer\n\nimport cloudpickle as cpkl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-18T22:50:47.792468Z","iopub.execute_input":"2023-08-18T22:50:47.792815Z","iopub.status.idle":"2023-08-18T22:51:01.797459Z","shell.execute_reply.started":"2023-08-18T22:50:47.792786Z","shell.execute_reply":"2023-08-18T22:51:01.796079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nDATA = INPUT / \"bengaliai-speech\"\nTRAIN = DATA / \"train_mp3s\"\nTEST = DATA / \"test_mp3s\"\n\nSAMPLING_RATE = 16_000\nMODEL_PATH = INPUT / \"bengali-sr-download-public-trained-models/indicwav2vec_v1_bengali/\"\nLM_PATH = INPUT / \"bengali-sr-download-public-trained-models/wav2vec2-xls-r-300m-bengali/language_model/\"","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:01.800743Z","iopub.execute_input":"2023-08-18T22:51:01.801151Z","iopub.status.idle":"2023-08-18T22:51:01.809166Z","shell.execute_reply.started":"2023-08-18T22:51:01.801110Z","shell.execute_reply":"2023-08-18T22:51:01.808062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### load model, processor, decoder","metadata":{}},{"cell_type":"code","source":"model = Wav2Vec2ForCTC.from_pretrained(MODEL_PATH)\nprocessor = Wav2Vec2Processor.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:01.810621Z","iopub.execute_input":"2023-08-18T22:51:01.811986Z","iopub.status.idle":"2023-08-18T22:51:17.497894Z","shell.execute_reply.started":"2023-08-18T22:51:01.811920Z","shell.execute_reply":"2023-08-18T22:51:17.496470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict = processor.tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = pyctcdecode.build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    str(LM_PATH / \"5gram.bin\"),\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:17.499830Z","iopub.execute_input":"2023-08-18T22:51:17.500267Z","iopub.status.idle":"2023-08-18T22:51:55.041571Z","shell.execute_reply.started":"2023-08-18T22:51:17.500220Z","shell.execute_reply":"2023-08-18T22:51:55.040274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor_with_lm = Wav2Vec2ProcessorWithLM(\n    feature_extractor=processor.feature_extractor,\n    tokenizer=processor.tokenizer,\n    decoder=decoder\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:55.043560Z","iopub.execute_input":"2023-08-18T22:51:55.044218Z","iopub.status.idle":"2023-08-18T22:51:55.055363Z","shell.execute_reply.started":"2023-08-18T22:51:55.044180Z","shell.execute_reply":"2023-08-18T22:51:55.053983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prepare dataloader","metadata":{}},{"cell_type":"code","source":"class BengaliSRTestDataset(torch.utils.data.Dataset):\n    \n    def __init__(\n        self,\n        audio_paths: list[str],\n        sampling_rate: int\n    ):\n        self.audio_paths = audio_paths\n        self.sampling_rate = sampling_rate\n        \n    def __len__(self,):\n        return len(self.audio_paths)\n    \n    def __getitem__(self, index: int):\n        audio_path = self.audio_paths[index]\n        sr = self.sampling_rate\n        w = librosa.load(audio_path, sr=sr, mono=False)[0]\n        \n        return w","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:55.057786Z","iopub.execute_input":"2023-08-18T22:51:55.058394Z","iopub.status.idle":"2023-08-18T22:51:55.067906Z","shell.execute_reply.started":"2023-08-18T22:51:55.058359Z","shell.execute_reply":"2023-08-18T22:51:55.066913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(DATA / \"sample_submission.csv\", dtype={\"id\": str})\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:51:55.069264Z","iopub.execute_input":"2023-08-18T22:51:55.069968Z","iopub.status.idle":"2023-08-18T22:51:55.096631Z","shell.execute_reply.started":"2023-08-18T22:51:55.069911Z","shell.execute_reply":"2023-08-18T22:51:55.095645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_audio_paths = [str(TEST / f\"{aid}.mp3\") for aid in test[\"id\"].values]","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:59:47.157119Z","iopub.execute_input":"2023-08-18T22:59:47.157542Z","iopub.status.idle":"2023-08-18T22:59:47.166022Z","shell.execute_reply.started":"2023-08-18T22:59:47.157508Z","shell.execute_reply":"2023-08-18T22:59:47.163867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = BengaliSRTestDataset(\n    test_audio_paths, SAMPLING_RATE\n)\n\ncollate_func = partial(\n    processor_with_lm.feature_extractor,\n    return_tensors=\"pt\", sampling_rate=SAMPLING_RATE,\n    padding=True,\n)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_dataset, batch_size=8, shuffle=False,\n    num_workers=2, collate_fn=collate_func, drop_last=False,\n    pin_memory=True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:59:47.556880Z","iopub.execute_input":"2023-08-18T22:59:47.558023Z","iopub.status.idle":"2023-08-18T22:59:47.565230Z","shell.execute_reply.started":"2023-08-18T22:59:47.557977Z","shell.execute_reply":"2023-08-18T22:59:47.563994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"if not torch.cuda.is_available():\n    device = torch.device(\"cpu\")\nelse:\n    device = torch.device(\"cuda\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:59:49.886868Z","iopub.execute_input":"2023-08-18T22:59:49.888033Z","iopub.status.idle":"2023-08-18T22:59:49.896023Z","shell.execute_reply.started":"2023-08-18T22:59:49.887985Z","shell.execute_reply":"2023-08-18T22:59:49.893680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model.to(device)\nmodel = model.eval()\nmodel = model.half()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T22:59:51.238653Z","iopub.execute_input":"2023-08-18T22:59:51.239058Z","iopub.status.idle":"2023-08-18T22:59:51.264078Z","shell.execute_reply.started":"2023-08-18T22:59:51.239025Z","shell.execute_reply":"2023-08-18T22:59:51.263080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_sentence_list = []\n\nwith torch.no_grad():\n    for batch in tqdm(test_loader):\n        x = batch[\"input_values\"]\n        x = x.to(device, non_blocking=True)\n        with torch.cuda.amp.autocast(True):\n            y = model(x).logits\n        y = y.detach().cpu().numpy()\n        \n        for l in y:  \n            sentence = processor_with_lm.decode(l, beam_width=512).text\n            pred_sentence_list.append(sentence)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:01:12.924422Z","iopub.execute_input":"2023-08-18T23:01:12.924804Z","iopub.status.idle":"2023-08-18T23:01:15.279092Z","shell.execute_reply.started":"2023-08-18T23:01:12.924773Z","shell.execute_reply":"2023-08-18T23:01:15.277795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Submission","metadata":{}},{"cell_type":"code","source":"bnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:01:20.898346Z","iopub.execute_input":"2023-08-18T23:01:20.898715Z","iopub.status.idle":"2023-08-18T23:01:20.906309Z","shell.execute_reply.started":"2023-08-18T23:01:20.898686Z","shell.execute_reply":"2023-08-18T23:01:20.905235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [\n    postprocess(s) for s in tqdm(pred_sentence_list)]","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:01:21.590896Z","iopub.execute_input":"2023-08-18T23:01:21.593483Z","iopub.status.idle":"2023-08-18T23:01:21.636405Z","shell.execute_reply.started":"2023-08-18T23:01:21.593450Z","shell.execute_reply":"2023-08-18T23:01:21.635391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\n\nprint(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-18T23:01:23.630558Z","iopub.execute_input":"2023-08-18T23:01:23.631371Z","iopub.status.idle":"2023-08-18T23:01:23.644641Z","shell.execute_reply.started":"2023-08-18T23:01:23.631334Z","shell.execute_reply":"2023-08-18T23:01:23.643396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EOF","metadata":{}}]}