{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n!rm -r python-packages2 jiwer normalizer pyctcdecode pypikenlm","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip list | grep tokenizers\n# !pip list | grep transformers\n# !pip list | grep datasets\n# !pip list | grep torchaudio","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom glob import glob\nfrom tqdm.auto import tqdm\nfrom transformers import pipeline, AutoModelForCTC, AutoTokenizer, AutoFeatureExtractor, AutoConfig\nfrom pyctcdecode import BeamSearchDecoderCTC, build_ctcdecoder\n\npath_to_model = \"/kaggle/input/bengali-2023-0017-1\"\npath_to_data = \"/kaggle/input/bengaliai-speech\"\npath_to_kenlm_model = \"//kaggle/input/bengali-kenlm-language-models/5gram_16G.bin\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-16T18:55:43.572669Z","iopub.execute_input":"2023-10-16T18:55:43.574809Z","iopub.status.idle":"2023-10-16T18:55:53.234796Z","shell.execute_reply.started":"2023-10-16T18:55:43.574775Z","shell.execute_reply":"2023-10-16T18:55:53.233844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head -n 5 $path_to_model/trainer_state.json","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:55:53.236483Z","iopub.execute_input":"2023-10-16T18:55:53.236775Z","iopub.status.idle":"2023-10-16T18:55:54.246585Z","shell.execute_reply.started":"2023-10-16T18:55:53.236753Z","shell.execute_reply":"2023-10-16T18:55:54.245441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = AutoConfig.from_pretrained(path_to_model)\ntokenizer=AutoTokenizer.from_pretrained(path_to_model)\nmodel = AutoModelForCTC.from_pretrained(path_to_model)\nconfig.vocab_size","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:55:57.601836Z","iopub.execute_input":"2023-10-16T18:55:57.602240Z","iopub.status.idle":"2023-10-16T18:56:10.069698Z","shell.execute_reply.started":"2023-10-16T18:55:57.602209Z","shell.execute_reply":"2023-10-16T18:56:10.068651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfeature_extractor=AutoFeatureExtractor.from_pretrained(path_to_model)\nfeature_extractor._set_processor_class(\"Wav2Vec2ProcessorWithLM\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:56:10.071407Z","iopub.execute_input":"2023-10-16T18:56:10.072326Z","iopub.status.idle":"2023-10-16T18:56:10.086049Z","shell.execute_reply.started":"2023-10-16T18:56:10.072291Z","shell.execute_reply":"2023-10-16T18:56:10.084949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:56:10.087490Z","iopub.execute_input":"2023-10-16T18:56:10.088321Z","iopub.status.idle":"2023-10-16T18:56:10.094551Z","shell.execute_reply.started":"2023-10-16T18:56:10.088289Z","shell.execute_reply":"2023-10-16T18:56:10.093589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvocab_dict = tokenizer.get_vocab()\nsorted_vocab_dict = {k: v for k, v in sorted(vocab_dict.items(), key=lambda item: item[1])}\n\ndecoder = build_ctcdecoder(\n    list(sorted_vocab_dict.keys()),\n    path_to_kenlm_model,\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:56:10.096393Z","iopub.execute_input":"2023-10-16T18:56:10.097192Z","iopub.status.idle":"2023-10-16T18:58:13.614980Z","shell.execute_reply.started":"2023-10-16T18:56:10.097159Z","shell.execute_reply":"2023-10-16T18:58:13.612955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_dict","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:13.619236Z","iopub.execute_input":"2023-10-16T18:58:13.619625Z","iopub.status.idle":"2023-10-16T18:58:13.630116Z","shell.execute_reply.started":"2023-10-16T18:58:13.619590Z","shell.execute_reply":"2023-10-16T18:58:13.629015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npipe = pipeline(\n    \"automatic-speech-recognition\",\n    model=model,\n    tokenizer=tokenizer,\n    feature_extractor=feature_extractor,\n    chunk_length_s=7,\n    decoder=decoder,\n    decoder_kwargs={\"beam_width\": 2048},\n    device=0)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:13.633511Z","iopub.execute_input":"2023-10-16T18:58:13.633764Z","iopub.status.idle":"2023-10-16T18:58:20.305410Z","shell.execute_reply.started":"2023-10-16T18:58:13.633742Z","shell.execute_reply":"2023-10-16T18:58:20.304386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Processing","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(f\"{path_to_data}/sample_submission.csv\")\ndf[\"audio_path\"] = df.id.apply(lambda x: f\"{path_to_data}/test_mp3s/{x}.mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:20.308392Z","iopub.execute_input":"2023-10-16T18:58:20.308918Z","iopub.status.idle":"2023-10-16T18:58:20.357129Z","shell.execute_reply.started":"2023-10-16T18:58:20.308865Z","shell.execute_reply":"2023-10-16T18:58:20.356289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"sentence\"] = [\n    out[\"text\"]\n    for out in tqdm(\n        pipe(df.audio_path.to_list(), batch_size=8), total=len(df))\n]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:20.358305Z","iopub.execute_input":"2023-10-16T18:58:20.360119Z","iopub.status.idle":"2023-10-16T18:58:27.873840Z","shell.execute_reply.started":"2023-10-16T18:58:20.360087Z","shell.execute_reply":"2023-10-16T18:58:27.872780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Postprocess\n\nhttps://github.com/xashru/punctuation-restoration/blob/master/src/inference.py","metadata":{}},{"cell_type":"markdown","source":"### Add punctuation model","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom transformers import XLMRobertaModel, XLMRobertaConfig, XLMRobertaTokenizer\n\n\npunctuation_dict = {'O': 0, 'COMMA': 1, 'PERIOD': 2, 'QUESTION': 3}\n\nclass DeepPunctuation(nn.Module):\n    def __init__(self, pretrained_model, freeze_bert=False, lstm_dim=-1):\n        super(DeepPunctuation, self).__init__()\n        self.output_dim = len(punctuation_dict)\n        \n        config = XLMRobertaConfig.from_pretrained(pretrained_model)\n        # self.bert_layer = XLMRobertaModel.from_pretrained(pretrained_model)\n        self.bert_layer = XLMRobertaModel(config)\n        # Freeze bert layers\n        if freeze_bert:\n            for p in self.bert_layer.parameters():\n                p.requires_grad = False\n        bert_dim = 1024\n        if lstm_dim == -1:\n            hidden_size = bert_dim\n        else:\n            hidden_size = lstm_dim\n        self.lstm = nn.LSTM(input_size=bert_dim, hidden_size=hidden_size, num_layers=1, bidirectional=True)\n        self.linear = nn.Linear(in_features=hidden_size*2, out_features=len(punctuation_dict))\n\n    def forward(self, x, attn_masks):\n        if len(x.shape) == 1:\n            x = x.view(1, x.shape[0])  # add dummy batch for single sample\n        # (B, N, E) -> (B, N, E)\n        x = self.bert_layer(x, attention_mask=attn_masks)[0]\n        # (B, N, E) -> (N, B, E)\n        x = torch.transpose(x, 0, 1)\n        x, (_, _) = self.lstm(x)\n        # (N, B, E) -> (B, N, E)\n        x = torch.transpose(x, 0, 1)\n        x = self.linear(x)\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:27.875321Z","iopub.execute_input":"2023-10-16T18:58:27.876111Z","iopub.status.idle":"2023-10-16T18:58:27.909230Z","shell.execute_reply.started":"2023-10-16T18:58:27.876077Z","shell.execute_reply":"2023-10-16T18:58:27.908410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"puntuation_model_path = \"/kaggle/input/xashru-punctuation-restoration\"","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:27.910392Z","iopub.execute_input":"2023-10-16T18:58:27.910723Z","iopub.status.idle":"2023-10-16T18:58:27.916499Z","shell.execute_reply.started":"2023-10-16T18:58:27.910694Z","shell.execute_reply":"2023-10-16T18:58:27.914222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"here\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:27.918562Z","iopub.execute_input":"2023-10-16T18:58:27.919032Z","iopub.status.idle":"2023-10-16T18:58:27.925213Z","shell.execute_reply.started":"2023-10-16T18:58:27.919000Z","shell.execute_reply":"2023-10-16T18:58:27.924175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = XLMRobertaTokenizer.from_pretrained(puntuation_model_path)\nroberta_token_ids = {\n    'START_SEQ': 0,\n    'PAD': 1,\n    'END_SEQ': 2,\n    'UNK': 3\n}\nsequence_len = 256","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:27.927486Z","iopub.execute_input":"2023-10-16T18:58:27.929090Z","iopub.status.idle":"2023-10-16T18:58:28.592022Z","shell.execute_reply.started":"2023-10-16T18:58:27.929057Z","shell.execute_reply":"2023-10-16T18:58:28.591028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndeep_punctuation = DeepPunctuation(puntuation_model_path)\ndeep_punctuation.load_state_dict(torch.load(os.path.join(puntuation_model_path, \"xlm-roberta-large-bn.pt\")))\ndeep_punctuation.eval()\ndeep_punctuation.to(device)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-16T18:58:28.594757Z","iopub.execute_input":"2023-10-16T18:58:28.595387Z","iopub.status.idle":"2023-10-16T18:58:55.995826Z","shell.execute_reply.started":"2023-10-16T18:58:28.595352Z","shell.execute_reply":"2023-10-16T18:58:55.994931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef predict_punctuations(text):\n    text = re.sub(r\"[,:\\-–.!;?]\", '', text)\n    words_original_case = text.split()\n    words = text.lower().split()\n\n    word_pos = 0\n    result = \"\"\n    decode_idx = 0\n    punctuation_map = {0: '', 1: ',', 2: '।', 3: '?'}\n\n    while word_pos < len(words):\n        x = [roberta_token_ids['START_SEQ']]\n        y_mask = [0]\n\n        while len(x) < sequence_len and word_pos < len(words):\n            tokens = tokenizer.tokenize(words[word_pos])\n            if len(tokens) + len(x) >= sequence_len:\n                break\n            else:\n                for i in range(len(tokens) - 1):\n                    x.append(tokenizer.convert_tokens_to_ids(tokens[i]))\n                    y_mask.append(0)\n                x.append(tokenizer.convert_tokens_to_ids(tokens[-1]))\n                y_mask.append(1)\n                word_pos += 1\n        x.append(roberta_token_ids['END_SEQ'])\n        y_mask.append(0)\n        if len(x) < sequence_len:\n            x = x + [roberta_token_ids['PAD'] for _ in range(sequence_len - len(x))]\n            y_mask = y_mask + [0 for _ in range(sequence_len - len(y_mask))]\n        attn_mask = [1 if token != roberta_token_ids['PAD'] else 0 for token in x]\n\n        x = torch.tensor(x).reshape(1,-1)\n        y_mask = torch.tensor(y_mask)\n        attn_mask = torch.tensor(attn_mask).reshape(1,-1)\n        x, attn_mask, y_mask = x.to(device), attn_mask.to(device), y_mask.to(device)\n\n        with torch.no_grad():\n            y_predict = deep_punctuation(x, attn_mask)\n            y_predict = y_predict.view(-1, y_predict.shape[2])\n            y_predict = torch.argmax(y_predict, dim=1).view(-1)\n        for i in range(y_mask.shape[0]):\n            if y_mask[i] == 1:\n                result += words_original_case[decode_idx] + punctuation_map[y_predict[i].item()] + ' '\n                decode_idx += 1\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:55.997328Z","iopub.execute_input":"2023-10-16T18:58:55.997918Z","iopub.status.idle":"2023-10-16T18:58:56.009040Z","shell.execute_reply.started":"2023-10-16T18:58:55.997857Z","shell.execute_reply":"2023-10-16T18:58:56.008043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_punctuations(\"কে আমি শিশুসুলভ বোকামি মিশ্রণ হিসাবে বর্ণনা করব তিনি প্রায়ই সময়ে ডিনারে এক হাজার ডলার খরচ করতেন তিনি তার বন্ধুদের সামনে সম্পদের শো অফ করতেন তিনি প্রকাশ্যে এবং উচ্চ স্বরে তার সম্পদ নিয়ে বড়াই করতেন প্রায় সময় মাতাল অবস্থায় হোটেল থেকে বের হতেন একদিন তিনি আমার এক সহকর্মীর হাতে নগদ কয়েক হাজার ডলার দিয়ে বলেন রাস্তার পাশে গয়নার দোকানে যাও এবং আমার জন্য কয়েক ডলার মূল্যের এক হাজারটি সোনার কয়েন কিনে আনো এক ঘণ্টা পর হাতে সোনার কয়েন নিয়ে প্রকৌশলী এবং তার বন্ধুরা প্রশান্ত মহা।\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:58:56.010677Z","iopub.execute_input":"2023-10-16T18:58:56.011530Z","iopub.status.idle":"2023-10-16T18:58:56.289463Z","shell.execute_reply.started":"2023-10-16T18:58:56.011497Z","shell.execute_reply":"2023-10-16T18:58:56.288370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"sentence\"] = [predict_punctuations(sentence) for sentence in tqdm(df.sentence)]","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:38.613722Z","iopub.execute_input":"2023-10-16T18:59:38.614094Z","iopub.status.idle":"2023-10-16T18:59:38.839163Z","shell.execute_reply.started":"2023-10-16T18:59:38.614066Z","shell.execute_reply":"2023-10-16T18:59:38.838203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# why do I have this...\n# df[\"sentence\"] = df.sentence.str.strip().apply(lambda x: x if x else \"।\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for special_token in tokenizer.all_special_tokens:\n    df.sentence = df.sentence.str.replace(special_token, \"\")","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:41.705454Z","iopub.execute_input":"2023-10-16T18:59:41.705795Z","iopub.status.idle":"2023-10-16T18:59:41.714036Z","shell.execute_reply.started":"2023-10-16T18:59:41.705768Z","shell.execute_reply":"2023-10-16T18:59:41.713065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:41.921394Z","iopub.execute_input":"2023-10-16T18:59:41.921702Z","iopub.status.idle":"2023-10-16T18:59:41.931572Z","shell.execute_reply.started":"2023-10-16T18:59:41.921676Z","shell.execute_reply":"2023-10-16T18:59:41.930476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"sentence_orig\"] = df[\"sentence\"]\ndf[\"sentence\"] = df[\"sentence\"].map(postprocess)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:43.493357Z","iopub.execute_input":"2023-10-16T18:59:43.494075Z","iopub.status.idle":"2023-10-16T18:59:43.516347Z","shell.execute_reply.started":"2023-10-16T18:59:43.494036Z","shell.execute_reply":"2023-10-16T18:59:43.515579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[[\"id\", \"sentence\"]].to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:44.296218Z","iopub.execute_input":"2023-10-16T18:59:44.296902Z","iopub.status.idle":"2023-10-16T18:59:44.315680Z","shell.execute_reply.started":"2023-10-16T18:59:44.296848Z","shell.execute_reply":"2023-10-16T18:59:44.314730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -n 4 submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-10-16T18:59:45.076454Z","iopub.execute_input":"2023-10-16T18:59:45.076790Z","iopub.status.idle":"2023-10-16T18:59:46.159007Z","shell.execute_reply.started":"2023-10-16T18:59:45.076762Z","shell.execute_reply":"2023-10-16T18:59:46.157841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}