{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Introduction\n\nThe purpose of this notebook is checking if words are recognized good by YellowKing DL Sprint 2022 model.\n","metadata":{}},{"cell_type":"code","source":"import time\nfrom random import *\nimport numpy as np\nimport pandas as pd\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GROUP_NO = 20\nGROUP_SIZE = 50\nGROUP_MAX = 0\n\nUSE_GPU = False\nBATCH_SIZE = 16\nRUN_TIME = 60 * 60 * 8\nSTART_TIME = time.time()\nDATA_FILE = '/kaggle/input/bengaliai-speech/train.csv'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_BN = [\n    '__W__ মানে কি?',\n    'আপনি কিভাবে __W__ বানান করবেন?',\n    'আপনি কিভাবে __W__ লিখবেন?',\n    'আপনি কিভাবে __W__ ব্যবহার করবেন?'\n]\nSAMPLE_EN = [\n    'What does __W__ mean?',\n    'How do you spell __W__?',\n    'How do you write __W__?',\n    'How do you use __W__?'\n]","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = pd.read_csv(DATA_FILE)\nsidx = (GROUP_NO - 1) * GROUP_SIZE\neidx = (GROUP_NO) * GROUP_SIZE\nif eidx >= len(data_df):\n    eidx = len(data_df)\ndata_df = data_df[sidx:eidx]\nif GROUP_MAX > 0:\n    vmax = GROUP_MAX\n    if vmax < len(data_df):\n        data_df = data_df[0:vmax]\ndata_df","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!apt-get update\n!apt-get install -y php-cli\n!apt-get install -y curl\n!apt-get install -y php7.4-curl","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:52:15.635854Z","iopub.execute_input":"2023-07-18T06:52:15.636244Z","iopub.status.idle":"2023-07-18T06:52:16.898344Z","shell.execute_reply.started":"2023-07-18T06:52:15.636129Z","shell.execute_reply":"2023-07-18T06:52:16.896805Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:52:16.905689Z","iopub.execute_input":"2023-07-18T06:52:16.908384Z","iopub.status.idle":"2023-07-18T06:53:58.122905Z","shell.execute_reply.started":"2023-07-18T06:52:16.908338Z","shell.execute_reply":"2023-07-18T06:53:58.121393Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/working/php\n!mkdir -p /kaggle/working/mp3","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile /kaggle/working/php/soundoftext.php\n<?php\n\n//print('argc = ' . $argc);\n//print('argv = ');\n//print_r($argv);\n\n$filename = '';\n$text = '';\nif ($argc > 2) {\n    $filename = $argv[1];\n    $text = $argv[2];\n}\n\nif (strlen($filename) > 0 && strlen($text) > 0) {\n    $voice = 'bn-IN';\n    $inputObj = array('engine' => 'Google', 'data' => array('text' => $text, 'voice' => $voice));\n    $inputJson = json_encode($inputObj);\n    $rs = g_curl_post_json('https://api.soundoftext.com/sounds', $inputJson, $msg); \n    if ($rs !== false) {\n        $outputObj = json_decode($rs, true);\n        if ($outputObj['success']) {\n            $id = $outputObj['id'];\n            $max = 5;\n            $count = 0;\n            while ($count < $max) {\n                $rs = g_curl('https://api.soundoftext.com/sounds/' . $id, $msg);\n                $outputObj = json_decode($rs, true);\n                if ($outputObj['status'] == 'Done') {\n                    $url = $outputObj['location'];\n                    $cmd = 'wget -O \"' . $filename . '\" \"' . $url . '\"';\n                    shell_exec($cmd);\n                    exit();\n                }\n                sleep(2);\n                $count++;\n            }\n        }\n    }    \n}\n\nfunction g_curl($url, &$msg) {\n    $ch = curl_init($url);\n\n    curl_setopt($ch, CURLOPT_CONNECTTIMEOUT, 30);\n    curl_setopt($ch, CURLOPT_TIMEOUT, 60);\n\n    curl_setopt($ch, CURLOPT_RETURNTRANSFER, 1);\n\t\n\tcurl_setopt($ch, CURLOPT_USERAGENT, \"Mozilla/5.0 (Windows; U; Windows NT 5.1; en-US; rv:1.8.1.1) Gecko/20061204 Firefox/2.0.0.1\");\n    curl_setopt($ch, CURLOPT_REFERER, \"\");\n    curl_setopt($ch, CURLOPT_FOLLOWLOCATION, TRUE);\t\n\n    $rs = curl_exec($ch);\n    if ($rs === false) {\n        $msg = '[' . curl_getinfo($ch, CURLINFO_HTTP_CODE) . '] ' . curl_error($ch);\n    }\n\t\n    curl_close($ch);\n\n    return $rs;\n}\n\nfunction g_curl_post_json($url, $json, &$msg) {\n    $ch = curl_init($url);\n\n    curl_setopt($ch, CURLOPT_CONNECTTIMEOUT, 30);\n    curl_setopt($ch, CURLOPT_TIMEOUT, 60);\n\n    curl_setopt($ch, CURLOPT_RETURNTRANSFER, 1);\n\t\n\tcurl_setopt($ch, CURLOPT_USERAGENT, \"Mozilla/5.0 (Windows; U; Windows NT 5.1; en-US; rv:1.8.1.1) Gecko/20061204 Firefox/2.0.0.1\");\n    curl_setopt($ch, CURLOPT_REFERER, \"\");\n    curl_setopt($ch, CURLOPT_FOLLOWLOCATION, TRUE);\t\n\n    curl_setopt($ch, CURLOPT_CUSTOMREQUEST, \"POST\");\n    curl_setopt($ch, CURLOPT_POSTFIELDS, $json);                                                                  \n    curl_setopt($ch, CURLOPT_HTTPHEADER, array(                                                                          \n      'Content-Type: application/json',                                                                                \n      'Content-Length: ' . strlen($json))                                                                       \n    );     \n\n    $rs = curl_exec($ch);\n    if ($rs === false) {\n        $msg = '[' . curl_getinfo($ch, CURLINFO_HTTP_CODE) . '] ' . curl_error($ch);\n    }\n\t\n    curl_close($ch);\n\n    return $rs;\n}\n\n?>","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\nimport lightgbm as lgb\nimport pickle \nimport regex\nimport jiwer\nimport jiwer.transforms as tr\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:53:58.124827Z","iopub.execute_input":"2023-07-18T06:53:58.125322Z","iopub.status.idle":"2023-07-18T06:54:08.8931Z","shell.execute_reply.started":"2023-07-18T06:53:58.12529Z","shell.execute_reply":"2023-07-18T06:54:08.891902Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tts(text, filename):\n    cmd = 'php /kaggle/working/php/soundoftext.php \"' + filename + '\" \"' + text + '\"';\n    os.system(cmd)\n    if os.path.exists(filename):\n        return True\n    else:\n        return False\n    \nfrom bnunicodenormalizer import Normalizer \nbnorm = Normalizer()\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def is_bengali(word):\n    return bool(regex.fullmatch(r'\\P{L}*\\p{Bengali}+(?:\\P{L}+\\p{Bengali}+)*\\P{L}*', word))\n\ndef split_words(sen):\n    words = []\n    trans = tr.Compose(\n        [\n            tr.RemoveMultipleSpaces(),\n            tr.Strip(),\n            tr.ReduceToListOfListOfWords(),\n        ]\n    )\n    oodc = '।abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789,.+×÷=/_<>[]!@#$%^&*()-\\'\":;,?`~\\|{}€£¥₩'\n    flds = trans(sen)[0]\n    for wi in range(len(flds)):\n        w = flds[wi]\n        nw = ''\n        has_oodc = False\n        for ci in range(len(w)):\n            c = w[ci:ci+1]\n            if c in oodc:\n                has_oodc = True\n                continue\n            nw += c\n        if len(nw) == 0:\n            continue\n        if not has_oodc:\n            if w != nw:\n                continue\n        w = nw\n        if not is_bengali(w):\n            continue\n        if len(w) > 30:\n            continue\n        words.append(w)\n    \n    return words\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    my_model_name = '../input/yellowking-dlsprint-model/YellowKing_model'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:54:08.909265Z","iopub.execute_input":"2023-07-18T06:54:08.909908Z","iopub.status.idle":"2023-07-18T06:54:08.923912Z","shell.execute_reply.started":"2023-07-18T06:54:08.909868Z","shell.execute_reply":"2023-07-18T06:54:08.922803Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:54:08.92559Z","iopub.execute_input":"2023-07-18T06:54:08.925948Z","iopub.status.idle":"2023-07-18T06:55:42.607866Z","shell.execute_reply.started":"2023-07-18T06:54:08.925912Z","shell.execute_reply":"2023-07-18T06:55:42.606792Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not USE_GPU:\n    my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=-1)\n    device = 'cpu'","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:55:42.609565Z","iopub.execute_input":"2023-07-18T06:55:42.609981Z","iopub.status.idle":"2023-07-18T06:56:02.015371Z","shell.execute_reply.started":"2023-07-18T06:55:42.609941Z","shell.execute_reply":"2023-07-18T06:56:02.014235Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_GPU:\n    my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)\n    device = 'cuda:0'","metadata":{"execution":{"iopub.status.busy":"2023-07-18T06:55:42.609565Z","iopub.execute_input":"2023-07-18T06:55:42.609981Z","iopub.status.idle":"2023-07-18T06:56:02.015371Z","shell.execute_reply.started":"2023-07-18T06:55:42.609941Z","shell.execute_reply":"2023-07-18T06:56:02.014235Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def match(sen1, sen2, wd):\n    sen1b = \" \".join(split_words(sen1))\n    sen2b = \" \".join(split_words(sen2))\n    if sen1b == sen2b:\n        return 1.0\n    else:\n        if wd in sen1 and wd in sen2:\n            return 0.75\n        else:\n            return 0.0","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = []\nsize = len(data_df)\nfor ri in range(len(data_df)):\n    if ri % 10 == 0:\n        print('L: ' + str(ri + 1) + ' / ' + str(size))\n        \n    sen = data_df['sentence'].iloc[ri]\n    fid = data_df['id'].iloc[ri]\n    words = split_words(sen)\n    for wi in range(len(words)):\n        rw = {'id': fid, 'wi': wi, 'word': words[wi], 'good': 1}\n        good = 1\n        for si in range(len(SAMPLE_BN)):\n            sp = SAMPLE_BN[si]\n            sp = sp.replace('__W__', words[wi])\n            rw['s' + str(si + 1) + '_text'] = sp\n            fn = '/kaggle/working/mp3/' + str(fid) + '_' + str(wi) + '_' + str(si) + '.mp3'\n            if not tts(sp, fn):\n                rw['s' + str(si + 1) + '_error'] = 'Cannot convert to mp3!'\n                rw['s' + str(si + 1) + '_predict'] = '_'\n            else:\n                rw['s' + str(si + 1) + '_error'] = 'Success'\n                speech, sr = librosa.load(fn, sr=processor.feature_extractor.sampling_rate)\n                p = my_asrLM(speech)\n                p = p['text']\n                p = normalize(p)\n                if p[-1] == \"।\":\n                    p = p[0:len(p) - 1]\n                rw['s' + str(si + 1) + '_predict'] = p\n                g = match(p, sp, words[wi])\n                rw['s' + str(si + 1) + '_good'] = g\n                if g <= 0.5:\n                    good = 0\n        rw['good'] = good\n        rows.append(rw)\n        \ndf = pd.DataFrame(rows)\ndf.to_csv('data-' + str(GROUP_NO) + '.csv', index=False)    ","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"_kg_hide-input":true},"execution_count":null,"outputs":[]}]}