{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### This notebook is a fork of [this notebook](https://www.kaggle.com/code/sameen53/yellowking-dlsprint-inference).\n\nThey won the [DL Sprint 2022](https://www.kaggle.com/competitions/dlsprint) Competition which was a mini version of this competition. They finetuned the \"facebook/wav2vec2-large-xlsr-53 model\" and used an N-gram language model.\n\nLink to their [training notebook](https://www.kaggle.com/code/sameen53/yellowking-dlsprint-training). ","metadata":{"execution":{"iopub.status.busy":"2022-08-31T14:41:32.385678Z","iopub.execute_input":"2022-08-31T14:41:32.386121Z","iopub.status.idle":"2022-08-31T14:41:33.84764Z","shell.execute_reply.started":"2022-08-31T14:41:32.386036Z","shell.execute_reply":"2022-08-31T14:41:33.84594Z"}}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport pandas as pd\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 1\nTEST_DIRECTORY = '/kaggle/input/bengaliai-speech/test_mp3s'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass CFG:\n    my_model_name = '../input/yellowking-dlsprint-model/YellowKing_model'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Following Sample Submission:**","metadata":{}},{"cell_type":"code","source":"def infer(audio_path):\n    speech, sr = librosa.load(audio_path, sr=processor.feature_extractor.sampling_rate)\n\n    my_LM_prediction = my_asrLM(\n                speech\n            )\n\n    return my_LM_prediction['text']\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batch_infer(audio_paths, batch_size=BATCH_SIZE):\n    '''\n    infers on a batch of audio\n    args:\n      audio_paths  : list of path to audio files <list of string>\n    returns:\n      bangla predicted texts <list of string>\n    '''\n    results = []\n    for path in audio_paths:\n        pred = \"\"\n        try:\n            pred = infer(path)\n        except:\n            pred = \"এ\"\n        if len(pred)==0:\n            pred = \"এ\"\n        results.append(pred)\n    \n    return results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer \n\n\nbnorm = Normalizer()\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_process_keys(str):\n    return str.replace(\"../input/test-wav-files-dl-sprint/test_files_wav/\",\"\").replace(\".wav\",\".mp3\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def directory_infer(audio_dir):\n    '''\n    infers on a directory that contains audio files\n    args:\n      audio_dir  : directory that contains some audio files <string>\n    returns:\n      a dataframe that contains 2 columns:\n        * path <string>\n        * sentence <string>\n    '''\n    # list all audio files\n\n    audio_paths=[audio_path for audio_path in tqdm(glob(os.path.join(audio_dir,\"*.*\")))]\n    files = os.listdir(\"/kaggle/input/bengaliai-speech/test_mp3s\")\n    paths = []\n    for i in files:\n        paths.append(i.split(\".\")[0])\n    sentences=[]\n    for idx in tqdm(range(0,len(audio_paths),BATCH_SIZE)):\n        batch_paths=audio_paths[idx:idx+BATCH_SIZE]\n        sentences+=batch_infer(batch_paths)\n        \n    df= pd.DataFrame({\"id\":paths,\"sentence\":sentences})\n    df.sentence= df.sentence.apply(lambda x:normalize(x))\n    #df.sentence= df.sentence.apply(lambda x:dari(x))\n    df['id'] = df['id'].apply(lambda x: post_process_keys(x))\n    \n    return df ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = directory_infer(TEST_DIRECTORY)\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check(sentence):\n    if len(sentence)==0:\n        return '।'\n    return sentence","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.sentence = submission.sentence.apply(lambda x:check(x))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}