{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"wav input has been cast to required mp3 keys using the following code\n  \n    df.path = df.path.apply(lambda x: post_process_keys(x))","metadata":{}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:28:50.055783Z","iopub.execute_input":"2023-07-21T11:28:50.056303Z","iopub.status.idle":"2023-07-21T11:28:51.262815Z","shell.execute_reply.started":"2023-07-21T11:28:50.056171Z","shell.execute_reply":"2023-07-21T11:28:51.261165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:28:51.268191Z","iopub.execute_input":"2023-07-21T11:28:51.268558Z","iopub.status.idle":"2023-07-21T11:30:30.934060Z","shell.execute_reply.started":"2023-07-21T11:28:51.268522Z","shell.execute_reply":"2023-07-21T11:30:30.932504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport pandas as pd\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:30:30.936397Z","iopub.execute_input":"2023-07-21T11:30:30.936841Z","iopub.status.idle":"2023-07-21T11:30:42.095360Z","shell.execute_reply.started":"2023-07-21T11:30:30.936793Z","shell.execute_reply":"2023-07-21T11:30:42.094210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 16\nTEST_DIRECTORY = '../input/test-wav-files-dl-sprint/test_files_wav'","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:30:42.098810Z","iopub.execute_input":"2023-07-21T11:30:42.100062Z","iopub.status.idle":"2023-07-21T11:30:42.110841Z","shell.execute_reply.started":"2023-07-21T11:30:42.100019Z","shell.execute_reply":"2023-07-21T11:30:42.107531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass CFG:\n    my_model_name = '../input/yellowking-dlsprint-model/YellowKing_model'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:30:42.113331Z","iopub.execute_input":"2023-07-21T11:30:42.114045Z","iopub.status.idle":"2023-07-21T11:30:42.128208Z","shell.execute_reply.started":"2023-07-21T11:30:42.114002Z","shell.execute_reply":"2023-07-21T11:30:42.127089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:30:42.129929Z","iopub.execute_input":"2023-07-21T11:30:42.130408Z","iopub.status.idle":"2023-07-21T11:32:12.023216Z","shell.execute_reply.started":"2023-07-21T11:30:42.130367Z","shell.execute_reply":"2023-07-21T11:32:12.022133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nmy_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:12.024876Z","iopub.execute_input":"2023-07-21T11:32:12.025295Z","iopub.status.idle":"2023-07-21T11:32:30.714671Z","shell.execute_reply.started":"2023-07-21T11:32:12.025246Z","shell.execute_reply":"2023-07-21T11:32:30.713617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Following Sample Submission:**","metadata":{}},{"cell_type":"code","source":"def infer(audio_path):\n    speech, sr = librosa.load(audio_path, sr=processor.feature_extractor.sampling_rate)\n\n    my_LM_prediction = my_asrLM(\n                speech, chunk_length_s=112, stride_length_s=None\n            )\n\n    return my_LM_prediction['text']\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.718198Z","iopub.execute_input":"2023-07-21T11:32:30.718513Z","iopub.status.idle":"2023-07-21T11:32:30.725332Z","shell.execute_reply.started":"2023-07-21T11:32:30.718483Z","shell.execute_reply":"2023-07-21T11:32:30.724248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batch_infer(audio_paths, batch_size=BATCH_SIZE):\n    '''\n    infers on a batch of audio\n    args:\n      audio_paths  : list of path to audio files <list of string>\n    returns:\n      bangla predicted texts <list of string>\n    '''\n    results = []\n    for path in audio_paths:\n        results.append(infer(path))\n    \n    return results","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.727376Z","iopub.execute_input":"2023-07-21T11:32:30.728039Z","iopub.status.idle":"2023-07-21T11:32:30.739810Z","shell.execute_reply.started":"2023-07-21T11:32:30.727995Z","shell.execute_reply":"2023-07-21T11:32:30.738839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer \n\n\nbnorm = Normalizer()\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.743995Z","iopub.execute_input":"2023-07-21T11:32:30.744443Z","iopub.status.idle":"2023-07-21T11:32:30.765407Z","shell.execute_reply.started":"2023-07-21T11:32:30.744398Z","shell.execute_reply":"2023-07-21T11:32:30.764080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_process_keys(str):\n    return str.replace(\"../input/test-wav-files-dl-sprint/test_files_wav/\",\"\").replace(\".wav\",\".mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.767169Z","iopub.execute_input":"2023-07-21T11:32:30.767557Z","iopub.status.idle":"2023-07-21T11:32:30.772836Z","shell.execute_reply.started":"2023-07-21T11:32:30.767521Z","shell.execute_reply":"2023-07-21T11:32:30.771647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def directory_infer(audio_dir):\n    '''\n    infers on a directory that contains audio files\n    args:\n      audio_dir  : directory that contains some audio files <string>\n    returns:\n      a dataframe that contains 2 columns:\n        * path <string>\n        * sentence <string>\n    '''\n    # list all audio files\n\n    audio_paths=[audio_path for audio_path in tqdm(glob(os.path.join(audio_dir,\"*.*\")))]\n    sentences=[]\n    for idx in tqdm(range(0,len(audio_paths),BATCH_SIZE)):\n        batch_paths=audio_paths[idx:idx+BATCH_SIZE]\n        sentences+=batch_infer(batch_paths)\n        \n    df= pd.DataFrame({\"path\":audio_paths,\"sentence\":sentences})\n    df.sentence= df.sentence.apply(lambda x:normalize(x))\n    df.sentence= df.sentence.apply(lambda x:dari(x))\n    df.path = df.path.apply(lambda x: post_process_keys(x))\n    \n    return df ","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.774774Z","iopub.execute_input":"2023-07-21T11:32:30.775152Z","iopub.status.idle":"2023-07-21T11:32:30.785697Z","shell.execute_reply.started":"2023-07-21T11:32:30.775098Z","shell.execute_reply":"2023-07-21T11:32:30.784344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = directory_infer(TEST_DIRECTORY)\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T11:32:30.787367Z","iopub.execute_input":"2023-07-21T11:32:30.787871Z","iopub.status.idle":"2023-07-21T12:14:32.549748Z","shell.execute_reply.started":"2023-07-21T11:32:30.787822Z","shell.execute_reply":"2023-07-21T12:14:32.548655Z"},"trusted":true},"execution_count":null,"outputs":[]}]}