{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview\n\n```\nThis notebook provides a baseline to train and infer wav2vec2 model on this competition data.\nWe train with 30k data and use 5k samples for validation set and trained them for 10 epochs. This achieves LB score 0.49.\nWe preprocess data and then train the model with them in a similar manner to the previous competition winners. \n```\n\nVersion history :\n\n> * **Version 1** - Training for first 5 epochs (LB 0.502)\n> * **Version 2** - Training another 5 epcohs (LB 0.49)\n> * **Version 3** - Inference ( LB 0.49)\n\n\n\n\n> [Data preprocessing Notebook](https://www.kaggle.com/code/mbmmurad/prepare-dataset-for-wav2vec2)\n\n> [Model weights](https://www.kaggle.com/datasets/mbmmurad/wav2vec2-demo) \n\nPlease upvote if you find this helpful and fork it,train further and explore! ","metadata":{"execution":{"iopub.execute_input":"2022-08-31T14:41:32.386121Z","iopub.status.busy":"2022-08-31T14:41:32.385678Z","iopub.status.idle":"2022-08-31T14:41:33.84764Z","shell.execute_reply":"2022-08-31T14:41:33.84594Z","shell.execute_reply.started":"2022-08-31T14:41:32.386036Z"},"papermill":{"duration":0.008937,"end_time":"2023-08-13T22:11:00.242510","exception":false,"start_time":"2023-08-13T22:11:00.233573","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Install Dependencies","metadata":{"papermill":{"duration":0.007684,"end_time":"2023-08-13T22:11:00.258196","exception":false,"start_time":"2023-08-13T22:11:00.250512","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!cp -r ../input/python-packages2 ./","metadata":{"papermill":{"duration":1.76376,"end_time":"2023-08-13T22:11:02.029995","exception":false,"start_time":"2023-08-13T22:11:00.266235","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:19:08.650062Z","iopub.execute_input":"2023-08-14T20:19:08.650419Z","iopub.status.idle":"2023-08-14T20:19:10.352470Z","shell.execute_reply.started":"2023-08-14T20:19:08.650390Z","shell.execute_reply":"2023-08-14T20:19:10.351185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture ts\n\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index\n!tar xvfz ./python-packages2/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index\n!tar xvfz ./python-packages2/pyctcdecode.tgz\n!pip install ./pyctcdecode/attrs-22.1.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/exceptiongroup-1.0.0rc9-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/hypothesis-6.54.4-py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/numpy-1.21.6-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pygtrie-2.5.0.tar.gz -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/sortedcontainers-2.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n!pip install ./pyctcdecode/pyctcdecode-0.4.0-py2.py3-none-any.whl -f ./ --no-index --no-deps\n\n!tar xvfz ./python-packages2/pypikenlm.tgz\n!pip install ./pypikenlm/pypi-kenlm-0.1.20220713.tar.gz -f ./ --no-index --no-deps\n\n","metadata":{"papermill":{"duration":102.264774,"end_time":"2023-08-13T22:12:44.302517","exception":false,"start_time":"2023-08-13T22:11:02.037743","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:19:10.354925Z","iopub.execute_input":"2023-08-14T20:19:10.356138Z","iopub.status.idle":"2023-08-14T20:20:21.209945Z","shell.execute_reply.started":"2023-08-14T20:19:10.356097Z","shell.execute_reply":"2023-08-14T20:20:21.208629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom glob import glob\nfrom transformers import AutoFeatureExtractor, pipeline\nimport pandas as pd\nimport librosa\nimport IPython\nfrom datasets import load_metric\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nimport torch\nimport gc\nimport wave\nfrom scipy.io import wavfile\nimport scipy.signal as sps\nimport pyctcdecode\n\ntqdm.pandas()\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"papermill":{"duration":11.642061,"end_time":"2023-08-13T22:12:55.960259","exception":false,"start_time":"2023-08-13T22:12:44.318198","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:20:21.212563Z","iopub.execute_input":"2023-08-14T20:20:21.213298Z","iopub.status.idle":"2023-08-14T20:20:36.787482Z","shell.execute_reply.started":"2023-08-14T20:20:21.213256Z","shell.execute_reply":"2023-08-14T20:20:36.786466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load model","metadata":{"papermill":{"duration":0.01497,"end_time":"2023-08-13T22:12:55.989943","exception":false,"start_time":"2023-08-13T22:12:55.974973","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# CHANGE ACCORDINGLY\nBATCH_SIZE = 1\nTEST_DIRECTORY = '/kaggle/input/bengaliai-speech/test_mp3s'","metadata":{"papermill":{"duration":0.025771,"end_time":"2023-08-13T22:12:56.030088","exception":false,"start_time":"2023-08-13T22:12:56.004317","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:20:36.790146Z","iopub.execute_input":"2023-08-14T20:20:36.791010Z","iopub.status.idle":"2023-08-14T20:20:36.799189Z","shell.execute_reply.started":"2023-08-14T20:20:36.790974Z","shell.execute_reply":"2023-08-14T20:20:36.798242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    my_model_name = '/kaggle/input/wav2vec2-demo/train_demo'\n    processor_name = '../input/yellowking-dlsprint-model/YellowKing_processor'","metadata":{"papermill":{"duration":0.022536,"end_time":"2023-08-13T22:12:56.066532","exception":false,"start_time":"2023-08-13T22:12:56.043996","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:20:36.800472Z","iopub.execute_input":"2023-08-14T20:20:36.801076Z","iopub.status.idle":"2023-08-14T20:20:36.813257Z","shell.execute_reply.started":"2023-08-14T20:20:36.801041Z","shell.execute_reply":"2023-08-14T20:20:36.812072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2ProcessorWithLM\n\nprocessor = Wav2Vec2ProcessorWithLM.from_pretrained(CFG.processor_name)\n","metadata":{"papermill":{"duration":94.127841,"end_time":"2023-08-13T22:14:30.209159","exception":false,"start_time":"2023-08-13T22:12:56.081318","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:20:36.814672Z","iopub.execute_input":"2023-08-14T20:20:36.815358Z","iopub.status.idle":"2023-08-14T20:22:13.755725Z","shell.execute_reply.started":"2023-08-14T20:20:36.815323Z","shell.execute_reply":"2023-08-14T20:22:13.754664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()\ngc.collect()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-14T20:22:13.757399Z","iopub.execute_input":"2023-08-14T20:22:13.757789Z","iopub.status.idle":"2023-08-14T20:22:43.830072Z","shell.execute_reply.started":"2023-08-14T20:22:13.757753Z","shell.execute_reply":"2023-08-14T20:22:43.828970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_asrLM = pipeline(\"automatic-speech-recognition\", model=CFG.my_model_name ,feature_extractor =processor.feature_extractor, tokenizer= processor.tokenizer,decoder=processor.decoder ,device=0)","metadata":{"papermill":{"duration":19.133684,"end_time":"2023-08-13T22:14:49.357075","exception":false,"start_time":"2023-08-13T22:14:30.223391","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:22:43.831557Z","iopub.execute_input":"2023-08-14T20:22:43.832009Z","iopub.status.idle":"2023-08-14T20:23:06.647014Z","shell.execute_reply.started":"2023-08-14T20:22:43.831974Z","shell.execute_reply":"2023-08-14T20:23:06.646008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalizer","metadata":{"papermill":{"duration":0.013643,"end_time":"2023-08-13T22:14:49.384953","exception":false,"start_time":"2023-08-13T22:14:49.371310","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def post_process_keys(str):\n    return str.replace(\"../input/test-wav-files-dl-sprint/test_files_wav/\",\"\").replace(\".wav\",\".mp3\")\n\nfrom bnunicodenormalizer import Normalizer \n\n\nbnorm = Normalizer()\ndef normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence","metadata":{"papermill":{"duration":0.036596,"end_time":"2023-08-13T22:14:49.435886","exception":false,"start_time":"2023-08-13T22:14:49.399290","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:06.648573Z","iopub.execute_input":"2023-08-14T20:23:06.648939Z","iopub.status.idle":"2023-08-14T20:23:06.663085Z","shell.execute_reply.started":"2023-08-14T20:23:06.648906Z","shell.execute_reply":"2023-08-14T20:23:06.662161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Function","metadata":{"papermill":{"duration":0.014077,"end_time":"2023-08-13T22:14:49.464596","exception":false,"start_time":"2023-08-13T22:14:49.450519","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"> **Inference on single audio**","metadata":{"papermill":{"duration":0.013803,"end_time":"2023-08-13T22:14:49.493012","exception":false,"start_time":"2023-08-13T22:14:49.479209","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"We'll trim the silences in the audios. This gives a slightly better performance.","metadata":{}},{"cell_type":"code","source":"def trim_silence(batch):\n    arr = batch\n    \n    try:\n        _max = max(max(arr), -min(arr))\n        old_length = len(arr)\n        \n        threshold = 30\n\n        for i,e in enumerate(arr):\n            if threshold*e>_max:\n                break\n\n        for j,e in enumerate(reversed(arr)):\n            if threshold*e>_max:\n                break\n\n        batch = arr[i:old_length-j]\n    except:\n        batch = arr\n    return batch","metadata":{"papermill":{"duration":0.028273,"end_time":"2023-08-13T22:14:49.536666","exception":false,"start_time":"2023-08-13T22:14:49.508393","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:06.667156Z","iopub.execute_input":"2023-08-14T20:23:06.667511Z","iopub.status.idle":"2023-08-14T20:23:06.675241Z","shell.execute_reply.started":"2023-08-14T20:23:06.667486Z","shell.execute_reply":"2023-08-14T20:23:06.674233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def infer(audio_path):\n    speech, sr = librosa.load(audio_path, sr=processor.feature_extractor.sampling_rate)\n \n    my_LM_prediction = my_asrLM(\n                speech\n            )\n\n    return my_LM_prediction['text']\n","metadata":{"papermill":{"duration":0.025123,"end_time":"2023-08-13T22:14:49.576326","exception":false,"start_time":"2023-08-13T22:14:49.551203","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:06.676566Z","iopub.execute_input":"2023-08-14T20:23:06.677107Z","iopub.status.idle":"2023-08-14T20:23:06.684812Z","shell.execute_reply.started":"2023-08-14T20:23:06.677050Z","shell.execute_reply":"2023-08-14T20:23:06.683894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(infer(\"/kaggle/input/bengaliai-speech/test_mp3s/bf36ea8b718d.mp3\"))\nprint(infer(\"/kaggle/input/bengaliai-speech/test_mp3s/0f3dac00655e.mp3\"))\nprint(infer(\"/kaggle/input/bengaliai-speech/test_mp3s/a9395e01ad21.mp3\"))","metadata":{"papermill":{"duration":11.922716,"end_time":"2023-08-13T22:15:01.513023","exception":false,"start_time":"2023-08-13T22:14:49.590307","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:06.687323Z","iopub.execute_input":"2023-08-14T20:23:06.688055Z","iopub.status.idle":"2023-08-14T20:23:20.349119Z","shell.execute_reply.started":"2023-08-14T20:23:06.688023Z","shell.execute_reply":"2023-08-14T20:23:20.348053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Batch inference**","metadata":{"papermill":{"duration":0.014312,"end_time":"2023-08-13T22:15:01.542580","exception":false,"start_time":"2023-08-13T22:15:01.528268","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def batch_infer(audio_paths, batch_size=BATCH_SIZE):\n    '''\n    infers on a batch of audio\n    args:\n      audio_paths  : list of path to audio files <list of string>\n    returns:\n      bangla predicted texts <list of string>\n    '''\n    results = []\n    for path in audio_paths:\n        pred = \"\"\n        try:\n            pred = infer(path)\n        except:\n            pred = \"।\"\n        if len(pred)==0:\n            pred = \"।\"\n        results.append(pred)\n    \n    return results","metadata":{"papermill":{"duration":0.028637,"end_time":"2023-08-13T22:15:01.586220","exception":false,"start_time":"2023-08-13T22:15:01.557583","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.350672Z","iopub.execute_input":"2023-08-14T20:23:20.351579Z","iopub.status.idle":"2023-08-14T20:23:20.357992Z","shell.execute_reply.started":"2023-08-14T20:23:20.351530Z","shell.execute_reply":"2023-08-14T20:23:20.357071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **Inference on a directory**","metadata":{"papermill":{"duration":0.014268,"end_time":"2023-08-13T22:15:01.615170","exception":false,"start_time":"2023-08-13T22:15:01.600902","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def directory_infer(audio_dir):\n    '''\n    infers on a directory that contains audio files\n    args:\n      audio_dir  : directory that contains some audio files <string>\n    returns:\n      a dataframe that contains 2 columns:\n        * path <string>\n        * sentence <string>\n    '''\n    # list all audio files\n\n    audio_paths=[audio_path for audio_path in tqdm(glob(os.path.join(audio_dir,\"*.*\")))]\n    files = os.listdir(\"/kaggle/input/bengaliai-speech/test_mp3s\")\n    paths = []\n    for i in files:\n        paths.append(i.split(\".\")[0])\n    sentences=[]\n    for idx in tqdm(range(0,len(audio_paths),BATCH_SIZE)):\n        batch_paths=audio_paths[idx:idx+BATCH_SIZE]\n        sentences+=batch_infer(batch_paths)\n        \n    df= pd.DataFrame({\"id\":paths,\"sentence\":sentences})\n    df.sentence= df.sentence.apply(lambda x:normalize(x))\n    df.sentence= df.sentence.apply(lambda x:dari(x))\n    df['id'] = df['id'].apply(lambda x: post_process_keys(x))\n    \n    return df ","metadata":{"papermill":{"duration":0.030013,"end_time":"2023-08-13T22:15:01.659859","exception":false,"start_time":"2023-08-13T22:15:01.629846","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.359420Z","iopub.execute_input":"2023-08-14T20:23:20.360007Z","iopub.status.idle":"2023-08-14T20:23:20.374823Z","shell.execute_reply.started":"2023-08-14T20:23:20.359974Z","shell.execute_reply":"2023-08-14T20:23:20.373799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{"papermill":{"duration":0.014798,"end_time":"2023-08-13T22:15:01.689611","exception":false,"start_time":"2023-08-13T22:15:01.674813","status":"completed"},"tags":[]}},{"cell_type":"code","source":"submission = directory_infer(TEST_DIRECTORY)\nsubmission.head()","metadata":{"papermill":{"duration":3.770201,"end_time":"2023-08-13T22:15:05.475112","exception":false,"start_time":"2023-08-13T22:15:01.704911","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.376130Z","iopub.execute_input":"2023-08-14T20:23:20.376557Z","iopub.status.idle":"2023-08-14T20:23:20.957946Z","shell.execute_reply.started":"2023-08-14T20:23:20.376524Z","shell.execute_reply":"2023-08-14T20:23:20.956868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check(sentence):\n    if len(sentence)==0:\n        return \"।\"\n    return sentence","metadata":{"papermill":{"duration":0.026415,"end_time":"2023-08-13T22:15:05.517069","exception":false,"start_time":"2023-08-13T22:15:05.490654","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.959366Z","iopub.execute_input":"2023-08-14T20:23:20.960411Z","iopub.status.idle":"2023-08-14T20:23:20.966041Z","shell.execute_reply.started":"2023-08-14T20:23:20.960376Z","shell.execute_reply":"2023-08-14T20:23:20.964823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.sentence = submission.sentence.apply(lambda x:check(x))","metadata":{"papermill":{"duration":0.027728,"end_time":"2023-08-13T22:15:05.560481","exception":false,"start_time":"2023-08-13T22:15:05.532753","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.967850Z","iopub.execute_input":"2023-08-14T20:23:20.968251Z","iopub.status.idle":"2023-08-14T20:23:20.976456Z","shell.execute_reply.started":"2023-08-14T20:23:20.968191Z","shell.execute_reply":"2023-08-14T20:23:20.975230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"papermill":{"duration":0.028426,"end_time":"2023-08-13T22:15:05.604067","exception":false,"start_time":"2023-08-13T22:15:05.575641","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-14T20:23:20.977774Z","iopub.execute_input":"2023-08-14T20:23:20.982417Z","iopub.status.idle":"2023-08-14T20:23:20.995653Z","shell.execute_reply.started":"2023-08-14T20:23:20.982382Z","shell.execute_reply":"2023-08-14T20:23:20.993843Z"},"trusted":true},"execution_count":null,"outputs":[]}]}