{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.display import clear_output\nimport sys, os\n\nif 0:\n    !pip download kenlm\n    \nif 0:\n    !pip install kenlm\n    !pip install pyctcdecode\n    !pip install jiwer\n    \ntry:\n    import jiwer\nexcept:     \n    !pip install jiwer pyctcdecode kenlm  bnunicodenormalizer --no-index --find-links /kaggle/input/hugging-face-extra-01\n    #!pip install /home/pip-packages/* \n    \n    \n#clear_output(wait=True) \nos.chdir('/kaggle/working')\n\nfrom bnunicodenormalizer import Normalizer\nimport numpy as np\nimport pandas as pd\nimport jiwer\nfrom glob import glob \n\n#-------------------------------------------------\nfrom timeit import default_timer as timer\n#helper\ndef time_to_str(t, mode='min'):\n    if mode=='min':\n        t  = int(t)/60\n        hr = t//60\n        min = t%60\n        return '%2d hr %02d min'%(hr,min)\n\n    elif mode=='sec':\n        t   = int(t)\n        min = t//60\n        sec = t%60\n        return '%2d min %02d sec'%(min,sec)\n\n    else:\n        raise NotImplementedError\n        \n        \nmode='submit'#'submit' #debug\n\n\nprint('mode', mode)\nprint('IMPORT OK !!!!')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:22.439323Z","iopub.execute_input":"2023-07-21T12:27:22.439712Z","iopub.status.idle":"2023-07-21T12:27:22.470029Z","shell.execute_reply.started":"2023-07-21T12:27:22.439680Z","shell.execute_reply":"2023-07-21T12:27:22.468993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import pipeline\n\nif mode=='debug':\n    device=-1\nif mode=='submit':\n    device=0\n\nlocal_file = \\\n'/kaggle/input/ai4bharat-indicwav2vec-v1-bengali'\npipe = pipeline(\n    'automatic-speech-recognition',\n    model=local_file,\n    #model='shahruk10/wav2vec2-xls-r-300m-bengali-commonvoice',\n    device=device,\n)\nprint('MODEL OK !!!!')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:22.471666Z","iopub.execute_input":"2023-07-21T12:27:22.472202Z","iopub.status.idle":"2023-07-21T12:27:26.528833Z","shell.execute_reply.started":"2023-07-21T12:27:22.472168Z","shell.execute_reply":"2023-07-21T12:27:26.527839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if mode=='debug':\n    mp3_dir = f'/kaggle/input/bengaliai-speech/train_mp3s'\n    valid_df = pd.read_csv('/kaggle/input/bengaliai-speech/train.csv')\n    valid_df = valid_df[:10]\n# if mode=='submit':\n#     mp3_dir = f'/kaggle/input/bengaliai-speech/test_mp3s'\n#     valid_df = pd.read_csv('/kaggle/input/bengaliai-speech/sample_submission.csv')\nif mode=='submit':\n    mp3_dir = f'/kaggle/input/bengaliai-speech/test_mp3s'\n    glob_file = glob(f'{mp3_dir}/*.mp3')\n    id = sorted([f[len(mp3_dir)+1:-4] for f in glob_file])\n    valid_df = pd.DataFrame({'id':id,'sentence':''})\n\n    \nprint(len(valid_df))\nprint(valid_df['id'][:5].tolist())\n\ndef data():\n    for t, d in valid_df.iterrows():\n        print('\\r', t, d['id'], end='')\n        mp3_file = f'{mp3_dir}/{d[\"id\"]}.mp3'\n        yield mp3_file\n\nprint('DATA OK !!!!')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:26.530223Z","iopub.execute_input":"2023-07-21T12:27:26.531859Z","iopub.status.idle":"2023-07-21T12:27:26.544547Z","shell.execute_reply.started":"2023-07-21T12:27:26.531822Z","shell.execute_reply":"2023-07-21T12:27:26.543410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#post processor\nbnorm = Normalizer()\ndef normalize(sentence):\n    word = [bnorm(word)['normalized']  for word in sentence.split()]\n    return \" \".join([w for w in word if w is not None])\n\ndef dari(sentence):\n    try:\n        if sentence[-1]!=\"।\":\n            sentence+=\"।\"\n    except:\n        print(sentence)\n    return sentence\n    \ndef check_text(sentence):\n    t = 'িনি এবং ও এই করে া। তার' #fake truth\n    #print(sentence)\n    try:\n        jiwer.wer(t, [sentence])\n    except:\n        sentence = '' #t\n    return sentence\n\n\n\n#----\nstart_timer=timer()\npredict = []\nfor out in pipe(data()):\n    predict.append( out['text'])\nprint('')\nprint(time_to_str(timer() - start_timer,'sec'))\nprint(valid_df.shape)\nprint(len(predict))\nassert (len(predict)==len(valid_df))\n\n    \nif mode=='debug':\n    score = jiwer.wer(valid_df['sentence'].to_list(), predict)\n    print('jiwer', score)\n    \nprint('')\nsubmit_df = valid_df.copy()\nsubmit_df.loc[:,'sentence'] = predict\n#submit_df.loc[:,'sentence'] = submit_df.sentence.apply(lambda x: x if type(x)==str else '') #???\nsubmit_df.loc[:,'sentence'] = submit_df.sentence.apply(check_text)\nsubmit_df.loc[:,'sentence'] = submit_df.sentence.apply(lambda x:normalize(x))\nsubmit_df.loc[:,'sentence'] = submit_df.sentence.apply(lambda x:dari(x))\n#print(submit_df.sentence.dtype) #object\n#submit_df.loc[:,'sentence'] = submit_df.sentence.astype('string')\nprint(submit_df.sentence.dtype)  \n\nsubmit_df.to_csv('submission.csv',index=False)\nprint(submit_df)\nprint('SUBMIT DONE !!!!!!!')\n\n'''\n\n'''","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:26.547767Z","iopub.execute_input":"2023-07-21T12:27:26.548185Z","iopub.status.idle":"2023-07-21T12:27:27.089974Z","shell.execute_reply.started":"2023-07-21T12:27:26.548152Z","shell.execute_reply":"2023-07-21T12:27:27.088555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -rf '/root/.cache/huggingface'\n# print( scan_cache_dir())","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:27.094405Z","iopub.execute_input":"2023-07-21T12:27:27.096851Z","iopub.status.idle":"2023-07-21T12:27:27.103683Z","shell.execute_reply.started":"2023-07-21T12:27:27.096815Z","shell.execute_reply":"2023-07-21T12:27:27.102530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -r /kaggle/working/*\n#!rm -rf /kaggle/working/hf_cache\nos.chdir('/kaggle/working')\n!pwd\n!ls","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:27.105034Z","iopub.execute_input":"2023-07-21T12:27:27.105716Z","iopub.status.idle":"2023-07-21T12:27:29.253419Z","shell.execute_reply.started":"2023-07-21T12:27:27.105682Z","shell.execute_reply":"2023-07-21T12:27:29.252177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 0:\n    import subprocess\n    from IPython.display import FileLink, display\n    import os\n\n    def download_file(path, download_zip_file_name):\n        os.chdir('/kaggle/working/')\n        zip_name = f\"/kaggle/working/{download_zip_file_name}\"\n        command = f\"zip {zip_name} {path} -r\"\n        print('zip ...')\n        result = subprocess.run(command, shell=True, capture_output=True, text=True)\n        if result.returncode != 0:\n            print(\"Unable to run zip command!\")\n            print(result.stderr)\n            return\n        display(FileLink(f'{download_zip_file_name}'))\n\n    download_file('/kaggle/working', 'working.out.zip')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:29.255278Z","iopub.execute_input":"2023-07-21T12:27:29.255803Z","iopub.status.idle":"2023-07-21T12:27:29.263739Z","shell.execute_reply.started":"2023-07-21T12:27:29.255762Z","shell.execute_reply":"2023-07-21T12:27:29.262629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 0:\n    import os\n    cache_dir = '/root/.cache/pip/wheels/4e/3a/01/9105a071c30781823efbd96a58279c16f948a87cafb1144042'\n    wheel = 'kenlm-0.1-cp310-cp310-linux_x86_64.whl'\n    wheel_file = f'{cache_dir}/{wheel}'\n \n    dst_file = f'/kaggle/working/{wheel}' \n    !cp $wheel_file $dst_file","metadata":{"execution":{"iopub.status.busy":"2023-07-21T12:27:29.265296Z","iopub.execute_input":"2023-07-21T12:27:29.265958Z","iopub.status.idle":"2023-07-21T12:27:29.278532Z","shell.execute_reply.started":"2023-07-21T12:27:29.265925Z","shell.execute_reply":"2023-07-21T12:27:29.277340Z"},"trusted":true},"execution_count":null,"outputs":[]}]}