{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Bengali Example Audio indicwav2vec\nhttps://www.kaggle.com/code/stpeteishii/english-audio-wav2vec2","metadata":{"papermill":{"duration":0.009089,"end_time":"2023-08-21T04:35:41.194748","exception":false,"start_time":"2023-08-21T04:35:41.185659","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## ai4bharat/indicwav2vec_v1_bengali \nThe ai4bharat/indicwav2vec_v1_bengali model is a speech recognition model trained on a large dataset of Bengali audio recordings. It is based on the Wav2vec 2.0 architecture, which is a deep learning model that can learn the long-term dependencies in speech signals. The model has been fine-tuned on a dataset of Bengali speech commands, and it can recognize a variety of words and phrases.\n","metadata":{"papermill":{"duration":0.007699,"end_time":"2023-08-21T04:35:41.211977","exception":false,"start_time":"2023-08-21T04:35:41.204278","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install moviepy","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:35:41.234740Z","iopub.status.busy":"2023-08-21T04:35:41.234078Z","iopub.status.idle":"2023-08-21T04:35:53.598177Z","shell.execute_reply":"2023-08-21T04:35:53.597321Z","shell.execute_reply.started":"2023-08-20T11:08:41.83061Z"},"papermill":{"duration":12.378265,"end_time":"2023-08-21T04:35:53.598356","exception":false,"start_time":"2023-08-21T04:35:41.220091","status":"completed"},"tags":[],"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2Tokenizer, Wav2Vec2ForCTC\nfrom IPython.display import Audio\nfrom IPython.display import Video\nimport moviepy.editor as mp\nimport torch\nimport librosa\nimport os\nimport pandas as pd","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:35:53.649930Z","iopub.status.busy":"2023-08-21T04:35:53.648742Z","iopub.status.idle":"2023-08-21T04:35:58.305879Z","shell.execute_reply":"2023-08-21T04:35:58.305273Z","shell.execute_reply.started":"2023-08-20T11:08:55.913276Z"},"papermill":{"duration":4.683807,"end_time":"2023-08-21T04:35:58.306083","exception":false,"start_time":"2023-08-21T04:35:53.622276","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = Wav2Vec2Tokenizer.from_pretrained(\"ai4bharat/indicwav2vec_v1_bengali\")\nmodel = Wav2Vec2ForCTC.from_pretrained(\"ai4bharat/indicwav2vec_v1_bengali\")","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:35:58.357411Z","iopub.status.busy":"2023-08-21T04:35:58.356397Z","iopub.status.idle":"2023-08-21T04:36:36.967958Z","shell.execute_reply":"2023-08-21T04:36:36.968731Z","shell.execute_reply.started":"2023-08-20T11:09:00.759286Z"},"papermill":{"duration":38.639682,"end_time":"2023-08-21T04:36:36.968935","exception":false,"start_time":"2023-08-21T04:35:58.329253","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clip_paths=[]\nfor dirname, _, filenames in os.walk('/kaggle/input/bengaliai-speech/examples'):\n    for filename in filenames:\n        clip_paths+=[(os.path.join(dirname, filename))]","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:36:37.024268Z","iopub.status.busy":"2023-08-21T04:36:37.023238Z","iopub.status.idle":"2023-08-21T04:36:37.049647Z","shell.execute_reply":"2023-08-21T04:36:37.050273Z","shell.execute_reply.started":"2023-08-20T11:09:18.602465Z"},"papermill":{"duration":0.056286,"end_time":"2023-08-21T04:36:37.050482","exception":false,"start_time":"2023-08-21T04:36:36.994196","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cc = []\nexdata=pd.DataFrame(columns=['path','sentence'],index=range(len(clip_paths)))\nfor i,path in enumerate(clip_paths):\n    print(path.split('/')[-1])\n    # Load the audio with the librosa library\n    input_audio, _ = librosa.load(path, sr=16000)\n    # Tokenize the audio\n    input_values = tokenizer(input_audio, return_tensors=\"pt\", padding=\"longest\").input_values\n    # Feed it through Wav2Vec & choose the most probable tokens\n    with torch.no_grad():\n        logits = model(input_values).logits\n        predicted_ids = torch.argmax(logits, dim=-1)\n    # Decode & add to our caption string\n    transcription = tokenizer.batch_decode(predicted_ids)[0]\n    #cc += [transcription]\n    exdata.loc[i,'sentence']=transcription\n    exdata.loc[i,'path']=path","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:36:37.108340Z","iopub.status.busy":"2023-08-21T04:36:37.107323Z","iopub.status.idle":"2023-08-21T04:51:12.096753Z","shell.execute_reply":"2023-08-21T04:51:12.097347Z","shell.execute_reply.started":"2023-08-20T11:09:43.076664Z"},"papermill":{"duration":875.02182,"end_time":"2023-08-21T04:51:12.097553","exception":false,"start_time":"2023-08-21T04:36:37.075733","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(clip_paths[0])","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:51:12.159750Z","iopub.status.busy":"2023-08-21T04:51:12.159114Z","iopub.status.idle":"2023-08-21T04:51:12.493932Z","shell.execute_reply":"2023-08-21T04:51:12.321796Z"},"papermill":{"duration":0.367026,"end_time":"2023-08-21T04:51:12.494135","exception":false,"start_time":"2023-08-21T04:51:12.127109","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(exdata)","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:51:13.103913Z","iopub.status.busy":"2023-08-21T04:51:13.103147Z","iopub.status.idle":"2023-08-21T04:51:13.122631Z","shell.execute_reply":"2023-08-21T04:51:13.121935Z"},"papermill":{"duration":0.31888,"end_time":"2023-08-21T04:51:13.122786","exception":false,"start_time":"2023-08-21T04:51:12.803906","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exdata.to_csv('exdata.csv',index=False)","metadata":{"execution":{"iopub.execute_input":"2023-08-21T04:51:13.674786Z","iopub.status.busy":"2023-08-21T04:51:13.673851Z","iopub.status.idle":"2023-08-21T04:51:13.680035Z","shell.execute_reply":"2023-08-21T04:51:13.680572Z"},"papermill":{"duration":0.286266,"end_time":"2023-08-21T04:51:13.680765","exception":false,"start_time":"2023-08-21T04:51:13.394499","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}