{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!pip install datasets -U\n#!pip install git+https://github.com/huggingface/transformers\n#!pip install librosa\n#!pip install evaluate>=0.30\n#!pip install jiwer\n#!pip install gradio\n#!pip install pyctcdecode\n#!pip install -q bnunicodenormalizer\n#!pip install -i https://test.pypi.org/simple/ bitsandbytes\n#!pip install -U git+https://github.com/huggingface/accelerate.git\n#!pip install -q git+https://github.com/huggingface/transformers.git@main git+https://github.com/huggingface/peft.git@main","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:37:57.292017Z","iopub.execute_input":"2023-09-19T13:37:57.292418Z","iopub.status.idle":"2023-09-19T13:37:57.297864Z","shell.execute_reply.started":"2023-09-19T13:37:57.292388Z","shell.execute_reply":"2023-09-19T13:37:57.296756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar xvfz /kaggle/input/python-packages/normalizer.tgz\n!pip install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:41:19.421906Z","iopub.execute_input":"2023-09-19T13:41:19.422313Z","iopub.status.idle":"2023-09-19T13:41:35.018731Z","shell.execute_reply.started":"2023-09-19T13:41:19.422280Z","shell.execute_reply":"2023-09-19T13:41:35.017494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport torch\nimport torchaudio\nimport numpy as np\n\nfrom transformers import WhisperTokenizer\nfrom transformers import WhisperProcessor\nfrom transformers import WhisperFeatureExtractor\nfrom transformers import WhisperForConditionalGeneration\nfrom tqdm import tqdm\nimport pandas as pd\nimport soundfile as sf\nfrom pydub import AudioSegment","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:41:37.557632Z","iopub.execute_input":"2023-09-19T13:41:37.558054Z","iopub.status.idle":"2023-09-19T13:41:37.565654Z","shell.execute_reply.started":"2023-09-19T13:41:37.558018Z","shell.execute_reply":"2023-09-19T13:41:37.564282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = '/kaggle/input/whisper-small-ver2/checkpoint-2000'\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nfeature_extractor = WhisperFeatureExtractor.from_pretrained(model_path)\ntokenizer = WhisperTokenizer.from_pretrained(model_path)\nprocessor = WhisperProcessor.from_pretrained(model_path)\nmodel = WhisperForConditionalGeneration.from_pretrained(model_path).to(device)\nforced_decoder_ids = processor.get_decoder_prompt_ids(language=\"bengali\", task=\"transcribe\")","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:41:39.192497Z","iopub.execute_input":"2023-09-19T13:41:39.192884Z","iopub.status.idle":"2023-09-19T13:41:43.092328Z","shell.execute_reply.started":"2023-09-19T13:41:39.192854Z","shell.execute_reply":"2023-09-19T13:41:43.091305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"|\"])\n    _words = [bnorm(word)['normalized'] for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"|\"\n    \n    except:\n        sentence = \"|\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:43:36.678331Z","iopub.execute_input":"2023-09-19T13:43:36.678740Z","iopub.status.idle":"2023-09-19T13:43:36.687787Z","shell.execute_reply.started":"2023-09-19T13:43:36.678708Z","shell.execute_reply":"2023-09-19T13:43:36.686549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference_fn(mp3_path):\n    speech_array, sampling_rate = sf.read(mp3_path)\n    speech_array = librosa.resample(np.asarray(speech_array), orig_sr=sampling_rate, target_sr=16000)\n    input_features = feature_extractor(speech_array, sampling_rate=16000, return_tensors=\"pt\").input_features\n    predicted_ids = model.generate(inputs=input_features.to(device), forced_decoder_ids=forced_decoder_ids)[0]\n    transcription = processor.decode(predicted_ids, skip_special_tokens=True)\n    return transcription","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:42:08.925494Z","iopub.execute_input":"2023-09-19T13:42:08.925916Z","iopub.status.idle":"2023-09-19T13:42:08.932152Z","shell.execute_reply.started":"2023-09-19T13:42:08.925881Z","shell.execute_reply":"2023-09-19T13:42:08.931127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\ntest_transcriptions = []\nfor i in tqdm(os.listdir('/kaggle/input/bengaliai-speech/test_mp3s')):\n    ids.append(i.split('.')[0])\n    test_transcriptions.append(postprocess(inference_fn('/kaggle/input/bengaliai-speech/test_mp3s/'+i)))","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:43:39.329010Z","iopub.execute_input":"2023-09-19T13:43:39.329394Z","iopub.status.idle":"2023-09-19T13:43:43.896769Z","shell.execute_reply.started":"2023-09-19T13:43:39.329364Z","shell.execute_reply":"2023-09-19T13:43:43.895870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({\n    'id':ids,\n    'sentence':test_transcriptions\n})\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:43:46.772594Z","iopub.execute_input":"2023-09-19T13:43:46.772977Z","iopub.status.idle":"2023-09-19T13:43:46.784654Z","shell.execute_reply.started":"2023-09-19T13:43:46.772948Z","shell.execute_reply":"2023-09-19T13:43:46.783545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-09-19T13:43:50.755746Z","iopub.execute_input":"2023-09-19T13:43:50.756129Z","iopub.status.idle":"2023-09-19T13:43:50.762528Z","shell.execute_reply.started":"2023-09-19T13:43:50.756098Z","shell.execute_reply":"2023-09-19T13:43:50.761533Z"},"trusted":true},"execution_count":null,"outputs":[]}]}