{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Install dependencies","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/python-libaries-pack-asr ./\n\n!tar xvfz ./python-libaries-pack-asr/normalizer.tgz\n!pip install -q ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index","metadata":{"_uuid":"88201c81-d8ba-4ed0-b79b-1fd7840a83fa","_cell_guid":"b95d3974-59b5-4d3e-aee1-ace0c5239434","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:54:09.343236Z","iopub.execute_input":"2023-09-18T13:54:09.343512Z","iopub.status.idle":"2023-09-18T13:54:28.692877Z","shell.execute_reply.started":"2023-09-18T13:54:09.343486Z","shell.execute_reply":"2023-09-18T13:54:28.691626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q ./python-libaries-pack-asr/accelerate-0.23.0-py3-none-any.whl\n!pip install -q ./python-libaries-pack-asr/peft-0.5.0-py3-none-any.whl\n!pip install -q ./python-libaries-pack-asr/bitsandbytes-0.41.1-py3-none-any.whl","metadata":{"_uuid":"0c2729cf-f220-4ef1-9715-1e1768688ab8","_cell_guid":"edb43a9e-ffec-4d74-a3a6-63c26bcbde07","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:54:28.695593Z","iopub.execute_input":"2023-09-18T13:54:28.696310Z","iopub.status.idle":"2023-09-18T13:56:05.938728Z","shell.execute_reply.started":"2023-09-18T13:54:28.696261Z","shell.execute_reply":"2023-09-18T13:56:05.937410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the model","metadata":{}},{"cell_type":"markdown","source":"As the fine-tuning with LoRA techniques only saves the LoRA weigths, it's needed the base model. To load it from a local file (internet connection is disabled), you need the change the field \"base_model_name_or_path\" from \"adapter_config.json\" file. ","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/whisper-small-bn/david/int8-whisper-small-asr-bengali/checkpoint-1000/ .","metadata":{"_uuid":"b244620c-3c79-4bbb-a446-2d35ef655dda","_cell_guid":"cab6cf1e-c794-431b-bf22-b29bd8294dd1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:05.940892Z","iopub.execute_input":"2023-09-18T13:56:05.941251Z","iopub.status.idle":"2023-09-18T13:56:07.477368Z","shell.execute_reply.started":"2023-09-18T13:56:05.941222Z","shell.execute_reply":"2023-09-18T13:56:07.476147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nconfig_file_path = '/kaggle/working/checkpoint-1000/adapter_model/adapter_config.json'\nwith open(config_file_path, 'r') as f:\n    adapter_config = json.load(f)\n    print(adapter_config)\nadapter_config['base_model_name_or_path'] = '/kaggle/input/openai-whisper-small'\n\nwith open(config_file_path, 'w') as f:\n    json.dump(adapter_config, f)","metadata":{"_uuid":"1d4ed455-d07a-4fba-9d5a-c75680477858","_cell_guid":"4925ceff-c5f3-43f4-80f2-d20a81b0331a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:07.480407Z","iopub.execute_input":"2023-09-18T13:56:07.480764Z","iopub.status.idle":"2023-09-18T13:56:07.492177Z","shell.execute_reply.started":"2023-09-18T13:56:07.480730Z","shell.execute_reply":"2023-09-18T13:56:07.491293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from peft import PeftModel, PeftConfig\nfrom transformers import WhisperForConditionalGeneration, WhisperTokenizer, WhisperProcessor\nimport torch","metadata":{"_uuid":"c366fa6c-44ad-4938-a4c8-ea68c3a43207","_cell_guid":"a03c9546-ba7a-4e93-9a36-45357dd285e5","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:07.493980Z","iopub.execute_input":"2023-09-18T13:56:07.494612Z","iopub.status.idle":"2023-09-18T13:56:19.724498Z","shell.execute_reply.started":"2023-09-18T13:56:07.494580Z","shell.execute_reply":"2023-09-18T13:56:19.723531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"peft_model_id = \"/kaggle/working/checkpoint-1000/adapter_model\"\nlanguage = \"Bengali\"\nlanguage_abbr = \"bn\"\ntask = \"transcribe\"\npeft_config = PeftConfig.from_pretrained(peft_model_id, inference_mode=True)\n\nmodel = WhisperForConditionalGeneration.from_pretrained(\n    peft_model_id, torch_dtype=torch.float16, device_map=\"auto\"\n)\nmodel = PeftModel.from_pretrained(model, peft_model_id)\ntokenizer = WhisperTokenizer.from_pretrained(peft_config.base_model_name_or_path, language=language, task=task)\nprocessor = WhisperProcessor.from_pretrained(peft_config.base_model_name_or_path, language=language, task=task)\nfeature_extractor = processor.feature_extractor\nforced_decoder_ids = processor.get_decoder_prompt_ids(language=language, task=task)","metadata":{"_uuid":"222b6ab2-d5f6-43db-ad31-ab1170a82a17","_cell_guid":"68fa0f24-21ce-4ab6-89bc-d9ac496619b1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:19.725764Z","iopub.execute_input":"2023-09-18T13:56:19.727118Z","iopub.status.idle":"2023-09-18T13:56:36.594385Z","shell.execute_reply.started":"2023-09-18T13:56:19.727081Z","shell.execute_reply":"2023-09-18T13:56:36.593395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge the LoRA and base model checkpoints\n\nMerging is suposed to speed up the inference ([https://github.com/huggingface/peft/issues/217#issuecomment-1506224612](https://github.com/huggingface/peft/issues/217#issuecomment-1506224612))","metadata":{}},{"cell_type":"code","source":"model = model.merge_and_unload()","metadata":{"_uuid":"00b2216e-6809-4917-933e-bacb437944ce","_cell_guid":"3109ab99-df3b-4cdd-ad37-e7a949a99f11","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:36.596024Z","iopub.execute_input":"2023-09-18T13:56:36.596372Z","iopub.status.idle":"2023-09-18T13:56:38.600075Z","shell.execute_reply.started":"2023-09-18T13:56:36.596340Z","shell.execute_reply":"2023-09-18T13:56:38.599106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"markdown","source":"### Load the pipeline","metadata":{}},{"cell_type":"code","source":"from transformers import AutomaticSpeechRecognitionPipeline\nimport os\n\naudios_path = '/kaggle/input/bengaliai-speech/test_mp3s'\naudios = list(map(lambda x: audios_path + '/' + x , os.listdir(audios_path)))\npipeline = AutomaticSpeechRecognitionPipeline(model=model, tokenizer=tokenizer, feature_extractor=feature_extractor)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:38.601357Z","iopub.execute_input":"2023-09-18T13:56:38.601688Z","iopub.status.idle":"2023-09-18T13:56:40.348745Z","shell.execute_reply.started":"2023-09-18T13:56:38.601658Z","shell.execute_reply":"2023-09-18T13:56:40.347767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make the predictions","metadata":{}},{"cell_type":"code","source":"import torch\n\nwith torch.cuda.amp.autocast():\n    text = pipeline(audios, generate_kwargs={\"forced_decoder_ids\": forced_decoder_ids}, max_new_tokens=255)#[\"text\"]","metadata":{"_uuid":"d316f727-2317-44aa-800d-39951815a355","_cell_guid":"4d2d3414-a6e0-46b0-87a8-dde886fbfcb0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-09-18T13:56:40.350069Z","iopub.execute_input":"2023-09-18T13:56:40.350408Z","iopub.status.idle":"2023-09-18T13:56:49.568178Z","shell.execute_reply.started":"2023-09-18T13:56:40.350376Z","shell.execute_reply":"2023-09-18T13:56:49.567141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Normalize the predicted sentences","metadata":{}},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        # print(sentence)\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:49.571699Z","iopub.execute_input":"2023-09-18T13:56:49.572188Z","iopub.status.idle":"2023-09-18T13:56:49.582389Z","shell.execute_reply.started":"2023-09-18T13:56:49.572150Z","shell.execute_reply":"2023-09-18T13:56:49.581272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences_output = [sentece['text'] for sentece in text]","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:49.583811Z","iopub.execute_input":"2023-09-18T13:56:49.584877Z","iopub.status.idle":"2023-09-18T13:56:49.598009Z","shell.execute_reply.started":"2023-09-18T13:56:49.584833Z","shell.execute_reply":"2023-09-18T13:56:49.597008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [postprocess(s) for s in sentences_output]","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:49.599684Z","iopub.execute_input":"2023-09-18T13:56:49.600098Z","iopub.status.idle":"2023-09-18T13:56:49.626691Z","shell.execute_reply.started":"2023-09-18T13:56:49.600066Z","shell.execute_reply":"2023-09-18T13:56:49.625704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create submission file ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ntest = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:49.628060Z","iopub.execute_input":"2023-09-18T13:56:49.628567Z","iopub.status.idle":"2023-09-18T13:56:49.640017Z","shell.execute_reply.started":"2023-09-18T13:56:49.628534Z","shell.execute_reply":"2023-09-18T13:56:49.639004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['id'] = [(audio_file.split('/'))[-1].split('.')[0] for audio_file in audios]\ntest[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T13:56:49.641676Z","iopub.execute_input":"2023-09-18T13:56:49.642401Z","iopub.status.idle":"2023-09-18T13:56:49.671027Z","shell.execute_reply.started":"2023-09-18T13:56:49.642368Z","shell.execute_reply":"2023-09-18T13:56:49.669997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## END","metadata":{}}]}