{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Install dependencies","metadata":{}},{"cell_type":"code","source":"!cp -r /kaggle/input/python-libaries-pack-asr ./\n\n!tar xvfz ./python-libaries-pack-asr/normalizer.tgz\n!pip -q install ./normalizer/bnunicodenormalizer-0.0.24.tar.gz -f ./ --no-index","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:05:22.019536Z","iopub.execute_input":"2023-09-20T10:05:22.019982Z","iopub.status.idle":"2023-09-20T10:05:41.734882Z","shell.execute_reply.started":"2023-09-20T10:05:22.019953Z","shell.execute_reply":"2023-09-20T10:05:41.733619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q ./python-libaries-pack-asr/accelerate-0.23.0-py3-none-any.whl\n!pip install -q ./python-libaries-pack-asr/peft-0.5.0-py3-none-any.whl\n!pip install -q ./python-libaries-pack-asr/bitsandbytes-0.41.1-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:05:41.737899Z","iopub.execute_input":"2023-09-20T10:05:41.738302Z","iopub.status.idle":"2023-09-20T10:07:20.767961Z","shell.execute_reply.started":"2023-09-20T10:05:41.738263Z","shell.execute_reply":"2023-09-20T10:07:20.766415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from peft import PeftModel, PeftConfig\nfrom transformers import WhisperForConditionalGeneration, WhisperTokenizer, WhisperProcessor","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:07:20.770062Z","iopub.execute_input":"2023-09-20T10:07:20.770437Z","iopub.status.idle":"2023-09-20T10:07:33.118399Z","shell.execute_reply.started":"2023-09-20T10:07:20.770398Z","shell.execute_reply":"2023-09-20T10:07:33.117450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the model","metadata":{}},{"cell_type":"code","source":"from transformers import WhisperForConditionalGeneration, WhisperProcessor, WhisperTokenizer\n\nmodel_path = '/kaggle/input/indicwhisper-bn/bengali_models/whisper-medium-bn_alldata_multigpu'\nlanguage = \"Bengali\"\ntask = \"transcribe\"\n\nmodel = WhisperForConditionalGeneration.from_pretrained(model_path, device_map=\"auto\")\nprocessor = WhisperProcessor.from_pretrained(model_path)\ntokenizer = WhisperTokenizer.from_pretrained(model_path, language=language, task=task)\nfeature_extractor = processor.feature_extractor\nforced_decoder_ids = processor.get_decoder_prompt_ids(language=language, task=task)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:07:33.121383Z","iopub.execute_input":"2023-09-20T10:07:33.122113Z","iopub.status.idle":"2023-09-20T10:07:59.011806Z","shell.execute_reply.started":"2023-09-20T10:07:33.122072Z","shell.execute_reply":"2023-09-20T10:07:59.010768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_extractor.sampling_rate","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:07:59.013482Z","iopub.execute_input":"2023-09-20T10:07:59.013836Z","iopub.status.idle":"2023-09-20T10:07:59.024218Z","shell.execute_reply.started":"2023-09-20T10:07:59.013801Z","shell.execute_reply":"2023-09-20T10:07:59.023102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.config.decoder_start_token_id","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:07:59.026186Z","iopub.execute_input":"2023-09-20T10:07:59.026673Z","iopub.status.idle":"2023-09-20T10:07:59.034112Z","shell.execute_reply.started":"2023-09-20T10:07:59.026640Z","shell.execute_reply":"2023-09-20T10:07:59.033083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create pipeline and running inference","metadata":{}},{"cell_type":"code","source":"from transformers import AutomaticSpeechRecognitionPipeline\nimport os\nimport torch\nfrom concurrent.futures import ThreadPoolExecutor\n\n# Path to audio files\naudios_path = '/kaggle/input/bengaliai-speech/test_mp3s'\naudios = [os.path.join(audios_path, audio_file) for audio_file in os.listdir(audios_path)]\n\n# Define the ASR pipeline\npipeline = AutomaticSpeechRecognitionPipeline(model=model, tokenizer=tokenizer, feature_extractor=feature_extractor)\n\n# Helper function to process a batch of audio files\ndef process_batch(start_index, end_index):\n    batch_audios = audios[start_index:end_index]\n    with torch.cuda.amp.autocast():\n        batch_transcripts = pipeline(batch_audios, generate_kwargs={\"forced_decoder_ids\": forced_decoder_ids}, max_new_tokens=255)\n    return batch_transcripts\n\n# Batch size for processing\nbatch_size = 16\nnum_audios = len(audios)\nstart_indices = list(range(0, num_audios, batch_size))\nend_indices = [min(start_index + batch_size, num_audios) for start_index in start_indices]\n\n# Process audio files in batches using ThreadPoolExecutor for asynchronous processing\nwith ThreadPoolExecutor() as executor:\n    transcripts = list(executor.map(process_batch, start_indices, end_indices))\n\n# Flatten the list of transcripts\ntranscripts = [trans for sublist in transcripts for trans in sublist]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:22:54.782562Z","iopub.execute_input":"2023-09-20T11:22:54.783049Z","iopub.status.idle":"2023-09-20T11:23:44.025278Z","shell.execute_reply.started":"2023-09-20T11:22:54.783007Z","shell.execute_reply":"2023-09-20T11:23:44.023775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Normalize the predicted sentences","metadata":{}},{"cell_type":"code","source":"from bnunicodenormalizer import Normalizer\n\nbnorm = Normalizer()\n\ndef postprocess(sentence):\n    period_set = set([\".\", \"?\", \"!\", \"।\"])\n    _words = [bnorm(word)['normalized']  for word in sentence.split()]\n    sentence = \" \".join([word for word in _words if word is not None])\n    try:\n        if sentence[-1] not in period_set:\n            sentence+=\"।\"\n    except:\n        sentence = \"।\"\n    return sentence","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:22:12.193162Z","iopub.execute_input":"2023-09-20T11:22:12.193596Z","iopub.status.idle":"2023-09-20T11:22:12.205338Z","shell.execute_reply.started":"2023-09-20T11:22:12.193561Z","shell.execute_reply":"2023-09-20T11:22:12.204109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentences_output = [sentece['text'] for sentece in transcripts]","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:22:02.212218Z","iopub.execute_input":"2023-09-20T11:22:02.212610Z","iopub.status.idle":"2023-09-20T11:22:02.219156Z","shell.execute_reply.started":"2023-09-20T11:22:02.212574Z","shell.execute_reply":"2023-09-20T11:22:02.217941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_pred_sentence_list = [postprocess(s) for s in sentences_output]","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:22:15.504337Z","iopub.execute_input":"2023-09-20T11:22:15.505586Z","iopub.status.idle":"2023-09-20T11:22:15.534847Z","shell.execute_reply.started":"2023-09-20T11:22:15.505539Z","shell.execute_reply":"2023-09-20T11:22:15.532012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create submission file","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ntest = pd.DataFrame()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:24:08.779833Z","iopub.execute_input":"2023-09-20T11:24:08.780633Z","iopub.status.idle":"2023-09-20T11:24:08.786821Z","shell.execute_reply.started":"2023-09-20T11:24:08.780597Z","shell.execute_reply":"2023-09-20T11:24:08.785673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['id'] = [(audio_file.split('/'))[-1].split('.')[0] for audio_file in audios]\ntest[\"sentence\"] = pp_pred_sentence_list\n\ntest.to_csv(\"submission.csv\", index=False)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T11:24:09.091683Z","iopub.execute_input":"2023-09-20T11:24:09.092104Z","iopub.status.idle":"2023-09-20T11:24:09.120607Z","shell.execute_reply.started":"2023-09-20T11:24:09.092071Z","shell.execute_reply":"2023-09-20T11:24:09.118165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## END","metadata":{}}]}