{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":6707460,"sourceType":"datasetVersion","datasetId":3865741},{"sourceId":8177500,"sourceType":"datasetVersion","datasetId":4840727}],"dockerImageVersionId":30528,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!cp /kaggle/input/bengali-eval-data/predict.py .\n\n!cp -r ../input/python-packages2 ./\n!tar xvfz ./python-packages2/jiwer.tgz\n!pip install ./jiwer/python-Levenshtein-0.12.2.tar.gz -f ./ --no-index\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T07:50:05.737890Z","iopub.execute_input":"2024-04-23T07:50:05.738186Z","iopub.status.idle":"2024-04-23T07:50:12.935186Z","shell.execute_reply.started":"2024-04-23T07:50:05.738159Z","shell.execute_reply":"2024-04-23T07:50:12.933869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport csv\nimport time\nimport glob\n\nMODEL = '/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium/'\nPUNCT_MODELS = [\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-6layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-8layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-11layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-12layers/'\n]\nCHUNK_LENGTH_S = 20.1\nENABLE_BEAM = True\n# none alone 0.9 0.037964275588318684\n# none alone 0.7 0.03592288063510065\n# none alone 0.4 0.03416501275871846\nPUNCT_WEIGHTS = [[1.0, 1.4, 1.0, 0.8]]\n\nif ENABLE_BEAM:\n    BATCH_SIZE = 4\nelse:\n    BATCH_SIZE = 8\n\nif len(glob.glob(\"/kaggle/input/bengaliai-speech/test_mp3s/*.mp3\")) > 10:\n    EVAL = False\n    DATASET_PATH = '/kaggle/input/bengaliai-speech/test_mp3s/'\nelse:\n    EVAL = True\n    DATASET_PATH = '/kaggle/input/bengaliai-speech/test_mp3s/'\n    \nimport csv\nimport glob\nimport shutil\nimport librosa\nimport argparse\nimport warnings\nfrom pathlib import Path\nimport transformers\nprint(transformers.__version__)\nfrom transformers import pipeline, AutoModelForTokenClassification, AutoTokenizer\n\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n\nfiles = list(glob.glob(DATASET_PATH + '/' + '*.wav'))\nfiles += list(glob.glob(DATASET_PATH + '/' + '*.mp3'))\nfiles.sort()\n\npipe = pipeline(task=\"automatic-speech-recognition\",\n                model=MODEL,\n                tokenizer=MODEL,\n                chunk_length_s=CHUNK_LENGTH_S, device=0, batch_size=BATCH_SIZE)\npipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\nprint(\"model loaded!\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T07:50:12.937337Z","iopub.execute_input":"2024-04-23T07:50:12.937671Z","iopub.status.idle":"2024-04-23T07:50:59.122858Z","shell.execute_reply.started":"2024-04-23T07:50:12.937642Z","shell.execute_reply":"2024-04-23T07:50:59.121932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def fix_repetition(text, max_count):\n#     uniq_word_counter = {}\n#     words = text.split()\n#     for word in text.split():\n#         if word not in uniq_word_counter:\n#             uniq_word_counter[word] = 1\n#         else:\n#             uniq_word_counter[word] += 1\n\n#     for word, count in uniq_word_counter.items():\n#         if count > max_count:\n#             words = [w for w in words if w != word]\n#     text = \" \".join(words)\n#     return text","metadata":{"execution":{"iopub.status.busy":"2024-04-23T07:50:59.124088Z","iopub.execute_input":"2024-04-23T07:50:59.124380Z","iopub.status.idle":"2024-04-23T07:50:59.128934Z","shell.execute_reply.started":"2024-04-23T07:50:59.124354Z","shell.execute_reply":"2024-04-23T07:50:59.128008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get install ffmpeg\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T07:50:59.130983Z","iopub.execute_input":"2024-04-23T07:50:59.131251Z","iopub.status.idle":"2024-04-23T07:55:04.008481Z","shell.execute_reply.started":"2024-04-23T07:50:59.131221Z","shell.execute_reply":"2024-04-23T07:55:04.007261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport subprocess\nfrom glob import glob\n\ndef convert_videos_to_audio(video_dir, audio_dir):\n    # Make sure the audio directory exists\n    os.makedirs(audio_dir, exist_ok=True)\n    \n    # Find all .mp4 files in the video directory\n    video_files = glob(os.path.join(video_dir, '*.mp4'))\n    \n    # Iterate through each video file\n    for video_file in video_files:\n        # Get the base name of the video file\n        video_name = os.path.basename(video_file)\n        \n        # Remove the file extension to get the audio file name\n        audio_name = os.path.splitext(video_name)[0] + '.mp3'\n        \n        # Construct the paths for input and output files\n        video_path = os.path.join(video_dir, video_name)\n        audio_path = os.path.join(audio_dir, audio_name)\n        \n        # Convert video to audio using ffmpeg\n        try:\n            subprocess.run(['ffmpeg', '-i', video_path, audio_path])\n            print(f\"Converted {video_name} to {audio_name}\")\n        except Exception as e:\n            print(f\"Error converting {video_name}: {str(e)}\")\n\n# Example usage:\nvideo_directory = \"/kaggle/input/kamrul-sir-test-news-audio/kamrul-sir-news-video\"\naudio_directory = \"/kaggle/working/audio\"\nconvert_videos_to_audio(video_directory, audio_directory)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T07:55:04.010386Z","iopub.execute_input":"2024-04-23T07:55:04.010798Z","iopub.status.idle":"2024-04-23T07:55:53.987577Z","shell.execute_reply.started":"2024-04-23T07:55:04.010760Z","shell.execute_reply":"2024-04-23T07:55:53.986589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport json\nfrom glob import glob\n\ndef process_audio_and_save_to_txt(audio_list, pipe, output_folder):\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n\n    for audio_path in audio_list:\n        audio_name = os.path.basename(audio_path)\n        try:\n            # Generate transcripts using the pipe object\n            transcript_dict = pipe(audio_path, generate_kwargs={\"max_length\": 260, \"num_beams\": 4})\n            transcript_text = transcript_dict[\"text\"]\n            \n            # Save transcript to a text file with the same name as the audio file\n            output_path = os.path.join(output_folder, os.path.splitext(audio_name)[0] + '.txt')\n            with open(output_path, 'w', encoding='utf-8') as txt_file:\n                txt_file.write(transcript_text)\n            \n            print(f\"Processed audio: {audio_name}\")\n        except Exception as e:\n            print(f\"Error processing audio {audio_name}: {str(e)}\")\n\n# Example usage:\naudio_directory = \"/kaggle/working/audio\"\naudio_list = glob(os.path.join(audio_directory, '*.mp3'))\n\n# Assuming you have a function or model named \"pipe\" to generate transcripts from audio\n# Replace it with your actual function or model\npipe = pipeline(task=\"automatic-speech-recognition\",\n                model=MODEL,\n                tokenizer=MODEL,\n                chunk_length_s=CHUNK_LENGTH_S, device=0, batch_size=BATCH_SIZE)\npipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\noutput_folder = \"/kaggle/working/transcripts\"\nprocess_audio_and_save_to_txt(audio_list, pipe, output_folder)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T07:55:53.989361Z","iopub.execute_input":"2024-04-23T07:55:53.989919Z","iopub.status.idle":"2024-04-23T08:06:39.501652Z","shell.execute_reply.started":"2024-04-23T07:55:53.989882Z","shell.execute_reply":"2024-04-23T08:06:39.500702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def process_audio_and_save_to_json(audio_list, pipe):\n#     results = []\n#     for audio_path in audio_list:\n#         audio_name = os.path.basename(audio_path)\n#         try:\n#             # Generate transcripts using the pipe object\n#             transcript_dict = pipe(audio_path, generate_kwargs={\"max_length\": 260, \"num_beams\": 4})\n#             transcript_text_unicode = transcript_dict[\"text\"]\n            \n#             # Convert Unicode transcript to human-readable Bengali text\n#             transcript_text = transcript_text_unicode.encode('utf-8').decode('unicode-escape')\n            \n#             results.append({\"id\": audio_name, \"transcript\": transcript_text})\n#             print(f\"Processed audio: {audio_name}\")\n#         except Exception as e:\n#             print(f\"Error processing audio {audio_name}: {str(e)}\")\n    \n#     # Save results to JSON\n#     with open('transcripts.json', 'w', encoding='utf-8') as json_file:\n#         json.dump(results, json_file, ensure_ascii=False, indent=4)\n\n# # Example usage:\n# audio_directory = \"/kaggle/working/audio\"\n# audio_list = glob(os.path.join(audio_directory, '*.mp3'))\n\n# # Assuming you have a function or model named \"pipe\" to generate transcripts from audio\n# # Replace it with your actual function or model\n# pipe = pipeline(task=\"automatic-speech-recognition\",\n#                 model=MODEL,\n#                 tokenizer=MODEL,\n#                 chunk_length_s=CHUNK_LENGTH_S, device=0, batch_size=BATCH_SIZE)\n# pipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\n# process_audio_and_save_to_json(audio_list, pipe)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.502963Z","iopub.execute_input":"2024-04-23T08:06:39.503261Z","iopub.status.idle":"2024-04-23T08:06:39.508849Z","shell.execute_reply.started":"2024-04-23T08:06:39.503234Z","shell.execute_reply":"2024-04-23T08:06:39.507793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# t0 = pipe(\"/kaggle/input/real-news-audio/-   (online-audio-converter.com).wav\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.509903Z","iopub.execute_input":"2024-04-23T08:06:39.510155Z","iopub.status.idle":"2024-04-23T08:06:39.523466Z","shell.execute_reply.started":"2024-04-23T08:06:39.510131Z","shell.execute_reply":"2024-04-23T08:06:39.522737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# t0","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.526241Z","iopub.execute_input":"2024-04-23T08:06:39.526506Z","iopub.status.idle":"2024-04-23T08:06:39.534884Z","shell.execute_reply.started":"2024-04-23T08:06:39.526483Z","shell.execute_reply":"2024-04-23T08:06:39.534134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# t1 = pipe(\"/kaggle/input/audio-dataset-meeting/Sequence 02.mp3\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.538195Z","iopub.execute_input":"2024-04-23T08:06:39.538538Z","iopub.status.idle":"2024-04-23T08:06:39.545900Z","shell.execute_reply.started":"2024-04-23T08:06:39.538514Z","shell.execute_reply":"2024-04-23T08:06:39.544987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# t1","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.547022Z","iopub.execute_input":"2024-04-23T08:06:39.547311Z","iopub.status.idle":"2024-04-23T08:06:39.557965Z","shell.execute_reply.started":"2024-04-23T08:06:39.547286Z","shell.execute_reply":"2024-04-23T08:06:39.557096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# t2 = pipe(\"/kaggle/input/audio-dataset-meeting/Sequence 02_1.mp3\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.559171Z","iopub.execute_input":"2024-04-23T08:06:39.559889Z","iopub.status.idle":"2024-04-23T08:06:39.570309Z","shell.execute_reply.started":"2024-04-23T08:06:39.559856Z","shell.execute_reply":"2024-04-23T08:06:39.569223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# texts0 = pipe(\"/kaggle/input/very-large-audio-3/output_file_3.mp3\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.571690Z","iopub.execute_input":"2024-04-23T08:06:39.571980Z","iopub.status.idle":"2024-04-23T08:06:39.582986Z","shell.execute_reply.started":"2024-04-23T08:06:39.571956Z","shell.execute_reply":"2024-04-23T08:06:39.582171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts0","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.584277Z","iopub.execute_input":"2024-04-23T08:06:39.584585Z","iopub.status.idle":"2024-04-23T08:06:39.592664Z","shell.execute_reply.started":"2024-04-23T08:06:39.584529Z","shell.execute_reply":"2024-04-23T08:06:39.591932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# texts1 = pipe(\"/kaggle/input/audio2/output_file_2.mp3\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.593628Z","iopub.execute_input":"2024-04-23T08:06:39.593894Z","iopub.status.idle":"2024-04-23T08:06:39.602441Z","shell.execute_reply.started":"2024-04-23T08:06:39.593870Z","shell.execute_reply":"2024-04-23T08:06:39.601512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts1","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.603373Z","iopub.execute_input":"2024-04-23T08:06:39.603627Z","iopub.status.idle":"2024-04-23T08:06:39.611058Z","shell.execute_reply.started":"2024-04-23T08:06:39.603604Z","shell.execute_reply":"2024-04-23T08:06:39.610060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# texts2 = pipe(\"/kaggle/input/audio1/output_file.mp3\", generate_kwargs={\"max_length\": 260, \"num_beams\": 4})","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.612237Z","iopub.execute_input":"2024-04-23T08:06:39.612585Z","iopub.status.idle":"2024-04-23T08:06:39.619626Z","shell.execute_reply.started":"2024-04-23T08:06:39.612533Z","shell.execute_reply":"2024-04-23T08:06:39.618849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts2","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.620525Z","iopub.execute_input":"2024-04-23T08:06:39.620820Z","iopub.status.idle":"2024-04-23T08:06:39.629637Z","shell.execute_reply.started":"2024-04-23T08:06:39.620795Z","shell.execute_reply":"2024-04-23T08:06:39.628867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if ENABLE_BEAM:\n#     texts = pipe(files, generate_kwargs={\"max_length\": 260, \"num_beams\": 4})\n# else:\n#     texts = pipe(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.630588Z","iopub.execute_input":"2024-04-23T08:06:39.630868Z","iopub.status.idle":"2024-04-23T08:06:39.638371Z","shell.execute_reply.started":"2024-04-23T08:06:39.630844Z","shell.execute_reply":"2024-04-23T08:06:39.637678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.639350Z","iopub.execute_input":"2024-04-23T08:06:39.639624Z","iopub.status.idle":"2024-04-23T08:06:39.648000Z","shell.execute_reply.started":"2024-04-23T08:06:39.639600Z","shell.execute_reply":"2024-04-23T08:06:39.647313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del pipe\n# import torch\n# models = [\n#     AutoModelForTokenClassification.from_pretrained(f).eval().cuda() for f in PUNCT_MODELS\n# ]\n# tokenizer = AutoTokenizer.from_pretrained(PUNCT_MODELS[0])\n# def punctuate(text):\n#     input_ids = tokenizer(text).input_ids\n#     with torch.no_grad():\n#         model = models[0]\n#         logits = torch.nn.functional.softmax(\n#             model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n#             dim=1).cpu()\n#         for model in models[1:]:\n#             logits += torch.nn.functional.softmax(\n#                 model(input_ids=torch.LongTensor([input_ids]).cuda()).logits[0, 1:-1],\n#                 dim=1).cpu()\n#         logits = logits / len(models)\n#         logits *= torch.FloatTensor(PUNCT_WEIGHTS)\n#         label_ids = torch.argmax(logits, dim=-1)\n\n#         tokens = tokenizer(text, add_special_tokens=False).input_ids\n#         punct_text = \"\"\n#         for index, token in enumerate(tokens):\n#             token_str = tokenizer.decode(token)\n#             if '##' not in token_str:\n#                 punct_text += \" \" + token_str\n#             else:\n#                 punct_text += token_str[2:]\n#             punct_text += ['', '।', ',', '?'][label_ids[index].item()]\n\n#     punct_text = punct_text.strip()\n#     return punct_text","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.648929Z","iopub.execute_input":"2024-04-23T08:06:39.649170Z","iopub.status.idle":"2024-04-23T08:06:39.657849Z","shell.execute_reply.started":"2024-04-23T08:06:39.649148Z","shell.execute_reply":"2024-04-23T08:06:39.656953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred=texts['text']","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.658951Z","iopub.execute_input":"2024-04-23T08:06:39.659793Z","iopub.status.idle":"2024-04-23T08:06:39.668508Z","shell.execute_reply.started":"2024-04-23T08:06:39.659767Z","shell.execute_reply":"2024-04-23T08:06:39.667716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pred[:500]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.669672Z","iopub.execute_input":"2024-04-23T08:06:39.670226Z","iopub.status.idle":"2024-04-23T08:06:39.681453Z","shell.execute_reply.started":"2024-04-23T08:06:39.670194Z","shell.execute_reply":"2024-04-23T08:06:39.680636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# punctuate(pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.682692Z","iopub.execute_input":"2024-04-23T08:06:39.683027Z","iopub.status.idle":"2024-04-23T08:06:39.690914Z","shell.execute_reply.started":"2024-04-23T08:06:39.682995Z","shell.execute_reply":"2024-04-23T08:06:39.690123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions = []\n# with open(\"submission.csv\", 'wt', encoding=\"utf8\") as csvfile:\n#     writer = csv.writer(csvfile)\n#     writer.writerow(['id', 'sentence'])\n#     for f, text in zip(files, texts):\n#         file_id = Path(f).stem\n#         pred = text['text'].strip()\n#         pred = fix_repetition(pred, max_count=8)\n#         pred = punctuate(pred)\n#         if pred[-1] not in ['।', '?', ',']:\n#             pred = pred + '।'\n#         # print(i, file_id, pred)\n#         prediction = [file_id, pred]\n#         writer.writerow(prediction)\n#         predictions.append(prediction)\n# print(\"inference finished!\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T08:06:39.691820Z","iopub.execute_input":"2024-04-23T08:06:39.692063Z","iopub.status.idle":"2024-04-23T08:06:39.699608Z","shell.execute_reply.started":"2024-04-23T08:06:39.692040Z","shell.execute_reply":"2024-04-23T08:06:39.698931Z"},"trusted":true},"execution_count":null,"outputs":[]}]}