{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-02T12:13:27.999737Z","iopub.execute_input":"2023-10-02T12:13:28.000179Z","iopub.status.idle":"2023-10-02T12:13:28.007797Z","shell.execute_reply.started":"2023-10-02T12:13:28.000145Z","shell.execute_reply":"2023-10-02T12:13:28.005786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Source\n- https://github.com/openai/whisper\n- https://deepgram.com/learn/guide-deepspeech-speech-to-text\n- https://www.assemblyai.com/blog/deepspeech-for-dummies-a-tutorial-and-overview-part-1/\n- https://deepspeech.readthedocs.io/en/r0.9/\n- https://github.com/alphacep/vosk-api/blob/v0.3.32/python/example/test_simple.py\n- https://alphacephei.com/vosk/\n- https://alphacephei.com/vosk/install","metadata":{}},{"cell_type":"code","source":"# !pip install google-cloud-speech","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:06:58.414282Z","iopub.execute_input":"2023-10-02T12:06:58.414719Z","iopub.status.idle":"2023-10-02T12:06:58.420791Z","shell.execute_reply.started":"2023-10-02T12:06:58.414699Z","shell.execute_reply":"2023-10-02T12:06:58.419313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install SpeechRecognition","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:11:13.300582Z","iopub.execute_input":"2023-10-02T12:11:13.301226Z","iopub.status.idle":"2023-10-02T12:11:23.471928Z","shell.execute_reply.started":"2023-10-02T12:11:13.301167Z","shell.execute_reply":"2023-10-02T12:11:23.469445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install google-cloud-speech --upgrade","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:07:09.453721Z","iopub.execute_input":"2023-10-02T12:07:09.454054Z","iopub.status.idle":"2023-10-02T12:07:09.458590Z","shell.execute_reply.started":"2023-10-02T12:07:09.454033Z","shell.execute_reply":"2023-10-02T12:07:09.457803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import speech_recognition as sr\n# import google.cloud.speech as speech\n\n# # Load the audio file\n# audio_file = sr.AudioFile('/kaggle/input/bengaliai-speech/train_mp3s/00004ed6485a.mp3')\n\n# # Create a recognizer object\n# r = sr.Recognizer()\n\n# # Create a speaker identification object\n# speaker_id = speech.SpeakerIdentification()\n\n# # Identify the speaker in the audio recording\n# speaker_id.identify(audio_file)\n\n# # Adapt the model to the speaker\n# r.adapt_to_speaker(speaker_id)\n\n# # Transcribe the audio using the Google Cloud Speech API\n# with audio_file as source:\n#     audio = r.record(source)\n\n# # Recognize the speech\n# transcript = r.recognize_google_cloud(audio, language_code='en-US')\n\n# # Print the transcript\n# print(transcript)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:07:09.459573Z","iopub.execute_input":"2023-10-02T12:07:09.459803Z","iopub.status.idle":"2023-10-02T12:07:09.474281Z","shell.execute_reply.started":"2023-10-02T12:07:09.459784Z","shell.execute_reply":"2023-10-02T12:07:09.473145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import speech_recognition as sr\nfrom os import path\nfrom pydub import AudioSegment\n\nAUDIO_FILE = \"/kaggle/input/bengaliai-speech/train_mp3s/000028220ab3.mp3\"\n\n# Convert MP3 to WAV\nsound = AudioSegment.from_mp3(AUDIO_FILE)\nsound.export(\"/kaggle/working/recording.wav\", format=\"wav\")\n\n# Create a speech recognizer object\nr = sr.Recognizer()\n\n# Open the WAV file and listen for the data\nwith sr.AudioFile(\"recording.wav\") as source:\n    audio = r.record(source)\n\n# Recognize the speech from the audio data\ntext = r.recognize_google(audio)\n\n# Print the transcription to the console\nprint(\"Transcription: \" + text)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:13:09.871403Z","iopub.execute_input":"2023-10-02T12:13:09.871937Z","iopub.status.idle":"2023-10-02T12:13:11.755169Z","shell.execute_reply.started":"2023-10-02T12:13:09.871900Z","shell.execute_reply":"2023-10-02T12:13:11.753577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import speech_recognition as sr\nfrom os import path\nfrom pydub import AudioSegment\n\n# Get the list of all audio files in the directory\naudio_files = [f for f in os.listdir() if f.endswith('.mp3' or '.wav')]\n\n# Create a speech recognizer object\nr = sr.Recognizer()\n\n# Iterate over the audio files and transcribe each one\nfor audio_file in audio_files:\n\n    # Convert MP3 to WAV if necessary\n    if audio_file.endswith('.mp3'):\n        sound = AudioSegment.from_mp3(audio_file)\n        sound.export(\"recording.wav\", format=\"wav\")\n        audio_file = \"recording.wav\"\n\n    # Open the WAV file and listen for the data\n    with sr.AudioFile(audio_file) as source:\n        audio = r.record(source)\n\n    # Recognize the speech from the audio data\n    try:\n        text = r.recognize_google(audio)\n    except sr.UnknownValueError:\n        print(f\"Could not recognize speech in {audio_file}.\")\n    except sr.RequestError as e:\n        print(f\"Could not transcribe audio file {audio_file}: {e}\")\n    else:\n        # Print the transcription to the console\n        if text is not None:\n            print(f\"Transcription of {audio_file}: {text}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:13:31.824848Z","iopub.execute_input":"2023-10-02T12:13:31.825310Z","iopub.status.idle":"2023-10-02T12:13:31.834031Z","shell.execute_reply.started":"2023-10-02T12:13:31.825275Z","shell.execute_reply":"2023-10-02T12:13:31.833145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import speech_recognition as sr\nimport os\n\n# Define a function to transcribe an audio file\ndef transcribe_audio_file(audio_file_path):\n  \"\"\"Transcribes an audio file to text.\n\n  Args:\n    audio_file_path: The path to the audio file.\n\n  Returns:\n    A string containing the transcribed text.\n  \"\"\"\n\n  # Create a speech recognition object\n  r = sr.Recognizer()\n\n  # Open the audio file\n  with sr.AudioFile(audio_file_path) as source:\n    # Read the entire audio file\n    audio = r.record(source)\n\n  # Recognize the speech in the audio file\n  text = r.recognize_google(audio)\n\n  return text\n\n# Get a list of all the audio files in the current directory\naudio_file_paths = os.listdir(\".\")\n\n# Transcribe each audio file and print the transcribed text to the console\nfor audio_file_path in audio_file_paths:\n  if audio_file_path.endswith(\".wav\"):\n    transcribed_text = transcribe_audio_file(audio_file_path)\n    print(transcribed_text)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:13:35.029776Z","iopub.execute_input":"2023-10-02T12:13:35.030271Z","iopub.status.idle":"2023-10-02T12:13:35.376718Z","shell.execute_reply.started":"2023-10-02T12:13:35.030230Z","shell.execute_reply":"2023-10-02T12:13:35.373908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install whisper\n# !pip install whisper --upgrade\n# !pip install -U openai-whisper\n# !pip install ffmpeg\n# !pip install ffmpeg-python","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:07:11.842913Z","iopub.execute_input":"2023-10-02T12:07:11.844112Z","iopub.status.idle":"2023-10-02T12:07:11.848990Z","shell.execute_reply.started":"2023-10-02T12:07:11.844069Z","shell.execute_reply":"2023-10-02T12:07:11.847585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import whisper\n\n# model = whisper.load_model(\"base\")\n\n# # load audio and pad/trim it to fit 30 seconds\n# audio = whisper.load_audio(\"/kaggle/input/bengaliai-speech/train_mp3s/000024b3d810.mp3\")\n# audio = whisper.pad_or_trim(audio)\n\n# # make log-Mel spectrogram and move to the same device as the model\n# mel = whisper.log_mel_spectrogram(audio).to(model.device)\n\n# # detect the spoken language\n# _, probs = model.detect_language(mel)\n# print(f\"Detected language: {max(probs, key=probs.get)}\")\n\n# # decode the audio\n# options = whisper.DecodingOptions()\n# result = whisper.decode(model, mel, options)\n\n# # print the recognized text\n# print(result.text)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:07:11.852880Z","iopub.execute_input":"2023-10-02T12:07:11.853410Z","iopub.status.idle":"2023-10-02T12:07:11.866079Z","shell.execute_reply.started":"2023-10-02T12:07:11.853381Z","shell.execute_reply":"2023-10-02T12:07:11.865077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install whisper","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:13:43.470271Z","iopub.execute_input":"2023-10-02T12:13:43.470660Z","iopub.status.idle":"2023-10-02T12:13:51.515585Z","shell.execute_reply.started":"2023-10-02T12:13:43.470632Z","shell.execute_reply":"2023-10-02T12:13:51.513146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transcribe all of the MP3 files in the current directory and save the transcriptions to corresponding TXT files\nimport whisper\nimport os\n\n# Create a list of all the MP3 files in the current directory\nmp3_files = [file for file in os.listdir() if file.endswith(\".mp3\")]\n\n# Transcribe each MP3 file and save the transcription to a corresponding TXT file\nfor mp3_file in mp3_files:\n  # Load the audio file\n  audio = whisper.load_audio(mp3_file)\n\n  # Transcribe the audio\n  transcription = whisper.transcribe(audio)\n\n  # Save the transcription to a TXT file\n  txt_file = mp3_file.replace(\".mp3\", \".txt\")\n  with open(txt_file, \"w\") as f:\n    f.write(transcription)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:13:58.570753Z","iopub.execute_input":"2023-10-02T12:13:58.571220Z","iopub.status.idle":"2023-10-02T12:13:58.612787Z","shell.execute_reply.started":"2023-10-02T12:13:58.571148Z","shell.execute_reply":"2023-10-02T12:13:58.610378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir whisper","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:07:21.857383Z","iopub.execute_input":"2023-10-02T12:07:21.857792Z","iopub.status.idle":"2023-10-02T12:07:22.110885Z","shell.execute_reply.started":"2023-10-02T12:07:21.857760Z","shell.execute_reply":"2023-10-02T12:07:22.108767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download Whisper model\n! curl -L https://cdn.openai.com/whisper/whisper-large-20230922.gz -o whisper.gz","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:07:22.112751Z","iopub.execute_input":"2023-10-02T12:07:22.113128Z","iopub.status.idle":"2023-10-02T12:07:22.898855Z","shell.execute_reply.started":"2023-10-02T12:07:22.113097Z","shell.execute_reply":"2023-10-02T12:07:22.897145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download Mozilla Deepspeech Model\n!wget https://github.com/mozilla/DeepSpeech/releases/download/v0.9.3/deepspeech-0.9.3-models.pbmm\n!wget https://github.com/mozilla/DeepSpeech/releases/download/v0.9.3/deepspeech-0.9.3-models.scorer","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:07:22.901528Z","iopub.execute_input":"2023-10-02T12:07:22.901985Z","iopub.status.idle":"2023-10-02T12:07:44.134966Z","shell.execute_reply.started":"2023-10-02T12:07:22.901953Z","shell.execute_reply":"2023-10-02T12:07:44.133620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget https://github.com/mozilla/DeepSpeech/releases/download/v0.8.2/deepspeech-0.8.2-models.scorer","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-02T12:07:44.136425Z","iopub.execute_input":"2023-10-02T12:07:44.136774Z","iopub.status.idle":"2023-10-02T12:08:02.767734Z","shell.execute_reply.started":"2023-10-02T12:07:44.136748Z","shell.execute_reply":"2023-10-02T12:08:02.766001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf\n# import tensorflow_io as tfio\n\n# # Load the MP3 audio file.\n# audio_file = tfio.audio.AudioIOTensor('/kaggle/input/bengaliai-speech/train_mp3s/000024b3d810.mp3')\n\n# # Resample the audio to 16kHz.\n# audio_data = tf.squeeze(tfio.audio.resample(audio_file, rate_in=44100, rate_out=16000), axis=-1)\n\n# # Normalize the audio volume.\n# audio_data = tf.audio.normalize(audio_data)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:08:02.769363Z","iopub.execute_input":"2023-10-02T12:08:02.769760Z","iopub.status.idle":"2023-10-02T12:08:02.777594Z","shell.execute_reply.started":"2023-10-02T12:08:02.769723Z","shell.execute_reply":"2023-10-02T12:08:02.775016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_io as tfio\n\n# Load the DeepSpeech acoustic model.\nacoustic_model = tf.io.gfile.GFile('/kaggle/working/deepspeech-0.9.3-models.pbmm', 'rb').read()\n\n# Load the DeepSpeech language model.\nlanguage_model = tf.io.gfile.GFile('/kaggle/working/deepspeech-0.9.3-models.scorer', 'rb').read()\n\n# # Create a DeepSpeech decoder.\n# decoder = tf.contrib.ctc.CTCDecoder(acoustic_model, language_model)\n\n# # Transcribe the audio data and get the decoded text.\n# decoded_text = decoder.decode(audio_data)\n\n# # Print the decoded text.\n# print(decoded_text)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:08:02.778722Z","iopub.execute_input":"2023-10-02T12:08:02.778991Z","iopub.status.idle":"2023-10-02T12:08:18.152438Z","shell.execute_reply.started":"2023-10-02T12:08:02.778972Z","shell.execute_reply":"2023-10-02T12:08:18.150122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers\n\n\n# Load the pre-trained speech recognition model\nmodel = transformers.AutoModelForCTC.from_pretrained(\"facebook/wav2vec2-base\")","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:08:18.153646Z","iopub.execute_input":"2023-10-02T12:08:18.154229Z","iopub.status.idle":"2023-10-02T12:08:28.005822Z","shell.execute_reply.started":"2023-10-02T12:08:18.154209Z","shell.execute_reply":"2023-10-02T12:08:28.003601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers\nimport torchaudio\n\n# Load the Wav2Vec 2.0 model\nmodel = transformers.Wav2Vec2ForCTC.from_pretrained(\"facebook/wav2vec2-base-960h\")","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:08:28.007649Z","iopub.execute_input":"2023-10-02T12:08:28.008001Z","iopub.status.idle":"2023-10-02T12:08:31.670114Z","shell.execute_reply.started":"2023-10-02T12:08:28.007971Z","shell.execute_reply":"2023-10-02T12:08:31.667733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchaudio\n# Prepare the input audio\naudio, sample_rate = torchaudio.load(\"/kaggle/input/bengaliai-speech/examples/Audiobook.wav\")\ninput_values = transformers.Wav2Vec2Tokenizer.from_pretrained(\"facebook/wav2vec2-base-960h\")(torch.flatten(audio), return_tensors=\"pt\").input_values\n\n# Transcribe the audio\nlogits = model(input_values).logits\ntranscript = transformers.Wav2Vec2Processor.from_pretrained(\"facebook/wav2vec2-base-960h\").decode(logits)","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:08:31.672843Z","iopub.execute_input":"2023-10-02T12:08:31.673275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import speech_recognition as sr\n\n# speech recognizer object\nr = sr.Recognizer()\n\n# audio file\nwith sr.AudioFile(\"/kaggle/input/bengaliai-speech/examples/Bangladeshi TV Drama.wav\") as source:\n    # Listen to audio and transcribe it\n    audio = r.listen(source)\n\n    # Try to recognize the speech\n    try:\n        # Recognize the speech using the Google Speech Recognition API\n        transcript = r.recognize_google(audio)\n\n        print(transcript)\n\n    except sr.UnknownValueError:\n        # If the speech recognition service could not understand the speech, print a message\n        print(\"Could not understand the speech\")\n    except sr.RequestError as e:\n        # If there was a problem with the speech recognition service, print the error message\n        print(e)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:25:20.073040Z","iopub.execute_input":"2023-10-02T12:25:20.073815Z","iopub.status.idle":"2023-10-02T12:25:42.385430Z","shell.execute_reply.started":"2023-10-02T12:25:20.073760Z","shell.execute_reply":"2023-10-02T12:25:42.382798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyaudio\n!pip install vosk","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:26:26.495382Z","iopub.execute_input":"2023-10-02T12:26:26.495843Z","iopub.status.idle":"2023-10-02T12:26:48.307833Z","shell.execute_reply.started":"2023-10-02T12:26:26.495809Z","shell.execute_reply":"2023-10-02T12:26:48.305361Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import vosk\nimport wave\nimport json","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:40:26.500019Z","iopub.execute_input":"2023-10-02T12:40:26.500511Z","iopub.status.idle":"2023-10-02T12:40:26.506285Z","shell.execute_reply.started":"2023-10-02T12:40:26.500487Z","shell.execute_reply":"2023-10-02T12:40:26.504857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/alphacep/vosk-api\n!cd vosk-api/python/example\n!python3 ./test_simple.py \"../input/bengaliai-speech/examples/Audiobook.wav\"","metadata":{"execution":{"iopub.status.busy":"2023-10-02T12:41:17.050540Z","iopub.execute_input":"2023-10-02T12:41:17.050969Z","iopub.status.idle":"2023-10-02T12:41:17.918671Z","shell.execute_reply.started":"2023-10-02T12:41:17.050936Z","shell.execute_reply":"2023-10-02T12:41:17.916411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}