{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  **Bengali.AI Speech Recognition : Detailed EDA**","metadata":{"execution":{"iopub.status.busy":"2023-07-25T17:01:14.942041Z","iopub.execute_input":"2023-07-25T17:01:14.942407Z","iopub.status.idle":"2023-07-25T17:01:14.949650Z","shell.execute_reply.started":"2023-07-25T17:01:14.942378Z","shell.execute_reply":"2023-07-25T17:01:14.948626Z"}}},{"cell_type":"code","source":"# importing libraries\n\nimport os\nimport librosa\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport soundfile as sf","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:19:41.049324Z","iopub.execute_input":"2023-07-25T20:19:41.050466Z","iopub.status.idle":"2023-07-25T20:19:41.137094Z","shell.execute_reply.started":"2023-07-25T20:19:41.050419Z","shell.execute_reply":"2023-07-25T20:19:41.135695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# path \nAUDIO_DIR_PATH = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\nTRANSCRIPTION_DIR_PATH = \"/kaggle/input/bengaliai-speech/train.csv\"\nTEST_DIR_PATH = \"/kaggle/input/bengaliai-speech/test_mp3s/\"\nEXAMPLE_AUDIO_DIR_PATH = \"/kaggle/input/bengaliai-speech/examples/\"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:19:41.139612Z","iopub.execute_input":"2023-07-25T20:19:41.140102Z","iopub.status.idle":"2023-07-25T20:19:41.146177Z","shell.execute_reply.started":"2023-07-25T20:19:41.140056Z","shell.execute_reply":"2023-07-25T20:19:41.145080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transcription = pd.read_csv(TRANSCRIPTION_DIR_PATH)\ndisplay(transcription.head(10))","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:19:41.147590Z","iopub.execute_input":"2023-07-25T20:19:41.148549Z","iopub.status.idle":"2023-07-25T20:19:47.601665Z","shell.execute_reply.started":"2023-07-25T20:19:41.148515Z","shell.execute_reply":"2023-07-25T20:19:47.600723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading the audio file\nsignal, sr = librosa.load(AUDIO_DIR_PATH + \"000005f3362c.mp3\")\n\n# Calculate the duration in seconds\nduration = len(signal) / sr\n\n# Print the duration\nprint(\"Duration of audio:\", duration, \"seconds\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:19:47.603953Z","iopub.execute_input":"2023-07-25T20:19:47.604455Z","iopub.status.idle":"2023-07-25T20:20:00.708772Z","shell.execute_reply.started":"2023-07-25T20:19:47.604421Z","shell.execute_reply":"2023-07-25T20:20:00.707455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the audio length via funtion\n\ndef get_audio_durations(audio_dir):\n    audio_files = sorted(os.listdir(audio_dir))\n    durations = []\n\n    for audio_file in audio_files:\n        audio_file_path = os.path.join(audio_dir, audio_file)\n        signal, sr = librosa.load(audio_file_path)\n        duration = len(signal) / sr\n        durations.append(duration)\n\n    return durations\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.710357Z","iopub.execute_input":"2023-07-25T20:20:00.711125Z","iopub.status.idle":"2023-07-25T20:20:00.720264Z","shell.execute_reply.started":"2023-07-25T20:20:00.711082Z","shell.execute_reply":"2023-07-25T20:20:00.718785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot audio length\n\ndef plot_audio_lengths_bar(audio_dir):\n    durations = get_audio_durations(audio_dir)\n    audio_files = sorted(os.listdir(audio_dir))\n\n    plt.figure(figsize=(10, 6))\n    plt.bar(range(len(audio_files)), durations)\n    plt.xticks(range(len(audio_files)), audio_files, rotation=45, ha=\"right\")\n    plt.xlabel(\"Audio Files\")\n    plt.ylabel(\"Duration (seconds)\")\n    plt.title(\"Audio File Durations\")\n    plt.tight_layout()\n    plt.show()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.722047Z","iopub.execute_input":"2023-07-25T20:20:00.722585Z","iopub.status.idle":"2023-07-25T20:20:00.736896Z","shell.execute_reply.started":"2023-07-25T20:20:00.722539Z","shell.execute_reply":"2023-07-25T20:20:00.735491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_audio_lengths_bar(AUDIO_DIR_PATH) # it will take time","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.738366Z","iopub.execute_input":"2023-07-25T20:20:00.738962Z","iopub.status.idle":"2023-07-25T20:20:00.748799Z","shell.execute_reply.started":"2023-07-25T20:20:00.738931Z","shell.execute_reply":"2023-07-25T20:20:00.747265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the audio","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pyplot as plt\nimport librosa.display","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.752240Z","iopub.execute_input":"2023-07-25T20:20:00.753651Z","iopub.status.idle":"2023-07-25T20:20:00.765032Z","shell.execute_reply.started":"2023-07-25T20:20:00.753603Z","shell.execute_reply":"2023-07-25T20:20:00.763603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audio analysis function\n\ndef plot_audio_analysis(audio_file_path):\n    # Load the audio file with duration=10 seconds\n    signal, sr = librosa.load(audio_file_path, duration=10)\n\n    # Create a figure with three subplots\n    fig, ax = plt.subplots(nrows=3, sharex=True, sharey=True, figsize=(10, 8))\n\n    # Plot the waveform of the monophonic audio\n    librosa.display.waveshow(signal, sr=sr, ax=ax[0])\n    ax[0].set(title='Monophonic')\n    ax[0].label_outer()\n\n    # Plot the waveform of the stereo audio\n    stereo_signal, sr = librosa.load(audio_file_path, mono=False, duration=10)\n    librosa.display.waveshow(stereo_signal, sr=sr, ax=ax[1])\n    ax[1].set(title='Stereo')\n    ax[1].label_outer()\n\n    # Calculate and plot the harmonic and percussive components\n    y_harm, y_perc = librosa.effects.hpss(signal)\n    librosa.display.waveshow(y_harm, sr=sr, alpha=0.25, ax=ax[2])\n    librosa.display.waveshow(y_perc, sr=sr, color='r', alpha=0.5, ax=ax[2])\n    ax[2].set(title='Harmonic + Percussive')\n\n    # Adjust layout and display the figure\n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.766533Z","iopub.execute_input":"2023-07-25T20:20:00.766875Z","iopub.status.idle":"2023-07-25T20:20:00.779653Z","shell.execute_reply.started":"2023-07-25T20:20:00.766845Z","shell.execute_reply":"2023-07-25T20:20:00.778686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for single entry audio analysis\n\nplot_audio_analysis(AUDIO_DIR_PATH + \"000005f3362c.mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:00.783877Z","iopub.execute_input":"2023-07-25T20:20:00.784758Z","iopub.status.idle":"2023-07-25T20:20:05.314816Z","shell.execute_reply.started":"2023-07-25T20:20:00.784721Z","shell.execute_reply":"2023-07-25T20:20:05.313881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# showcasing waveform\n\ndef show_audio_waveform(audio_file_path):\n    # Load the audio file\n    signal, sr = librosa.load(audio_file_path)\n\n    # Plot the waveform of the audio signal\n    plt.figure(figsize=(10, 4))\n    librosa.display.waveshow(signal, sr=sr)\n    plt.xlabel(\"Time\")\n    plt.ylabel(\"Amplitude\")\n    plt.title(\"Audio Signal Waveform\")\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:05.315929Z","iopub.execute_input":"2023-07-25T20:20:05.316317Z","iopub.status.idle":"2023-07-25T20:20:05.323869Z","shell.execute_reply.started":"2023-07-25T20:20:05.316286Z","shell.execute_reply":"2023-07-25T20:20:05.322554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for single entry audio signal waveform\n\nshow_audio_waveform(AUDIO_DIR_PATH + \"000005f3362c.mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:05.325121Z","iopub.execute_input":"2023-07-25T20:20:05.326049Z","iopub.status.idle":"2023-07-25T20:20:06.008513Z","shell.execute_reply.started":"2023-07-25T20:20:05.326006Z","shell.execute_reply":"2023-07-25T20:20:06.007156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# showcasing spectogram\n\ndef plot_spectrogram(audio_file_path):\n    # Load the audio file\n    signal, sr = librosa.load(audio_file_path)\n\n    # Compute the spectrogram using Short-Time Fourier Transform (STFT)\n    X = librosa.stft(signal)\n    Xdb = librosa.amplitude_to_db(abs(X))\n\n    # Plot the spectrogram\n    plt.figure(figsize=(14, 5))\n    librosa.display.specshow(Xdb, sr=sr, x_axis='time', y_axis='hz')\n    plt.colorbar(format='%+2.0f dB')\n    plt.xlabel(\"Time\")\n    plt.ylabel(\"Frequency (Hz)\")\n    plt.title(\"Spectrogram\")\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:06.010629Z","iopub.execute_input":"2023-07-25T20:20:06.011000Z","iopub.status.idle":"2023-07-25T20:20:06.018968Z","shell.execute_reply.started":"2023-07-25T20:20:06.010969Z","shell.execute_reply":"2023-07-25T20:20:06.017628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for single entry sepectrogram\n\nplot_spectrogram(AUDIO_DIR_PATH + \"000005f3362c.mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:06.020527Z","iopub.execute_input":"2023-07-25T20:20:06.020867Z","iopub.status.idle":"2023-07-25T20:20:06.600807Z","shell.execute_reply.started":"2023-07-25T20:20:06.020837Z","shell.execute_reply":"2023-07-25T20:20:06.599518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# listening the audio\n\nimport IPython.display as ipd\nipd.Audio(AUDIO_DIR_PATH+ \"000005f3362c.mp3\")","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:06.602239Z","iopub.execute_input":"2023-07-25T20:20:06.602611Z","iopub.status.idle":"2023-07-25T20:20:06.614965Z","shell.execute_reply.started":"2023-07-25T20:20:06.602577Z","shell.execute_reply":"2023-07-25T20:20:06.613893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \ndef play_audio_with_sentence(audio_file_path, csv_file_path):\n    # Play the audio\n    ipd.display(ipd.Audio(audio_file_path))\n\n    # Get the audio file name without the extension\n    audio_file_name = audio_file_path.split('/')[-1].split('.')[0]\n\n    # Load the CSV file containing audio file names and sentences\n    df = pd.read_csv(csv_file_path)\n\n    # Search for an exact match of the audio file name in the CSV file\n    get_matched_sentence = df.loc[df['id'] == audio_file_name, 'sentence'].iloc[0]\n\n    # If a match is found, print the sentence\n    if not pd.isna(get_matched_sentence):\n        print(\"Sentence for the audio:\")\n        print(get_matched_sentence)\n    else:\n        print(\"No matching sentence found for the audio.\")","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:06.616619Z","iopub.execute_input":"2023-07-25T20:20:06.616950Z","iopub.status.idle":"2023-07-25T20:20:06.625678Z","shell.execute_reply.started":"2023-07-25T20:20:06.616922Z","shell.execute_reply":"2023-07-25T20:20:06.624474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play_audio_with_sentence(AUDIO_DIR_PATH + \"00004ed6485a.mp3\", TRANSCRIPTION_DIR_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:06.627376Z","iopub.execute_input":"2023-07-25T20:20:06.627734Z","iopub.status.idle":"2023-07-25T20:20:11.127757Z","shell.execute_reply.started":"2023-07-25T20:20:06.627703Z","shell.execute_reply":"2023-07-25T20:20:11.126599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nThis EDA aims to provide a preliminary understanding of the dataset and should be considered as a starting point for more in-depth analyses and modeling.\nThe insights gained here can be valuable for future decision-making and exploration of Dataset Speech Recognition. It's my submission for EDA. I will update this. Any suggestion will be highly appreciated.\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:20:11.129390Z","iopub.execute_input":"2023-07-25T20:20:11.129850Z","iopub.status.idle":"2023-07-25T20:20:11.137353Z","shell.execute_reply.started":"2023-07-25T20:20:11.129801Z","shell.execute_reply":"2023-07-25T20:20:11.136265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}