{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1.Imports and introduction","metadata":{}},{"cell_type":"code","source":"!pip install tensorflow-io==0.32.0 ","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:29:59.437697Z","iopub.execute_input":"2023-08-06T14:29:59.438820Z","iopub.status.idle":"2023-08-06T14:30:16.035143Z","shell.execute_reply.started":"2023-08-06T14:29:59.438771Z","shell.execute_reply":"2023-08-06T14:30:16.033979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport librosa\nimport numpy as np\nimport IPython.display as ipd\nimport torchaudio\nimport librosa.display\nimport matplotlib.pyplot as plt\nimport random\nimport string\nimport nltk\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import PorterStemmer\nfrom collections import Counter\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.decomposition import LatentDirichletAllocation\nfrom gensim import models\nimport tensorflow as tf\nfrom tensorflow.keras import models, layers, callbacks, Sequential\nfrom tensorflow.keras.layers.experimental import preprocessing\nfrom tqdm import tqdm\nimport tensorflow_io as tfio\nimport fractions\nimport re\nimport gensim\nfrom gensim.models import Word2Vec\nfrom sklearn.decomposition import PCA\nimport keras","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:30:16.038305Z","iopub.execute_input":"2023-08-06T14:30:16.038942Z","iopub.status.idle":"2023-08-06T14:30:29.068896Z","shell.execute_reply.started":"2023-08-06T14:30:16.038901Z","shell.execute_reply":"2023-08-06T14:30:29.067890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.Directory path and training dataframe","metadata":{}},{"cell_type":"code","source":"# Define the paths to the data directories\ntrain_data_dir = \"/kaggle/input/bengaliai-speech/train_mp3s/\"  \ntest_data_dir = \"/kaggle/input/bengaliai-speech/test_mp3s/\" \ntrain_csv_path = \"/kaggle/input/bengaliai-speech/train.csv\" \ndomains = \"/kaggle/input/bengaliai-speech/examples/\" \n\n# Load the train.csv file using pandas\ntrain_df = pd.read_csv(train_csv_path, index_col = False)\n\n# Preview the first few rows of the DataFrame\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:30:29.070377Z","iopub.execute_input":"2023-08-06T14:30:29.071186Z","iopub.status.idle":"2023-08-06T14:30:33.754059Z","shell.execute_reply.started":"2023-08-06T14:30:29.071147Z","shell.execute_reply":"2023-08-06T14:30:33.752090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3.Examples of Data","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/bengaliai-speech/examples/","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:30:33.757132Z","iopub.execute_input":"2023-08-06T14:30:33.757527Z","iopub.status.idle":"2023-08-06T14:30:34.744706Z","shell.execute_reply.started":"2023-08-06T14:30:33.757491Z","shell.execute_reply":"2023-08-06T14:30:34.743464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4.Creating numpy array from the audio data\nIn this section we are going to put 5 audio data into an array and corresponding transcriptions based on that data. then we will convert the audio data into an numpy array","metadata":{}},{"cell_type":"code","source":"# Load audio files and corresponding transcriptions\naudio_data = []  # List to store audio data\ntranscriptions = []  # List to store corresponding transcriptions\n\nfor idx, row in train_df.head(20).iterrows():\n    audio_file_path = os.path.join(train_data_dir, f\"{row['id']}.mp3\")\n\n    # Load the audio file using librosa\n    audio, sr = librosa.load(audio_file_path, sr=None)\n\n    # Append audio data and transcription to lists\n    audio_data.append(audio)\n    transcriptions.append(row['sentence'])\n    \naudio_data = np.array(audio_data,dtype = 'object')\ntranscriptions = np.array(transcriptions,dtype = 'object')\n\n# Check the shapes of the loaded data\nprint(\"Audio data shape:\", audio_data.shape)\nprint(\"Transcriptions shape:\", transcriptions.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:30:34.748292Z","iopub.execute_input":"2023-08-06T14:30:34.748659Z","iopub.status.idle":"2023-08-06T14:30:43.984773Z","shell.execute_reply.started":"2023-08-06T14:30:34.748628Z","shell.execute_reply":"2023-08-06T14:30:43.983439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5.Checking basic information of the training audio files\nIn this section we will get the total duraion of thefirst 10 training data file, total no of train files and test files.","metadata":{}},{"cell_type":"code","source":"# Get the total number of audio files in the training and test directories\ntrain_audio_files = os.listdir(train_data_dir)\ntest_audio_files = os.listdir(test_data_dir)\n\n# Get the total duration of audio data in the training set (in seconds)\ntrain_total_duration = 0\n# for idx, row in train_df.head(10).iterrows():\n#     audio_file_path = os.path.join(train_data_dir, f\"{row['id']}.mp3\")\n#     audio_info = torchaudio.info(audio_file_path)\n#     duration = audio_info.num_frames / audio_info.sample_rate\n#     train_total_duration += duration\n\n# Get the number of unique domains present in the training data\nunique_domains = train_df['split'].unique()\n\n# Get the total number of samples in the training data\ntotal_samples = train_df.shape[0]\n\n# Print the data summary\nprint(\"Data Summary:\")\nprint(f\"Total number of audio files in the training directory: {len(train_audio_files)}\")\nprint(f\"Total number of audio files in the test directory: {len(test_audio_files)}\")\nprint(f\"Total duration of audio data in the training set (in seconds): {train_total_duration:.2f}\")\nprint(f\"Number of unique domains in the training data: {unique_domains}\")\nprint(f\"Total number of samples in the training data: {total_samples}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:30:43.986192Z","iopub.execute_input":"2023-08-06T14:30:43.987021Z","iopub.status.idle":"2023-08-06T14:31:25.569987Z","shell.execute_reply.started":"2023-08-06T14:30:43.986981Z","shell.execute_reply":"2023-08-06T14:31:25.567929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib\n\nDATASET_SPLIT = (0.7, 0.2, 0.1)\n# data_dir = pathlib.Path(_DATASET_DIRECTORY_PATH)\n# label_names = tf.io.gfile.glob(str(data_dir) + '/*')\n# label_names = [name for name in label_names if not name.endswith('_background_noise_')]\n\n# filenames = tf.io.gfile.glob(str(data_dir) + '/*/*')\n# filenames = [filename for filename in filenames if '_background_noise_' not in filename]\n# filenames = tf.random.shuffle(filenames)\n\ntrain_samples = int(total_samples * DATASET_SPLIT[0])\nval_samples = int(total_samples * DATASET_SPLIT[1])\n\ntrain, remainder = train_df[:train_samples], train_df[train_samples:]\nval, test = remainder[:val_samples], remainder[val_samples:]","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:25.571570Z","iopub.execute_input":"2023-08-06T14:31:25.571959Z","iopub.status.idle":"2023-08-06T14:31:25.578818Z","shell.execute_reply.started":"2023-08-06T14:31:25.571924Z","shell.execute_reply":"2023-08-06T14:31:25.577805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\nprint(val.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:25.580499Z","iopub.execute_input":"2023-08-06T14:31:25.581228Z","iopub.status.idle":"2023-08-06T14:31:25.596971Z","shell.execute_reply.started":"2023-08-06T14:31:25.581194Z","shell.execute_reply":"2023-08-06T14:31:25.596030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6.Checking some audio data","metadata":{}},{"cell_type":"code","source":"# Hello Hello , mic check\n# Choose some random indices for checking\nrandom_indices = [0, 10, 20, 30, 40]\n\nfor idx in random_indices:\n    row = train_df.iloc[idx]\n    audio_file_path = os.path.join(train_data_dir, f\"{row['id']}.mp3\")\n\n    # Load the audio file using librosa\n    audio, sr = librosa.load(audio_file_path, sr=None)\n\n    # Print the transcription and play the audio\n    print(\"Transcription:\", row['sentence'])\n    ipd.display(ipd.Audio(audio, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:25.598486Z","iopub.execute_input":"2023-08-06T14:31:25.598858Z","iopub.status.idle":"2023-08-06T14:31:25.713587Z","shell.execute_reply.started":"2023-08-06T14:31:25.598824Z","shell.execute_reply":"2023-08-06T14:31:25.712785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7.Data Distributions","metadata":{}},{"cell_type":"code","source":"domain_counts = train_df['split'].value_counts()\n\n# Plot the data distribution\nplt.figure(figsize=(10, 6))\ndomain_counts.plot(kind='bar', color='skyblue')\nplt.title(\"Data Distribution Across Domains\")\nplt.xlabel(\"Domain\")\nplt.ylabel(\"Number of Recordings\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:25.717928Z","iopub.execute_input":"2023-08-06T14:31:25.718810Z","iopub.status.idle":"2023-08-06T14:31:26.258799Z","shell.execute_reply.started":"2023-08-06T14:31:25.718773Z","shell.execute_reply":"2023-08-06T14:31:26.257791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8. waveform and log mel spectrum for the audio data","metadata":{}},{"cell_type":"code","source":"for idx in random_indices:\n    row = train_df.iloc[idx]\n    audio_file_path = os.path.join(train_data_dir, f\"{row['id']}.mp3\")\n\n    # Load the audio file using librosa\n    audio, sr = librosa.load(audio_file_path, sr=None)\n\n    # Plot the waveform\n    plt.figure(figsize=(10, 4))\n    librosa.display.waveshow(audio, sr=sr)\n    plt.title(f\"Waveform - Audio File ID: {row['id']}\")\n    plt.xlabel(\"Time (s)\")\n    plt.ylabel(\"Amplitude\")\n    plt.tight_layout()\n    plt.show()\n\n    # Plot the log Mel spectrogram\n    plt.figure(figsize=(10, 4))\n    mel_spec = librosa.feature.melspectrogram(y=audio, sr=sr)\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    librosa.display.specshow(mel_spec_db, sr=sr, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f\"Log Mel Spectrogram - Audio File ID: {row['id']}\")\n    plt.xlabel(\"Time (s)\")\n    plt.ylabel(\"Mel Frequency\")\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:26.260286Z","iopub.execute_input":"2023-08-06T14:31:26.260931Z","iopub.status.idle":"2023-08-06T14:31:33.616590Z","shell.execute_reply.started":"2023-08-06T14:31:26.260894Z","shell.execute_reply":"2023-08-06T14:31:33.615685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 9.Basic Preprocess of first 5 data points of transcription","metadata":{}},{"cell_type":"code","source":"# Select the first 5 transcriptions\ntranscriptions = train_df['sentence'][:5].tolist()\n\n# Convert transcriptions to lowercase\ntranscriptions_lower = [transcription.lower() for transcription in transcriptions]\n\n# Remove punctuation\ntranslator = str.maketrans(\"\", \"\", string.punctuation)\ntranscriptions_no_punct = [transcription.translate(translator) for transcription in transcriptions_lower]\n\n# Tokenization\nnltk.download('punkt')  \ntranscriptions_tokens = [word_tokenize(transcription) for transcription in transcriptions_no_punct]\n\n\nnltk.download('stopwords')  \nstop_words = set(stopwords.words('bengali'))\ntranscriptions_no_stopwords = [\n    [word for word in tokens if word not in stop_words]\n    for tokens in transcriptions_tokens\n]\n\nnltk.download('wordnet')  \nstemmer = PorterStemmer()\ntranscriptions_stemmed = [\n    [stemmer.stem(word) for word in tokens]\n    for tokens in transcriptions_no_stopwords\n]\n\n# Print the preprocessed transcriptions\nfor i, transcription in enumerate(transcriptions_stemmed):\n    print(f\"Preprocessed transcription {i+1}: {' '.join(transcription)}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:33.618042Z","iopub.execute_input":"2023-08-06T14:31:33.619065Z","iopub.status.idle":"2023-08-06T14:31:33.844263Z","shell.execute_reply.started":"2023-08-06T14:31:33.619029Z","shell.execute_reply":"2023-08-06T14:31:33.843202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 10.First 100 data points of transcription withour removing stopwords","metadata":{}},{"cell_type":"code","source":"# Select the first 100 transcriptions \ntranscriptions = train_df['sentence'][:100].tolist()\n\n# Tokenization\nnltk.download('punkt')  # Download the Punkt tokenizer\ntranscriptions_tokens = [word_tokenize(transcription) for transcription in transcriptions]\n\n# Print the tokenized transcriptions\nfor i, transcription_tokens in enumerate(transcriptions_tokens):\n    if i%10 == 0:\n        print(f\"Tokenized transcription {i+1}: {transcription_tokens}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:33.845826Z","iopub.execute_input":"2023-08-06T14:31:33.846429Z","iopub.status.idle":"2023-08-06T14:31:33.874410Z","shell.execute_reply.started":"2023-08-06T14:31:33.846394Z","shell.execute_reply":"2023-08-06T14:31:33.872905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 11.Building a vocabulary","metadata":{}},{"cell_type":"code","source":"# Build vocabulary\nvocabulary = set()\nfor transcription_tokens in transcriptions_tokens:\n    vocabulary.update(transcription_tokens)\n\n# print(\"Vocabulary:\")\n# print(vocabulary)\nprint(f\"Vocabulary Size: {len(vocabulary)}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:33.876496Z","iopub.execute_input":"2023-08-06T14:31:33.877179Z","iopub.status.idle":"2023-08-06T14:31:33.884529Z","shell.execute_reply.started":"2023-08-06T14:31:33.877140Z","shell.execute_reply":"2023-08-06T14:31:33.883432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lens = train_df.sentence.apply(lambda x: len(x))\nplt.hist(lens)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:33.886211Z","iopub.execute_input":"2023-08-06T14:31:33.887008Z","iopub.status.idle":"2023-08-06T14:31:34.891942Z","shell.execute_reply.started":"2023-08-06T14:31:33.886974Z","shell.execute_reply":"2023-08-06T14:31:34.890949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 12.Basic Information and analysis on sentence lengths","metadata":{}},{"cell_type":"code","source":"# Calculate sentence lengths\nsentence_lengths = [len(tokens) for tokens in transcriptions_tokens]\n\n# Calculate average and maximum sentence lengths\naverage_length = sum(sentence_lengths) / len(sentence_lengths)\nmax_length = max(sentence_lengths)\n\n# Print the sentence length analysis\nprint(f\"Average Sentence Length: {average_length}\")\nprint(f\"Maximum Sentence Length: {max_length}\")\n\nplt.plot(sentence_lengths)\nplt.axhline(average_length,c='red')","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:34.893419Z","iopub.execute_input":"2023-08-06T14:31:34.893822Z","iopub.status.idle":"2023-08-06T14:31:35.152666Z","shell.execute_reply.started":"2023-08-06T14:31:34.893784Z","shell.execute_reply":"2023-08-06T14:31:35.151694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Flatten the list of tokens\nall_tokens = [token for tokens in transcriptions_tokens for token in tokens]\n\n# Count word frequency\nword_frequency = Counter(all_tokens)\n\n# Print the word frequency analysis\nprint(\"Word Frequency:\")\nplt.plot(word_frequency.values())","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:35.154136Z","iopub.execute_input":"2023-08-06T14:31:35.154595Z","iopub.status.idle":"2023-08-06T14:31:35.430077Z","shell.execute_reply.started":"2023-08-06T14:31:35.154559Z","shell.execute_reply":"2023-08-06T14:31:35.429132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute descriptive statistics\nsentence_lengths = [len(tokens) for tokens in transcriptions_tokens]\nmin_length = min(sentence_lengths)\nmax_length = max(sentence_lengths)\nmean_length = sum(sentence_lengths) / len(sentence_lengths)\nmedian_length = sorted(sentence_lengths)[len(sentence_lengths) // 2]\n\n# Plot the distribution of sentence lengths\nplt.figure(figsize=(10, 6))\nplt.hist(sentence_lengths, bins=50, color='skyblue', edgecolor='black')\nplt.axvline(mean_length, color='red', linestyle='dashed', linewidth=2, label='Mean')\nplt.axvline(median_length, color='green', linestyle='dashed', linewidth=2, label='Median')\nplt.xlabel('Sentence Length')\nplt.ylabel('Frequency')\nplt.title('Distribution of Sentence Lengths')\nplt.legend()\nplt.show()\n\n# Print the descriptive statistics\nprint(\"Descriptive Statistics:\")\nprint(f\"Minimum Sentence Length: {min_length}\")\nprint(f\"Maximum Sentence Length: {max_length}\")\nprint(f\"Mean Sentence Length: {mean_length:.2f}\")\nprint(f\"Median Sentence Length: {median_length}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:35.431652Z","iopub.execute_input":"2023-08-06T14:31:35.432285Z","iopub.status.idle":"2023-08-06T14:31:35.835064Z","shell.execute_reply.started":"2023-08-06T14:31:35.432250Z","shell.execute_reply":"2023-08-06T14:31:35.834113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 13.Basic Analysis on Words of transcription tokens","metadata":{}},{"cell_type":"code","source":"# Flatten the list of tokens\nall_tokens = [token for tokens in transcriptions_tokens for token in tokens]\n\n# Count word frequency\nword_frequency = Counter(all_tokens)\n\n# Print the word frequency analysis\nprint(\"Word Frequency:\")\nplt.plot(word_frequency.values())","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:35.836736Z","iopub.execute_input":"2023-08-06T14:31:35.837585Z","iopub.status.idle":"2023-08-06T14:31:36.110885Z","shell.execute_reply.started":"2023-08-06T14:31:35.837547Z","shell.execute_reply":"2023-08-06T14:31:36.109825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of unique words in each sentence\nsentences = train_df['sentence'].tolist()\nunique_word_counts = [len(set(sentence.split())) for sentence in sentences]\n\n# Compute descriptive statistics\nmin_unique_words = min(unique_word_counts)\nmax_unique_words = max(unique_word_counts)\nmean_unique_words = sum(unique_word_counts) / len(unique_word_counts)\nmedian_unique_words = sorted(unique_word_counts)[len(unique_word_counts) // 2]\n\n# Plot the distribution of unique word counts\nplt.figure(figsize=(10, 6))\nplt.hist(unique_word_counts, bins=50, color='lightcoral', edgecolor='black')\nplt.axvline(mean_unique_words, color='red', linestyle='dashed', linewidth=2, label='Mean')\nplt.axvline(median_unique_words, color='green', linestyle='dashed', linewidth=2, label='Median')\nplt.xlabel('Number of Unique Words')\nplt.ylabel('Frequency')\nplt.title('Distribution of Unique Words in Transcriptions')\nplt.legend()\nplt.show()\n\n# Print the descriptive statistics\nprint(\"Descriptive Statistics:\")\nprint(f\"Minimum Number of Unique Words: {min_unique_words}\")\nprint(f\"Maximum Number of Unique Words: {max_unique_words}\")\nprint(f\"Mean Number of Unique Words: {mean_unique_words:.2f}\")\nprint(f\"Median Number of Unique Words: {median_unique_words}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:36.112269Z","iopub.execute_input":"2023-08-06T14:31:36.113060Z","iopub.status.idle":"2023-08-06T14:31:42.752982Z","shell.execute_reply.started":"2023-08-06T14:31:36.113021Z","shell.execute_reply":"2023-08-06T14:31:42.752024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 14.Performing LDA topic modeling on first 1000 transcriptions and analyzing top words based on topic","metadata":{}},{"cell_type":"code","source":"# Select the first 1000 sentences \nsentences = train_df['sentence'][:10000].tolist()\n\n# Tokenization using NLTK\nnltk.download('punkt')  # Download the Punkt tokenizer\nsentences_tokens = [word_tokenize(sentence) for sentence in sentences]\n\n\n# Remove Bengali stopwords\nsentences_no_stopwords = [\n    [word for word in tokens if word not in stop_words]\n    for tokens in sentences_tokens\n]\n\n# Convert tokenized sentences back to strings\nsentences_processed = [' '.join(tokens) for tokens in sentences_no_stopwords]\n\n# Create a CountVectorizer to convert text data to a bag-of-words representation\nvectorizer = CountVectorizer(max_features=1000)\nX = vectorizer.fit_transform(sentences_processed)\n\n# Perform LDA topic modeling\nn_topics = 10  # Number of topics to discover\nlda_model = LatentDirichletAllocation(n_components=n_topics, random_state=42)\nlda_model.fit(X)\n\n# Get the top words for each topic\nfeature_names = vectorizer.get_feature_names_out()\ntop_words_per_topic = []\nfor topic_idx, topic in enumerate(lda_model.components_):\n    top_words = [feature_names[i] for i in topic.argsort()[:-4:-1]]\n    top_words_per_topic.append(top_words)\n\n# Print the top words for each topic\nfor i, top_words in enumerate(top_words_per_topic):\n    print(f\"Topic {i + 1}: {' '.join(top_words)}\")\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:31:42.754672Z","iopub.execute_input":"2023-08-06T14:31:42.755384Z","iopub.status.idle":"2023-08-06T14:32:01.087151Z","shell.execute_reply.started":"2023-08-06T14:31:42.755347Z","shell.execute_reply":"2023-08-06T14:32:01.086047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lda_model.components_","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:01.088808Z","iopub.execute_input":"2023-08-06T14:32:01.089181Z","iopub.status.idle":"2023-08-06T14:32:01.096819Z","shell.execute_reply.started":"2023-08-06T14:32:01.089146Z","shell.execute_reply":"2023-08-06T14:32:01.095866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DOMAINS = !ls /kaggle/input/bengaliai-speech/examples/","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:01.098359Z","iopub.execute_input":"2023-08-06T14:32:01.099155Z","iopub.status.idle":"2023-08-06T14:32:01.120799Z","shell.execute_reply.started":"2023-08-06T14:32:01.099096Z","shell.execute_reply":"2023-08-06T14:32:01.119906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in np.arange(5):\n    audio_file_path = f'{domains}/{DOMAINS[idx]}'\n\n    # Load the audio file using librosa\n    audio, sr = librosa.load(audio_file_path, sr=None)\n\n    # Print DOMAIN\n    print(DOMAINS[idx])\n    ipd.display(ipd.Audio(audio, rate=sr))","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:01.122693Z","iopub.execute_input":"2023-08-06T14:32:01.123400Z","iopub.status.idle":"2023-08-06T14:32:02.444986Z","shell.execute_reply.started":"2023-08-06T14:32:01.123366Z","shell.execute_reply":"2023-08-06T14:32:02.443326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on domains we can check the log-mel spectrum here","metadata":{}},{"cell_type":"code","source":"# Define the list of domains and their corresponding audio files\nDOMAINS = [\n    'Audiobook.wav', 'Parliament Session.wav', 'Bangladeshi TV Drama.wav',\n    'Poem Recital.wav', 'Bengali Advertisement.wav', 'Puthi Literature.wav',\n    'Cartoon.wav', 'Slang Profanity.mp3', 'Debate.wav', 'Stage Drama Jatra.wav',\n    'Indian TV Drama.wav', 'Talk Show Interview.wav', 'Movie.wav', 'Telemedicine.mp3',\n    'News Presentation.wav', 'Waz Islamic Sermon.wav', 'Online Class.wav'\n]\n\n# Visualize the audio files and play them\nfor idx in np.arange(5):\n    audio_file_path = os.path.join(domains, DOMAINS[idx])\n\n    # Load the audio file using librosa\n    audio, sr = librosa.load(audio_file_path, sr=None)\n\n    # Plot the waveform\n    plt.figure(figsize=(10, 4))\n    librosa.display.waveshow(audio, sr=sr)\n    plt.title(f'Waveform - {DOMAINS[idx]}')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Amplitude')\n    plt.show()\n\n    # Plot the spectrogram\n    plt.figure(figsize=(10, 4))\n    D = librosa.amplitude_to_db(np.abs(librosa.stft(audio)), ref=np.max)\n    librosa.display.specshow(D, sr=sr, x_axis='time', y_axis='linear')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(f'Spectrogram - {DOMAINS[idx]}')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Frequency (Hz)')\n    plt.show()\n\n    # Play the audio\n    print(f\"Audio: {DOMAINS[idx]}\")\n    ipd.display(ipd.Audio(audio, rate=sr))","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-08-06T14:32:02.446843Z","iopub.execute_input":"2023-08-06T14:32:02.447833Z","iopub.status.idle":"2023-08-06T14:32:19.150335Z","shell.execute_reply.started":"2023-08-06T14:32:02.447795Z","shell.execute_reply":"2023-08-06T14:32:19.148943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 16. Creating a sample daatset to work with","metadata":{}},{"cell_type":"code","source":"sample_train_df = train_df[['id', 'sentence']].copy()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.152174Z","iopub.execute_input":"2023-08-06T14:32:19.153364Z","iopub.status.idle":"2023-08-06T14:32:19.281264Z","shell.execute_reply.started":"2023-08-06T14:32:19.153304Z","shell.execute_reply":"2023-08-06T14:32:19.279994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_df.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.282556Z","iopub.execute_input":"2023-08-06T14:32:19.283471Z","iopub.status.idle":"2023-08-06T14:32:19.288225Z","shell.execute_reply.started":"2023-08-06T14:32:19.283433Z","shell.execute_reply":"2023-08-06T14:32:19.287226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_df","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.298718Z","iopub.execute_input":"2023-08-06T14:32:19.299531Z","iopub.status.idle":"2023-08-06T14:32:19.316018Z","shell.execute_reply.started":"2023-08-06T14:32:19.299497Z","shell.execute_reply":"2023-08-06T14:32:19.315217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**renaming column sentence to transcription**","metadata":{}},{"cell_type":"code","source":"sample_train_df.rename(columns = {'sentence':'transcriptions'}, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.317186Z","iopub.execute_input":"2023-08-06T14:32:19.318073Z","iopub.status.idle":"2023-08-06T14:32:19.323354Z","shell.execute_reply.started":"2023-08-06T14:32:19.318041Z","shell.execute_reply":"2023-08-06T14:32:19.322302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.324737Z","iopub.execute_input":"2023-08-06T14:32:19.325858Z","iopub.status.idle":"2023-08-06T14:32:19.337511Z","shell.execute_reply.started":"2023-08-06T14:32:19.325822Z","shell.execute_reply":"2023-08-06T14:32:19.336587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_20_train = sample_train_df[:20]","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.339063Z","iopub.execute_input":"2023-08-06T14:32:19.340063Z","iopub.status.idle":"2023-08-06T14:32:19.350623Z","shell.execute_reply.started":"2023-08-06T14:32:19.340032Z","shell.execute_reply":"2023-08-06T14:32:19.349678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_20_train.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.352059Z","iopub.execute_input":"2023-08-06T14:32:19.353124Z","iopub.status.idle":"2023-08-06T14:32:19.374339Z","shell.execute_reply.started":"2023-08-06T14:32:19.353086Z","shell.execute_reply":"2023-08-06T14:32:19.373091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 17.Hyperparameters for for futher assitance","metadata":{}},{"cell_type":"code","source":"\"\"\"Hyperparameters for the CNN\"\"\"\n\n\nBATCH_SIZE = 100\nEPOCHS = 20\nLEARNING_RATE = 1e-3\nDROPOUT = 0.2\n\nSAMPLE_RATE = 400000\nFFT_SIZE = 400\nHOP_SIZE = 600\nFRAMES = 16\nMEL_BINS = 26\nMEL_MIN_HZ = 125.0\nMEL_MAX_HZ = SAMPLE_RATE / 2\nLOG_OFFSET = 1e-3","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:26:03.701866Z","iopub.execute_input":"2023-08-06T15:26:03.702980Z","iopub.status.idle":"2023-08-06T15:26:03.709396Z","shell.execute_reply.started":"2023-08-06T15:26:03.702934Z","shell.execute_reply":"2023-08-06T15:26:03.708192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 18.Creating a Logmelspectrum input layer for feeding the model","metadata":{}},{"cell_type":"code","source":"class LogMelgramLayer(tf.keras.layers.Layer):\n\n    def __init__(\n        self, sample_rate, num_fft, hop_length, num_mels, f_min=125.0, f_max=3800.0, ** kwargs\n    ):\n        super(LogMelgramLayer, self).__init__(**kwargs)\n        self.num_fft = num_fft\n        self.hop_length = hop_length\n        self.num_mels = num_mels\n        self.sample_rate = sample_rate\n        self.f_min = f_min\n        self.f_max = f_max\n        self.num_spectrogram_bins = num_fft // 2 + 1\n        mel_filterbank = tf.signal.linear_to_mel_weight_matrix(\n            num_mel_bins=self.num_mels,\n            num_spectrogram_bins=self.num_spectrogram_bins,\n            sample_rate=self.sample_rate,\n            lower_edge_hertz=self.f_min,\n            upper_edge_hertz=self.f_max,\n        )\n\n        self.mel_filterbank = mel_filterbank\n\n    def build(self, input_shape):\n        self.non_trainable_weights.append(self.mel_filterbank)\n        super(LogMelgramLayer, self).build(input_shape)\n\n    def call(self, input):\n        \"\"\"\n        Args:\n            input (tensor): Batch of mono waveform, shape: (None, N)\n        Returns:\n            log_melgrams (tensor): Batch of log mel-spectrograms, shape: (None, num_frame, mel_bins, channel=1)\n        \"\"\"\n\n        def power_to_db(S, amin=1e-16, top_db=80.0):\n            \"\"\"Convert a power-spectrogram (magnitude squared) to decibel (dB) units.\n            Computes the scaling ``10 * log10(S / max(S))`` in a numerically\n            stable way.\n\n            Based on:\n            http://man.hubwiz.com/docset/LibROSA.docset/Contents/Resources/Documents/generated/librosa.core.power_to_db.html\n            \"\"\"\n            def _tf_log10(x):\n                numerator = tf.math.log(x)\n                denominator = tf.math.log(\n                    tf.constant(10, dtype=numerator.dtype))\n                return numerator / denominator\n\n            # Scale magnitude relative to maximum value in S. Zeros in the output\n            # correspond to positions where S == ref.\n            ref = tf.reduce_max(S)\n\n            log_spec = 10.0 * _tf_log10(tf.maximum(amin, S))\n            log_spec -= 10.0 * _tf_log10(tf.maximum(amin, ref))\n\n            log_spec = tf.maximum(log_spec, tf.reduce_max(log_spec) - top_db)\n\n            return log_spec\n\n        # Compute short time fourier transform\n        spectrogram = tf.signal.stft(\n            input, frame_length=self.num_fft, frame_step=self.hop_length, fft_length=self.num_fft)\n\n        # Compute magnitudes to avoid complex values\n        magnitude_spectrogram = tf.abs(spectrogram)\n\n        # Transform the linear-scale magnitude-spectrograms to mel-scale\n        mel_power_spectrograms = tf.matmul(tf.square(magnitude_spectrogram),\n                                           self.mel_filterbank)\n\n        # Transform magnitudes to log-scale\n        log_magnitude_mel_spectrograms = power_to_db(\n            mel_power_spectrograms)\n        log_magnitude_mel_spectrograms = tf.expand_dims(\n            log_magnitude_mel_spectrograms, axis=-1)\n\n        # MinMax scale the features\n        log_magnitude_mel_spectrograms = tf.math.divide(tf.math.subtract(log_magnitude_mel_spectrograms, tf.math.reduce_min(\n            log_magnitude_mel_spectrograms)), tf.math.subtract(tf.math.reduce_max(log_magnitude_mel_spectrograms),\n                                                               tf.math.reduce_min(log_magnitude_mel_spectrograms)))\n        return log_magnitude_mel_spectrograms\n\n    def get_config(self):\n        config = {\n            'num_fft': self.num_fft,\n            'hop_length': self.hop_length,\n            'num_mels': self.num_mels,\n            'sample_rate': self.sample_rate,\n            'f_min': self.f_min,\n            'f_max': self.f_max,\n        }\n        base_config = super(LogMelgramLayer, self).get_config()\n        return dict(list(config.items()) + list(base_config.items()))","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.384812Z","iopub.execute_input":"2023-08-06T14:32:19.386033Z","iopub.status.idle":"2023-08-06T14:32:19.412885Z","shell.execute_reply.started":"2023-08-06T14:32:19.385997Z","shell.execute_reply":"2023-08-06T14:32:19.411684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 19.Another way of creating log-mel-spectrum from the input signal and then feed it to the model","metadata":{}},{"cell_type":"code","source":"def power_to_db(S, amin=1e-16, top_db=80.0):\n    \"\"\"Convert a power-spectrogram (magnitude squared) to decibel (dB) units.\n       Computes the scaling ``10 * log10(S / max(S))`` in a numerically\n       stable way.\n\n       Based on:\n       http://man.hubwiz.com/docset/LibROSA.docset/Contents/Resources/Documents/generated/librosa.core.power_to_db.html\n    \"\"\"\n    def _tf_log10(x):\n        numerator = tf.math.log(x)\n        denominator = tf.math.log(tf.constant(10, dtype=numerator.dtype))\n        return numerator / denominator\n\n    # Scale magnitude relative to maximum value in S. Zeros in the output\n    # correspond to positions where S == ref.\n    ref = tf.reduce_max(S)\n\n    log_spec = 10.0 * _tf_log10(tf.maximum(amin, S))\n    log_spec -= 10.0 * _tf_log10(tf.maximum(amin, ref))\n\n    log_spec = tf.maximum(log_spec, tf.reduce_max(log_spec) - top_db)\n\n    return log_spec\n\ndef generate_log_mel_spectrograms(signal, w2v_array):\n    \"\"\"Method for feature extraction. Calculates the log mel-spectrogram representation of the input signal.\n\n    Args:\n        signal (tf.Tensor[float]): [description]. The input signal to extract features on.\n    Returns:\n        [tf.Tensor[float]]: [description] The log mel-spectrogram representation of the input signal\n        [tf.Tensor[int]]: [description]   The encoded label of the sample\n    \"\"\"\n#     print(tf.shape(signal))\n#     print(\"from generate_log_mel_spectrograms\")\n    # Compute short time fourier transform\n    spectrogram = tf.signal.stft(signal, \n                                 frame_length=FFT_SIZE, \n                                 frame_step=HOP_SIZE, \n                                 fft_length=FFT_SIZE)\n\n    # Compute magnitudes to avoid complex values\n    magnitude_spectrogram = tf.abs(spectrogram)\n\n    # Set the filter bank\n    mel_filterbank = tf.signal.linear_to_mel_weight_matrix(num_mel_bins=MEL_BINS, \n                                                           num_spectrogram_bins=int(FFT_SIZE / 2 + 1),\n                                                           sample_rate=SAMPLE_RATE)\n\n    # Transform the linear-scale magnitude-spectrograms to mel-scale\n    mel_power_spectrograms = tf.matmul(tf.square(magnitude_spectrogram),\n                                       mel_filterbank)\n\n    # Transform magnitudes to log-scale\n    log_magnitude_mel_spectrograms = power_to_db(mel_power_spectrograms)\n    log_magnitude_mel_spectrograms = tf.expand_dims(\n        log_magnitude_mel_spectrograms, axis=-1)\n\n    # MinMax scale the features\n    log_magnitude_mel_spectrograms = tf.math.divide(tf.math.subtract(log_magnitude_mel_spectrograms, tf.math.reduce_min(\n        log_magnitude_mel_spectrograms)), tf.math.subtract(tf.math.reduce_max(log_magnitude_mel_spectrograms),\n                                                           tf.math.reduce_min(log_magnitude_mel_spectrograms)))\n#     print(tf.shape(log_magnitude_mel_spectrograms))\n    \n#     print(\"returning spec\")\n    return log_magnitude_mel_spectrograms, w2v_array\n\n\ndef waveform_to_log_mel_spectrogram(signal):\n    \"\"\"Method for feature extraction. Calculates the log mel-spectrogram representation of the input signal.\n\n    Args:\n        signal (tf.Tensor[float]): [description]. The input signal to extract features on.\n    Returns:\n        [tf.Tensor[float]]: [description] The log mel-spectrogram representation of the input signal\n        [tf.Tensor[int]]: [description]   The encoded label of the sample\n    \"\"\"\n    # Compute short time fourier transform\n    spectrogram = tf.signal.stft(signal, \n                                 frame_length=FFT_SIZE, \n                                 frame_step=HOP_SIZE, \n                                 fft_length=FFT_SIZE)\n\n    # Compute magnitudes to avoid complex values\n    magnitude_spectrogram = tf.abs(spectrogram)\n\n    # Set the filter bank\n    mel_filterbank = tf.signal.linear_to_mel_weight_matrix(num_mel_bins=MEL_BINS, \n                                                           num_spectrogram_bins=int(FFT_SIZE / 2 + 1),\n                                                           sample_rate=SAMPLE_RATE)\n\n    # Transform the linear-scale magnitude-spectrograms to mel-scale\n    mel_power_spectrograms = tf.matmul(tf.square(magnitude_spectrogram),\n                                       mel_filterbank)\n\n    # Transform magnitudes to log-scale\n    log_magnitude_mel_spectrograms = power_to_db(mel_power_spectrograms)\n    log_magnitude_mel_spectrograms = tf.expand_dims(\n        log_magnitude_mel_spectrograms, axis=-1)\n\n    # MinMax scale the features\n    log_magnitude_mel_spectrograms = tf.math.divide(tf.math.subtract(log_magnitude_mel_spectrograms, tf.math.reduce_min(\n        log_magnitude_mel_spectrograms)), tf.math.subtract(tf.math.reduce_max(log_magnitude_mel_spectrograms),\n                                                           tf.math.reduce_min(log_magnitude_mel_spectrograms)))\n    print(tf.shape(log_magnitude_mel_spectrograms))\n    \n    print(\"returning spec\")\n    return log_magnitude_mel_spectrograms","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:26:53.632054Z","iopub.execute_input":"2023-08-06T15:26:53.632422Z","iopub.status.idle":"2023-08-06T15:26:53.651821Z","shell.execute_reply.started":"2023-08-06T15:26:53.632392Z","shell.execute_reply":"2023-08-06T15:26:53.650797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 20.Removing stopwords for bengali words and creating a vector from that using word2vec","metadata":{}},{"cell_type":"code","source":"def remove_stopwords_from_sentence(text):\n    whitespace = re.compile(u\"[\\s\\u0020\\u00a0\\u1680\\u180e\\u202f\\u205f\\u3000\\u2000-\\u200a]+\", re.UNICODE)\n    bangla_fullstop = u\"\\u0964\"\n    punctSeq   = u\"['\\\"“”‘’]+|[.?!,…]+|[:;]+\"\n    punc = u\"[(),$%^&*+={}\\[\\]:\\\"|\\'\\~`<>/,¦!?½£¶¼©⅐⅑⅒⅓⅔⅕⅖⅗⅘⅙⅚⅛⅜⅝⅞⅟↉¤¿º;-]+\"\n    text= whitespace.sub(\" \",text).strip()\n    text = re.sub(punctSeq, \" \", text)\n    text = re.sub(bangla_fullstop, \" \",text)\n    text = re.sub(punc, \" \", text)\n#     print(text)\n    return text\n\ndef create_vector_from_sentence(text):\n    doc=[text.split()]\n    w2v_model = Word2Vec(doc, min_count=1)\n    w2v_model.train(doc, total_examples=len(doc), epochs=20)\n    words = list(w2v_model.wv.key_to_index.keys())\n    word2vec = [w2v_model.wv[word] for word in words]\n#     n_components_text = min(n_samples, n_features) // 2 + 1\n    pca = PCA(n_components=1)\n    result = pca.fit_transform(word2vec)\n    result = result.reshape(result.shape[0] * result.shape[1])\n#     word2vec = word2vec.reshape(word2vec.shape[0] * word2vec.shape[1])\n    return result\n\n\ndef find_max_list(list):\n    list_len = [len(i) for i in list]\n    print(max(list_len))\n    return max(list_len)\n\ndef find_max_shape(list):\n    list_len = [tf.shape(i) for i in list]\n    return max(list_len)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:14:29.682547Z","iopub.execute_input":"2023-08-06T15:14:29.682938Z","iopub.status.idle":"2023-08-06T15:14:29.694328Z","shell.execute_reply.started":"2023-08-06T15:14:29.682904Z","shell.execute_reply":"2023-08-06T15:14:29.693220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 21.Handy functions for creating a dataframe of transcription of audio files(very useful for short dataset)","metadata":{}},{"cell_type":"code","source":"def find_transcription_using_label_single(audio_file):\n    label_1 = single_file.split(\"/\")[-1]\n    label_1 = label_1.split(\".\")[0]\n    transcription = sample_train_df[sample_train_df['id']==label_1]['transcriptions']\n    return label_1, transcription\n\ndef find_transcription_using_label(audio_files):\n    semi_files_transcriptions = pd.DataFrame(columns=['id', 'transcriptions'], index=range(len(audio_files)))\n    i = 0\n    for single_file in audio_files:\n        label_1 = single_file.split(\"/\")[-1]\n        label_1 = label_1.split(\".\")[0]\n        transcription = sample_train_df[sample_train_df['id']==label_1]['transcriptions']\n        df_element = {'id': label_1, 'transcriptions': transcription}\n        df_temp = pd.DataFrame(df_element)\n        semi_files_transcriptions.iloc[i] = df_temp\n        i += 1\n    return semi_files_transcriptions","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.460325Z","iopub.execute_input":"2023-08-06T14:32:19.460594Z","iopub.status.idle":"2023-08-06T14:32:19.472725Z","shell.execute_reply.started":"2023-08-06T14:32:19.460569Z","shell.execute_reply":"2023-08-06T14:32:19.471508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def genarate_log_mel_spectrogram(audio_file):\n    y, sr = librosa.load(audio_file)\n    # trim silent edges\n    signal, _ = librosa.effects.trim(y)\n    mel_spect = librosa.feature.melspectrogram(signal, sr=sr, n_fft=n_fft, hop_length=hop_length, n_mels=n_mels)\n    mel_spect = librosa.power_to_db(mel_spect, ref=np.max)\n    return mel_spect","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.476858Z","iopub.execute_input":"2023-08-06T14:32:19.477266Z","iopub.status.idle":"2023-08-06T14:32:19.489707Z","shell.execute_reply.started":"2023-08-06T14:32:19.477218Z","shell.execute_reply":"2023-08-06T14:32:19.488568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 22.Preprocessing audio file for creating datastream ","metadata":{}},{"cell_type":"code","source":"@tf.function\ndef decode_audio(audio_binary):\n    \"\"\"Decodes a 16-bit MP3 file to a float tensor, values scaled between -1.0 and 1.0\n\n    Args:\n        audio_binary (tf.Tensor[string]): [description] The MP3-encoded audio from a file.\n    Returns:\n        [tf.Tensor[float]]: [description] A float tensor with values between -1.0 and 1.0 representing the audio.\n    \"\"\"  \n    audio = tfio.audio.decode_mp3(audio_binary)\n    return tf.squeeze(audio, axis=-1)\n\ndef encode_label(transcription_list):\n    \"\"\"Encodes the label with a number\"\"\"  \n    vector_list = []\n    padded_vector_list = []\n    for transcription in transcription_list:\n        preprocessed_text = remove_stopwords_from_sentence(transcription)\n        vector_list.append(create_vector_from_sentence(preprocessed_text))\n        \n    \n    padding_len = find_max_list(vector_list)\n    for vect in vector_list:\n        padding = tf.zeros(padding_len + 1 - len(vect), dtype=tf.float32)\n        padded_vector = tf.concat([vect, padding], 0)\n        padded_vector_list.append(padded_vector)\n\n    return padded_vector_list\n\ndef get_sample_normal_spec(source_file, w2v_array):\n    \"\"\"Reads source audio file and get the label based on that label fetch the transcription.\n    apply word tokenization on that transcription and create a bag of word of that\n\n    Args:\n        source_file (tf.Tensor[String]): [description]. The path to the source audio file\n    Returns:\n        [tf.Tensor[float]]: [description] The decoded and padded audio signal\n        [tf.Tensor[string]]: [description] The label of the sample\n    \"\"\"\n    audio_binary = tf.io.read_file(source_file)\n    signal = decode_audio(audio_binary)\n    signal = tf.cast(signal, tf.float32)\n    spectrogram = tr.signal.stft(signal, \n                                 frame_length=FFT_SIZE, \n                                 frame_step=HOP_SIZE, \n                                 fft_length=FFT_SIZE)\n    \n    magnitude_spectrogram = tf.abs(spectrogram)\n    magnitude_spectrogram = tr.math.pow(magnitude_spectrogram, 0.5)\n    \n    means = tf.math.reduce_mean(magnitude_spectrogram, 1, keepdims = True)\n    stddevs = tf.math.reduce_std(magnitude_spectrogram, 1, keepdims = True)\n    \n    magnitude_spectrogram = (magnitude_spectrogram - mean) / (stddevs + 1e-10)\n    \n    # Add padding in case the source file has less than _SAMPLE_RATE samples\n    padding = tf.zeros([SAMPLE_RATE] - tf.shape(signal), dtype=tf.float32)\n    \n    padded_signal = tf.concat([signal, padding], 0)\n\n    return padded_signal, w2v_array\n\ndef get_sample_audio(source_file):\n    \"\"\"Reads source audio file and get the label based on that label fetch the transcription.\n    apply word tokenization on that transcription and create a bag of word of that\n\n    Args:\n        source_file (tf.Tensor[String]): [description]. The path to the source audio file\n    Returns:\n        [tf.Tensor[float]]: [description] The decoded and padded audio signal\n        [tf.Tensor[string]]: [description] The label of the sample\n    \"\"\"\n    audio_binary = tf.io.read_file(source_file)\n    signal = decode_audio(audio_binary)\n\n    # Add padding in case the source file has less than _SAMPLE_RATE samples\n    padding = tf.zeros([SAMPLE_RATE] - tf.shape(signal), dtype=tf.float32)\n    signal = tf.cast(signal, tf.float32)\n    padded_signal = tf.concat([signal, padding], 0)\n#     label_1 = source_file.split(\"/\")[-1]\n#     label_1 = label_1.split(\".\")[0]\n    label = tf.strings.split(source_file, os.path.sep)[-1]\n    return padded_signal, label\n\ndef get_sample(source_file, w2v_array):\n    \"\"\"Reads source audio file and get the label based on that label fetch the transcription.\n    apply word tokenization on that transcription and create a bag of word of that\n\n    Args:\n        source_file (tf.Tensor[String]): [description]. The path to the source audio file\n    Returns:\n        [tf.Tensor[float]]: [description] The decoded and padded audio signal\n        [tf.Tensor[string]]: [description] The label of the sample\n    \"\"\"\n#     print(source_file)\n    audio_binary = tf.io.read_file(source_file)\n    signal = decode_audio(audio_binary)\n\n    # Add padding in case the source file has less than _SAMPLE_RATE samples\n    padding = tf.zeros([SAMPLE_RATE] - tf.shape(signal), dtype=tf.float32)\n    signal = tf.cast(signal, tf.float32)\n    padded_signal = tf.concat([signal, padding], 0)\n#     print(tf.shape(signal))\n#     print(\"from get_sample\")\n    return padded_signal, w2v_array\n\ndef configure_data_stream(audio_files):\n    \"\"\"Creates an input pipeline by reading from the source data, applying transformations for\n       preprocessing and (optionally) data augmentation. Output data consists of a feature set\n       and the encoded labels.\n\n    Args:\n        audio_files (tf.Tensor[string]): [description]. The list of files to use as input data\n        data_augmentation (boolean, optional): [description]. Whether to apply data augmenting transformations or not.                                                                                      Defaults to false.\n    Returns:\n        [tf.data.Dataset]: A source dataset for the given input data\n    \"\"\"\n    semi_files_transcriptions_data = find_transcription_using_label(audio_files)\n    transcriptions_list = list(semi_files_transcriptions_data[\"transcriptions\"])\n    w2v_list = encode_label(transcriptions_list)\n    print(len(audio_files))\n    print(len(w2v_list))\n    ds = tf.data.Dataset.from_tensor_slices((audio_files, w2v_list))\n    ds = ds.map(get_sample, num_parallel_calls=tf.data.AUTOTUNE)\n    feature_ds =ds.map(generate_log_mel_spectrograms,num_parallel_calls=tf.data.experimental.AUTOTUNE)\n#     if data_augmentation and (random.uniform(0, 1) >= 0.5):\n#         ds = ds.map(add_noise, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    #ds = ds.map(encode_label, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n    return feature_ds\n\n# def configure_data_stream_without_tf(audio_files):\n#     \"\"\"Creates an input pipeline by reading from the source data, applying transformations for\n#        preprocessing and (optionally) data augmentation. Output data consists of a feature set\n#        and the encoded labels.\n\n#     Args:\n#         audio_files (tf.Tensor[string]): [description]. The list of files to use as input data\n#         data_augmentation (boolean, optional): [description]. Whether to apply data augmenting transformations or not.                                                                                      Defaults to false.\n#     Returns:\n#         [tf.data.Dataset]: A source dataset for the given input data\n#     \"\"\"\n#     for file in audio_files:\n        \n#     semi_files_transcriptions_data = find_transcription_using_label(audio_files)\n#     transcriptions_list = list(semi_files_transcriptions_data[\"transcriptions\"])\n#     w2v_list = encode_label(transcriptions_list)\n#     print(len(audio_files))\n#     print(len(w2v_list))\n#     ds = tf.data.Dataset.from_tensor_slices((audio_files, w2v_list))\n#     ds = ds.map(get_sample, num_parallel_calls=tf.data.AUTOTUNE)\n#     feature_ds =ds.map(generate_log_mel_spectrograms,num_parallel_calls=tf.data.experimental.AUTOTUNE)\n# #     if data_augmentation and (random.uniform(0, 1) >= 0.5):\n# #         ds = ds.map(add_noise, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n#     #ds = ds.map(encode_label, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n#     return feature_ds","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.491401Z","iopub.execute_input":"2023-08-06T14:32:19.491848Z","iopub.status.idle":"2023-08-06T14:32:19.517900Z","shell.execute_reply.started":"2023-08-06T14:32:19.491809Z","shell.execute_reply":"2023-08-06T14:32:19.516918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 23.Loss function for the model in our case using CTC loss","metadata":{}},{"cell_type":"code","source":"def decode_batch_predictions(pred):\n    input_len = np.ones(pred.shape[0]) * pred.shape[1]\n    results = keras.backend.ctc_decode(pred, input_length = input_len, greedy = True)[0][0]\n    \n    output_text = []\n    for result in results:\n        result = tf.strings.reduce_join(num_to_char(result)).numpy().decode(\"utf-8\")\n        output_text.append(result)\n    return output_text","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.520181Z","iopub.execute_input":"2023-08-06T14:32:19.520792Z","iopub.status.idle":"2023-08-06T14:32:19.532572Z","shell.execute_reply.started":"2023-08-06T14:32:19.520718Z","shell.execute_reply":"2023-08-06T14:32:19.531555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cnn():\n    \"\"\"Defines the core model\"\"\"\n    model = Sequential(layers=[\n        LogMelgramLayer(input_shape=((SAMPLE_RATE,)), sample_rate= SAMPLE_RATE,\n                        num_fft= FFT_SIZE, hop_length= HOP_SIZE, num_mels= MEL_BINS), \n        layers.Conv2D(16, 3, activation='relu',\n                      kernel_initializer='he_normal'),\n        layers.Conv2D(32, 2, activation='relu',\n                      kernel_initializer='he_normal'),\n        layers.Conv2D(64, 2, activation='relu',\n                      kernel_initializer='he_normal'),\n        \n        layers.Conv2D(128, 2, activation='relu',\n                      kernel_initializer='he_normal'),\n        layers.MaxPooling2D(),\n        layers.Flatten(),\n        layers.Dropout(DROPOUT),\n        layers.Dense(64, activation='relu', kernel_initializer='he_normal'),\n        layers.Dense(len(WORDS), activation='softmax'),\n    ], name=\"CNN\")\n\n    return model\n\n\ndef cnn_tflite_compatible_model(input_dim):\n\n    \n    \n    net = layers.Conv2D(32, 2, activation='relu',\n                        kernel_initializer='he_normal')(net)\n    net = layers.MaxPooling2D()(net)\n    net = layers.Conv2D(64, 2, activation='relu',\n                        kernel_initializer='he_normal')(net)\n    net = layers.MaxPooling2D()(net)\n    net = layers.Flatten()(net)\n    net = layers.Dense(128, activation='relu',\n                       kernel_initializer='he_normal')(net)\n    net = layers.Dropout(DROPOUT)(net)\n    net = layers.Dense(len(WORDS), activation='softmax')(net)\n\n    model = Model(name=\"TFLITE_COMPATIBLE_CNN\",\n                  inputs=waveform, outputs=net)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:32:19.535732Z","iopub.execute_input":"2023-08-06T14:32:19.536041Z","iopub.status.idle":"2023-08-06T14:32:19.547482Z","shell.execute_reply.started":"2023-08-06T14:32:19.536016Z","shell.execute_reply":"2023-08-06T14:32:19.546379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 24.For using gpu find it","metadata":{}},{"cell_type":"code","source":"device_name = tf.test.gpu_device_name()\nif len(device_name) > 0:\n    print(\"Found GPU at: {}\".format(device_name))\nelse:\n    device_name = \"/device:CPU:0\"\n    print(\"No GPU, using {}.\".format(device_name))","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.284673Z","iopub.execute_input":"2023-08-06T15:00:29.285612Z","iopub.status.idle":"2023-08-06T15:00:29.300756Z","shell.execute_reply.started":"2023-08-06T15:00:29.285574Z","shell.execute_reply":"2023-08-06T15:00:29.299631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# 25.Fetching all the audio data files from training directory to a list","metadata":{}},{"cell_type":"code","source":"# with tf.device(device_name):\n#     audio_data_files = tf.io.gfile.glob(str(train_data_dir) + '*')\naudio_data_files = tf.io.gfile.glob(str(train_data_dir) + '*')","metadata":{"execution":{"iopub.status.busy":"2023-08-06T14:51:59.992100Z","iopub.execute_input":"2023-08-06T14:51:59.992469Z","iopub.status.idle":"2023-08-06T15:00:29.112767Z","shell.execute_reply.started":"2023-08-06T14:51:59.992438Z","shell.execute_reply":"2023-08-06T15:00:29.111706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(audio_data_files)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.114782Z","iopub.execute_input":"2023-08-06T15:00:29.115224Z","iopub.status.idle":"2023-08-06T15:00:29.122178Z","shell.execute_reply.started":"2023-08-06T15:00:29.115190Z","shell.execute_reply":"2023-08-06T15:00:29.121263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 26.Train test split of audio dataset","metadata":{}},{"cell_type":"code","source":"train, remainder = audio_data_files[:train_samples], audio_data_files[train_samples:]\nval, test = remainder[:val_samples], remainder[val_samples:]","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.123792Z","iopub.execute_input":"2023-08-06T15:00:29.124458Z","iopub.status.idle":"2023-08-06T15:00:29.179469Z","shell.execute_reply.started":"2023-08-06T15:00:29.124423Z","shell.execute_reply":"2023-08-06T15:00:29.178307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.182434Z","iopub.execute_input":"2023-08-06T15:00:29.182816Z","iopub.status.idle":"2023-08-06T15:00:29.193273Z","shell.execute_reply.started":"2023-08-06T15:00:29.182779Z","shell.execute_reply":"2023-08-06T15:00:29.192141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**For basic testing taking only 20 train, 10 validation, 5 test files**","metadata":{}},{"cell_type":"code","source":"semi_files = train[:500]\nsemi_val_files = val[:200]\nsemi_test_files = val[:5]","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:23:26.654542Z","iopub.execute_input":"2023-08-06T15:23:26.654924Z","iopub.status.idle":"2023-08-06T15:23:26.659661Z","shell.execute_reply.started":"2023-08-06T15:23:26.654891Z","shell.execute_reply":"2023-08-06T15:23:26.658528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"singlefile = semi_files[:1]","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.207036Z","iopub.execute_input":"2023-08-06T15:00:29.207407Z","iopub.status.idle":"2023-08-06T15:00:29.217689Z","shell.execute_reply.started":"2023-08-06T15:00:29.207373Z","shell.execute_reply":"2023-08-06T15:00:29.216617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(semi_files)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:23:33.011671Z","iopub.execute_input":"2023-08-06T15:23:33.012889Z","iopub.status.idle":"2023-08-06T15:23:33.019753Z","shell.execute_reply.started":"2023-08-06T15:23:33.012848Z","shell.execute_reply":"2023-08-06T15:23:33.018685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"single_ds = configure_data_stream(singlefile).prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_input_and_label(audio_files):\n    semi_files_transcriptions_data = find_transcription_using_label(audio_files)\n    transcriptions_list = list(semi_files_transcriptions_data[\"transcriptions\"])\n    w2v_list = encode_label(transcriptions_list)\n    mel_spec_list = []\n    w2v_final = []\n    i = 0\n    for file in audio_files:\n        signal, w2v = get_sample(file, w2v_list[i])\n        mel_spec, w2v = generate_log_mel_spectrograms(signal, w2v)\n        mel_spec_list.append(mel_spec)\n        w2v_final.append(w2v)\n        i += 1\n    return mel_spec_list, w2v_final\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.302332Z","iopub.execute_input":"2023-08-06T15:00:29.302699Z","iopub.status.idle":"2023-08-06T15:00:29.312901Z","shell.execute_reply.started":"2023-08-06T15:00:29.302660Z","shell.execute_reply":"2023-08-06T15:00:29.311811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def config_my_datastream(audio_files):\n    mel_spec_list, w2v_list = preprocess_input_and_label(audio_files)\n    ds = tf.data.Dataset.from_tensor_slices((mel_spec_list, w2v_list))\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.318096Z","iopub.execute_input":"2023-08-06T15:00:29.318406Z","iopub.status.idle":"2023-08-06T15:00:29.325797Z","shell.execute_reply.started":"2023-08-06T15:00:29.318379Z","shell.execute_reply":"2023-08-06T15:00:29.324570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec_list, w2v_list = preprocess_input_and_label(singlefile)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.shape(w2v_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.shape(mel_spec_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in mel_spec_list:\n    print(i)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in w2v_list:\n    print(i)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = tf.data.Dataset.from_tensor_slices((singlefile, w2v_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transcription = sample_train_df[sample_train_df['id']==label_1]['transcriptions']\nprint(transcription)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"semi_files_transcriptions_data = find_transcription_using_label(semi_files)\ntranscriptions_list = list(semi_files_transcriptions_data[\"transcriptions\"])\nw2v_list = encode_label(transcriptions_list)\nds = tf.data.Dataset.from_tensor_slices((semi_files, w2v_list))\n# ds = ds.(list(sample_20_train[\"id\"]), list(sample_20_train[\"transcriptions\"]))\nds = ds.map(get_sample, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 27.Visual representation of melspectrogram","metadata":{}},{"cell_type":"code","source":"# Visual representation of the source audio and the corresponding features extracted\noriginal_ds = tf.data.Dataset.from_tensor_slices(semi_files).map(get_sample_audio)\nfeature_ds = tf.data.Dataset.from_tensor_slices(semi_files).map(get_sample_audio).map(generate_log_mel_spectrograms,num_parallel_calls=tf.data.experimental.AUTOTUNE)\n\nfig, axes = plt.subplots(8, 2, figsize=(12,10))\nfig.suptitle('Input Signal and the corresponding Log Melspectrogram')\nfor idx, ((signal, label),(feature, _)) in enumerate(zip(original_ds.take(8), feature_ds.take(8))):\n    ax = axes[idx][0]\n    ax.plot(np.linspace(0, 1, len(signal.numpy())), signal.numpy())\n#     ax.set_title(label.numpy().decode('UTF-8'))\n    ax.set_xlim(0, 1)\n    ax.set_ylabel(\"Amplitude\")\n    ax.set_xlabel(\"Time (Seconds)\")\n\n    ax = axes[idx][1]\n    img = librosa.display.specshow(np.squeeze(feature.numpy()), sr=SAMPLE_RATE,\n                                   hop_length=HOP_SIZE, x_axis='time', y_axis='mel',\n                                   cmap='viridis', ax=ax);\n    fig.colorbar(img, ax=ax)\n    ax.grid(False)\n#     ax.set_title(label.numpy().decode('UTF-8'))\n    ax.set_ylabel(\"Hz\")\n    ax.set_xlabel(\"Time\")\n\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:00:29.327521Z","iopub.execute_input":"2023-08-06T15:00:29.327928Z","iopub.status.idle":"2023-08-06T15:00:36.428530Z","shell.execute_reply.started":"2023-08-06T15:00:29.327892Z","shell.execute_reply":"2023-08-06T15:00:36.427568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# t1 = list(semi_files_transcriptions_data[\"transcriptions\"])\n# t1 = t1[0]\n# doc=[t1.split()]\n# w2v_model = Word2Vec(doc, min_count=1)\n# w2v_model.train(doc, total_examples=len(doc), epochs=20)\n# words = list(w2v_model.wv.key_to_index.keys())\n# word2vec = [w2v_model.wv[word] for word in words]\n# pca = PCA(n_components=1)\n# result = pca.fit_transform(word2vec)\n# result = result.reshape(12)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for spec , w2v in enumerate(single_ds):\n#     print(spec)\n#     librosa.display.specshow(spec, sr=SAMPLE_RATE, x_axis='time', y_axis='mel')\n    print(w2v)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# padding_len = 16\n# padding = tf.zeros(padding_len + 1 - len(result), dtype=tf.float32)\n# print(padding)\n# padded_vector = tf.concat([result, padding], 0)\n# padded_vector","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# semi_files_transcriptions_data = find_transcription_using_label(semi_files)\n# transcriptions_list = list(semi_files_transcriptions_data[\"transcriptions\"])\n# lda_model = encode_label(transcriptions_list)\n# #lda_list = lda_model.components_\n# ds = tf.data.Dataset.from_tensor_slices((semi_files, X))\n# #ds = ds.(list(sample_20_train[\"id\"]), list(sample_20_train[\"transcriptions\"]))\n# ds = ds.map(get_sample, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 28.creating a Datastream for all 3 dataset(for basic flow using semi data)","metadata":{}},{"cell_type":"code","source":"train_ds = config_my_datastream(semi_files).prefetch(tf.data.AUTOTUNE).batch(BATCH_SIZE).repeat(EPOCHS)\n\n# Cache validation set since it will be used repeatedly and wont change due to any data augmentation\nval_ds = config_my_datastream(semi_val_files).prefetch(tf.data.AUTOTUNE).batch(BATCH_SIZE).repeat(EPOCHS)\nval_ds.cache()\n\n# Test samples will be evaluated individually, once only\ntest_ds = config_my_datastream(semi_test_files).batch(1)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:27:27.163965Z","iopub.execute_input":"2023-08-06T15:27:27.164341Z","iopub.status.idle":"2023-08-06T15:30:04.905282Z","shell.execute_reply.started":"2023-08-06T15:27:27.164309Z","shell.execute_reply":"2023-08-06T15:30:04.903570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2023-08-06T16:20:52.534806Z","iopub.execute_input":"2023-08-06T16:20:52.535195Z","iopub.status.idle":"2023-08-06T16:20:52.542937Z","shell.execute_reply.started":"2023-08-06T16:20:52.535165Z","shell.execute_reply":"2023-08-06T16:20:52.541939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for element in train_ds:\n    print(element)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(input_dim, output_dim, rnn_layers = 5, rnn_units = 128):\n    waveform = layers.Input(batch_shape=(1,input_dim), name = \"input\")\n    logMelgram = waveform_to_log_mel_spectrogram(waveform)\n    print(logMelgram.shape)\n#     net = layers.Reshape((-1, input_dim, 1))(logMelgram)\n    net = layers.Conv2D(filters = 32, \n                        kernel_size = [11,41],\n                        strides = [2, 2],\n                        padding = \"same\",\n                        use_bias = False,\n                        name = \"conv_1\",)(logMelgram)\n    net = layers.BatchNormalization(name = \"conv1_bn\")(net)\n    net = layers.ReLU(name = \"conv1_relu\")(net)\n    print(net.shape)\n    #convolution layer 2\n    \n    net = layers.Conv2D(filters = 32, \n                        kernel_size = [11,21],\n                        strides = [1, 2],\n                        padding = \"same\",\n                        use_bias = False,\n                        name = \"conv_2\",)(net)\n    net = layers.BatchNormalization(name = \"conv2_bn\")(net)\n    net = layers.ReLU(name = \"conv2_relu\")(net)\n    print(net.shape)\n    net = layers.Reshape((-1, net.shape[-2] * net.shape[-1]))(net)\n    \n    for i in range(1, rnn_layers + 1):\n        recurrent = layers.GRU(\n            units = rnn_units,\n            activation = \"tanh\",\n            recurrent_activation = \"sigmoid\",\n            use_bias=True,\n            return_sequences = True,\n            reset_after = True,\n            name = f\"gru_{i}\",\n        )\n        net = layers.Bidirectional(\n                recurrent, name=f\"birectional_{i}\", merge_mode = \"concat\")(net)\n        if i < rnn_layers:\n            net = layers.Dropout(rate = 0.5)(net)\n          \n    net = layers.Dense(units = rnn_units * 2, name = \"dense_1\")(net)\n    net = layers.ReLU(name=\"dense_1_relu\")(net)\n    net = layers.Dropout(rate = 0.5)(net)\n    \n    output = layers.Dense(units = output_dim + 1, activation = \"softmax\")(net)\n    \n    model = keras.Model(logMelgram, output, name = \"DeepSpeech\")\n    \n    opt =  keras.optimizers.Adam(learning_rate=1e-4)\n    \n    model.compile(optimizer = opt, loss = CTCLoss)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:31:42.069617Z","iopub.execute_input":"2023-08-06T15:31:42.070582Z","iopub.status.idle":"2023-08-06T15:31:42.084796Z","shell.execute_reply.started":"2023-08-06T15:31:42.070544Z","shell.execute_reply":"2023-08-06T15:31:42.083608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def CTCLoss(y_true, y_pred):\n    batch_len = tf.cast(tf.shape(y_true)[0], dtype = \"int64\")\n    input_length = tf.cast(tf.shape(y_pred)[1], dtype = \"int64\")\n    transcription_length = tf.cast(tf.shape(y_true)[1], dtype = \"int64\")\n    \n    input_length = input_length * tf.ones(shape = (batch_len, 1), dtype = \"int64\")\n    transcription_length = transcription_length * tf.ones(shape = (batch_len, 1), dtype = \"int64\")\n    \n    loss = keras.backend.ctc_batch_cost(y_true, y_pred, input_length, transcription_length)\n    \n    return loss\n    ","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:31:48.722471Z","iopub.execute_input":"2023-08-06T15:31:48.722888Z","iopub.status.idle":"2023-08-06T15:31:48.730128Z","shell.execute_reply.started":"2023-08-06T15:31:48.722838Z","shell.execute_reply":"2023-08-06T15:31:48.729181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CallbackEval(keras.callbacks.Callback):\n    def __init__(self, dataset):\n        super().__init__()\n        self.dataset = dataset\n    \n    def on_epoch_end(self, epoch: int, logs = None):\n        predictions = []\n        targets = []\n        for batch in self.dataset:\n            X, y = batch\n            batch_predictions = model.predict(X)\n            batch_predictions = decode_batch_predictions(batch_predictions)\n            predictions.extend(batch_predictions)\n            for label in y:\n                label = (tf.strings.reduce_join(num_to_chars(label)).numpy().decode(\"utf-8\"))\n                targets.append(label)\n        \n        wer_score = wer(targets, predictions)\n        print(f\"wer score:{wer_score:.4f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tf.device(device_name):\n    model = build_model(\n            input_dim = SAMPLE_RATE,\n            output_dim = 20,\n            rnn_units = 512\n    )\n\nmodel.summary()\n\n#input_dim = FFT_SIZE // 2 + 1,\n# # Callbacks\n# checkpoint = callbacks.ModelCheckpoint(\n#     _PATH_TO_MODEL, save_best_only=True, monitor='val_loss', mode='min')\n\n# # reduce_lr = callbacks.ReduceLROnPlateau(\n# #     monitor='val_loss', factor=0.1, patience=2, verbose=1, min_lr=1e-5, mode='min')\n\n# early_stopping = callbacks.EarlyStopping(verbose=1, patience=3)\n\n# tensorboard = callbacks.TensorBoard(\n#     log_dir=_PATH_TO_RESULTS+'/logs', histogram_freq=1)\n\n\n# history = model.fit(\n#     train_ds, \n#     validation_data=val_ds,\n#     epochs=EPOCHS,\n# #     steps_per_epoch=int(train_samples/BATCH_SIZE),\n#     validation_steps=int(val_samples/BATCH_SIZE),\n#     class_weight=class_weights,\n# #     callbacks=[checkpoint, early_stopping, tensorboard]\n# )\n\n# model.save(_PATH_TO_MODEL)\n\n\n# test_labels = []\n# for _, label in test_ds:\n#   test_labels.append(label.numpy()[0])\n\n# test_labels = np.array(test_labels)\n# y_pred = np.argmax(model.predict(test_ds, verbose=1), axis=1)\n# y_true = test_labels\n\n# test_acc = sum(y_pred == y_true) / len(y_true)\n# print(\"Test set accuracy: {:.0%}\".format(test_acc))","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:31:58.324726Z","iopub.execute_input":"2023-08-06T15:31:58.325137Z","iopub.status.idle":"2023-08-06T15:32:02.232994Z","shell.execute_reply.started":"2023-08-06T15:31:58.325104Z","shell.execute_reply":"2023-08-06T15:32:02.232095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    validation_data=val_ds,\n    epochs=EPOCHS,\n#     steps_per_epoch=int(train_samples/BATCH_SIZE),\n    validation_steps=int(val_samples/BATCH_SIZE),\n#     class_weight=class_weights,\n#     callbacks=[checkpoint, early_stopping, tensorboard]\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T15:32:09.628793Z","iopub.execute_input":"2023-08-06T15:32:09.629165Z","iopub.status.idle":"2023-08-06T16:16:22.514392Z","shell.execute_reply.started":"2023-08-06T15:32:09.629133Z","shell.execute_reply":"2023-08-06T16:16:22.513402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Plot results\n# plt.figure(figsize=(20.0, 10.0))\n# plt.suptitle('{}'.format(model.name))\n# plt.subplot(1, 2, 1, label='Loss plot')\n# plt.plot(np.arange(1, len(history.history['loss'])+1), history.history['loss'])\n# plt.plot(\n#     np.arange(1, len(history.history['val_loss'])+1), history.history['val_loss'])\n# plt.ylabel('Loss')\n# plt.xlabel('Epoch')\n# plt.legend(['Training set', 'Validation set'], loc='upper left')\n\n# plt.subplot(1, 2, 2, label='Accuracy plot')\n# plt.plot(np.arange(\n#     1, len(history.history['acc'])+1), history.history['acc'])\n# plt.plot(np.arange(\n#     1, len(history.history['val_acc'])+1), history.history['val_acc'])\n# plt.ylim([0, 1])\n# plt.ylabel('Accuracy')\n# plt.xlabel('Epoch')\n# plt.legend(['Training set', 'Validation set'], loc='upper left')\n# plt.savefig(_PATH_TO_RESULTS+'/images/training_process.png')\n\n\n# confusion_matrix = tf.math.confusion_matrix(y_true, y_pred)\n# plt.figure(figsize=(10, 8))\n# sns.heatmap(confusion_matrix, cmap=\"PuBu\", robust=True,\n#             xticklabels=WORDS, yticklabels=WORDS, annot=True, fmt='g')\n# plt.xlabel('Prediction')\n# plt.ylabel('Label')\n# plt.savefig(_PATH_TO_RESULTS+'/images/confusion_matrix.png')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}