{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install jiwer\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-06T11:33:56.014731Z","iopub.execute_input":"2023-09-06T11:33:56.015486Z","iopub.status.idle":"2023-09-06T11:34:10.522997Z","shell.execute_reply.started":"2023-09-06T11:33:56.015451Z","shell.execute_reply":"2023-09-06T11:34:10.521945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom jiwer import wer\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom IPython import display\nimport tensorflow_io as tfio","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:10.525956Z","iopub.execute_input":"2023-09-06T11:34:10.526634Z","iopub.status.idle":"2023-09-06T11:34:18.491632Z","shell.execute_reply.started":"2023-09-06T11:34:10.526595Z","shell.execute_reply":"2023-09-06T11:34:18.490581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b_ai = pd.read_csv('/kaggle/input/bengaliai-speech/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:18.493154Z","iopub.execute_input":"2023-09-06T11:34:18.493875Z","iopub.status.idle":"2023-09-06T11:34:23.524237Z","shell.execute_reply.started":"2023-09-06T11:34:18.493838Z","shell.execute_reply":"2023-09-06T11:34:23.523262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b_ai['id'].head()","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:23.525575Z","iopub.execute_input":"2023-09-06T11:34:23.526003Z","iopub.status.idle":"2023-09-06T11:34:23.540502Z","shell.execute_reply.started":"2023-09-06T11:34:23.525970Z","shell.execute_reply":"2023-09-06T11:34:23.539545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"b_ai['sentence'].head(1)","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:23.544095Z","iopub.execute_input":"2023-09-06T11:34:23.544972Z","iopub.status.idle":"2023-09-06T11:34:23.554147Z","shell.execute_reply.started":"2023-09-06T11:34:23.544937Z","shell.execute_reply":"2023-09-06T11:34:23.553075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = int(len(b_ai)*0.9)\ndf_train = b_ai[:split]\ndf_val = b_ai[split:]\n\nprint(f\"size of the training set: {len(df_train)}\")\nprint(f\"size of the validation set: {len(df_val)}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:23.555595Z","iopub.execute_input":"2023-09-06T11:34:23.555847Z","iopub.status.idle":"2023-09-06T11:34:23.564302Z","shell.execute_reply.started":"2023-09-06T11:34:23.555824Z","shell.execute_reply":"2023-09-06T11:34:23.563224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"characters = [x for x in \"অআইঈউঊঋঌএঐওঔকখগঘঙচছজঝঞটঠডঢণতথদধনপফবভমযরলশষসহৎড়ঢ়য়০১২৩৪৫৬৭৮৯ৰৱৠৡ৲৳.।॥৺ঀঽ◌়◌ৢ◌ৣ ী া ি◌ূ◌ৃ◌ৄ ে ৈ! ো ৌ?◌্ ৗ◌ঁ ং ঃ\"]\nunique_characters = list(set(characters))  # Remove duplicates\n\n# Mapping characters to integers\nchar_to_num = keras.layers.StringLookup(vocabulary=unique_characters, oov_token=\"\")\n# Mapping integers back to original characters\nnum_to_char = keras.layers.StringLookup(\n    vocabulary=char_to_num.get_vocabulary(), oov_token=\"\", invert=True\n)\n\nprint(\n    f\"The vocabulary is: {char_to_num.get_vocabulary()} \"\n    f\"(size ={char_to_num.vocabulary_size()})\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:23.565948Z","iopub.execute_input":"2023-09-06T11:34:23.566303Z","iopub.status.idle":"2023-09-06T11:34:26.598507Z","shell.execute_reply.started":"2023-09-06T11:34:23.566273Z","shell.execute_reply":"2023-09-06T11:34:26.596669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# An integer scalar Tensor. The window length in samples.\nframe_length = 256\n# An integer scalar Tensor. The number of samples to step.\nframe_step = 160\n# An integer scalar Tensor. The size of the FFT to apply.\n# If not provided, uses the smallest power of 2 enclosing frame_length.\nfft_length = 384\nwave_path = '/kaggle/input/bengaliai-speech/train_mp3s/'\ndef encode_signal_sample(wav_file,label):\n     ###########################################\n    ##  Process the Audio\n    ##########################################\n    # 1. Read wav file\n    file = tf.io.read_file(wave_path+wav_file+\".mp3\")\n    # 2. Decode the wav file\n    audio = tfio.audio.decode_mp3(file)  # Set desired_channels based on your needs desired_channels=1 means monoaudio\n    audio = tf.squeeze(audio, axis=-1)\n    # 3. Change type to float\n    audio = tf.cast(audio, tf.float32)\n    # 4. Get the spectrogram\n    spectrogram = tf.signal.stft(\n        audio, frame_length=frame_length, frame_step=frame_step, fft_length=fft_length\n    )\n    # 5. We only need the magnitude, which can be derived by applying tf.abs\n    spectrogram = tf.abs(spectrogram)\n    spectrogram = tf.math.pow(spectrogram, 0.5)\n    # 6. normalisation\n    means = tf.math.reduce_mean(spectrogram, 1, keepdims=True)\n    stddevs = tf.math.reduce_std(spectrogram, 1, keepdims=True)\n    spectrogram = (spectrogram - means) / (stddevs + 1e-10)\n    ###########################################\n    ##  Process the label\n    ##########################################\n    # 7. Convert label to Lower case\n    label = tf.strings.lower(label)\n    # 8. Split the label\n    label = tf.strings.unicode_split(label, input_encoding=\"UTF-8\")\n    # 9. Map the characters in label to numbers\n    label = char_to_num(label)\n    # 10. Return a dict as our model is expecting two inputs\n    return spectrogram, label\n","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:26.600000Z","iopub.execute_input":"2023-09-06T11:34:26.600366Z","iopub.status.idle":"2023-09-06T11:34:26.609871Z","shell.execute_reply.started":"2023-09-06T11:34:26.600333Z","shell.execute_reply":"2023-09-06T11:34:26.608965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\n# Define the training dataset\ntrain_dataset = tf.data.Dataset.from_tensor_slices(\n    (list(df_train['id']), list(df_train['sentence']))\n)\ntrain_dataset = (\n    train_dataset.map(encode_signal_sample, num_parallel_calls=tf.data.AUTOTUNE)\n    .padded_batch(batch_size)\n    .prefetch(buffer_size=tf.data.AUTOTUNE)\n)\n\n# Define the validation dataset\nvalidation_dataset = tf.data.Dataset.from_tensor_slices(\n    (list(df_val['id']), list(df_val['sentence']))\n)\nvalidation_dataset = (\n    validation_dataset.map(encode_signal_sample, num_parallel_calls=tf.data.AUTOTUNE)\n    .padded_batch(batch_size)\n    .prefetch(buffer_size=tf.data.AUTOTUNE)\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:26.611421Z","iopub.execute_input":"2023-09-06T11:34:26.612102Z","iopub.status.idle":"2023-09-06T11:34:34.403736Z","shell.execute_reply.started":"2023-09-06T11:34:26.612070Z","shell.execute_reply":"2023-09-06T11:34:34.402761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(8, 5))\nfor batch in train_dataset.take(1):\n    spectrogram = batch[0][0].numpy()\n    spectrogram = np.array([np.trim_zeros(x) for x in np.transpose(spectrogram)])\n    label = batch[1][0]\n    # Spectrogram\n    label = tf.strings.reduce_join(num_to_char(label)).numpy().decode(\"utf-8\")\n    ax = plt.subplot(2, 1, 1)\n    ax.imshow(spectrogram, vmax=1)\n    ax.set_title(label)\n    ax.axis(\"off\")\n    # Wav\n    file = tf.io.read_file(wave_path + list(df_train['id'])[0] + \".mp3\")\n    audio = tfio.audio.decode_mp3(file)\n    audio = audio.numpy()\n    ax = plt.subplot(2, 1, 2)\n    plt.plot(audio)\n    ax.set_title(\"Signal Wave\")\n    ax.set_xlim(0, len(audio))\n    display.display(display.Audio(np.transpose(audio), rate=16000))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:34.405382Z","iopub.execute_input":"2023-09-06T11:34:34.405793Z","iopub.status.idle":"2023-09-06T11:34:37.835632Z","shell.execute_reply.started":"2023-09-06T11:34:34.405756Z","shell.execute_reply":"2023-09-06T11:34:37.834695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def CTCLoss(y_true,y_pred):\n    # compute the treaining-time loss value\n    batch_len = tf.cast(tf.shape(y_true)[0], dtype='int64')\n    input_length = tf.cast(tf.shape(y_pred)[1], dtype = 'int64')\n    label_length = tf.cast(tf.shape(y_true)[1],dtype='int64')\n    \n    input_length = input_length*tf.ones(shape=(batch_len,1), dtype='int64')\n    label_length = label_length*tf.ones(shape=(batch_len,1), dtype='int64')\n    \n    loss = keras.backend.ctc_batch_cost(y_true,y_pred,input_length,label_length)\n    return loss","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:37.836853Z","iopub.execute_input":"2023-09-06T11:34:37.838387Z","iopub.status.idle":"2023-09-06T11:34:37.846155Z","shell.execute_reply.started":"2023-09-06T11:34:37.838352Z","shell.execute_reply":"2023-09-06T11:34:37.844660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef build_model(input_dim, output_dim, rnn_layers=5, rnn_units=128):\n    \"\"\"Model similar to DeepSpeech2.\"\"\"\n    # Model's input\n    input_spectrogram = layers.Input((None, input_dim), name=\"input\")\n    # Expand the dimension to use 2D CNN.\n    x = layers.Reshape((-1, input_dim, 1), name=\"expand_dim\")(input_spectrogram)\n    # Convolution layer 1\n    x = layers.Conv2D(\n        filters=32,\n        kernel_size=[11, 41],\n        strides=[2, 2],\n        padding=\"same\",\n        use_bias=False,\n        name=\"conv_1\",\n    )(x)\n    x = layers.BatchNormalization(name=\"conv_1_bn\")(x)\n    x = layers.ReLU(name=\"conv_1_relu\")(x)\n    # Convolution layer 2\n    x = layers.Conv2D(\n        filters=32,\n        kernel_size=[11, 21],\n        strides=[1, 2],\n        padding=\"same\",\n        use_bias=False,\n        name=\"conv_2\",\n    )(x)\n    x = layers.BatchNormalization(name=\"conv_2_bn\")(x)\n    x = layers.ReLU(name=\"conv_2_relu\")(x)\n    # Reshape the resulted volume to feed the RNNs layers\n    x = layers.Reshape((-1, x.shape[-2] * x.shape[-1]))(x)\n    # RNN layers\n    for i in range(1, rnn_layers + 1):\n        recurrent = layers.GRU(\n            units=rnn_units,\n            activation=\"tanh\",\n            recurrent_activation=\"sigmoid\",\n            use_bias=True,\n            return_sequences=True,\n            reset_after=True,\n            name=f\"gru_{i}\",\n        )\n        x = layers.Bidirectional(\n            recurrent, name=f\"bidirectional_{i}\", merge_mode=\"concat\"\n        )(x)\n        if i < rnn_layers:\n            x = layers.Dropout(rate=0.5)(x)\n    # Dense layer\n    x = layers.Dense(units=rnn_units * 2, name=\"dense_1\")(x)\n    x = layers.ReLU(name=\"dense_1_relu\")(x)\n    x = layers.Dropout(rate=0.5)(x)\n    # Classification layer\n    output = layers.Dense(units=output_dim + 1, activation=\"softmax\")(x)\n    # Model\n    model = keras.Model(input_spectrogram, output, name=\"DeepSpeech_2\")\n    # Optimizer\n    opt = keras.optimizers.Adam(learning_rate=1e-4)\n    # Compile the model and return\n    model.compile(optimizer=opt, loss=CTCLoss)\n    return model\n\n\n# Get the model\nmodel = build_model(\n    input_dim=fft_length // 2 + 1,\n    output_dim=char_to_num.vocabulary_size(),\n    rnn_units=512,\n)\nmodel.summary(line_length=110)","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:37.847813Z","iopub.execute_input":"2023-09-06T11:34:37.848174Z","iopub.status.idle":"2023-09-06T11:34:41.053543Z","shell.execute_reply.started":"2023-09-06T11:34:37.848124Z","shell.execute_reply":"2023-09-06T11:34:41.052812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# A utility function to decode the output of the network\ndef decode_batch_predictions(pred):\n    input_len = np.ones(pred.shape[0]) * pred.shape[1]\n    # Use greedy search. For complex tasks, you can use beam search\n    results = keras.backend.ctc_decode(pred, input_length=input_len, greedy=True)[0][0]\n    # Iterate over the results and get back the text\n    output_text = []\n    for result in results:\n        result = tf.strings.reduce_join(num_to_char(result)).numpy().decode(\"utf-8\")\n        output_text.append(result)\n    return output_text\n\n\n# A callback class to output a few transcriptions during training\nclass CallbackEval(keras.callbacks.Callback):\n    \"\"\"Displays a batch of outputs after every epoch.\"\"\"\n\n    def __init__(self, dataset):\n        super().__init__()\n        self.dataset = dataset\n\n    def on_epoch_end(self, epoch: int, logs=None):\n        predictions = []\n        targets = []\n        for batch in self.dataset:\n            X, y = batch\n            batch_predictions = model.predict(X)\n            batch_predictions = decode_batch_predictions(batch_predictions)\n            predictions.extend(batch_predictions)\n            for label in y:\n                label = (\n                    tf.strings.reduce_join(num_to_char(label)).numpy().decode(\"utf-8\")\n                )\n                targets.append(label)\n        wer_score = wer(targets, predictions)\n        print(\"-\" * 100)\n        print(f\"Word Error Rate: {wer_score:.4f}\")\n        print(\"-\" * 100)\n        for i in np.random.randint(0, len(predictions), 2):\n            print(f\"Target    : {targets[i]}\")\n            print(f\"Prediction: {predictions[i]}\")\n            print(\"-\" * 100)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:41.054699Z","iopub.execute_input":"2023-09-06T11:34:41.055102Z","iopub.status.idle":"2023-09-06T11:34:41.076812Z","shell.execute_reply.started":"2023-09-06T11:34:41.055075Z","shell.execute_reply":"2023-09-06T11:34:41.076055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the number of epochs.\nepochs = 1\n# Callback function to check transcription on the val set.\nvalidation_callback = CallbackEval(validation_dataset)\n# Train the model\nhistory = model.fit(\n    train_dataset,\n    validation_data=validation_dataset,\n    epochs=epochs,\n    callbacks=[validation_callback],\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-06T11:34:41.080738Z","iopub.execute_input":"2023-09-06T11:34:41.081056Z","iopub.status.idle":"2023-09-06T13:20:07.775930Z","shell.execute_reply.started":"2023-09-06T11:34:41.081024Z","shell.execute_reply":"2023-09-06T13:20:07.773709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's check results on more validation samples\npredictions = []\ntargets = []\nfor batch in validation_dataset:\n    X, y = batch\n    batch_predictions = model.predict(X)\n    batch_predictions = decode_batch_predictions(batch_predictions)\n    predictions.extend(batch_predictions)\n    for label in y:\n        label = tf.strings.reduce_join(num_to_char(label)).numpy().decode(\"utf-8\")\n        targets.append(label)\nwer_score = wer(targets, predictions)\nprint(\"-\" * 100)\nprint(f\"Word Error Rate: {wer_score:.4f}\")\nprint(\"-\" * 100)\nfor i in np.random.randint(0, len(predictions), 5):\n    print(f\"Target    : {targets[i]}\")\n    print(f\"Prediction: {predictions[i]}\")\n    print(\"-\" * 100)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-06T13:20:07.776846Z","iopub.status.idle":"2023-09-06T13:20:07.777254Z","shell.execute_reply.started":"2023-09-06T13:20:07.777035Z","shell.execute_reply":"2023-09-06T13:20:07.777054Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}