{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21685,"databundleVersionId":1357306,"sourceType":"competition"}],"dockerImageVersionId":30458,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        pass\n        # print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-25T23:18:52.060216Z","iopub.execute_input":"2024-11-25T23:18:52.06069Z","iopub.status.idle":"2024-11-25T23:19:17.30811Z","shell.execute_reply.started":"2024-11-25T23:18:52.06065Z","shell.execute_reply":"2024-11-25T23:19:17.306768Z"},"trusted":true},"outputs":[],"execution_count":2},{"cell_type":"markdown","source":"## **Install Required Libraries**","metadata":{}},{"cell_type":"code","source":"!pip install librosa torch torchaudio transformers datasets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:19:19.517012Z","iopub.execute_input":"2024-11-25T23:19:19.518415Z","iopub.status.idle":"2024-11-25T23:26:33.476212Z","shell.execute_reply.started":"2024-11-25T23:19:19.518367Z","shell.execute_reply":"2024-11-25T23:26:33.474665Z"}},"outputs":[{"name":"stdout","text":"Requirement already satisfied: librosa in /opt/conda/lib/python3.7/site-packages (0.10.0.post2)\nRequirement already satisfied: torch in /opt/conda/lib/python3.7/site-packages (1.13.0+cpu)\nRequirement already satisfied: torchaudio in /opt/conda/lib/python3.7/site-packages (0.13.0+cpu)\nRequirement already satisfied: transformers in /opt/conda/lib/python3.7/site-packages (4.27.4)\nRequirement already satisfied: datasets in /opt/conda/lib/python3.7/site-packages (2.1.0)\nRequirement already satisfied: scikit-learn>=0.20.0 in /opt/conda/lib/python3.7/site-packages (from librosa) (1.0.2)\nRequirement already satisfied: msgpack>=1.0 in /opt/conda/lib/python3.7/site-packages (from librosa) (1.0.4)\nRequirement already satisfied: soxr>=0.3.2 in /opt/conda/lib/python3.7/site-packages (from librosa) (0.3.4)\nRequirement already satisfied: pooch<1.7,>=1.0 in /opt/conda/lib/python3.7/site-packages (from librosa) (1.6.0)\n\u001b[33mWARNING: Retrying (Retry(total=4, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc4150>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/soundfile/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=3, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75ffba90>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/soundfile/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=2, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc0290>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/soundfile/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=1, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc4310>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/soundfile/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=0, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc4290>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/soundfile/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=4, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75ff0350>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/librosa/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=3, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc4290>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/librosa/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=2, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc4850>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/librosa/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=1, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75fc41d0>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/librosa/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=0, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b0d75ff04d0>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/librosa/\u001b[0m\u001b[33m\n\u001b[0m\u001b[31mERROR: Could not find a version that satisfies the requirement soundfile>=0.12.1 (from librosa) (from versions: none)\u001b[0m\u001b[31m\n\u001b[0m\u001b[31mERROR: No matching distribution found for soundfile>=0.12.1\u001b[0m\u001b[31m\n\u001b[0m\u001b[33mWARNING: There was an error checking the latest version of pip.\u001b[0m\u001b[33m\n\u001b[0m","output_type":"stream"}],"execution_count":3},{"cell_type":"markdown","source":"## **Preprocess the Data**","metadata":{}},{"cell_type":"code","source":"import librosa\nimport numpy as np\nimport torchaudio\n\n# Preprocess audio: Resample, normalize, and extract features\ndef preprocess_audio(audio_path, target_sr=16000):\n    # Load the audio file\n    y, sr = librosa.load(audio_path, sr=target_sr, mono=True)\n    \n    # Normalize audio\n    y = librosa.util.normalize(y)\n    \n    # Extract Mel-spectrogram\n    mel_spec = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128, fmax=8000)\n    log_mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n    \n    return log_mel_spec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:26:33.478949Z","iopub.execute_input":"2024-11-25T23:26:33.479415Z","iopub.status.idle":"2024-11-25T23:26:36.852834Z","shell.execute_reply.started":"2024-11-25T23:26:33.479367Z","shell.execute_reply":"2024-11-25T23:26:36.851306Z"}},"outputs":[],"execution_count":4},{"cell_type":"markdown","source":"## **Define the Model**","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torchaudio.transforms as transforms\n\nclass QuranSTTModel(nn.Module):\n    def __init__(self, input_dim, hidden_dim, output_dim):\n        super(QuranSTTModel, self).__init__()\n        # Encoder\n        self.lstm = nn.LSTM(input_dim, hidden_dim, num_layers=3, batch_first=True, bidirectional=True)\n        # Fully connected layer\n        self.fc = nn.Linear(hidden_dim * 2, output_dim)  # Bi-directional doubles the hidden_dim\n        self.softmax = nn.LogSoftmax(dim=2)  # For CTC loss\n\n    def forward(self, x):\n        x, _ = self.lstm(x)\n        x = self.fc(x)\n        x = self.softmax(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:26:49.954371Z","iopub.execute_input":"2024-11-25T23:26:49.955662Z","iopub.status.idle":"2024-11-25T23:26:49.964466Z","shell.execute_reply.started":"2024-11-25T23:26:49.955618Z","shell.execute_reply":"2024-11-25T23:26:49.963088Z"}},"outputs":[],"execution_count":5},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\n# Custom Dataset for Quranic STT\nclass QuranDataset(Dataset):\n    def __init__(self, audio_paths, transcriptions, vocab, sample_rate=16000):\n        self.audio_paths = audio_paths\n        self.transcriptions = transcriptions\n        self.char_to_index = {char: idx for idx, char in enumerate(vocab)}\n        self.sample_rate = sample_rate\n\n    def __len__(self):\n        return len(self.audio_paths)\n\n    def __getitem__(self, idx):\n        # Load and preprocess audio\n        audio_path = self.audio_paths[idx]\n        y, sr = librosa.load(audio_path, sr=self.sample_rate, mono=True)\n        y = librosa.util.normalize(y)\n        mel_spec = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128, fmax=8000)\n        log_mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n\n        # Convert transcription to indices\n        transcription = self.transcriptions[idx]\n        target = torch.tensor([self.char_to_index[char] for char in transcription if char in self.char_to_index])\n\n        return torch.tensor(log_mel_spec.T, dtype=torch.float32), target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:27:17.636313Z","iopub.execute_input":"2024-11-25T23:27:17.637246Z","iopub.status.idle":"2024-11-25T23:27:17.647465Z","shell.execute_reply.started":"2024-11-25T23:27:17.637198Z","shell.execute_reply":"2024-11-25T23:27:17.646227Z"}},"outputs":[],"execution_count":7},{"cell_type":"code","source":"# Define the Quranic vocabulary\nquran_vocab = sorted(list(\" ءابةتثجحخدذرزسشصضطظعغفقكلمنهوي\"))  # Include Arabic characters and spaces\nchar_to_index = {char: idx for idx, char in enumerate(quran_vocab)}\nindex_to_char = {idx: char for idx, char in enumerate(quran_vocab)}\n\n\n# Model parameters\ninput_dim = 128  # Number of Mel frequency bins\nhidden_dim = 256\noutput_dim = len(quran_vocab)  # Vocabulary size\n\n# Define training parameters\nnum_epochs = 10  # Set the number of training epochs\nbatch_size = 32  # Adjust based on your GPU/CPU memory capacity\n\n# Example audio paths and transcriptions\naudio_paths = [r'/kaggle/input/quran-asr-challenge/test_set/18250218.mp3']  # Replace with your actual file paths\ntranscriptions = [\"كلا إن كتاب الأبرار لفي عليين\"]  # Replace with actual transcriptions\n\n# Create Dataset and DataLoader\ndataset = QuranDataset(audio_paths, transcriptions, quran_vocab)\ntrain_loader = DataLoader(dataset, batch_size=batch_size, shuffle=True, collate_fn=None)\n\ndef collate_fn(batch):\n    features, targets = zip(*batch)\n    input_lengths = [feat.shape[0] for feat in features]\n    target_lengths = [len(tgt) for tgt in targets]\n    padded_features = torch.nn.utils.rnn.pad_sequence(features, batch_first=True)\n    padded_targets = torch.nn.utils.rnn.pad_sequence(targets, batch_first=True)\n    return padded_features, padded_targets, input_lengths, target_lengths\n\ntrain_loader = DataLoader(dataset, batch_size=batch_size, shuffle=True, collate_fn=collate_fn)\n\n## **Train the Model**\n\n# Instantiate the model\nmodel = QuranSTTModel(input_dim, hidden_dim, output_dim)\n\n# Define loss and optimizer\ncriterion = nn.CTCLoss(blank=0)  # Use CTC loss\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:27:59.074829Z","iopub.execute_input":"2024-11-25T23:27:59.076257Z","iopub.status.idle":"2024-11-25T23:27:59.133335Z","shell.execute_reply.started":"2024-11-25T23:27:59.076199Z","shell.execute_reply":"2024-11-25T23:27:59.132214Z"}},"outputs":[],"execution_count":10},{"cell_type":"code","source":"# Training loop\nfor epoch in range(num_epochs):\n    model.train()\n    epoch_loss = 0.0\n\n    for audio_features, targets, input_lengths, target_lengths in train_loader:\n        # Forward pass\n        outputs = model(audio_features)\n\n        # Compute loss\n        loss = criterion(outputs.transpose(0, 1), targets, input_lengths, target_lengths)\n\n        # Backward pass\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        # Track loss\n        epoch_loss += loss.item()\n\n    print(f\"Epoch {epoch + 1}/{num_epochs}, Loss: {epoch_loss / len(train_loader)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:27:59.987379Z","iopub.execute_input":"2024-11-25T23:27:59.988345Z","iopub.status.idle":"2024-11-25T23:28:22.39743Z","shell.execute_reply.started":"2024-11-25T23:27:59.988299Z","shell.execute_reply":"2024-11-25T23:28:22.395838Z"}},"outputs":[{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/10, Loss: 32.77109146118164\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 2/10, Loss: 25.595293045043945\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/10, Loss: 12.602422714233398\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 4/10, Loss: 3.0008976459503174\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 5/10, Loss: 1.7528173923492432\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 6/10, Loss: 2.3378512859344482\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 7/10, Loss: 2.5876474380493164\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 8/10, Loss: 2.5240557193756104\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 9/10, Loss: 2.2735626697540283\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.7/site-packages/ipykernel_launcher.py:16: UserWarning: PySoundFile failed. Trying audioread instead.\n  \n/opt/conda/lib/python3.7/site-packages/librosa/core/audio.py:184: FutureWarning: librosa.core.audio.__audioread_load\n\tDeprecated as of librosa version 0.10.0.\n\tIt will be removed in librosa version 1.0.\n  y, sr_native = __audioread_load(path, offset, duration, dtype)\n","output_type":"stream"},{"name":"stdout","text":"Epoch 10/10, Loss: 1.9296514987945557\n","output_type":"stream"}],"execution_count":11},{"cell_type":"code","source":"!pip install jiwer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:28:53.753764Z","iopub.execute_input":"2024-11-25T23:28:53.755058Z","iopub.status.idle":"2024-11-25T23:32:47.492508Z","shell.execute_reply.started":"2024-11-25T23:28:53.754999Z","shell.execute_reply":"2024-11-25T23:32:47.491033Z"}},"outputs":[{"name":"stdout","text":"\u001b[33mWARNING: Retrying (Retry(total=4, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x785680f5e390>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/jiwer/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=3, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x785680f5e850>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/jiwer/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=2, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x785680f5ee10>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/jiwer/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=1, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x785680f4c150>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/jiwer/\u001b[0m\u001b[33m\n\u001b[0m\u001b[33mWARNING: Retrying (Retry(total=0, connect=None, read=None, redirect=None, status=None)) after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x785680f4c5d0>: Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/jiwer/\u001b[0m\u001b[33m\n\u001b[0m\u001b[31mERROR: Could not find a version that satisfies the requirement jiwer (from versions: none)\u001b[0m\u001b[31m\n\u001b[0m\u001b[31mERROR: No matching distribution found for jiwer\u001b[0m\u001b[31m\n\u001b[0m\u001b[33mWARNING: There was an error checking the latest version of pip.\u001b[0m\u001b[33m\n\u001b[0m","output_type":"stream"}],"execution_count":13},{"cell_type":"code","source":"## **Evaluate the Model**\n\nfrom jiwer import wer, cer\n\n# Evaluation\nmodel.eval()\npredictions, references = [], []\n\nwith torch.no_grad():\n    for batch in test_loader:\n        audio_features, targets, input_lengths, target_lengths = batch\n        \n        # Get model predictions\n        outputs = model(audio_features)\n        decoded_preds = decode_predictions(outputs)\n        decoded_targets = decode_targets(targets)\n        \n        predictions.extend(decoded_preds)\n        references.extend(decoded_targets)\n\n# Compute WER and CER\nprint(\"Word Error Rate (WER):\", wer(references, predictions))\nprint(\"Character Error Rate (CER):\", cer(references, predictions))\n\n## **Transcribe New Audio**\n\nfrom jiwer import wer, cer\n\n# Evaluation\nmodel.eval()\npredictions, references = [], []\n\nwith torch.no_grad():\n    for batch in test_loader:\n        audio_features, targets, input_lengths, target_lengths = batch\n        \n        # Get model predictions\n        outputs = model(audio_features)\n        decoded_preds = decode_predictions(outputs)\n        decoded_targets = decode_targets(targets)\n        \n        predictions.extend(decoded_preds)\n        references.extend(decoded_targets)\n\n# Compute WER and CER\nprint(\"Word Error Rate (WER):\", wer(references, predictions))\nprint(\"Character Error Rate (CER):\", cer(references, predictions))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T23:32:54.601918Z","iopub.execute_input":"2024-11-25T23:32:54.603062Z","iopub.status.idle":"2024-11-25T23:32:54.657399Z","shell.execute_reply.started":"2024-11-25T23:32:54.602999Z","shell.execute_reply":"2024-11-25T23:32:54.655352Z"}},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mModuleNotFoundError\u001b[0m                       Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_27/1392252994.py\u001b[0m in \u001b[0;36m<module>\u001b[0;34m\u001b[0m\n\u001b[1;32m      1\u001b[0m \u001b[0;31m## **Evaluate the Model**\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      2\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m----> 3\u001b[0;31m \u001b[0;32mfrom\u001b[0m \u001b[0mjiwer\u001b[0m \u001b[0;32mimport\u001b[0m \u001b[0mwer\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcer\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m      4\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      5\u001b[0m \u001b[0;31m# Evaluation\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mModuleNotFoundError\u001b[0m: No module named 'jiwer'"],"ename":"ModuleNotFoundError","evalue":"No module named 'jiwer'","output_type":"error"}],"execution_count":14},{"cell_type":"code","source":"# Transcription example\naudio_path = \"/kaggle/input/quran-asr-challenge/test_set/18250244.mp3\"\nlog_mel_spec = preprocess_audio(audio_path)\n\n# Convert to tensor and add batch dimension\ninput_tensor = torch.tensor(log_mel_spec).unsqueeze(0)\n\n# Get transcription\nmodel.eval()\nwith torch.no_grad():\n    output = model(input_tensor)\n    transcription = decode_predictions(output)\nprint(\"Transcription:\", transcription)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}