{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"datasetVersion","sourceId":6707460,"datasetId":3865741,"databundleVersionId":6791840},{"sourceType":"datasetVersion","sourceId":14893332,"datasetId":9528796,"databundleVersionId":15757409},{"sourceType":"datasetVersion","sourceId":14916160,"datasetId":9544112,"databundleVersionId":15782283},{"sourceType":"datasetVersion","sourceId":4143520,"datasetId":2447262,"databundleVersionId":4200057},{"sourceType":"datasetVersion","sourceId":14916503,"datasetId":9544348,"databundleVersionId":15782664}],"dockerImageVersionId":31259,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport torch\nimport os\nimport base64\nimport pandas as pd\nimport io\nimport re","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import pipeline\nimport torch\n\nMODEL = \"/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/bengali-whisper-medium\"\n\npipe = pipeline(\n    task=\"automatic-speech-recognition\",\n    model=MODEL,\n    tokenizer=MODEL,\n    chunk_length_s=20.1,\n    device=0 if torch.cuda.is_available() else -1,\n    batch_size=1\n)\n\npipe.model.config.forced_decoder_ids = (\n    pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n)\n\nprint(\"ASR Loaded\")\n\nfrom transformers import AutoModelForTokenClassification, AutoTokenizer\nimport torch\nimport torch.nn.functional as F\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nprint(\"Punctuation Device:\", DEVICE)\n\nPUNCT_MODELS = [\n    '/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/punct-model-6layers/',\n    '/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/punct-model-8layers/',\n    '/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/punct-model-11layers/',\n    '/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/punct-model-12layers/'\n]\n\npunct_models = [\n    AutoModelForTokenClassification.from_pretrained(f).to(DEVICE).eval()\n    for f in PUNCT_MODELS\n]\n\ntokenizer = AutoTokenizer.from_pretrained(PUNCT_MODELS[0])\n\nPUNCT_WEIGHTS = torch.FloatTensor([[1.0,1.4,1.0,0.8]]).to(DEVICE)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef fix_repetition(text, max_count=8):\n    uniq={}\n    words=text.split()\n    for w in words:\n        uniq[w]=uniq.get(w,0)+1\n    for w,c in uniq.items():\n        if c>max_count:\n            words=[x for x in words if x!=w]\n    return \" \".join(words)\n\ndef punctuate(text):\n\n    input_ids = tokenizer(text).input_ids\n\n    with torch.no_grad():\n\n        logits = F.softmax(\n            punct_models[0](\n                input_ids=torch.LongTensor([input_ids]).to(DEVICE)\n            ).logits[0,1:-1],\n            dim=1\n        )\n\n        for model in punct_models[1:]:\n            logits += F.softmax(\n                model(\n                    input_ids=torch.LongTensor([input_ids]).to(DEVICE)\n                ).logits[0,1:-1],\n                dim=1\n            )\n\n        logits = logits / len(punct_models)\n        logits *= PUNCT_WEIGHTS\n\n        label_ids = torch.argmax(logits, dim=-1)\n\n        tokens = tokenizer(text,add_special_tokens=False).input_ids\n\n        punct_text=\"\"\n\n        for i,t in enumerate(tokens):\n\n            tok=tokenizer.decode(t)\n\n            if '##' not in tok:\n                punct_text+=\" \"+tok\n            else:\n                punct_text+=tok[2:]\n\n            punct_text+=['','।',',','?'][label_ids[i].item()]\n\n    punct_text=punct_text.strip()\n\n    if punct_text[-1] not in ['।','? ',',']:\n        punct_text+='।'\n\n    return punct_text\n\ndef write_base64_to_wav(encoded_audio, sample_id):\n    \"\"\"\n    Converts base64 audio to temporary WAV file\n    and returns filepath for Whisper pipeline.\n    \"\"\"\n\n    audio_bytes = base64.b64decode(encoded_audio)\n\n    tmp_path = f\"/kaggle/working/tmp_{sample_id}.wav\"\n\n    with open(tmp_path, \"wb\") as f:\n        f.write(audio_bytes)\n\n    return tmp_path\n\n\n\n############################################\n# -------------- WER -------------------- #\n############################################\n\ndef wer(reference, hypothesis):\n    r = reference.split()\n    h = hypothesis.split()\n\n    d = [[0] * (len(h) + 1) for _ in range(len(r) + 1)]\n\n    for i in range(len(r) + 1):\n        d[i][0] = i\n    for j in range(len(h) + 1):\n        d[0][j] = j\n\n    for i in range(1, len(r) + 1):\n        for j in range(1, len(h) + 1):\n            if r[i - 1] == h[j - 1]:\n                cost = 0\n            else:\n                cost = 1\n            d[i][j] = min(\n                d[i - 1][j] + 1,\n                d[i][j - 1] + 1,\n                d[i - 1][j - 1] + cost,\n            )\n\n    return d[len(r)][len(h)] / max(len(r), 1)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef score(solution: pd.DataFrame,\n          submission: pd.DataFrame,\n          row_id_column_name: str) -> float:\n    \"\"\"\n    Computes the Mean Word Error Rate (WER) between\n    ground-truth Bengali sentences and ASR transcriptions\n    generated from base64-encoded audio in the submission.\n\n    Expected submission format:\n        id,audio_base64\n\n    Parameters\n    ----------\n    solution : pd.DataFrame\n        Ground truth dataframe containing reference sentences.\n        Must include:\n            - row_id_column_name\n            - 'sentence' column\n\n    submission : pd.DataFrame\n        Submission dataframe containing:\n            - row_id_column_name\n            - 'audio_base64' column (base64-encoded WAV audio)\n\n    row_id_column_name : str\n        Column name used as the unique identifier for each sample.\n\n    Returns\n    -------\n    float\n        Mean Word Error Rate (lower is better).\n        Returns 1.0 if submission is invalid or empty.\n    \"\"\"\n\n    if \"base64_audio\" not in submission.columns:\n        return 1.0\n\n    sub=dict(zip(submission[row_id_column_name],submission[\"base64_audio\"]))\n\n    total=0.0\n    count=0\n\n    for _,row in solution.iterrows():\n\n        if \"Usage\" in solution.columns and row[\"Usage\"]==\"Ignored\":\n            continue\n\n        sid=(row[row_id_column_name])\n\n        if sid not in sub:\n            print(type(sid))\n            continue\n\n        try:\n            wav=write_base64_to_wav(sub[sid],sid)\n\n            text=pipe(\n                wav,\n                generate_kwargs={\"max_length\":260,\"num_beams\":4}\n            )[\"text\"]\n\n            os.remove(wav)\n\n            text=fix_repetition(text)\n            text=punctuate(text)\n\n            print(row['sentence'])\n            print(text)\n\n            total+=wer(row[\"sentence\"],text)\n            count+=1\n\n        except Exception as e:\n            print(\"FAIL:\",e)\n            continue\n\n    if count==0:\n        return 1.0\n\n    return float(total/count)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport librosa\nfrom transformers import WhisperProcessor, WhisperModel\n\n# You can use the local Kaggle path from your notebook here\nMODEL_PATH = \"/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/bengali-whisper-medium\"\n\n# 1. Load the Processor and the base Model\nprint(\"Loading model and processor...\")\nprocessor = WhisperProcessor.from_pretrained(MODEL_PATH)\nmodel = WhisperModel.from_pretrained(MODEL_PATH)\nmodel.eval() # Set to evaluation mode\n\n# Move to GPU if available\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\ndef get_audio_embedding(audio_path, processor, model, device):\n    \"\"\"\n    Loads an audio file, processes it through Whisper's encoder, \n    and returns a single mean-pooled embedding vector.\n    \"\"\"\n    # Whisper natively expects audio sampled at 16kHz\n    speech_array, sampling_rate = librosa.load(audio_path, sr=16000)\n    \n    # Extract log-mel spectrogram features\n    inputs = processor(\n        speech_array, \n        sampling_rate=16000, \n        return_tensors=\"pt\"\n    )\n    \n    input_features = inputs.input_features.to(device)\n    \n    # Pass features through the encoder only (we don't need the decoder for embeddings)\n    with torch.no_grad():\n        encoder_outputs = model.encoder(input_features)\n        \n    # encoder_outputs.last_hidden_state shape: (batch_size, sequence_length, hidden_size)\n    # Average across the sequence length (dim=1) to get a single vector per audio file\n    embedding = encoder_outputs.last_hidden_state.mean(dim=1)\n    \n    return embedding\n\n# --- Example Usage ---\n# Assuming you have two saved audio files:\ntruth_path = \"/kaggle/input/datasets/tajulislamtarek/debug-data/1.wav\"\n#pred_path = \"predicted.wav\"\n\nemb_truth = get_audio_embedding(truth_path, processor, model, device)\n#emb_pred = get_audio_embedding(pred_path, processor, model, device)\n\n# 2. Calculate Cosine Similarity\n#cos_sim = F.cosine_similarity(emb_truth, emb_pred)\nprint(f\"Cosine Similarity: \",emb_truth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T05:41:21.864006Z","iopub.execute_input":"2026-02-22T05:41:21.865037Z","iopub.status.idle":"2026-02-22T05:42:49.435440Z","shell.execute_reply.started":"2026-02-22T05:41:21.864976Z","shell.execute_reply":"2026-02-22T05:42:49.434448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"emb_truth = get_audio_embedding(truth_path, processor, model, device)\n#emb_pred = get_audio_embedding(pred_path, processor, model, device)\n\n# 2. Calculate Cosine Similarity\n#cos_sim = F.cosine_similarity(emb_truth, emb_pred)\nprint(f\"Cosine Similarity: \",emb_truth)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T05:43:39.164283Z","iopub.execute_input":"2026-02-22T05:43:39.164623Z","iopub.status.idle":"2026-02-22T05:43:50.568413Z","shell.execute_reply.started":"2026-02-22T05:43:39.164589Z","shell.execute_reply":"2026-02-22T05:43:50.567456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ni_path=\"/kaggle/input/datasets/rafidadib/niloy-voice/n.wav\"\n\nniga_pred = get_audio_embedding(ni_path, processor, model, device)\n\n\ncos_sim = F.cosine_similarity(emb_truth, niga_pred)\nprint(f\"Cosine Similarity same: {cos_sim.item():.4f}\")\n\n\ndiff_path=\"/kaggle/input/datasets/tajulislamtarek/debug-data/2.wav\"\n\ndiff_pred = get_audio_embedding(diff_path, processor, model, device)\n\n\ncos_sim_diff = F.cosine_similarity(emb_truth, diff_pred)\nprint(f\"Cosine Similarity diff: {cos_sim_diff.item():.4f}\")\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T05:53:01.235505Z","iopub.execute_input":"2026-02-22T05:53:01.236021Z","iopub.status.idle":"2026-02-22T05:53:44.203549Z","shell.execute_reply.started":"2026-02-22T05:53:01.235967Z","shell.execute_reply":"2026-02-22T05:53:44.202840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"diff_path=\"/kaggle/input/datasets/tajulislamtarek/debug-data/3.wav\"\n\ndiff_pred = get_audio_embedding(diff_path, processor, model, device)\n\n\ncos_sim_diff = F.cosine_similarity(emb_truth, diff_pred)\nprint(f\"Cosine Similarity diff: {cos_sim_diff.item():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T05:54:31.985376Z","iopub.execute_input":"2026-02-22T05:54:31.986170Z","iopub.status.idle":"2026-02-22T05:54:43.401663Z","shell.execute_reply.started":"2026-02-22T05:54:31.986125Z","shell.execute_reply":"2026-02-22T05:54:43.400610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import librosa\nimport librosa.sequence\nimport numpy as np\n\ndef calculate_mcd_librosa(truth_path, pred_path, sr=16000, n_mfcc=13):\n    \"\"\"\n    Calculates the Mel Cepstral Distortion (MCD) using librosa and DTW.\n    \"\"\"\n    # 1. Load the audio files\n    y_truth, _ = librosa.load(truth_path, sr=sr)\n    y_pred, _ = librosa.load(pred_path, sr=sr)\n\n    # 2. Extract MFCCs\n    # We extract n_mfcc + 1, then drop the 0th coefficient (energy)\n    mfcc_truth = librosa.feature.mfcc(y=y_truth, sr=sr, n_mfcc=n_mfcc + 1)[1:]\n    mfcc_pred = librosa.feature.mfcc(y=y_pred, sr=sr, n_mfcc=n_mfcc + 1)[1:]\n\n    # 3. Align the sequences using Dynamic Time Warping (DTW)\n    # librosa.sequence.dtw returns the accumulated cost matrix (D) and the warping path (wp)\n    D, wp = librosa.sequence.dtw(X=mfcc_truth, Y=mfcc_pred, metric='euclidean')\n\n    # 4. Calculate the MCD formula\n    # MCD scaling factor: (10 * sqrt(2)) / ln(10)\n    scale = (10.0 * np.sqrt(2.0)) / np.log(10.0)\n    \n    total_distance = 0.0\n    \n    # wp is a list of (truth_index, pred_index) pairs\n    for truth_idx, pred_idx in wp:\n        # Get the feature vectors for the aligned frames\n        frame_truth = mfcc_truth[:, truth_idx]\n        frame_pred = mfcc_pred[:, pred_idx]\n        \n        # Euclidean distance between the aligned frames\n        diff = frame_truth - frame_pred\n        distance = np.sqrt(np.sum(diff ** 2))\n        \n        total_distance += distance\n        \n    # Average the distance over the length of the warping path\n    mcd = scale * (total_distance / len(wp))\n    \n    return mcd\n\n# Example Usage:\nscore = calculate_mcd_librosa(\"/kaggle/input/datasets/tajulislamtarek/debug-data/1.wav\", \"/kaggle/input/datasets/rafidadib/silsila/sil.wav\")\nprint(f\"MCD Score: {score:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T06:39:51.657078Z","iopub.execute_input":"2026-02-22T06:39:51.658254Z","iopub.status.idle":"2026-02-22T06:39:51.726190Z","shell.execute_reply.started":"2026-02-22T06:39:51.658207Z","shell.execute_reply":"2026-02-22T06:39:51.724188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}