{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":129276,"databundleVersionId":15506988}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[],"machine_shape":"hm","gpuType":"A100"},"accelerator":"GPU","widgets":{"application/vnd.jupyter.widget-state+json":{"3b13d52f827b452b9bf7ccee3436abb3":{"model_module":"@jupyter-widgets/controls","model_name":"HBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_7825b6a74d174e16b8ffeb77798824d5","IPY_MODEL_6108b7c4ccf04a0c8e7a062ea79c7769","IPY_MODEL_b6fd6e5b196548a8949bdab882c72236"],"layout":"IPY_MODEL_27ae4340778b42d19d70f7813cf0081d"}},"7825b6a74d174e16b8ffeb77798824d5":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f397dc70f04742c983f94d7e44b86143","placeholder":"​","style":"IPY_MODEL_d8d1f5079dfc43e69e16d2c0b7db13b0","value":"Loading weights: 100%"}},"6108b7c4ccf04a0c8e7a062ea79c7769":{"model_module":"@jupyter-widgets/controls","model_name":"FloatProgressModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_a949e8ab7d134db3ae9dc5f65cd4e0ee","max":947,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f911e324a4634054b05320407db81c29","value":947}},"b6fd6e5b196548a8949bdab882c72236":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_dc25b8f9ca094634a4662108746bb946","placeholder":"​","style":"IPY_MODEL_3039ff65513d4f8e8820ddedfbf23911","value":" 947/947 [00:01&lt;00:00, 798.65it/s, Materializing param=model.encoder.layers.23.self_attn_layer_norm.weight]"}},"27ae4340778b42d19d70f7813cf0081d":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f397dc70f04742c983f94d7e44b86143":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d8d1f5079dfc43e69e16d2c0b7db13b0":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"a949e8ab7d134db3ae9dc5f65cd4e0ee":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f911e324a4634054b05320407db81c29":{"model_module":"@jupyter-widgets/controls","model_name":"ProgressStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"dc25b8f9ca094634a4662108746bb946":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3039ff65513d4f8e8820ddedfbf23911":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# PLEASE USE Kaggle GPU T4 (x2) as here both of them are used simultaneously for faster inference","metadata":{"id":"DgHsc1QIgMjh"}},{"cell_type":"markdown","source":"## 1. Dependency Installation and Environment Setup\n","metadata":{}},{"cell_type":"markdown","source":"# Please restart after installations to avoid import errors","metadata":{}},{"cell_type":"code","source":"%%capture\n# ── Install all required packages ──\n# transformers + accelerate for HuggingFace Whisper inference\n# silero-vad for VAD segmentation, bnunicodenormalizer for Bengali text normalization\n# datasets for efficient batched GPU inference via Dataset + KeyDataset\n!pip install -q transformers accelerate silero-vad bnunicodenormalizer jiwer librosa soundfile torchaudio huggingface_hub datasets","metadata":{"trusted":true,"id":"eDipTAxDgMji"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Imports","metadata":{}},{"cell_type":"code","source":"# ── Core Imports ──\nimport os\nimport gc\nimport re\nimport sys\nimport glob\nimport time\nimport logging\nimport warnings\nimport unicodedata\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport librosa\nfrom transformers import AutoModelForSpeechSeq2Seq, AutoProcessor\nfrom transformers import pipeline as hf_pipeline\nfrom transformers.pipelines.pt_utils import KeyDataset\nfrom datasets import Dataset\nfrom bnunicodenormalizer import Normalizer\n\n# ── Logging Configuration ──\nlogging.basicConfig(\n    level=logging.INFO,\n    format=\"[%(asctime)s] [%(levelname)s] %(message)s\",\n    datefmt=\"%H:%M:%S\",\n    handlers=[logging.StreamHandler(sys.stdout)],\n)\nlogger = logging.getLogger(\"BengaliASR\")\n\nwarnings.filterwarnings(\"ignore\")\nprint(\"All imports successful. Environment ready.\")","metadata":{"trusted":true,"id":"hc3V5TbLgMjj","outputId":"10e9fd23-2c7c-4703-dc64-e7f34c57ecfd"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Directories\n\n- **Silero VAD**: Cached via `torch.hub` to a local directory\n","metadata":{"id":"cBr8lFzEgMjj"}},{"cell_type":"code","source":"# ══════════════════════════════════════════════════════════════════════════════\n# MODEL & DATA PATHS\n# ══════════════════════════════════════════════════════════════════════════════\n\n# Whisper model — downloaded from HuggingFace at runtime\n# Using mozilla-ai/whisper-large-v3-bn (Bengali fine-tuned) for best Bengali quality\nWHISPER_MODEL_ID = \"zarifmahir21/whisper-medium-bangla\"\n\n# Competition data directories\nTEST_AUDIO_DIR = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio\"\nTRAIN_AUDIO_DIR = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/audio\"\nTRAIN_ANNOTATION_DIR = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/annotation\"\nSAMPLE_SUBMISSION_PATH = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/sample_submission .csv\"\n\n# Submission output path\nSUBMISSION_PATH = \"./submission.csv\"\n\n# Silero VAD local cache directory\nVAD_CACHE_DIR = \"./silero_vad_cache\"\n\n# ══════════════════════════════════════════════════════════════════════════════\n# TRANSCRIPTION HYPERPARAMETERS (tuned for Bengali long-form)\n# ══════════════════════════════════════════════════════════════════════════════\nLANGUAGE = \"bn\"\nBEAM_SIZE = 4\nBEST_OF = 4\nTEMPERATURE = (0.0, 0.2, 0.4, 0.6, 0.8, 1.0)  # Fallback temperatures\nREPETITION_PENALTY = 1.1\nCONDITION_ON_PREVIOUS_TEXT = False    # Prevent hallucination propagation\nCOMPRESSION_RATIO_THRESHOLD = 2.4    # Detect repetitive hallucinations\nLOG_PROB_THRESHOLD = -1.0\nNO_SPEECH_THRESHOLD = 0.6\nBATCH_SIZE = 13                  # GPU batch size for batched pipeline inference\n\n# ── Download and cache Silero VAD ──\nos.makedirs(VAD_CACHE_DIR, exist_ok=True)\nos.environ[\"TORCH_HOME\"] = VAD_CACHE_DIR\n\nprint(\"Loading Silero VAD model...\")\nvad_model, vad_utils = torch.hub.load(\n    repo_or_dir=\"snakers4/silero-vad\",\n    model=\"silero_vad\",\n    force_reload=False,\n    trust_repo=True,\n)\n(get_speech_timestamps, save_audio, read_audio, VADIterator, collect_chunks) = vad_utils\nprint(\"Silero VAD model loaded and cached successfully.\")\n\nprint(f\"Whisper model: {WHISPER_MODEL_ID} (will download from HuggingFace)\")\nprint(f\"Beam={BEAM_SIZE}, RepPenalty={REPETITION_PENALTY}, BatchSize={BATCH_SIZE}\")\nprint(f\"Temperatures={TEMPERATURE}\")","metadata":{"trusted":true,"id":"1zSZyvJ9gMjj","outputId":"0d3e110c-219b-45ed-f322-437004e3b453"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. GPU Configuration and RTF Optimizations","metadata":{"id":"Vbnvii8qgMjj"}},{"cell_type":"code","source":"# ── GPU & cuDNN Optimizations ──\ntorch.backends.cudnn.benchmark = True\ntorch.backends.cudnn.deterministic = False\n\nif torch.cuda.is_available():\n    gpu_name = torch.cuda.get_device_name(0)\n    gpu_mem = torch.cuda.get_device_properties(0).total_memory / (1024 ** 3)\n    print(f\"GPU: {gpu_name} | Memory: {gpu_mem:.1f} GB\")\n    print(\"cuDNN benchmark enabled for optimized kernel selection.\")\nelse:\n    logger.warning(\"No CUDA GPU detected. Pipeline will be slow on CPU.\")\n\n\ndef clear_cuda_cache():\n    \"\"\"Flush GPU memory to prevent OOM on long files.\"\"\"\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n        torch.cuda.synchronize()\n\n\nprint(\"GPU configuration complete.\")","metadata":{"trusted":true,"id":"sfyC3YXigMjk","outputId":"69b4ab19-217c-459b-a892-c8057a016af7"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Dual T4 GPU — Threading-Based Parallel Inference\n","metadata":{}},{"cell_type":"code","source":"import torch\n\nNUM_GPUS = torch.cuda.device_count()\nprint(f\"Available GPUs: {NUM_GPUS}\")\nfor i in range(NUM_GPUS):\n    props = torch.cuda.get_device_properties(i)\n    print(f\"  GPU {i}: {props.name} | {props.total_memory / 1024**3:.1f} GB\")\n\nUSE_DUAL_GPU = NUM_GPUS >= 2\nprint(f\"\\nDual-GPU mode: {'ENABLED' if USE_DUAL_GPU else 'DISABLED — single-GPU fallback'}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Audio Loading and Signal Processing Utility","metadata":{"id":"1QSqZZQzgMjk"}},{"cell_type":"code","source":"import torchaudio\nimport torchaudio.transforms as T\n\nTARGET_SR = 16000\n\ndef load_audio(filepath: str) -> np.ndarray:\n    \"\"\"Fast audio loading using PyTorch C++ backend\"\"\"\n    try:\n        # Load audio instantly\n        waveform, sr = torchaudio.load(filepath)\n\n        # Resample only if necessary\n        if sr != TARGET_SR:\n            resampler = T.Resample(orig_freq=sr, new_freq=TARGET_SR)\n            waveform = resampler(waveform)\n\n        # Convert stereo to mono if necessary\n        if waveform.shape[0] > 1:\n            waveform = torch.mean(waveform, dim=0, keepdim=True)\n\n        return waveform.squeeze().numpy()\n    except Exception as e:\n        logger.error(f\"Failed to load audio {filepath}: {e}\")\n        return np.array([], dtype=np.float32)","metadata":{"trusted":true,"id":"mAIpUFk9gMjk"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Silero VAD — Tuned for Bengali Speech\n\nSilero VAD identifies speech regions with a configurable probability threshold.\nParameters tuned for Bengali: lower threshold (0.35) catches softer speech common in Bengali,\nshorter min_speech (200ms) and longer silence detection (150ms) for natural Bengali pauses.","metadata":{"id":"P-6nqzh9gMjk"}},{"cell_type":"code","source":"# ── Optimized VAD parameters for Bengali speech ──\nVAD_THRESHOLD = 0.35       # Lowered from 0.6 — catches softer Bengali speech\nVAD_MIN_SPEECH_MS = 200    # Shorter minimum to capture short utterances\nVAD_MIN_SILENCE_MS = 150   # Slightly longer silence detection for Bengali pauses\nVAD_SPEECH_PAD_MS = 200     # Padding around detected speech\n\n\ndef get_speech_segments_vad(\n    audio: np.ndarray,\n    sample_rate: int = TARGET_SR,\n) -> list:\n    \"\"\"\n    Run Silero VAD on audio waveform and return speech segments as\n    a list of dicts with 'start' and 'end' keys (in samples).\n    \"\"\"\n    # Reset VAD model state for a fresh file\n    vad_model.reset_states()\n\n    # Convert numpy to torch tensor\n    audio_tensor = torch.from_numpy(audio).float()\n\n    # Get speech timestamps using Silero utility\n    speech_timestamps = get_speech_timestamps(\n        audio_tensor,\n        vad_model,\n        threshold=VAD_THRESHOLD,\n        sampling_rate=sample_rate,\n        min_speech_duration_ms=VAD_MIN_SPEECH_MS,\n        min_silence_duration_ms=VAD_MIN_SILENCE_MS,\n        speech_pad_ms=VAD_SPEECH_PAD_MS,\n        return_seconds=False,  # Return in samples\n    )\n\n    return speech_timestamps\n\n\nprint(\n    f\"VAD segmenter configured: threshold={VAD_THRESHOLD}, \"\n    f\"min_speech={VAD_MIN_SPEECH_MS}ms, min_silence={VAD_MIN_SILENCE_MS}ms, \"\n    f\"speech_pad={VAD_SPEECH_PAD_MS}ms\"\n)","metadata":{"trusted":true,"id":"ZRsllqgbgMjk","outputId":"2d6bfa33-e17c-4b05-af95-45ab96ed47e9"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. VAD Cut & Merge Chunking Strategy\n\nGreedily merge consecutive VAD speech segments into chunks of ≤29 seconds.\nChunk boundaries occur only during silence (between VAD segments), preventing mid-word cuts.","metadata":{"id":"CIZMVjEGgMjk"}},{"cell_type":"code","source":"MAX_CHUNK_DURATION = 29.0  # seconds — fits within Whisper's 30s window\nPADDING_SAMPLES = int(0.1 * TARGET_SR)  # 100ms silence padding per side\n\n\ndef merge_segments_into_chunks(\n    segments: list,\n    audio: np.ndarray,\n    sr: int = TARGET_SR,\n    max_duration: float = MAX_CHUNK_DURATION,\n) -> list:\n    \"\"\"\n    Merge VAD speech segments into chunks of at most `max_duration` seconds.\n    Each chunk is a contiguous numpy array of audio samples containing only\n    speech regions, with boundaries falling during silence.\n\n    Returns:\n        List of numpy arrays, each a chunk ready for ASR inference.\n    \"\"\"\n    if not segments:\n        return []\n\n    max_samples = int(max_duration * sr)\n    chunks = []\n    current_chunk_segments = []\n    current_chunk_len = 0\n\n    for seg in segments:\n        seg_start = max(0, seg[\"start\"] - PADDING_SAMPLES)\n        seg_end = min(len(audio), seg[\"end\"] + PADDING_SAMPLES)\n        seg_len = seg_end - seg_start\n\n        # If a single segment exceeds max duration, split it\n        if seg_len > max_samples:\n            # Flush current accumulator first\n            if current_chunk_segments:\n                chunk_audio = np.concatenate(\n                    [audio[s:e] for s, e in current_chunk_segments]\n                )\n                chunks.append(chunk_audio)\n                current_chunk_segments = []\n                current_chunk_len = 0\n\n            # Split the oversized segment into sub-chunks\n            for offset in range(seg_start, seg_end, max_samples):\n                sub_end = min(offset + max_samples, seg_end)\n                chunks.append(audio[offset:sub_end])\n            continue\n\n        # Check if adding this segment exceeds max chunk duration\n        if current_chunk_len + seg_len > max_samples:\n            # Flush current chunk\n            if current_chunk_segments:\n                chunk_audio = np.concatenate(\n                    [audio[s:e] for s, e in current_chunk_segments]\n                )\n                chunks.append(chunk_audio)\n            current_chunk_segments = [(seg_start, seg_end)]\n            current_chunk_len = seg_len\n        else:\n            current_chunk_segments.append((seg_start, seg_end))\n            current_chunk_len += seg_len\n\n    # Flush remaining segments\n    if current_chunk_segments:\n        chunk_audio = np.concatenate(\n            [audio[s:e] for s, e in current_chunk_segments]\n        )\n        chunks.append(chunk_audio)\n\n    return chunks\n\n\nprint(f\"Chunking strategy defined: max_duration={MAX_CHUNK_DURATION}s\")","metadata":{"trusted":true,"id":"A43n-E2DgMjl","outputId":"1993048b-9e78-491c-f41c-7537be4da49b"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Batched ASR Inference with Dataset + Anti-Hallucination\n","metadata":{"id":"-l-B33FWgMjl"}},{"cell_type":"code","source":"# ── Load mozilla-ai/whisper-large-v3-bn with HuggingFace Transformers ──\n# Using transformers directly (not CTranslate2) to ensure task=\"transcribe\"\n# is correctly applied and we get Bengali text, not English translation.\n\nfrom transformers import GenerationConfig\n\n_device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n_torch_dtype = torch.float16 if _device == \"cuda\" else torch.float32\n\nprint(f\"Loading {WHISPER_MODEL_ID} with HuggingFace Transformers on {_device}...\")\n\nhf_model = AutoModelForSpeechSeq2Seq.from_pretrained(\n    WHISPER_MODEL_ID,\n    torch_dtype=_torch_dtype,\n    low_cpu_mem_usage=True,\n    use_safetensors=True,\n    attn_implementation=\"sdpa\"  # <--- THIS IS THE MAGIC SPEEDUP LINE\n).to(_device)\n\n# ── FIX: Update generation config BEFORE creating the pipeline ──\nprint(\"Checking generation config for language support...\")\nif getattr(hf_model.generation_config, \"lang_to_id\", None) is None:\n    print(\"Missing language mapping. Pulling base config from openai/whisper-medium...\")\n    base_config = GenerationConfig.from_pretrained(\"openai/whisper-medium\")\n    hf_model.generation_config = base_config\n    print(\"Generation config updated successfully!\")\n\n# Force extra_special_tokens to be a valid dictionary to bypass the config bug\nprocessor = AutoProcessor.from_pretrained(\n    WHISPER_MODEL_ID,\n    extra_special_tokens={}\n)\n\n# Create ASR pipeline — referenced as `whisper_model` throughout the notebook\nwhisper_model = hf_pipeline(\n    \"automatic-speech-recognition\",\n    model=hf_model,\n    tokenizer=processor.tokenizer,\n    feature_extractor=processor.feature_extractor,\n    torch_dtype=_torch_dtype,\n    device=_device,\n)\n\nprint(f\"Model loaded: {WHISPER_MODEL_ID} on {_device} ({_torch_dtype})\")\n\n@torch.inference_mode()\ndef transcribe_chunks(model, chunks: list, batch_size: int = BATCH_SIZE) -> str:\n    \"\"\"\n    Transcribe audio chunks using batched HuggingFace ASR pipeline.\n\n    Passes a generator of {\"raw\": np.ndarray, \"sampling_rate\": int} dicts\n    to the pipeline with `batch_size`, enabling its internal DataLoader to\n    batch multiple chunks onto the GPU simultaneously for higher throughput.\n\n    Returns concatenated transcription text.\n    \"\"\"\n    if not chunks:\n        return \"\"\n\n    # Filter out empty chunks\n    valid_chunks = [c for c in chunks if len(c) > 0]\n    if not valid_chunks:\n        return \"\"\n\n    print(\n        f\"  Batched inference: {len(valid_chunks)} chunks, \"\n        f\"batch_size={batch_size}\"\n    )\n    t_start = time.time()\n\n    # ── Build a generator of audio dicts for the pipeline ──\n    # The pipeline accepts any iterable; with batch_size>1 it batches\n    # multiple items for parallel GPU forward passes.\n    def audio_generator():\n        for chunk in valid_chunks:\n            yield {\"raw\": chunk, \"sampling_rate\": TARGET_SR}\n\n    generate_kwargs = {\n        \"language\": \"bengali\",\n        \"task\": \"transcribe\",\n        \"num_beams\": BEAM_SIZE,\n        \"repetition_penalty\": REPETITION_PENALTY,\n    }\n\n    all_texts = []\n    for i, result in enumerate(\n        model(\n            audio_generator(),\n            batch_size=batch_size,\n            generate_kwargs=generate_kwargs,\n            # chunk_length_s=30,\n        )\n    ):\n        text = result[\"text\"].strip()\n        chunk_dur = len(valid_chunks[i]) / TARGET_SR\n        if text:\n            all_texts.append(text)\n            print(f\"    Chunk {i+1}/{len(valid_chunks)} ({chunk_dur:.1f}s): '{text[:70]}...'\")\n        else:\n            logger.warning(f\"    Chunk {i+1}/{len(valid_chunks)} ({chunk_dur:.1f}s): empty transcription\")\n\n    elapsed = time.time() - t_start\n    total_audio = sum(len(c) / TARGET_SR for c in valid_chunks)\n    batch_rtf = elapsed / total_audio if total_audio > 0 else 0\n\n    print(\n        f\"  Batched inference done in {elapsed:.1f}s \"\n        f\"({len(all_texts)}/{len(valid_chunks)} successful, \"\n        f\"batch RTF={batch_rtf:.3f})\"\n    )\n    return \" \".join(all_texts)\n\nprint(\n    f\"ASR inference ready: model={WHISPER_MODEL_ID}, beam={BEAM_SIZE}, \"\n    f\"rep_penalty={REPETITION_PENALTY}, batch_size={BATCH_SIZE}, \"\n    f\"task=transcribe, language={LANGUAGE}\"\n)","metadata":{"trusted":true,"id":"4HBunZaPgMjl","outputId":"b8d8ad28-81d3-45af-933c-64d3d82dc739"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Bengali Number-to-Word Conversion\n\nMaps all digit characters (ASCII 0–9 and Bengali Unicode ০–৯) to their Bengali word forms.","metadata":{"id":"y0oRG-e7gMjl"}},{"cell_type":"code","source":"# ── Digit-to-Bengali-Word Mapping ──\nDIGIT_TO_BENGALI_WORD = {\n    \"0\": \"শূন্য\", \"1\": \"এক\", \"2\": \"দুই\", \"3\": \"তিন\", \"4\": \"চার\",\n    \"5\": \"পাঁচ\", \"6\": \"ছয়\", \"7\": \"সাত\", \"8\": \"আট\", \"9\": \"নয়\",\n    # Bengali Unicode digits\n    \"০\": \"শূন্য\", \"১\": \"এক\", \"২\": \"দুই\", \"৩\": \"তিন\", \"৪\": \"চার\",\n    \"৫\": \"পাঁচ\", \"৬\": \"ছয়\", \"৭\": \"সাত\", \"৮\": \"আট\", \"৯\": \"নয়\",\n}\n\n# Pre-compiled regex for all digit characters (ASCII + Bengali)\n_DIGIT_PATTERN = re.compile(r\"[0-9০-৯]\")\n\n\ndef digits_to_bengali_words(text: str) -> str:\n    \"\"\"\n    Replace every digit (ASCII and Bengali Unicode) in the text with\n    the corresponding Bengali word, separated by spaces.\n    \"\"\"\n    def _replace_digit(match):\n        return DIGIT_TO_BENGALI_WORD.get(match.group(0), match.group(0))\n\n    return _DIGIT_PATTERN.sub(_replace_digit, text)\n\n\n# Quick test\nassert digits_to_bengali_words(\"আমি 3টা বই পড়ি\") == \"আমি তিনটা বই পড়ি\"\nassert digits_to_bengali_words(\"২০২৬\") == \"দুইশূন্যদুইছয়\"\nprint(\"Bengali digit-to-word converter ready.\")","metadata":{"trusted":true,"id":"GUPMhxKogMjl","outputId":"5197809c-749b-4309-8841-82aaf9806989"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Unicode Normalization and Text Cleaning Engine\n\nEvery word is passed through `bnunicodenormalizer.Normalizer` to fix Unicode inconsistencies.\nA regex-based cleaner strips all punctuation, and digits are converted to Bengali words.","metadata":{"id":"7LkYQQkvgMjm"}},{"cell_type":"code","source":"# ── Initialize Bengali Unicode Normalizer ──\nbn_normalizer = Normalizer()\n\n# ── Compiled regex: remove all specified punctuation ──\n# Characters: ! \" # $ % & ' ( ) * + , - . / : ; < = > ? @ [ \\ ] ^ _ ` { | } ~ ।\n_PUNCTUATION_PATTERN = re.compile(\n    r'[!\"#$%&\\'()*+,\\-./:;<=>?@\\[\\\\\\]^_`{|}~।]'\n)\n\n# Collapse multiple whitespace into single space\n_MULTI_SPACE = re.compile(r\"\\s+\")\n\n\ndef normalize_transcript(text: str) -> str:\n    \"\"\"\n    Full post-processing pipeline for a raw Whisper transcript:\n      1. Remove all punctuation characters\n      2. Normalize each word with bnunicodenormalizer\n      3. Convert digits to Bengali words\n      4. Strip whitespace and collapse multi-spaces\n    \"\"\"\n    if not text or not text.strip():\n        return \"\"\n\n    # Step 1: Remove punctuation\n    text = _PUNCTUATION_PATTERN.sub(\" \", text)\n\n    # Step 2: Word-level Bengali Unicode normalization\n    words = text.split()\n    normalized_words = []\n    for word in words:\n        normalized = bn_normalizer(word)\n        # bn_normalizer returns a dict with 'normalized' key or the string directly\n        if isinstance(normalized, dict):\n            norm_text = normalized.get(\"normalized\", word)\n        elif isinstance(normalized, tuple):\n            norm_text = normalized[0] if normalized[0] else word\n        else:\n            norm_text = str(normalized) if normalized else word\n\n        if norm_text and norm_text.strip():\n            normalized_words.append(norm_text.strip())\n\n    text = \" \".join(normalized_words)\n\n    # Step 3: Convert digits to Bengali words\n    text = digits_to_bengali_words(text)\n\n    # Step 4: Final whitespace cleanup\n    text = _MULTI_SPACE.sub(\" \", text).strip()\n\n    return text\n\n\n# Quick validation\nprint(\"Normalization engine initialized and tested.\")","metadata":{"trusted":true,"id":"1KC1xaPWgMjm","outputId":"771948d6-f68a-41ed-c59c-d605f334ac3f"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10. Advanced Post-Processing","metadata":{"id":"Lt3ucXP-gMjm"}},{"cell_type":"code","source":"class BengaliTextProcessor:\n    \"\"\"\n    Advanced Bengali text processor with repetition removal.\n    Fixes Whisper hallucinations that produce repeated words/phrases.\n    \"\"\"\n\n    def __init__(self):\n        self.whitespace_pattern = re.compile(r\"\\s+\")\n        self.zero_width_pattern = re.compile(r\"[\\u200b\\u200c\\u200d\\ufeff\\u200e\\u200f]\")\n\n    def normalize_unicode(self, text: str) -> str:\n        \"\"\"Normalize Unicode to NFC form.\"\"\"\n        return unicodedata.normalize(\"NFC\", text)\n\n    def remove_zero_width(self, text: str) -> str:\n        \"\"\"Remove zero-width characters.\"\"\"\n        return self.zero_width_pattern.sub(\"\", text)\n\n    def normalize_whitespace(self, text: str) -> str:\n        \"\"\"Normalize whitespace to single spaces.\"\"\"\n        return self.whitespace_pattern.sub(\" \", text).strip()\n\n    def remove_repeated_words(self, text: str, max_repeats: int = 2) -> str:\n        \"\"\"\n        Remove consecutively repeated words.\n        Keeps at most max_repeats consecutive occurrences.\n        \"\"\"\n        words = text.split()\n        if len(words) <= 1:\n            return text\n\n        result = []\n        repeat_count = 1\n\n        for i, word in enumerate(words):\n            if i == 0:\n                result.append(word)\n            elif word == words[i - 1]:\n                repeat_count += 1\n                if repeat_count <= max_repeats:\n                    result.append(word)\n            else:\n                repeat_count = 1\n                result.append(word)\n\n        return \" \".join(result)\n\n    def remove_repeated_phrases(\n        self, text: str, min_phrase_len: int = 2, max_phrase_len: int = 8\n    ) -> str:\n        \"\"\"\n        Remove repeated n-gram phrases from text.\n        Critical for fixing Whisper hallucinations like\n        'এটা পোরেবেন সেটে মানারে এটা পোরেবেন সেটে মানারে'\n        \"\"\"\n        words = text.split()\n        if len(words) < min_phrase_len * 2:\n            return text\n\n        for phrase_len in range(max_phrase_len, min_phrase_len - 1, -1):\n            i = 0\n            new_words = []\n\n            while i < len(words):\n                if i + phrase_len * 2 <= len(words):\n                    phrase1 = words[i : i + phrase_len]\n                    phrase2 = words[i + phrase_len : i + phrase_len * 2]\n\n                    if phrase1 == phrase2:\n                        # Count total repetitions and skip them\n                        j = i + phrase_len\n                        while (\n                            j + phrase_len <= len(words)\n                            and words[j : j + phrase_len] == phrase1\n                        ):\n                            j += phrase_len\n\n                        new_words.extend(phrase1)  # Keep one occurrence\n                        i = j\n                        continue\n\n                new_words.append(words[i])\n                i += 1\n\n            words = new_words\n\n        return \" \".join(words)\n\n    def remove_excessive_repetition(self, text: str, threshold: float = 0.3) -> str:\n        \"\"\"\n        Detect and fix texts with excessive repetition.\n        If a single word appears > threshold of the time, aggressively deduplicate.\n        \"\"\"\n        words = text.split()\n        if len(words) < 10:\n            return text\n\n        word_counts = Counter(words)\n        most_common_word, most_common_count = word_counts.most_common(1)[0]\n        repetition_ratio = most_common_count / len(words)\n\n        if repetition_ratio > threshold:\n            seen_bigrams = set()\n            result = []\n            for i, word in enumerate(words):\n                if i == 0:\n                    result.append(word)\n                else:\n                    bigram = (words[i - 1], word)\n                    if bigram not in seen_bigrams or len(seen_bigrams) < 100:\n                        result.append(word)\n                        seen_bigrams.add(bigram)\n            return \" \".join(result)\n\n        return text\n\n    def process(self, text: str) -> str:\n        \"\"\"Full post-processing pipeline.\"\"\"\n        if not text:\n            return \"\"\n\n        text = self.normalize_unicode(text)\n        text = self.remove_zero_width(text)\n        text = self.normalize_whitespace(text)\n        text = self.remove_repeated_phrases(text)\n        text = self.remove_repeated_words(text, max_repeats=2)\n        text = self.remove_excessive_repetition(text)\n        text = self.normalize_whitespace(text)\n        return text\n\n\n# Initialize processor\ntext_processor = BengaliTextProcessor()\n\n# Test with known hallucination patterns\ntest_cases = [\n    \"বাই বাই বাই বাই বাই বাই বাই বাই বাই বাই বাই\",\n    \"একটো তারা তারি করে একটো তারা তারি করে একটো তারা\",\n    \"এটা পোরেবেন সেটে মানারে এটা পোরেবেন সেটে মানারে এটা পোরেবেন সেটে মানারে\",\n    \"আজ আমরা দীর্ঘ অডিও ট্রান্সক্রিপশন নিয়ে আলোচনা করব\",\n]\n\nprint(\"Testing repetition removal:\")\nprint(\"=\" * 60)\nfor test in test_cases:\n    result = text_processor.process(test)\n    inp = f\"'{test[:50]}...'\" if len(test) > 50 else f\"'{test}'\"\n    print(f\"Input:  {inp}\")\n    print(f\"Output: '{result}'\")\n    print(\"-\" * 40)\n\nprint(\"BengaliTextProcessor initialized and tested.\")","metadata":{"trusted":true,"id":"DiPcJU_ZgMjm","outputId":"427107cb-4417-44bf-f26e-8e89486874a3"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11. Processing single file code","metadata":{"id":"l35rNOxWgMjn"}},{"cell_type":"code","source":"@torch.inference_mode()\ndef process_single_file(\n    filepath: str,\n    vad_model_ref,\n    whisper_model_ref,\n    normalizer_fn,\n    post_processor=None,\n) -> str:\n    \"\"\"\n    Full pipeline for a single audio file:\n      1. Load & resample audio → 16 kHz mono\n      2. VAD segmentation\n      3. Merge segments into ≤29s chunks\n      4. Whisper ASR inference (with anti-hallucination)\n      5. Normalize & clean transcript\n      6. Advanced post-processing (n-gram dedup)\n      7. Clear CUDA cache\n    \"\"\"\n    fname = os.path.basename(filepath)\n    t0 = time.time()\n\n    # ── Stage 1: Load Audio ──\n    print(f\"[{fname}] Loading audio...\")\n    audio = load_audio(filepath)\n    if len(audio) == 0:\n        logger.warning(f\"[{fname}] Empty/corrupt audio → returning empty string\")\n        return \"\"\n\n    duration = len(audio) / TARGET_SR\n    print(f\"[{fname}] Loaded: {duration:.1f}s ({len(audio)} samples)\")\n\n    # ── Stage 2: VAD Segmentation ──\n    print(f\"[{fname}] Running VAD segmentation...\")\n    segments = get_speech_segments_vad(audio, TARGET_SR)\n    print(f\"[{fname}] VAD segments found: {len(segments)}\")\n\n    if len(segments) == 0:\n        logger.warning(f\"[{fname}] No speech detected → returning empty string\")\n        clear_cuda_cache()\n        return \"\"\n\n    # Log individual segment durations\n    for si, seg in enumerate(segments):\n        seg_dur = (seg[\"end\"] - seg[\"start\"]) / TARGET_SR\n        # print(f\"[{fname}]   Segment {si+1}: {seg_dur:.2f}s\")\n\n    # ── Stage 3: Chunk Merging ──\n    print(f\"[{fname}] Merging segments into chunks (max {MAX_CHUNK_DURATION}s)...\")\n    chunks = merge_segments_into_chunks(segments, audio, TARGET_SR)\n    total_chunk_dur = sum(len(c) / TARGET_SR for c in chunks)\n    print(\n        f\"[{fname}] Chunks: {len(chunks)} | \"\n        f\"Total speech: {total_chunk_dur:.1f}s / {duration:.1f}s\"\n    )\n    # for ci, c in enumerate(chunks):\n    #     print(f\"[{fname}]   Chunk {ci+1}: {len(c)/TARGET_SR:.2f}s ({len(c)} samples)\")\n\n    # ── Stage 4: ASR Inference ──\n    print(f\"[{fname}] Starting ASR inference on {len(chunks)} chunks...\")\n    t_asr = time.time()\n    raw_transcript = transcribe_chunks(whisper_model_ref, chunks)\n    asr_elapsed = time.time() - t_asr\n    print(f\"[{fname}] ASR completed in {asr_elapsed:.1f}s\")\n    print(f\"[{fname}] Raw transcript ({len(raw_transcript.split())} words): {raw_transcript[:120]}...\")\n\n    # ── Stage 5: Normalization ──\n    # print(f\"[{fname}] Normalizing transcript...\")\n    # final_transcript = normalizer_fn(raw_transcript)\n    # print(f\"[{fname}] Normalized ({len(final_transcript.split())} words): {final_transcript}...\")\n\n    final_transcript = raw_transcript\n    # ── Stage 6: Advanced Post-Processing (n-gram dedup) ──\n    if post_processor is not None:\n        original_len = len(final_transcript.split())\n        final_transcript = post_processor.process(final_transcript)\n        clean_len = len(final_transcript.split())\n        removed = original_len - clean_len\n        if original_len > 0 and removed > 0:\n            print(\n                f\"[{fname}] Dedup: removed {removed} repeated words \"\n                f\"({100*removed/original_len:.1f}%) → {clean_len} words remaining\"\n            )\n        else:\n            print(f\"[{fname}] Dedup: no repetitions removed\")\n\n    print(f\"[{fname}] Final transcript ({len(final_transcript.split())} words): {final_transcript}...\")\n\n    # ── Stage 7: Cleanup ──\n    clear_cuda_cache()\n\n    elapsed = time.time() - t0\n    rtf = elapsed / duration if duration > 0 else 0\n    print(f\"[{fname}] DONE in {elapsed:.1f}s | RTF={rtf:.3f}\")\n    print(f\"  ✓ {fname}: {len(final_transcript.split())} words | {elapsed:.1f}s | RTF={rtf:.3f}\")\n\n    return final_transcript\n\n\nprint(\"Pipeline orchestration function ready.\")","metadata":{"trusted":true,"id":"LyTVsqBagMjn","outputId":"168e0711-aeae-4a5d-d5df-722ad5e0abd3"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 12. Main Inference","metadata":{}},{"cell_type":"code","source":"# ══════════════════════════════════════════════════════════════════════════════\n# MAIN EXECUTION — Dual T4 parallel inference via threading\n#\n# Architecture:\n#   1. Load model_gpu0 on cuda:0  \\  sequential in main thread (no file-lock race)\n#   2. Load model_gpu1 on cuda:1  /\n#   3. Thread-0 runs inference on cuda:0 for files[0:half]\n#   4. Thread-1 runs inference on cuda:1 for files[half:]\n#      GIL is released during CUDA ops → both GPUs run truly in parallel\n#   5. Merge results in original file order\n# ══════════════════════════════════════════════════════════════════════════════\nimport os, gc, glob, time, threading\nimport numpy as np\nimport pandas as pd\nimport torch\nimport librosa\nfrom transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, GenerationConfig\nfrom transformers import pipeline as hf_pipeline\n\n# ── Discover test files ──\naudio_extensions = [\"*.mp3\", \"*.wav\", \"*.flac\", \"*.ogg\", \"*.m4a\", \"*.webm\"]\ntest_files = []\nfor ext in audio_extensions:\n    test_files.extend(glob.glob(os.path.join(TEST_AUDIO_DIR, ext)))\nif not test_files:\n    test_files = [f for f in glob.glob(os.path.join(TEST_AUDIO_DIR, \"*\"))\n                  if os.path.isfile(f)]\ntest_files = sorted(test_files)\nprint(f\"Found {len(test_files)} test audio files\")\n\npipeline_start = time.time()\n\n# ══════════════════════════════════════════════════════════════════════════════\nif USE_DUAL_GPU and len(test_files) >= 2:\n\n    # ── Step 1: Load both models SEQUENTIALLY (no file-lock contention) ──\n    print(\"\\nLoading models sequentially (avoids safetensors file-lock races)...\")\n\n    def load_whisper_on_device(device_str):\n        \"\"\"Load Whisper onto a specific device string e.g. 'cuda:0'\"\"\"\n        _dtype = torch.float16\n        model = AutoModelForSpeechSeq2Seq.from_pretrained(\n            WHISPER_MODEL_ID,\n            torch_dtype=_dtype,\n            low_cpu_mem_usage=True,\n            use_safetensors=True,\n            attn_implementation=\"sdpa\",\n        ).to(device_str)\n        if getattr(model.generation_config, \"lang_to_id\", None) is None:\n            base_cfg = GenerationConfig.from_pretrained(\"openai/whisper-medium\")\n            model.generation_config = base_cfg\n        proc = AutoProcessor.from_pretrained(WHISPER_MODEL_ID, extra_special_tokens={})\n        pipe = hf_pipeline(\n            \"automatic-speech-recognition\",\n            model=model,\n            tokenizer=proc.tokenizer,\n            feature_extractor=proc.feature_extractor,\n            torch_dtype=_dtype,\n            device=device_str,\n        )\n        return pipe\n\n    print(\"  Loading model on cuda:0 ...\")\n    pipe0 = load_whisper_on_device(\"cuda:0\")\n    print(\"  cuda:0 ready ✓\")\n\n    print(\"  Loading model on cuda:1 ...\")\n    pipe1 = load_whisper_on_device(\"cuda:1\")\n    print(\"  cuda:1 ready ✓\")\n\n    print(\"Both models loaded. Starting parallel inference via threads.\\n\")\n\n    # ── Step 2: Thread worker — uses a pre-loaded pipeline ──\n    def thread_worker(pipe, file_paths, thread_id, out_list):\n        \"\"\"\n        Runs on one thread. Uses the pre-loaded pipeline (already on its GPU).\n        GIL is released during CUDA forward passes so both threads run in parallel.\n        \"\"\"\n        generate_kwargs = {\n            \"language\": \"bengali\",\n            \"task\": \"transcribe\",\n            \"num_beams\": BEAM_SIZE,\n            \"repetition_penalty\": REPETITION_PENALTY,\n        }\n\n        for idx, fpath in enumerate(file_paths):\n            fname = os.path.basename(fpath)\n            print(f\"  [Thread {thread_id}] [{idx+1}/{len(file_paths)}] {fname}\")\n            t0 = time.time()\n\n            # Load audio\n            audio = load_audio(fpath)\n            if len(audio) == 0:\n                out_list.append({\"filename\": fname, \"transcript\": \"\"})\n                continue\n\n            # VAD — note: vad_model is shared but reset_states + inference\n            # is protected by a lock since it holds internal state\n            with vad_lock:\n                segments = get_speech_segments_vad(audio, TARGET_SR)\n\n            if not segments:\n                out_list.append({\"filename\": fname, \"transcript\": \"\"})\n                continue\n\n            chunks = merge_segments_into_chunks(segments, audio, TARGET_SR)\n            valid  = [c for c in chunks if len(c) > 0]\n\n            def audio_gen():\n                for c in valid:\n                    yield {\"raw\": c, \"sampling_rate\": TARGET_SR}\n\n            texts = []\n            with torch.inference_mode():\n                for result in pipe(\n                    audio_gen(),\n                    batch_size=BATCH_SIZE,\n                    generate_kwargs=generate_kwargs,\n                    # chunk_length_s=30,\n                ):\n                    t = result[\"text\"].strip()\n                    if t:\n                        texts.append(t)\n\n            raw   = \" \".join(texts)\n            final = text_processor.process(raw)\n            dur   = len(audio) / TARGET_SR\n            elapsed = time.time() - t0\n            print(f\"  [Thread {thread_id}] ✓ {fname} | {elapsed:.1f}s | \"\n                  f\"{len(final.split())} words | RTF={elapsed/dur:.3f}\")\n            # print(final)\n            out_list.append({\"filename\": fname, \"transcript\": final})\n\n    # VAD model is stateful — protect reset_states + inference with a lock\n    vad_lock = threading.Lock()\n\n    half   = len(test_files) // 2\n    split0 = test_files[:half]\n    split1 = test_files[half:]\n\n    # Thread-safe result lists (each thread appends only to its own list)\n    results0, results1 = [], []\n\n    t0_thread = threading.Thread(target=thread_worker, args=(pipe0, split0, 0, results0))\n    t1_thread = threading.Thread(target=thread_worker, args=(pipe1, split1, 1, results1))\n\n    print(f\"🚀 Launching threads: Thread-0 → {len(split0)} files on cuda:0 | \"\n          f\"Thread-1 → {len(split1)} files on cuda:1\")\n    t0_thread.start()\n    t1_thread.start()\n    t0_thread.join()\n    t1_thread.join()\n    print(\"\\n✅ Both threads complete.\")\n\n    # Merge results in original file order\n    all_results_map = {item[\"filename\"]: item[\"transcript\"]\n                       for item in results0 + results1}\n    results = [\n        {\"filename\": os.path.splitext(os.path.basename(fp))[0],\n         \"transcript\": all_results_map.get(os.path.basename(fp), \"\")}\n        for fp in test_files\n    ]\n    print(f\"Merged {len(results)} results in original file order.\")\n\n    # Free GPU memory\n    del pipe0, pipe1\n    torch.cuda.empty_cache()\n    gc.collect()\n\n# ══════════════════════════════════════════════════════════════════════════════\nelse:\n    # Single-GPU fallback — original pipeline, unchanged\n    print(\"\\n⚡ Single-GPU mode (original pipeline)\")\n    results = []\n    with torch.inference_mode():\n        for idx, fpath in enumerate(test_files):\n            print(f\"{'='*60}\")\n            print(f\"Processing [{idx+1}/{len(test_files)}]: {os.path.basename(fpath)}\")\n            transcription = process_single_file(\n                filepath=fpath,\n                vad_model_ref=vad_model,\n                whisper_model_ref=whisper_model,\n                normalizer_fn=normalize_transcript,\n                post_processor=text_processor,\n            )\n            results.append({\"filename\": os.path.splitext(os.path.basename(fpath))[0],\n                            \"transcript\": transcription})\n            if torch.cuda.is_available() and (idx + 1) % 3 == 0:\n                torch.cuda.empty_cache()\n                gc.collect()\n\n# ══════════════════════════════════════════════════════════════════════════════\n# STATS & SUBMISSION\n# ══════════════════════════════════════════════════════════════════════════════\npipeline_elapsed = time.time() - pipeline_start\ntotal_audio_duration = sum(\n    librosa.get_duration(path=fp) for fp in test_files if os.path.isfile(fp)\n)\navg_rtf = pipeline_elapsed / total_audio_duration if total_audio_duration > 0 else 0\n\nprint(f\"\\n{'='*60}\")\nprint(f\"PIPELINE COMPLETE\")\nprint(f\"Total files:           {len(test_files)}\")\nprint(f\"Total wall-clock time: {pipeline_elapsed:.1f}s\")\nprint(f\"Total audio duration:  {total_audio_duration:.1f}s\")\nprint(f\"Average RTF:           {avg_rtf:.4f}\")\n\nsubmission_df = pd.DataFrame(results)\nif os.path.exists(SAMPLE_SUBMISSION_PATH):\n    sample_sub = pd.read_csv(SAMPLE_SUBMISSION_PATH)\n    target_cols = list(sample_sub.columns)\n    if len(target_cols) >= 2:\n        submission_df.columns = target_cols[:2]\n\n# submission_df.to_csv(SUBMISSION_PATH, index=False)\n# print(f\"Submission saved: {SUBMISSION_PATH} | shape: {submission_df.shape}\")\nsubmission_df.head(10)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 13. Saving","metadata":{"id":"eKLo_3O-gMjn"}},{"cell_type":"code","source":"import pandas as pd\nimport re\n\n# ── ASCII Digit-to-Bengali Word Mapping ──\nASCII_TO_BENGALI = {\n    \"0\": \"শূন্য\", \"1\": \"এক\", \"2\": \"দুই\", \"3\": \"তিন\", \"4\": \"চার\",\n    \"5\": \"পাঁচ\", \"6\": \"ছয়\", \"7\": \"সাত\", \"8\": \"আট\", \"9\": \"নয়\",\n}\n\n# Regex Explanation:\n# (?<=[\\u0980-\\u09FF])\\d : Digit preceded by a Bengali character\n# \\d(?=[\\u0980-\\u09FF]) : Digit followed by a Bengali character\n_CONTEXTUAL_DIGIT_PATTERN = re.compile(r\"(?<=[\\u0980-\\u09FF])[0-9]|[0-9](?=[\\u0980-\\u09FF])\")\n_PUNCTUATION_PATTERN = re.compile(\n    r'[!\"#$%&\\'()*+,\\-./:;<=>?@\\[\\\\\\]^_`{|}~।]'\n)\n_MULTI_SPACE = re.compile(r\"\\s+\")\n\ndef normalize_contextual_digits(text: str) -> str:\n    \"\"\"\n    Replace ASCII digits with Bengali words ONLY if they are \n    attached to Bengali text (e.g., '3টা' -> 'তিনটা').\n    Standalone numbers (e.g., '40') remain unchanged.\n    \"\"\"\n    def _replace_digit(match):\n        return ASCII_TO_BENGALI.get(match.group(0), match.group(0))\n\n    text = _PUNCTUATION_PATTERN.sub(\" \", text)\n    text = _MULTI_SPACE.sub(\" \", text).strip()\n    return _CONTEXTUAL_DIGIT_PATTERN.sub(_replace_digit, text)\n\n# ── Verification ──\n\n# Case 1: Attached to Bengali (Should convert)\nres1 = normalize_contextual_digits(\"আমি 3টা বই পড়ি\")\nprint(f\"Case 1: {res1}\") # আমি তিনটা বই পড়ি\n\n# Case 2: Standalone Number (Should stay)\nres2 = normalize_contextual_digits(\"আমার বয়স 20\")\nprint(f\"Case 2: {res2}\") # আমার বয়স 20\n\n# Case 3: Mixed (Should stay)\nres3 = normalize_contextual_digits(\"২০২৬\")\nprint(f\"Case 3: {res3}\") # ২০২৬\n\n# Assertions\nassert res1 == \"আমি তিনটা বই পড়ি\"\nassert res2 == \"আমার বয়স 20\"\nassert res3 == \"২০২৬\"\n\n\n# 3. Apply the function to the specific column containing the audio text\n# Replace 'transcript' with your actual column name if it's different\ntext_column_name = 'transcript'\n\n# The .apply() method runs your function on every single row in that column\nsubmission_df[text_column_name] = submission_df[text_column_name].apply(normalize_contextual_digits)\n\n# 4. Save the updated dataframe to a new CSV\noutput_csv_path = \"submission.csv\"\nsubmission_df.to_csv(output_csv_path, index=False)\n\nprint(f\"✓ Function applied successfully! Saved to {output_csv_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}