{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":129276,"databundleVersionId":15506988},{"sourceType":"datasetVersion","sourceId":6707460,"datasetId":3865741,"databundleVersionId":6791840},{"sourceType":"datasetVersion","sourceId":14896738,"datasetId":9510682,"databundleVersionId":15761109}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":28383.957436,"end_time":"2026-02-19T04:31:49.510219","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-18T20:38:45.552783","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Bengali ASR Pipeline with VAD for Music Filtering\n\nThis notebook improves the 1st-place solution by adding **Silero VAD** to prevent music hallucinations.\n\n## Key Improvements:\n1. ✅ **VAD Pre-filtering** - Detects speech segments before transcription\n2. ✅ **Skip Music Sections** - Only transcribes actual speech\n3. ✅ **No More Hallucinations** - Music sections are filtered out\n\n## How It Works:\n- Use Silero VAD to detect where speech occurs\n- Extract only speech segments\n- Transcribe speech segments with Whisper\n- Skip music/silence completely\n\n---","metadata":{"papermill":{"duration":0.005714,"end_time":"2026-02-18T20:38:48.323233","exception":false,"start_time":"2026-02-18T20:38:48.317519","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ============================================================\n# INSTALL (Kaggle)\n# ============================================================\nimport os\n\n# Core ASR stack\nos.system(\"pip install -q transformers accelerate\")\n\n# Text normalization for Bangla (Unicode cleaning)\nos.system(\"pip install -q bnunicodenormalizer\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:11.546440Z","iopub.execute_input":"2026-02-20T12:27:11.546778Z","iopub.status.idle":"2026-02-20T12:27:18.861254Z","shell.execute_reply.started":"2026-02-20T12:27:11.546747Z","shell.execute_reply":"2026-02-20T12:27:18.860530Z"},"papermill":{"duration":8.218089,"end_time":"2026-02-18T20:38:56.546102","exception":false,"start_time":"2026-02-18T20:38:48.328013","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.system(\"pip install -q silero-vad\")","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:18.862607Z","iopub.execute_input":"2026-02-20T12:27:18.862903Z","iopub.status.idle":"2026-02-20T12:27:22.565576Z","shell.execute_reply.started":"2026-02-20T12:27:18.862877Z","shell.execute_reply":"2026-02-20T12:27:22.564700Z"},"papermill":{"duration":5.473813,"end_time":"2026-02-18T20:39:02.024883","exception":false,"start_time":"2026-02-18T20:38:56.551070","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install -q pyannote.audio","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.566755Z","iopub.execute_input":"2026-02-20T12:27:22.567516Z","iopub.status.idle":"2026-02-20T12:27:22.570779Z","shell.execute_reply.started":"2026-02-20T12:27:22.567487Z","shell.execute_reply":"2026-02-20T12:27:22.569971Z"},"papermill":{"duration":0.010355,"end_time":"2026-02-18T20:39:02.040330","exception":false,"start_time":"2026-02-18T20:39:02.029975","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport csv\nimport glob\nimport subprocess\nimport tempfile\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport torch\nimport torchaudio\n# import pvcobra\nfrom transformers import pipeline, AutoModelForTokenClassification, AutoTokenizer\n\n\n# NEW - Add pyannote\n# from pyannote.audio import Pipeline  # ← NEW","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.572594Z","iopub.execute_input":"2026-02-20T12:27:22.572959Z","iopub.status.idle":"2026-02-20T12:27:22.585349Z","shell.execute_reply.started":"2026-02-20T12:27:22.572917Z","shell.execute_reply":"2026-02-20T12:27:22.584522Z"},"papermill":{"duration":34.752484,"end_time":"2026-02-18T20:39:36.797387","exception":false,"start_time":"2026-02-18T20:39:02.044903","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATASET_PATH = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio\"\n# DATASET_PATH = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/audio\"\nOUTPUT_PATH = \"/kaggle/working/submission.csv\"\n\nMODEL = \"/kaggle/input/datasets/tugstugi/bengali-ai-asr-submission/bengali-whisper-medium\"\n\nPUNCT_MODELS = [\n    \"/kaggle/input/bengali-ai-asr-submission/punct-model-6layers/\",\n    \"/kaggle/input/bengali-ai-asr-submission/punct-model-8layers/\",\n    \"/kaggle/input/bengali-ai-asr-submission/punct-model-11layers/\",\n    \"/kaggle/input/bengali-ai-asr-submission/punct-model-12layers/\"\n]\nPUNCT_WEIGHTS = [[1.0, 1.4, 1.0, 0.8]]\n\n# Whisper works best when inputs look like training: 16kHz mono, ~30s chunks.\nCHUNK_SEC = 30\nOVERLAP_SEC = 2\nGEN_KWARGS = {\"max_length\": 260, \"num_beams\": 4}\nBATCH_SIZE = 4\n\n# =========================\n# PREPROCESSING TOGGLES\n# =========================\n# (1) Fast, robust ffmpeg preprocessing (recommended)\nUSE_FFMPEG_PREPROCESS = True\n# Lightweight bandpass to reduce rumble + hiss (safe defaults)\nUSE_BANDPASS = True\n\n# (2) Heavy dereverb (often slow; may or may not help). Disabled by default.\nUSE_DEREVERB = False\n\n# (3) Optional speed adjustment for hard-to-hear fast speech (try 0.90–1.00)\nSEGMENT_SPEED = 1.00\n\n# (4) Final text normalization: Unicode normalize + punctuation cleanup\nFINAL_TEXT_NORMALIZE = True\nREMOVE_PUNCTUATION = True\n\n# =========================\n# VAD CONFIG (Silero)\n# =========================\n# Key knobs to tune:\n# - threshold: higher => fewer false positives (noise/music), lower => more recall\n# - speech_pad_ms: add context at segment borders (helps not cutting syllables)\n# - min_silence_duration_ms: how much silence to \"end\" a segment\n# - max_speech_duration_s: force-split long segments (Whisper prefers <=30s)\nVAD_CONFIG = {\n    \"threshold\": 0.55,\n    \"min_speech_duration_ms\": 250,\n    \"min_silence_duration_ms\": 300,\n    \"speech_pad_ms\": 200,\n    \"max_speech_duration_s\": 25,\n    \"window_size_samples\": 512,\n}\n\n# For heavy music/noise: be stricter (reduce false positives)\nVAD_CONFIG_MUSIC_SUPPRESS = {\n    \"threshold\": 0.70,\n    \"min_speech_duration_ms\": 300,\n    \"min_silence_duration_ms\": 600,\n    \"speech_pad_ms\": 150,\n    \"max_speech_duration_s\": 20,\n    \"window_size_samples\": 512,\n}\n\nUSE_MUSIC_SUPPRESS_VAD = False  # set True for files with lots of music/noise\n\nprint(\"✓ Config loaded\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.586471Z","iopub.execute_input":"2026-02-20T12:27:22.586760Z","iopub.status.idle":"2026-02-20T12:27:22.601928Z","shell.execute_reply.started":"2026-02-20T12:27:22.586725Z","shell.execute_reply":"2026-02-20T12:27:22.601225Z"},"papermill":{"duration":0.015106,"end_time":"2026-02-18T20:39:36.817623","exception":false,"start_time":"2026-02-18T20:39:36.802517","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nMANIFEST_CSV = \"/kaggle/input/datasets/sajibcuet/whisper-finetuned-reverb/pa_preprocessed_all_vad_segments.csv\"\nvad_df = pd.read_csv(MANIFEST_CSV)\n\nBASE_PATH = \"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio/\"\nvad_df[\"audio_path\"] = BASE_PATH + vad_df[\"filename\"]\nvad_df.head()\n\n# index: audio_path -> list of segments dicts\n_vad_index = {}\nfor ap, g in vad_df.groupby(\"audio_path\", sort=False):\n    g = g.sort_values(\"start\")\n    _vad_index[ap] = [\n        {\"start\": float(r.start), \"end\": float(r.end)}\n        for r in g.itertuples(index=False)\n    ]\n\ndef detect_speech_segments_using_csv(audio_path: str, aggressive=None):\n    \"\"\"CSV-backed VAD segments. (aggressive ignored)\"\"\"\n    return _vad_index.get(str(audio_path), [])\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.602951Z","iopub.execute_input":"2026-02-20T12:27:22.603264Z","iopub.status.idle":"2026-02-20T12:27:22.689505Z","shell.execute_reply.started":"2026-02-20T12:27:22.603241Z","shell.execute_reply":"2026-02-20T12:27:22.688878Z"},"papermill":{"duration":0.154519,"end_time":"2026-02-18T20:39:36.977944","exception":false,"start_time":"2026-02-18T20:39:36.823425","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Loading Silero VAD model...\")\nvad_model, vad_utils = torch.hub.load(\n    repo_or_dir='snakers4/silero-vad',\n    model='silero_vad',\n    force_reload=False,\n    onnx=False\n)\n(get_speech_timestamps, save_audio, read_audio, VADIterator, collect_chunks) = vad_utils\nprint(\"✓ VAD model loaded\")","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.690410Z","iopub.execute_input":"2026-02-20T12:27:22.690699Z","iopub.status.idle":"2026-02-20T12:27:22.973129Z","shell.execute_reply.started":"2026-02-20T12:27:22.690664Z","shell.execute_reply":"2026-02-20T12:27:22.972359Z"},"papermill":{"duration":1.13224,"end_time":"2026-02-18T20:39:38.115388","exception":false,"start_time":"2026-02-18T20:39:36.983148","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_speech_segments(audio_path: str, music_suppress: bool = False):\n    \"\"\"Use Silero VAD to detect speech segments (timestamps in seconds).\"\"\"\n    wav = read_audio(audio_path, sampling_rate=16000)\n    config = VAD_CONFIG_MUSIC_SUPPRESS if music_suppress else VAD_CONFIG\n\n    speech_timestamps = get_speech_timestamps(\n        wav, vad_model,\n        threshold=config[\"threshold\"],\n        min_speech_duration_ms=config[\"min_speech_duration_ms\"],\n        max_speech_duration_s=config[\"max_speech_duration_s\"],\n        min_silence_duration_ms=config[\"min_silence_duration_ms\"],\n        speech_pad_ms=config[\"speech_pad_ms\"],\n        window_size_samples=config[\"window_size_samples\"],\n        return_seconds=False,\n    )\n\n    return [{\"start\": s[\"start\"] / 16000.0, \"end\": s[\"end\"] / 16000.0} for s in speech_timestamps]\n\n\ndef merge_close_segments(segments, max_gap_sec=0.4):\n    \"\"\"Merge segments that are close together (reduces ASR boundary errors).\"\"\"\n    if not segments:\n        return []\n    merged = [segments[0].copy()]\n    for seg in segments[1:]:\n        if seg[\"start\"] - merged[-1][\"end\"] <= max_gap_sec:\n            merged[-1][\"end\"] = seg[\"end\"]\n        else:\n            merged.append(seg.copy())\n    return merged\n\n\ndef filter_short_segments(segments, min_duration_sec=0.25):\n    \"\"\"Remove very short segments.\"\"\"\n    return [s for s in segments if (s[\"end\"] - s[\"start\"]) >= min_duration_sec]\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.974070Z","iopub.execute_input":"2026-02-20T12:27:22.974381Z","iopub.status.idle":"2026-02-20T12:27:22.982191Z","shell.execute_reply.started":"2026-02-20T12:27:22.974355Z","shell.execute_reply":"2026-02-20T12:27:22.981525Z"},"papermill":{"duration":0.015236,"end_time":"2026-02-18T20:39:38.136032","exception":false,"start_time":"2026-02-18T20:39:38.120796","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_duration_sec(audio_path: str) -> float:\n    cmd = [\"ffprobe\", \"-v\", \"error\", \"-show_entries\", \"format=duration\",\n           \"-of\", \"default=noprint_wrappers=1:nokey=1\", str(audio_path)]\n    return float(subprocess.check_output(cmd).decode().strip())\n\n\ndef preprocess_audio_ffmpeg(src_path: str, out_wav: str):\n    \"\"\"Fast preprocessing: force 16kHz mono + optional gentle bandpass.\"\"\"\n    af = []\n    if USE_BANDPASS:\n        # gentle speech-focused band (keeps most Bengali phonetics)\n        af += [\"highpass=f=60\", \"lowpass=f=7800\"]\n    af_str = \",\".join(af) if af else \"anull\"\n\n    cmd = [\n        \"ffmpeg\", \"-hide_banner\", \"-loglevel\", \"error\",\n        \"-i\", str(src_path),\n        \"-ac\", \"1\", \"-ar\", \"16000\",\n        \"-af\", af_str,\n        \"-vn\", str(out_wav),\n        \"-y\",\n    ]\n    subprocess.check_call(cmd)\n\n\ndef extract_audio_segment(src_path: str, start_sec: float, end_sec: float, out_wav: str):\n    \"\"\"Extract a segment and re-encode as 16kHz mono WAV.\"\"\"\n    duration = max(0.0, end_sec - start_sec)\n    cmd = [\n        \"ffmpeg\", \"-hide_banner\", \"-loglevel\", \"error\",\n        \"-ss\", str(start_sec), \"-t\", str(duration),\n        \"-i\", str(src_path),\n        \"-ac\", \"1\", \"-ar\", \"16000\",\n        \"-vn\", str(out_wav), \"-y\"\n    ]\n    subprocess.check_call(cmd)\n\n\ndef merge_with_overlap(prev_text: str, next_text: str, max_overlap_words: int = 25) -> str:\n    prev_text, next_text = (prev_text or \"\").strip(), (next_text or \"\").strip()\n    if not prev_text:\n        return next_text\n    if not next_text:\n        return prev_text\n\n    prev_words, next_words = prev_text.split(), next_text.split()\n    max_k = min(max_overlap_words, len(prev_words), len(next_words))\n    best_k = 0\n    for k in range(1, max_k + 1):\n        if prev_words[-k:] == next_words[:k]:\n            best_k = k\n    if best_k > 0:\n        next_words = next_words[best_k:]\n    return (prev_text + \" \" + \" \".join(next_words)).strip()\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:22.983448Z","iopub.execute_input":"2026-02-20T12:27:22.983985Z","iopub.status.idle":"2026-02-20T12:27:22.999131Z","shell.execute_reply.started":"2026-02-20T12:27:22.983960Z","shell.execute_reply":"2026-02-20T12:27:22.998285Z"},"papermill":{"duration":0.01534,"end_time":"2026-02-18T20:39:38.156487","exception":false,"start_time":"2026-02-18T20:39:38.141147","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"files = sorted(glob.glob(DATASET_PATH + \"/*.wav\"))\n# files = (glob.glob(DATASET_PATH + \"/*.wav\"))\nprint(f\"✓ Found {len(files)} files\")","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:23.001944Z","iopub.execute_input":"2026-02-20T12:27:23.002567Z","iopub.status.idle":"2026-02-20T12:27:23.013932Z","shell.execute_reply.started":"2026-02-20T12:27:23.002542Z","shell.execute_reply":"2026-02-20T12:27:23.013129Z"},"papermill":{"duration":0.019846,"end_time":"2026-02-18T20:39:38.181424","exception":false,"start_time":"2026-02-18T20:39:38.161578","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe = pipeline(\n    task=\"automatic-speech-recognition\",\n    model=MODEL,\n    tokenizer=MODEL,\n    device=0,\n    batch_size=BATCH_SIZE\n)\npipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\nprint(\"✓ ASR pipeline loaded\")","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:23.014790Z","iopub.execute_input":"2026-02-20T12:27:23.015090Z","iopub.status.idle":"2026-02-20T12:27:25.628298Z","shell.execute_reply.started":"2026-02-20T12:27:23.015068Z","shell.execute_reply":"2026-02-20T12:27:25.627382Z"},"papermill":{"duration":18.102342,"end_time":"2026-02-18T20:39:56.289162","exception":false,"start_time":"2026-02-18T20:39:38.186820","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\n\ndef change_speed_ffmpeg(input_wav: str, output_wav: str, speed: float = 1.0):\n    \"\"\"Change speed with ffmpeg. speed<1 => slower, speed>1 => faster.\"\"\"\n    if abs(speed - 1.0) < 1e-6:\n        # no-op copy (still ensures wav exists)\n        cmd = [\"ffmpeg\", \"-y\", \"-loglevel\", \"error\", \"-i\", input_wav, output_wav]\n        subprocess.run(cmd, check=True)\n        return\n\n    # atempo supports 0.5–2.0 per filter; chain if needed\n    tempos = []\n    s = speed\n    while s < 0.5:\n        tempos.append(0.5); s /= 0.5\n    while s > 2.0:\n        tempos.append(2.0); s /= 2.0\n    tempos.append(s)\n    atempo = \",\".join([f\"atempo={t:.5f}\" for t in tempos])\n\n    cmd = [\"ffmpeg\", \"-y\", \"-loglevel\", \"error\", \"-i\", input_wav, \"-filter:a\", atempo, output_wav]\n    subprocess.run(cmd, check=True)\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.629451Z","iopub.execute_input":"2026-02-20T12:27:25.629780Z","iopub.status.idle":"2026-02-20T12:27:25.636402Z","shell.execute_reply.started":"2026-02-20T12:27:25.629756Z","shell.execute_reply":"2026-02-20T12:27:25.635613Z"},"papermill":{"duration":0.014686,"end_time":"2026-02-18T20:39:56.309379","exception":false,"start_time":"2026-02-18T20:39:56.294693","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nAudio Dereverb Pipeline for ASR Enhancement\n============================================\nRemoves echoing/reverb from audio files to improve ASR accuracy\n\nAuthor: Audio Processing Pipeline\nEnvironment: Kaggle Notebook\n\"\"\"\n\n# ====================================================================\n# SECTION 1: INSTALLATION & IMPORTS\n# ====================================================================\n\n# Install required packages (uncomment if needed)\n# !pip install numpy scipy soundfile librosa\n\nimport numpy as np\nimport scipy.signal as signal\nfrom scipy.io import wavfile\nimport wave\nimport os\nimport glob\nfrom pathlib import Path\nimport matplotlib.pyplot as plt\nimport librosa\nimport librosa.display\n\nprint(\"✓ All libraries imported successfully!\")\n\n# ====================================================================\n# SECTION 2: DEREVERB PIPELINE CLASS\n# ====================================================================\n\nclass DereverbPipeline:\n    \"\"\"\n    Complete pipeline for removing reverb/echo from audio files\n    Optimized for ASR preprocessing\n    \"\"\"\n    \n    def __init__(self, sample_rate=16000):\n        self.sample_rate = sample_rate\n        \n    def load_audio(self, filepath):\n        \"\"\"Load audio file (supports WAV)\"\"\"\n        with wave.open(filepath, 'rb') as wf:\n            params = wf.getparams()\n            frames = wf.readframes(params.nframes)\n            audio = np.frombuffer(frames, dtype=np.int16).astype(np.float32)\n            audio = audio / 32768.0  # Normalize to [-1, 1]\n            self.sample_rate = params.framerate\n        return audio\n    \n    def save_audio(self, audio, filepath):\n        \"\"\"Save audio file as WAV\"\"\"\n        # Create directory if it doesn't exist\n        os.makedirs(os.path.dirname(filepath), exist_ok=True)\n        \n        # Denormalize and convert back to int16\n        audio = np.clip(audio, -1.0, 1.0)\n        audio_int = (audio * 32767).astype(np.int16)\n        \n        with wave.open(filepath, 'wb') as wf:\n            wf.setnchannels(1)\n            wf.setsampwidth(2)\n            wf.setframerate(self.sample_rate)\n            wf.writeframes(audio_int.tobytes())\n    \n    def high_pass_filter(self, audio, cutoff=80):\n        \"\"\"Apply high-pass filter to remove low-frequency rumble\"\"\"\n        nyquist = self.sample_rate / 2\n        normalized_cutoff = cutoff / nyquist\n        sos = signal.butter(4, normalized_cutoff, btype='highpass', output='sos')\n        filtered = signal.sosfilt(sos, audio)\n        return filtered\n    \n    def spectral_subtraction(self, audio, noise_floor_percentile=10, alpha=2.0):\n        \"\"\"\n        Apply spectral subtraction to reduce reverb tail\n        \n        Parameters:\n        - noise_floor_percentile: percentile to estimate noise floor\n        - alpha: oversubtraction factor (higher = more aggressive)\n        \"\"\"\n        # Compute STFT\n        f, t, Zxx = signal.stft(audio, fs=self.sample_rate, nperseg=512, noverlap=384)\n        magnitude = np.abs(Zxx)\n        phase = np.angle(Zxx)\n        \n        # Estimate noise floor from low-energy frames\n        frame_energy = np.sum(magnitude**2, axis=0)\n        noise_threshold = np.percentile(frame_energy, noise_floor_percentile)\n        noise_frames = magnitude[:, frame_energy < noise_threshold]\n        \n        if noise_frames.shape[1] > 0:\n            noise_spectrum = np.mean(noise_frames, axis=1, keepdims=True)\n        else:\n            noise_spectrum = np.percentile(magnitude, noise_floor_percentile, axis=1, keepdims=True)\n        \n        # Spectral subtraction\n        magnitude_clean = magnitude - alpha * noise_spectrum\n        magnitude_clean = np.maximum(magnitude_clean, 0.01 * magnitude)  # Flooring\n        \n        # Reconstruct signal\n        Zxx_clean = magnitude_clean * np.exp(1j * phase)\n        _, audio_clean = signal.istft(Zxx_clean, fs=self.sample_rate, nperseg=512, noverlap=384)\n        \n        return audio_clean\n    \n    def wiener_filter(self, audio, noise_reduction=0.5):\n        \"\"\"\n        Apply Wiener filtering for reverb reduction\n        \n        Parameters:\n        - noise_reduction: strength of filtering (0-1)\n        \"\"\"\n        # Compute STFT\n        f, t, Zxx = signal.stft(audio, fs=self.sample_rate, nperseg=512, noverlap=384)\n        magnitude = np.abs(Zxx)\n        phase = np.angle(Zxx)\n        \n        # Estimate signal and noise power\n        signal_power = magnitude ** 2\n        noise_power = np.percentile(signal_power, 20, axis=1, keepdims=True)\n        \n        # Wiener gain\n        wiener_gain = signal_power / (signal_power + noise_reduction * noise_power)\n        wiener_gain = np.clip(wiener_gain, 0.1, 1.0)\n        \n        # Apply filter\n        magnitude_filtered = magnitude * wiener_gain\n        Zxx_filtered = magnitude_filtered * np.exp(1j * phase)\n        \n        # Reconstruct\n        _, audio_filtered = signal.istft(Zxx_filtered, fs=self.sample_rate, nperseg=512, noverlap=384)\n        \n        return audio_filtered\n    \n    def adaptive_dereverb(self, audio, frame_size=0.03, overlap=0.5):\n        \"\"\"\n        Adaptive dereverb using envelope-based processing\n        Reduces tail energy while preserving speech onsets\n        \"\"\"\n        frame_samples = int(frame_size * self.sample_rate)\n        hop_samples = int(frame_samples * (1 - overlap))\n        \n        # Compute energy envelope\n        envelope = np.abs(signal.hilbert(audio))\n        \n        # Smooth envelope\n        window_size = int(0.01 * self.sample_rate)  # 10ms\n        window = np.ones(window_size) / window_size\n        envelope_smooth = np.convolve(envelope, window, mode='same')\n        \n        # Detect speech/silence\n        threshold = np.percentile(envelope_smooth, 30)\n        speech_mask = envelope_smooth > threshold\n        \n        # Apply gain reduction to low-energy regions (reverb tail)\n        gain = np.ones_like(audio)\n        gain[~speech_mask] *= 0.3  # Reduce reverb tail by 70%\n        \n        # Smooth gain transitions\n        gain_smooth = np.convolve(gain, window, mode='same')\n        \n        return audio * gain_smooth\n    \n    def process(self, audio, aggressive=False):\n        \"\"\"\n        Main processing pipeline\n        \n        Parameters:\n        - aggressive: if True, apply more aggressive dereverb\n        \"\"\"\n        print(\"  → High-pass filtering...\")\n        audio = self.high_pass_filter(audio, cutoff=80)\n        \n        print(\"  → Spectral subtraction...\")\n        alpha = 2.5 if aggressive else 2.0\n        audio = self.spectral_subtraction(audio, alpha=alpha)\n        \n        print(\"  → Wiener filtering...\")\n        noise_reduction = 0.7 if aggressive else 0.5\n        audio = self.wiener_filter(audio, noise_reduction=noise_reduction)\n        \n        # print(\"  → Adaptive dereverb...\")\n        # audio = self.adaptive_dereverb(audio)\n        \n        # Normalize\n        audio = audio / (np.max(np.abs(audio)) + 1e-8) * 0.95\n        \n        return audio\n\n\n# ====================================================================\n# SECTION 3: VISUALIZATION FUNCTIONS\n# ====================================================================\n\ndef plot_audio_comparison(original, processed, sr=16000, duration=5):\n    \"\"\"\n    Plot waveform and spectrogram comparison\n    \"\"\"\n    # Limit to first N seconds for visualization\n    n_samples = int(duration * sr)\n    original = original[:n_samples]\n    processed = processed[:n_samples]\n    \n    fig, axes = plt.subplots(2, 2, figsize=(15, 8))\n    \n    # Waveforms\n    times = np.arange(len(original)) / sr\n    axes[0, 0].plot(times, original, alpha=0.7)\n    axes[0, 0].set_title('Original Waveform', fontsize=12, fontweight='bold')\n    axes[0, 0].set_xlabel('Time (s)')\n    axes[0, 0].set_ylabel('Amplitude')\n    axes[0, 0].grid(True, alpha=0.3)\n    \n    axes[0, 1].plot(times, processed, alpha=0.7, color='green')\n    axes[0, 1].set_title('Dereverbed Waveform', fontsize=12, fontweight='bold')\n    axes[0, 1].set_xlabel('Time (s)')\n    axes[0, 1].set_ylabel('Amplitude')\n    axes[0, 1].grid(True, alpha=0.3)\n    \n    # Spectrograms\n    D_orig = librosa.amplitude_to_db(np.abs(librosa.stft(original)), ref=np.max)\n    D_proc = librosa.amplitude_to_db(np.abs(librosa.stft(processed)), ref=np.max)\n    \n    img1 = librosa.display.specshow(D_orig, sr=sr, x_axis='time', y_axis='hz', ax=axes[1, 0])\n    axes[1, 0].set_title('Original Spectrogram', fontsize=12, fontweight='bold')\n    fig.colorbar(img1, ax=axes[1, 0], format='%+2.0f dB')\n    \n    img2 = librosa.display.specshow(D_proc, sr=sr, x_axis='time', y_axis='hz', ax=axes[1, 1])\n    axes[1, 1].set_title('Dereverbed Spectrogram', fontsize=12, fontweight='bold')\n    fig.colorbar(img2, ax=axes[1, 1], format='%+2.0f dB')\n    \n    plt.tight_layout()\n    plt.savefig('audio_comparison.png', dpi=150, bbox_inches='tight')\n    plt.show()\n    \n    print(\"✓ Visualization saved as 'audio_comparison.png'\")\n\n\n# ====================================================================\n# SECTION 4: BATCH PROCESSING FUNCTION\n# ====================================================================\n\ndef batch_process_audio(input_dir, output_dir, aggressive=False, visualize=False):\n    \"\"\"\n    Process all WAV files in a directory\n    \n    Parameters:\n    - input_dir: Directory containing input WAV files\n    - output_dir: Directory to save processed files\n    - aggressive: Use aggressive dereverb mode\n    - visualize: Create visualization for first file\n    \"\"\"\n    # Create output directory\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Find all WAV files\n    wav_files = glob.glob(os.path.join(input_dir, \"*.wav\"))\n    \n    if not wav_files:\n        print(f\"⚠ No WAV files found in {input_dir}\")\n        return\n    \n    print(f\"Found {len(wav_files)} WAV files to process\")\n    print(f\"Mode: {'AGGRESSIVE' if aggressive else 'NORMAL'}\")\n    print(\"=\"*60)\n    \n    pipeline = DereverbPipeline()\n    \n    for i, input_path in enumerate(wav_files, 1):\n        filename = os.path.basename(input_path)\n        output_path = os.path.join(output_dir, filename)\n        \n        print(f\"\\n[{i}/{len(wav_files)}] Processing: {filename}\")\n        \n        try:\n            # Load audio\n            audio = pipeline.load_audio(input_path)\n            duration = len(audio) / pipeline.sample_rate\n            print(f\"  Duration: {duration:.2f}s | Sample rate: {pipeline.sample_rate}Hz\")\n            \n            # Process\n            audio_clean = pipeline.process(audio, aggressive=aggressive)\n            \n            # Save\n            pipeline.save_audio(audio_clean, output_path)\n            print(f\"  ✓ Saved to: {output_path}\")\n            \n            # Visualize first file\n            if visualize and i == 1:\n                print(\"\\n  Creating visualization...\")\n                plot_audio_comparison(audio, audio_clean, sr=pipeline.sample_rate)\n            \n        except Exception as e:\n            print(f\"  ✗ Error processing {filename}: {str(e)}\")\n    \n    print(\"\\n\" + \"=\"*60)\n    print(f\"✓ Batch processing complete! Processed {len(wav_files)} files\")\n    print(f\"✓ Output directory: {output_dir}\")\n\n\n# ====================================================================\n# SECTION 5: SINGLE FILE PROCESSING\n# ====================================================================\n\ndef process_single_file(input_path, output_path=None, aggressive=False, visualize=True):\n    \"\"\"\n    Process a single audio file\n    \n    Parameters:\n    - input_path: Path to input WAV file\n    - output_path: Path for output file (optional)\n    - aggressive: Use aggressive dereverb mode\n    - visualize: Show before/after comparison\n    \"\"\"\n    if output_path is None:\n        base, ext = os.path.splitext(input_path)\n        output_path = f\"{base}_dereverb{ext}\"\n    \n    print(f\"Input: {input_path}\")\n    print(f\"Output: {output_path}\")\n    print(f\"Mode: {'AGGRESSIVE' if aggressive else 'NORMAL'}\")\n    print(\"=\"*60)\n    \n    pipeline = DereverbPipeline()\n    \n    # Load\n    print(\"\\n1. Loading audio...\")\n    audio = pipeline.load_audio(input_path)\n    duration = len(audio) / pipeline.sample_rate\n    print(f\"   Duration: {duration:.2f}s | Sample rate: {pipeline.sample_rate}Hz\")\n    \n    # Process\n    print(\"\\n2. Processing audio...\")\n    audio_clean = pipeline.process(audio, aggressive=aggressive)\n    \n    # Save\n    print(\"\\n3. Saving processed audio...\")\n    pipeline.save_audio(audio_clean, output_path)\n    print(f\"   ✓ Saved to: {output_path}\")\n    \n    # Visualize\n    if visualize:\n        print(\"\\n4. Creating visualization...\")\n        plot_audio_comparison(audio, audio_clean, sr=pipeline.sample_rate)\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"✓ Processing complete!\")\n    \n    return output_path\n\n\n# ====================================================================\n# SECTION 6: KAGGLE-SPECIFIC SETUP\n# ====================================================================\n\ndef setup_kaggle_paths():\n    \"\"\"\n    Setup standard Kaggle paths\n    \"\"\"\n    paths = {\n        'input': '/kaggle/input',\n        'working': '/kaggle/working',\n        'output': '/kaggle/working/dereverb_output'\n    }\n    \n    # Create output directory\n    os.makedirs(paths['output'], exist_ok=True)\n    \n    return paths","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.637466Z","iopub.execute_input":"2026-02-20T12:27:25.637783Z","iopub.status.idle":"2026-02-20T12:27:25.676987Z","shell.execute_reply.started":"2026-02-20T12:27:25.637748Z","shell.execute_reply":"2026-02-20T12:27:25.676308Z"},"papermill":{"duration":0.25533,"end_time":"2026-02-18T20:39:56.570200","exception":false,"start_time":"2026-02-18T20:39:56.314870","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transcribe_file_with_vad(audio_path: str) -> str:\n    \"\"\"Transcribe with VAD pre-filtering.\"\"\"\n\n    with tempfile.TemporaryDirectory() as tmpd:\n        # 1) Preprocess audio (fast + consistent)\n        work_wav = os.path.join(tmpd, \"work_16k.wav\")\n        if USE_FFMPEG_PREPROCESS:\n            preprocess_audio_ffmpeg(audio_path, work_wav)\n        else:\n            work_wav = audio_path\n\n        # 2) Optional dereverb (slow; disabled by default)\n        if USE_DEREVERB:\n            dereverb_out = os.path.join(tmpd, \"dereverb.wav\")\n            process_single_file(work_wav, dereverb_out, aggressive=False, visualize=False)\n            work_wav = dereverb_out\n\n        # 3) VAD on the same audio used for ASR\n        print(f\"  Detecting speech segments...\")\n        speech_segments = detect_speech_segments(work_wav, music_suppress=USE_MUSIC_SUPPRESS_VAD)\n        # speech_segments = filter_short_segments(speech_segments, min_duration_sec=0.25)\n        # speech_segments = merge_close_segments(speech_segments, max_gap_sec=0.4)\n\n        print(f\"  Found {len(speech_segments)} speech segments\")\n        if not speech_segments:\n            print(\"  ⚠ No speech detected!\")\n            return \"\"\n\n        # 4) ASR per segment (keeps memory stable)\n        merged = \"\"\n        for i, seg in enumerate(speech_segments):\n            raw_wav = os.path.join(tmpd, f\"segment_{i:04d}_raw.wav\")\n            seg_wav = os.path.join(tmpd, f\"segment_{i:04d}.wav\")\n\n            extract_audio_segment(work_wav, seg[\"start\"], seg[\"end\"], raw_wav)\n            change_speed_ffmpeg(raw_wav, seg_wav, speed=1)\n\n            out = pipe(seg_wav, generate_kwargs=GEN_KWARGS, return_timestamps=False)\n            text = (out.get(\"text\", \"\") or \"\").strip()\n\n            merged = merge_with_overlap(merged, text, max_overlap_words=25)\n\n            # free CUDA memory proactively\n            del out\n            if torch.cuda.is_available():\n                torch.cuda.empty_cache()\n\n        return merged.strip()\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.678127Z","iopub.execute_input":"2026-02-20T12:27:25.678415Z","iopub.status.idle":"2026-02-20T12:27:25.692717Z","shell.execute_reply.started":"2026-02-20T12:27:25.678385Z","shell.execute_reply":"2026-02-20T12:27:25.691938Z"},"papermill":{"duration":0.016493,"end_time":"2026-02-18T20:39:56.591928","exception":false,"start_time":"2026-02-18T20:39:56.575435","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, tempfile\n\n# ── Patch: make sure generation_config has the timestamp token ids that\n#    Whisper's _set_return_timestamps() requires, so passing\n#    return_timestamps=False never triggers the ValueError.\nif not hasattr(pipe.model.generation_config, \"no_timestamps_token_id\"):\n    pipe.model.generation_config.no_timestamps_token_id = (\n        pipe.tokenizer.convert_tokens_to_ids(\"<|notimestamps|>\")\n    )\nif not hasattr(pipe.model.generation_config, \"begin_suppress_tokens\"):\n    pipe.model.generation_config.begin_suppress_tokens = []\n\n\ndef transcribe_file_with_vad_csv(audio_path: str) -> str:\n    \"\"\"Transcribe using precomputed VAD segments from CSV.\"\"\"\n\n    with tempfile.TemporaryDirectory() as tmpd:\n        # 1) Preprocess audio for consistent VAD/ASR input\n        work_wav = os.path.join(tmpd, \"work_16k.wav\")\n        if USE_FFMPEG_PREPROCESS:\n            preprocess_audio_ffmpeg(audio_path, work_wav)\n        else:\n            work_wav = audio_path\n\n        if USE_DEREVERB:\n            dereverb_out = os.path.join(tmpd, \"dereverb.wav\")\n            process_single_file(work_wav, dereverb_out, aggressive=False, visualize=False)\n            work_wav = dereverb_out\n\n        print(f\"  Detecting speech segments (from CSV)...\")\n        speech_segments = detect_speech_segments_using_csv(audio_path, aggressive=None)\n        speech_segments = filter_short_segments(speech_segments, min_duration_sec=0.25)\n        speech_segments = merge_close_segments(speech_segments, max_gap_sec=0.4)\n\n        print(f\"  Found {len(speech_segments)} speech segments\")\n        if not speech_segments:\n            print(\"  ⚠ No speech detected!\")\n            return \"\"\n\n        merged = \"\"\n        for i, seg in enumerate(speech_segments):\n            raw_wav = os.path.join(tmpd, f\"segment_{i:04d}_raw.wav\")\n            seg_wav = os.path.join(tmpd, f\"segment_{i:04d}.wav\")\n\n            extract_audio_segment(work_wav, seg[\"start\"], seg[\"end\"], raw_wav)\n            change_speed_ffmpeg(raw_wav, seg_wav, speed=1)\n\n            out = pipe(seg_wav, generate_kwargs=GEN_KWARGS, return_timestamps=False)\n            text = (out.get(\"text\", \"\") or \"\").strip()\n\n            # print(text)\n            \n            merged = merge_with_overlap(merged, text, max_overlap_words=25)\n\n            del out\n            if torch.cuda.is_available():\n                torch.cuda.empty_cache()\n\n        return merged.strip()\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.693835Z","iopub.execute_input":"2026-02-20T12:27:25.694144Z","iopub.status.idle":"2026-02-20T12:27:25.709201Z","shell.execute_reply.started":"2026-02-20T12:27:25.694109Z","shell.execute_reply":"2026-02-20T12:27:25.708314Z"},"papermill":{"duration":0.017759,"end_time":"2026-02-18T20:39:56.615106","exception":false,"start_time":"2026-02-18T20:39:56.597347","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# t = transcribe_file_with_vad_csv(\"/kaggle/input/competitions/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio/test_004.wav\")","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.710308Z","iopub.execute_input":"2026-02-20T12:27:25.710620Z","iopub.status.idle":"2026-02-20T12:27:25.724468Z","shell.execute_reply.started":"2026-02-20T12:27:25.710588Z","shell.execute_reply":"2026-02-20T12:27:25.723734Z"},"papermill":{"duration":0.012014,"end_time":"2026-02-18T20:39:56.632712","exception":false,"start_time":"2026-02-18T20:39:56.620698","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Bangla text normalization (helps WER in many Bengali ASR setups)\n# - Unicode normalization (fixes composing chars, invalid sequences)\n# - Optional punctuation cleanup\n# ============================================================\nfrom bnunicodenormalizer import Normalizer\n\n_bn_norm = Normalizer()\n\n# keep Bengali letters (ঀ-৿), Bengali digits (০-৯), ASCII digits, space\n_BN_KEEP = re.compile(r\"[^ঀ-৿0-9০-৯ ]+\")\n\ndef normalize_bn_text(text: str) -> str:\n    text = (text or \"\").strip()\n    if not text:\n        return \"\"\n\n    # Normalize each token (bnunicodenormalizer is token-oriented)\n    toks = []\n    for w in text.split():\n        try:\n            out = _bn_norm(w)\n            w2 = out.get(\"normalized\") or w\n        except Exception:\n            w2 = w\n        toks.append(w2)\n    text = \" \".join(toks)\n\n    if REMOVE_PUNCTUATION:\n        text = _BN_KEEP.sub(\" \", text)\n\n    # Collapse spaces\n    text = re.sub(r\"\\s+\", \" \", text).strip()\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.725625Z","iopub.execute_input":"2026-02-20T12:27:25.725947Z","iopub.status.idle":"2026-02-20T12:27:25.737631Z","shell.execute_reply.started":"2026-02-20T12:27:25.725919Z","shell.execute_reply":"2026-02-20T12:27:25.736751Z"},"papermill":{"duration":0.018339,"end_time":"2026-02-18T20:39:56.656581","exception":false,"start_time":"2026-02-18T20:39:56.638242","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nrows = []\n\nprint(\"\\nStarting transcription with VAD filtering...\")\nfor f in tqdm(files, desc=\"VAD+ASR\"):\n    fname = Path(f).stem  # submission expects stem\n    print(f\"\\nProcessing: {fname}\")\n\n    t = transcribe_file_with_vad_csv(f)\n\n    if FINAL_TEXT_NORMALIZE:\n        t = normalize_bn_text(t)\n\n    print(t)\n\n    rows.append({\n        \"filename\": fname,\n        \"transcript\": t\n    })\n\nsubmission = pd.DataFrame(rows, columns=[\"filename\", \"transcript\"])\nprint(\"\\n✓ ASR done\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:27:25.738666Z","iopub.execute_input":"2026-02-20T12:27:25.738931Z","iopub.status.idle":"2026-02-20T12:44:50.570264Z","shell.execute_reply.started":"2026-02-20T12:27:25.738908Z","shell.execute_reply":"2026-02-20T12:44:50.569008Z"},"papermill":{"duration":28292.995428,"end_time":"2026-02-19T04:31:29.657629","exception":false,"start_time":"2026-02-18T20:39:56.662201","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission_no_postprocess.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.571240Z","iopub.status.idle":"2026-02-20T12:44:50.571569Z","shell.execute_reply.started":"2026-02-20T12:44:50.571440Z","shell.execute_reply":"2026-02-20T12:44:50.571456Z"},"papermill":{"duration":0.913482,"end_time":"2026-02-19T04:31:31.321162","exception":false,"start_time":"2026-02-19T04:31:30.407680","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport unicodedata\n\nSPACE_RE = re.compile(r\"\\s+\")\nNBSP_RE = re.compile(r\"[\\u00A0\\u2007\\u202F]\")  # common non-breaking spaces\n\ndef basic_normalize(text: str) -> str:\n    text = unicodedata.normalize(\"NFC\", text)\n    text = NBSP_RE.sub(\" \", text)\n    text = SPACE_RE.sub(\" \", text).strip()\n    return text\n\n\ndef remove_consecutive_duplicates(text):\n    words = text.split()\n    result = []\n    prev_word = None\n    repeat_count = 0\n    \n    for word in words:\n        if word == prev_word:\n            repeat_count += 1\n            # Keep first occurrence, skip subsequent unless valid pattern\n            # Allow: \"শত শত\", \"লাখ লাখ\", \"কোটি কোটি\" (quantity expressions)\n            if word in ['শত', 'লাখ', 'কোটি', 'হাজার', 'মানে'] and repeat_count == 1:\n                result.append(word)\n            if word in ['না '] and repeat_count == 2:\n                result.append(word)\n            if word in ['না '] and repeat_count == 3:\n                result.append(word)\n        else:\n            result.append(word)\n            repeat_count = 0\n        prev_word = word\n    \n    return ' '.join(result)\n\n# Standardize number words\n\ndef normalize_numbers(text):\n    replacements = {\n    'একশ': 'একশো',\n    'দুইশ': 'দুইশো',\n    'তিনশ': 'তিনশো',\n    'চারশ': 'চারশো',\n    'পাঁচশ': 'পাঁচশো',\n    'ছয়শ': 'ছয়শো',\n    'সাতশ': 'সাতশো',\n    'আটশ': 'আটশো',\n    'নয়শ': 'নয়শো',\n}\n    \n    for old, new in replacements.items():\n        text = text.replace(old, new)\n    \n    # Optionally convert to digits if reference has digits\n    # number_map = {'এক': '১', 'দুই': '২', ...}\n    \n    return text\n\ndef collapse_repeated_bigrams(text: str, max_repeats: int = 1) -> str:\n    w = text.split()\n    if len(w) < 4:\n        return text\n    out = []\n    i = 0\n    while i < len(w):\n        if i+3 < len(w) and w[i] == w[i+2] and w[i+1] == w[i+3]:\n            # bigram repeated at least twice\n            out.extend([w[i], w[i+1]])\n            # skip repeats\n            reps = 1\n            j = i + 2\n            while j+1 < len(w) and w[j] == w[i] and w[j+1] == w[i+1]:\n                reps += 1\n                if reps <= max_repeats:\n                    out.extend([w[j], w[j+1]])\n                j += 2\n            i = j\n        else:\n            out.append(w[i])\n            i += 1\n    return \" \".join(out)\n\n\ndef remove_long_words(text: str, max_len: int = 25) -> str:\n    words = text.split()\n    return \" \".join(w for w in words if len(w) <= max_len)\n\ndef postprocess(text: str) -> str:\n    text = basic_normalize(text)\n    text = remove_consecutive_duplicates(text)\n    text = collapse_repeated_bigrams(text, max_repeats=1)\n    text = remove_long_words(text, max_len=25)\n    text = normalize_numbers(text)\n    return text\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.573065Z","iopub.status.idle":"2026-02-20T12:44:50.573587Z","shell.execute_reply.started":"2026-02-20T12:44:50.573452Z","shell.execute_reply":"2026-02-20T12:44:50.573469Z"},"papermill":{"duration":0.7635,"end_time":"2026-02-19T04:31:32.821327","exception":false,"start_time":"2026-02-19T04:31:32.057827","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission[\"transcript\"] = submission[\"transcript\"].apply(postprocess)","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.574442Z","iopub.status.idle":"2026-02-20T12:44:50.574779Z","shell.execute_reply.started":"2026-02-20T12:44:50.574622Z","shell.execute_reply":"2026-02-20T12:44:50.574647Z"},"papermill":{"duration":0.947207,"end_time":"2026-02-19T04:31:34.630481","exception":false,"start_time":"2026-02-19T04:31:33.683274","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.575787Z","iopub.status.idle":"2026-02-20T12:44:50.576176Z","shell.execute_reply.started":"2026-02-20T12:44:50.575974Z","shell.execute_reply":"2026-02-20T12:44:50.575994Z"},"papermill":{"duration":0.777204,"end_time":"2026-02-19T04:31:36.277526","exception":false,"start_time":"2026-02-19T04:31:35.500322","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.577461Z","iopub.status.idle":"2026-02-20T12:44:50.577736Z","shell.execute_reply.started":"2026-02-20T12:44:50.577611Z","shell.execute_reply":"2026-02-20T12:44:50.577628Z"},"papermill":{"duration":0.869458,"end_time":"2026-02-19T04:31:37.881543","exception":false,"start_time":"2026-02-19T04:31:37.012085","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# del pipe\n# # del vad_model\n# del vad_pipeline  \n# torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.578574Z","iopub.status.idle":"2026-02-20T12:44:50.578922Z","shell.execute_reply.started":"2026-02-20T12:44:50.578733Z","shell.execute_reply":"2026-02-20T12:44:50.578752Z"},"papermill":{"duration":0.904279,"end_time":"2026-02-19T04:31:39.532218","exception":false,"start_time":"2026-02-19T04:31:38.627939","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install jiwer --quiet","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.580125Z","iopub.status.idle":"2026-02-20T12:44:50.580399Z","shell.execute_reply.started":"2026-02-20T12:44:50.580273Z","shell.execute_reply":"2026-02-20T12:44:50.580288Z"},"papermill":{"duration":0.868782,"end_time":"2026-02-19T04:31:41.140953","exception":false,"start_time":"2026-02-19T04:31:40.272171","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import jiwer\n\n# def wer_from_two_files(reference_path: str, hypothesis_path: str) -> float:\n#     \"\"\"\n#     Compute WER between two text files using jiwer.\n\n#     - If files have multiple lines, compares line-by-line (weighted WER).\n#     - If line counts differ, it falls back to comparing the full text as one string.\n#     \"\"\"\n#     with open(reference_path, \"r\", encoding=\"utf-8\") as f:\n#         ref_lines = [ln.strip() for ln in f.read().splitlines()]\n\n#     with open(hypothesis_path, \"r\", encoding=\"utf-8\") as f:\n#         hyp_lines = [ln.strip() for ln in f.read().splitlines()]\n\n#     # Drop completely empty lines on both sides (optional but usually sensible)\n#     ref_lines = [x for x in ref_lines if x != \"\"]\n#     hyp_lines = [x for x in hyp_lines if x != \"\"]\n\n#     # If same number of lines, do list WER (weighted by ref length)\n#     if len(ref_lines) == len(hyp_lines) and len(ref_lines) > 0:\n#         return float(jiwer.wer(ref_lines, hyp_lines))\n\n#     # Otherwise compare as one big string\n#     ref_text = \" \".join(ref_lines).strip()\n#     hyp_text = \" \".join(hyp_lines).strip()\n#     return float(jiwer.wer(ref_text, hyp_text))\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.581205Z","iopub.status.idle":"2026-02-20T12:44:50.581575Z","shell.execute_reply.started":"2026-02-20T12:44:50.581385Z","shell.execute_reply":"2026-02-20T12:44:50.581405Z"},"papermill":{"duration":0.859301,"end_time":"2026-02-19T04:31:42.746960","exception":false,"start_time":"2026-02-19T04:31:41.887659","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# score = wer_from_two_files(\"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/annotation/train_001.txt\", \n#                           \"/kaggle/working/transcript_0.txt\"\n#                           )\n# print(\"WER:\", score)\n","metadata":{"execution":{"iopub.status.busy":"2026-02-20T12:44:50.583331Z","iopub.status.idle":"2026-02-20T12:44:50.584161Z","shell.execute_reply.started":"2026-02-20T12:44:50.583791Z","shell.execute_reply":"2026-02-20T12:44:50.584006Z"},"papermill":{"duration":0.856705,"end_time":"2026-02-19T04:31:44.335690","exception":false,"start_time":"2026-02-19T04:31:43.478985","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.855658,"end_time":"2026-02-19T04:31:45.921488","exception":false,"start_time":"2026-02-19T04:31:45.065830","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}