{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":129276,"databundleVersionId":15506988,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n!pip install -U transformers accelerate","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:26:26.338136Z","iopub.execute_input":"2026-02-09T19:26:26.338416Z","iopub.status.idle":"2026-02-09T19:26:42.132637Z","shell.execute_reply.started":"2026-02-09T19:26:26.338390Z","shell.execute_reply":"2026-02-09T19:26:42.131690Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os, glob\nfrom pathlib import Path\n\nimport torch\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\n\nfrom transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:26:42.134347Z","iopub.execute_input":"2026-02-09T19:26:42.134651Z","iopub.status.idle":"2026-02-09T19:27:01.369858Z","shell.execute_reply.started":"2026-02-09T19:26:42.134624Z","shell.execute_reply":"2026-02-09T19:27:01.369069Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Paths + sanity checks + read sample submission","metadata":{}},{"cell_type":"code","source":"ROOT = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition\"\n\nTRAIN_AUDIO_DIR = f\"{ROOT}/transcription/transcription/train/audio\"\nTRAIN_ANNO_DIR  = f\"{ROOT}/transcription/transcription/train/annotation\"\nTEST_AUDIO_DIR  = f\"{ROOT}/transcription/transcription/test/audio\"\nSAMPLE_SUB_PATH = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/sample_submission .csv\"\n\n# --- sanity checks\nfor p in [TRAIN_AUDIO_DIR, TRAIN_ANNO_DIR, TEST_AUDIO_DIR, SAMPLE_SUB_PATH]:\n    print((\"OK  \" if os.path.exists(p) else \"MISS\"), p)\n\ntrain_wavs = sorted(glob.glob(f\"{TRAIN_AUDIO_DIR}/*.wav\"))\ntrain_txts = sorted(glob.glob(f\"{TRAIN_ANNO_DIR}/*.txt\"))\ntest_wavs  = sorted(glob.glob(f\"{TEST_AUDIO_DIR}/*.wav\"))\n\nprint(\"\\nCounts\")\nprint(\"train wav:\", len(train_wavs))\nprint(\"train txt:\", len(train_txts))\nprint(\"test  wav:\", len(test_wavs))\n\nprint(\"\\nExamples\")\nprint(\"train wav ex:\", os.path.basename(train_wavs[0]) if train_wavs else None)\nprint(\"train txt ex:\", os.path.basename(train_txts[0]) if train_txts else None)\nprint(\"test  wav ex:\", os.path.basename(test_wavs[0])  if test_wavs  else None)\n\nSAMPLE_SUB_PATH = SAMPLE_SUB_PATH.strip()  # protects against invisible whitespace\nsample_sub = pd.read_csv(SAMPLE_SUB_PATH)\ndisplay(sample_sub.head())\nprint(\"sample_submission columns:\", sample_sub.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:01.370836Z","iopub.execute_input":"2026-02-09T19:27:01.371293Z","iopub.status.idle":"2026-02-09T19:27:01.436190Z","shell.execute_reply.started":"2026-02-09T19:27:01.371259Z","shell.execute_reply":"2026-02-09T19:27:01.435618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model + chunking config","metadata":{}},{"cell_type":"code","source":"# Pick ONE model\nMODEL_ID = \"bengaliAI/tugstugi_bengaliai-asr_whisper-medium\"\n# MODEL_ID = \"bengaliAI/bengali-asr_tugstugi_whisper-medium\"\n# MODEL_ID = \"mozilla-ai/whisper-large-v3-turbo-bn\"\n\nLANG = \"bn\"\nTASK = \"transcribe\"     # keep this unless competition wants English\n\n# Long-audio chunking\nCHUNK_LENGTH_S = 60\nSTRIDE_S = (5, 2)       # (left, right) seconds overlap\nBATCH_SIZE = 8         # lower if VRAM issues (4/2/1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:01.437804Z","iopub.execute_input":"2026-02-09T19:27:01.438123Z","iopub.status.idle":"2026-02-09T19:27:01.442270Z","shell.execute_reply.started":"2026-02-09T19:27:01.438101Z","shell.execute_reply":"2026-02-09T19:27:01.441636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# ---- unify variable names\nCHUNK_S      = globals().get(\"CHUNK_S\", globals().get(\"CHUNK_LENGTH_S\", 30))\nOVERLAP_S    = globals().get(\"OVERLAP_S\", (globals().get(\"STRIDE_S\", (5, 2))[0] if isinstance(globals().get(\"STRIDE_S\", (5,2)), (tuple, list)) else 5))\nBATCH_CHUNKS = globals().get(\"BATCH_CHUNKS\", globals().get(\"BATCH_SIZE\", 8))\n\nMAX_NEW_TOKENS = globals().get(\"MAX_NEW_TOKENS\", 256)\nNUM_BEAMS      = globals().get(\"NUM_BEAMS\", 1)\n\n# ---- define _chunk_audio if you don't already have it\nif \"_chunk_audio\" not in globals():\n    def _chunk_audio(y: np.ndarray, sr: int, chunk_s: int, overlap_s: int):\n        chunk_len = int(chunk_s * sr)\n        hop_len   = int((chunk_s - overlap_s) * sr)\n        if hop_len <= 0:\n            raise ValueError(\"OVERLAP_S must be smaller than CHUNK_S\")\n\n        n = len(y)\n        if n <= chunk_len:\n            return [y.astype(np.float32)]\n\n        chunks = []\n        start = 0\n        while start < n:\n            end = start + chunk_len\n            chunk = y[start:end]\n            if len(chunk) < chunk_len:\n                chunk = np.pad(chunk, (0, chunk_len - len(chunk)), mode=\"constant\")\n            chunks.append(chunk.astype(np.float32))\n            start += hop_len\n        return chunks\n\nprint(\"CONFIG:\",\n      \"CHUNK_S=\", CHUNK_S,\n      \"OVERLAP_S=\", OVERLAP_S,\n      \"BATCH_CHUNKS=\", BATCH_CHUNKS,\n      \"MAX_NEW_TOKENS=\", MAX_NEW_TOKENS,\n      \"NUM_BEAMS=\", NUM_BEAMS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:01.443051Z","iopub.execute_input":"2026-02-09T19:27:01.443277Z","iopub.status.idle":"2026-02-09T19:27:01.460445Z","shell.execute_reply.started":"2026-02-09T19:27:01.443257Z","shell.execute_reply":"2026-02-09T19:27:01.459735Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load model + processor + pipeline","metadata":{}},{"cell_type":"code","source":"device = 0 if torch.cuda.is_available() else -1\ndtype  = torch.float16 if torch.cuda.is_available() else torch.float32\n\nprint(\"CUDA:\", torch.cuda.is_available())\nif torch.cuda.is_available():\n    print(\"GPU:\", torch.cuda.get_device_name(0))\n\nmodel = AutoModelForSpeechSeq2Seq.from_pretrained(\n    MODEL_ID,\n    torch_dtype=dtype,\n    low_cpu_mem_usage=True,\n    use_safetensors=True,\n)\nprocessor = AutoProcessor.from_pretrained(MODEL_ID)\n\nasr = pipeline(\n    \"automatic-speech-recognition\",\n    model=model,\n    tokenizer=processor.tokenizer,\n    feature_extractor=processor.feature_extractor,\n    device=device,\n    dtype=dtype,\n)\n\nprint(\"Loaded:\", MODEL_ID)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:01.461398Z","iopub.execute_input":"2026-02-09T19:27:01.461735Z","iopub.status.idle":"2026-02-09T19:27:17.781776Z","shell.execute_reply.started":"2026-02-09T19:27:01.461713Z","shell.execute_reply":"2026-02-09T19:27:17.780929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quick sanity test: transcribe only the first 2 minutes of train_001.wav","metadata":{}},{"cell_type":"code","source":"import librosa\nimport torch\ndef _chunk_audio(y, sr, chunk_s, overlap_s):\n    import numpy as np\n    chunk_len = int(chunk_s * sr)\n    hop_len   = int((chunk_s - overlap_s) * sr)\n    if hop_len <= 0:\n        raise ValueError(\"OVERLAP_S must be smaller than CHUNK_S\")\n\n    n = len(y)\n    if n <= chunk_len:\n        return [y.astype(np.float32)]\n\n    chunks = []\n    start = 0\n    while start < n:\n        end = start + chunk_len\n        chunk = y[start:end]\n        if len(chunk) < chunk_len:\n            chunk = np.pad(chunk, (0, chunk_len - len(chunk)), mode=\"constant\")\n        chunks.append(chunk.astype(np.float32))\n        start += hop_len\n    return chunks\n\nTEST_LONG_WAV = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/audio/train_001.wav\"\n\n# 1) Load ONLY first 120s (2 minutes)\ny, sr = librosa.load(TEST_LONG_WAV, sr=16000, mono=True, offset=0.0, duration=120.0)\nprint(\"Loaded seconds:\", round(len(y)/sr, 2), \"| sr:\", sr)\n\n# 2) Convert to input features\ninputs = processor(y, sampling_rate=sr, return_tensors=\"pt\")\ninput_features = inputs.input_features.to(model.device)\nif input_features.dtype != next(model.parameters()).dtype:\n    input_features = input_features.to(next(model.parameters()).dtype)\n\n# 3) Generate + decode (NO language/task args)\nwith torch.inference_mode():\n    pred_ids = model.generate(\n        input_features,\n        max_new_tokens=128,\n        num_beams=1\n    )\n\ntext = processor.batch_decode(pred_ids, skip_special_tokens=True)[0].strip()\n\nprint(\"\\n--- 2-min transcript preview ---\\n\")\nprint(text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:17.782828Z","iopub.execute_input":"2026-02-09T19:27:17.783181Z","iopub.status.idle":"2026-02-09T19:27:43.518206Z","shell.execute_reply.started":"2026-02-09T19:27:17.783156Z","shell.execute_reply":"2026-02-09T19:27:43.517551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install webrtcvad","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:43.519039Z","iopub.execute_input":"2026-02-09T19:27:43.519539Z","iopub.status.idle":"2026-02-09T19:27:51.211305Z","shell.execute_reply.started":"2026-02-09T19:27:43.519514Z","shell.execute_reply":"2026-02-09T19:27:51.210568Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Long-audio chunked transcription","metadata":{}},{"cell_type":"code","source":"import webrtcvad\nimport librosa\n\n# ---- gates (tune)\nSILENCE_DB = -45.0          # if chunk RMS is below this, likely silence\nMIN_NONSILENT_S = 0.8       # require at least this much non-silent audio\nVAD_MODE = 2                # 0=loose ... 3=aggressive\nFRAME_MS = 30               # 10/20/30 only\nMIN_SPEECH_S = 0.6          # require at least this much speech per chunk\n\ndef _rms_db(y: np.ndarray):\n    rms = np.sqrt(np.mean(np.square(y)) + 1e-12)\n    return 20 * np.log10(rms + 1e-12)\n\ndef _nonsilent_seconds(y: np.ndarray, sr: int):\n    # librosa split uses top_db relative to peak; good enough for gating\n    intervals = librosa.effects.split(y, top_db=25)\n    total = sum((end - start) for start, end in intervals) / sr\n    return float(total)\n\ndef _float_to_pcm16(y: np.ndarray):\n    y = np.clip(y, -1.0, 1.0)\n    return (y * 32767.0).astype(np.int16)\n\ndef _webrtc_speech_seconds(y: np.ndarray, sr: int, mode=2, frame_ms=30):\n    assert sr in (8000, 16000, 32000, 48000), \"webrtcvad supports 8/16/32/48 kHz\"\n    vad = webrtcvad.Vad(mode)\n\n    pcm16 = _float_to_pcm16(y)\n    frame_len = int(sr * frame_ms / 1000)\n    if frame_len <= 0:\n        return 0.0\n\n    voiced_frames = 0\n    total_frames = 0\n\n    # bytes per frame\n    for i in range(0, len(pcm16) - frame_len + 1, frame_len):\n        frame = pcm16[i:i+frame_len].tobytes()\n        total_frames += 1\n        if vad.is_speech(frame, sr):\n            voiced_frames += 1\n\n    return (voiced_frames * frame_ms) / 1000.0\n\ndef chunk_should_transcribe(y: np.ndarray, sr: int) -> bool:\n    # 1) silence gate\n    if _rms_db(y) < SILENCE_DB and _nonsilent_seconds(y, sr) < MIN_NONSILENT_S:\n        return False\n\n    # 2) VAD gate\n    speech_s = _webrtc_speech_seconds(y, sr, mode=VAD_MODE, frame_ms=FRAME_MS)\n    return speech_s >= MIN_SPEECH_S\n\n# ---- Updated transcribe_long_wav: skips chunks failing the gate\ndef transcribe_long_wav(wav_path: str) -> str:\n    y, sr = librosa.load(wav_path, sr=16000, mono=True)\n    chunks = _chunk_audio(y, sr, CHUNK_S, OVERLAP_S)\n\n    # pre-filter chunks (skip noise/silence/music-ish)\n    kept = [c for c in chunks if chunk_should_transcribe(c, sr)]\n    if not kept:\n        return \"\"  # nothing worth transcribing\n\n    texts = []\n    param_dtype = next(model.parameters()).dtype\n\n    for i in range(0, len(kept), BATCH_CHUNKS):\n        batch = kept[i:i + BATCH_CHUNKS]\n        inputs = processor(batch, sampling_rate=sr, return_tensors=\"pt\", padding=True)\n\n        feats = inputs.input_features.to(model.device)\n        if feats.dtype != param_dtype:\n            feats = feats.to(param_dtype)\n\n        with torch.inference_mode():\n            with torch.cuda.amp.autocast(enabled=torch.cuda.is_available(), dtype=torch.float16):\n                pred_ids = model.generate(\n                    feats,\n                    max_new_tokens=MAX_NEW_TOKENS,\n                    num_beams=NUM_BEAMS,\n                )\n\n        batch_text = processor.batch_decode(pred_ids, skip_special_tokens=True)\n        texts.extend([t.strip() for t in batch_text])\n\n    return \" \".join([t for t in texts if t]).strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:51.212841Z","iopub.execute_input":"2026-02-09T19:27:51.213413Z","iopub.status.idle":"2026-02-09T19:27:51.413852Z","shell.execute_reply.started":"2026-02-09T19:27:51.213382Z","shell.execute_reply":"2026-02-09T19:27:51.413269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_LONG_WAV = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/audio/train_001.wav\"\n\n# Load 2 minutes\ny2, sr2 = librosa.load(TEST_LONG_WAV, sr=16000, mono=True, offset=0.0, duration=120.0)\nchunks2 = _chunk_audio(y2, sr2, CHUNK_S, OVERLAP_S)\n\nkept_mask = [chunk_should_transcribe(c, sr2) for c in chunks2]\nkept_n = sum(kept_mask)\ntotal_n = len(chunks2)\n\nprint(f\"2-min chunks: total={total_n}, kept={kept_n}, skipped={total_n-kept_n}\")\n\n# Transcribe ONLY kept chunks (same logic as transcribe_long_wav, but for this 2-min clip)\nkept_chunks = [c for c, keep in zip(chunks2, kept_mask) if keep]\nif not kept_chunks:\n    print(\"All chunks skipped. Your thresholds are too aggressive.\")\nelse:\n    texts = []\n    param_dtype = next(model.parameters()).dtype\n\n    for i in range(0, len(kept_chunks), BATCH_CHUNKS):\n        batch = kept_chunks[i:i + BATCH_CHUNKS]\n        inputs = processor(batch, sampling_rate=sr2, return_tensors=\"pt\", padding=True)\n\n        feats = inputs.input_features.to(model.device)\n        if feats.dtype != param_dtype:\n            feats = feats.to(param_dtype)\n\n        with torch.inference_mode():\n            with torch.cuda.amp.autocast(enabled=torch.cuda.is_available(), dtype=torch.float16):\n                pred_ids = model.generate(\n                    feats,\n                    max_new_tokens=MAX_NEW_TOKENS,\n                    num_beams=NUM_BEAMS,\n                )\n\n        batch_text = processor.batch_decode(pred_ids, skip_special_tokens=True)\n        texts.extend([t.strip() for t in batch_text])\n\n    print(\"\\n--- 2-min transcript (VAD/silence-gated) ---\\n\")\n    print(\" \".join([t for t in texts if t]).strip())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:51.415789Z","iopub.execute_input":"2026-02-09T19:27:51.416065Z","iopub.status.idle":"2026-02-09T19:27:58.089254Z","shell.execute_reply.started":"2026-02-09T19:27:51.416045Z","shell.execute_reply":"2026-02-09T19:27:58.088526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"import glob\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport pandas as pd\nimport torch\n\n# Load sample submission (has: filename, transcript)\nSAMPLE_SUB_PATH = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/sample_submission .csv\".strip()\nsub = pd.read_csv(SAMPLE_SUB_PATH)\n\nassert \"filename\" in sub.columns and \"transcript\" in sub.columns, sub.columns.tolist()\nprint(\"sample rows:\", len(sub))\n\n# Map filename -> row index\nidx = {str(v): i for i, v in enumerate(sub[\"filename\"].astype(str).tolist())}\n\n# Collect test wavs\nTEST_AUDIO_DIR = f\"{ROOT}/transcription/transcription/test/audio\"\ntest_wavs = sorted(glob.glob(f\"{TEST_AUDIO_DIR}/*.wav\"))\nprint(\"test wavs:\", len(test_wavs), \"| ex:\", Path(test_wavs[0]).name if test_wavs else None)\n\nerrors = []\n\nfor wav_path in tqdm(test_wavs, desc=\"Transcribing test\"):\n    stem = Path(wav_path).stem  # e.g., test_001\n    try:\n        text = transcribe_long_wav(wav_path)  # <-- uses VAD/silence skip version\n    except torch.cuda.OutOfMemoryError:\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n        # brutal fallback\n        old_bs = BATCH_CHUNKS\n        BATCH_CHUNKS = 8\n        try:\n            text = transcribe_long_wav(wav_path)\n            print(text, \"\\n\")\n        finally:\n            BATCH_CHUNKS = old_bs\n    except Exception as e:\n        errors.append((stem, str(e)))\n        text = \"\"\n\n    if stem in idx:\n        sub.at[idx[stem], \"transcript\"] = text\n    else:\n        errors.append((stem, \"stem not found in sample_submission filename\"))\n\nOUT_PATH = \"/kaggle/working/submission.csv\"\nsub.to_csv(OUT_PATH, index=False, encoding=\"utf-8\")\nprint(\"Saved:\", OUT_PATH)\n\ndisplay(sub.head())\n\nif errors:\n    print(\"\\nErrors (first 10):\")\n    for e in errors[:10]:\n        print(e)\n    print(\"Total errors:\", len(errors))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:27:58.090287Z","iopub.execute_input":"2026-02-09T19:27:58.090692Z"}},"outputs":[],"execution_count":null}]}