{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":129276,"databundleVersionId":15506988,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# HF 404 Error and Version conflict resolve","metadata":{}},{"cell_type":"code","source":"!pip -q install -U transformers peft","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:05:35.917789Z","iopub.execute_input":"2026-02-10T00:05:35.918077Z","iopub.status.idle":"2026-02-10T00:05:51.211238Z","shell.execute_reply.started":"2026-02-10T00:05:35.918042Z","shell.execute_reply":"2026-02-10T00:05:51.210403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setup + build full file list (skip train_089) + fixed dev split","metadata":{}},{"cell_type":"code","source":"import os, glob, re\nfrom pathlib import Path\nimport pandas as pd\nimport soundfile as sf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:05:56.789097Z","iopub.execute_input":"2026-02-10T00:05:56.789727Z","iopub.status.idle":"2026-02-10T00:05:57.089813Z","shell.execute_reply.started":"2026-02-10T00:05:56.789693Z","shell.execute_reply":"2026-02-10T00:05:57.089270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition\"\nTRAIN_AUDIO_DIR = f\"{ROOT}/transcription/transcription/train/audio\"\nTRAIN_ANNO_DIR  = f\"{ROOT}/transcription/transcription/train/annotation\"\n\nfor p in [TRAIN_AUDIO_DIR, TRAIN_ANNO_DIR]:\n    print((\"OK  \" if os.path.exists(p) else \"MISS\"), p)\n\nBAD_IDS = {\"train_089\"}  # corrupted\n\n# fixed dev set (keep it stable across all training)\nDEV_IDS = {\"train_120\", \"train_121\"}\n\ndef normalize_bn(text: str) -> str:\n    t = str(text) if text is not None else \"\"\n    t = re.sub(r\"[\\u200B-\\u200D\\uFEFF]\", \"\", t)  # zero-width\n    t = re.sub(r\"\\s+\", \" \", t).strip()\n    t = t.replace(\" ।\", \"।\")\n    t = re.sub(r\"\\s+([।,?.!])\", r\"\\1\", t)\n    t = re.sub(r\"([।,?.!])([^\\s])\", r\"\\1 \\2\", t)\n    return t.strip()\n\ntrain_wavs = sorted(glob.glob(f\"{TRAIN_AUDIO_DIR}/*.wav\"))\ntrain_txts = sorted(glob.glob(f\"{TRAIN_ANNO_DIR}/*.txt\"))\ntxt_map = {Path(p).stem: p for p in train_txts}\n\nrows = []\nfor wav in train_wavs:\n    sid = Path(wav).stem\n    if sid in BAD_IDS:\n        continue\n    txtp = txt_map.get(sid)\n    if not txtp:\n        continue\n    raw = Path(txtp).read_text(encoding=\"utf-8\", errors=\"ignore\")\n    rows.append({\n        \"id\": sid,\n        \"audio\": wav,\n        \"text\": normalize_bn(raw),\n        \"duration\": float(sf.info(wav).duration),\n    })\n\ndf = pd.DataFrame(rows).sort_values(\"id\").reset_index(drop=True)\nprint(\"paired kept:\", len(df))\nprint(\"contains train_089?\", (\"train_089\" in set(df[\"id\"])))\n\ndev_df = df[df[\"id\"].isin(DEV_IDS)].reset_index(drop=True)\ntrain_df = df[~df[\"id\"].isin(DEV_IDS)].reset_index(drop=True)\n\nprint(\"train_df:\", len(train_df), \"| dev_df:\", len(dev_df))\nprint(\"dev ids:\", dev_df[\"id\"].tolist())\ndisplay(df.head(2))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:06:02.471314Z","iopub.execute_input":"2026-02-10T00:06:02.471788Z","iopub.status.idle":"2026-02-10T00:06:04.298875Z","shell.execute_reply.started":"2026-02-10T00:06:02.471765Z","shell.execute_reply":"2026-02-10T00:06:04.298306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# VAD to timestamps JSONL\n\nOutputs:\n\n```/kaggle/working/prep_ts/train_vad_timestamps.jsonl```\n```/kaggle/working/prep_ts/dev_vad_timestamps.jsonl```","metadata":{}},{"cell_type":"code","source":"!pip -q install webrtcvad scipy\nimport json, math, shutil\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport soundfile as sf\nimport webrtcvad\nfrom scipy.signal import resample_poly\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:08:29.298350Z","iopub.execute_input":"2026-02-10T00:08:29.298674Z","iopub.status.idle":"2026-02-10T00:08:37.756826Z","shell.execute_reply.started":"2026-02-10T00:08:29.298648Z","shell.execute_reply":"2026-02-10T00:08:37.756140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"OUT_ROOT = Path(\"/kaggle/working/prep_ts\")\nif OUT_ROOT.exists():\n    shutil.rmtree(OUT_ROOT)\nOUT_ROOT.mkdir(parents=True, exist_ok=True)\n\nTRAIN_TS = OUT_ROOT / \"train_vad_timestamps.jsonl\"\nDEV_TS   = OUT_ROOT / \"dev_vad_timestamps.jsonl\"\n\n# VAD params (good defaults)\nVAD_SR = 16000\nVAD_MODE = 2\nFRAME_MS = 30\nPAD_MS = 200\nMERGE_GAP_MS = 300\nMIN_SEG_MS = 800\nMAX_SEG_S = 25\n\nvad = webrtcvad.Vad(VAD_MODE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:08:40.880479Z","iopub.execute_input":"2026-02-10T00:08:40.881521Z","iopub.status.idle":"2026-02-10T00:08:40.886624Z","shell.execute_reply.started":"2026-02-10T00:08:40.881477Z","shell.execute_reply":"2026-02-10T00:08:40.885890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_mono(x):\n    return x if x.ndim == 1 else x.mean(axis=1)\n\ndef resample_to(x, sr, target_sr):\n    if sr == target_sr:\n        return x.astype(np.float32), sr\n    g = math.gcd(sr, target_sr)\n    x = resample_poly(x, target_sr // g, sr // g).astype(np.float32)\n    return x, target_sr\n\ndef float_to_pcm16(x: np.ndarray) -> np.ndarray:\n    x = np.clip(x, -1.0, 1.0)\n    return (x * 32767.0).astype(np.int16)\n\ndef vad_intervals_seconds(x16: np.ndarray, sr=16000):\n    frame_len = int(sr * FRAME_MS / 1000)\n    if len(x16) < frame_len:\n        return []\n\n    pcm16 = float_to_pcm16(x16)\n    voiced = []\n    for i in range(0, len(pcm16) - frame_len + 1, frame_len):\n        voiced.append(vad.is_speech(pcm16[i:i+frame_len].tobytes(), sr))\n\n    # flags -> sample intervals\n    intervals = []\n    in_seg = False\n    seg_start = 0\n    for idx, v in enumerate(voiced):\n        if v and not in_seg:\n            in_seg = True\n            seg_start = idx * frame_len\n        elif (not v) and in_seg:\n            in_seg = False\n            intervals.append((seg_start, idx * frame_len))\n    if in_seg:\n        intervals.append((seg_start, len(x16)))\n\n    # pad + merge + filter\n    pad = int(sr * PAD_MS / 1000)\n    merge_gap = int(sr * MERGE_GAP_MS / 1000)\n    min_len = int(sr * MIN_SEG_MS / 1000)\n\n    padded = []\n    for s, e in intervals:\n        s2 = max(0, s - pad)\n        e2 = min(len(x16), e + pad)\n        if e2 - s2 >= min_len:\n            padded.append((s2, e2))\n    if not padded:\n        return []\n\n    merged = [padded[0]]\n    for s, e in padded[1:]:\n        ps, pe = merged[-1]\n        if s - pe <= merge_gap:\n            merged[-1] = (ps, max(pe, e))\n        else:\n            merged.append((s, e))\n\n    # split long segments\n    max_len = int(sr * MAX_SEG_S)\n    final = []\n    for s, e in merged:\n        cur = s\n        while cur < e:\n            ee = min(e, cur + max_len)\n            if ee - cur >= min_len:\n                final.append((cur, ee))\n            cur = ee\n\n    return [(round(s/sr, 3), round(e/sr, 3)) for s, e in final]\n\ndef build_ts_manifest(df_in, out_path: Path, split_name: str):\n    nrows = 0\n    with open(out_path, \"w\", encoding=\"utf-8\") as f:\n        for r in tqdm(df_in.to_dict(\"records\"), desc=f\"VAD timestamps -> {split_name}\"):\n            pid = r[\"id\"]\n            wav = r[\"audio\"]\n\n            x, sr = sf.read(wav, dtype=\"float32\", always_2d=False)\n            x = to_mono(x)\n            x16, _ = resample_to(x, sr, VAD_SR)\n\n            segs = vad_intervals_seconds(x16, sr=VAD_SR)\n            for k, (s_s, e_s) in enumerate(segs):\n                item = {\n                    \"parent_id\": pid,\n                    \"seg_id\": f\"{pid}_{k:04d}\",\n                    \"audio\": wav,  # original file path\n                    \"start_s\": s_s,\n                    \"end_s\": e_s,\n                    \"duration_s\": round(e_s - s_s, 3),\n                }\n                f.write(json.dumps(item, ensure_ascii=False) + \"\\n\")\n                nrows += 1\n    return nrows\n\nn_train = build_ts_manifest(train_df, TRAIN_TS, \"train\")\nn_dev   = build_ts_manifest(dev_df,   DEV_TS,   \"dev\")\n\nprint(\"Saved:\", TRAIN_TS, \"| rows:\", n_train)\nprint(\"Saved:\", DEV_TS,   \"| rows:\", n_dev)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:08:44.333172Z","iopub.execute_input":"2026-02-10T00:08:44.333691Z","iopub.status.idle":"2026-02-10T00:12:27.879387Z","shell.execute_reply.started":"2026-02-10T00:08:44.333666Z","shell.execute_reply":"2026-02-10T00:12:27.878472Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Regex","metadata":{}},{"cell_type":"code","source":"import re\n\n_bn_re = re.compile(r\"[ঁ-৿]+\")  # Bengali block (rough but works)\n\ndef keep_pseudolabel(text: str, dur_s: float) -> bool:\n    t = (text or \"\").strip()\n    if len(t) == 0:\n        return False\n\n    bn_chars = _bn_re.findall(t)\n    bn_len = sum(len(x) for x in bn_chars)\n\n    # if it's mostly not Bengali, skip\n    if bn_len < 0.6 * max(1, len(re.sub(r\"\\s+\", \"\", t))):\n        return False\n\n    # long segment but almost nothing said\n    if dur_s >= 2.0 and bn_len < 4:\n        return False\n\n    # nonsense speed (tune if needed)\n    cps = bn_len / max(dur_s, 1e-3)\n    if cps > 25 or cps < 0.2:\n        return False\n\n    # repetition detector (very crude but catches loops)\n    tokens = t.split()\n    if len(tokens) >= 8 and (len(set(tokens)) / len(tokens)) < 0.25:\n        return False\n\n    return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:13:10.961528Z","iopub.execute_input":"2026-02-10T00:13:10.962355Z","iopub.status.idle":"2026-02-10T00:13:10.968295Z","shell.execute_reply.started":"2026-02-10T00:13:10.962324Z","shell.execute_reply":"2026-02-10T00:13:10.967631Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pseudo-label ONCE (fixed teacher) to labeled JSONL\nOutputs:\n\n```/kaggle/working/prep_ts/train_labeled_manifest.jsonl```\n```/kaggle/working/prep_ts/dev_labeled_manifest.jsonl```","metadata":{}},{"cell_type":"code","source":"import json, math\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport soundfile as sf\nimport torch\nfrom scipy.signal import resample_poly\nfrom transformers import AutoModelForSpeechSeq2Seq, AutoProcessor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:12:27.880596Z","iopub.execute_input":"2026-02-10T00:12:27.880834Z","iopub.status.idle":"2026-02-10T00:12:44.112342Z","shell.execute_reply.started":"2026-02-10T00:12:27.880811Z","shell.execute_reply":"2026-02-10T00:12:44.111588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fixed teacher model for pseudo labels\n#TEACHER_ID = \"mozilla-ai/whisper-large-v3-turbo-bn\"\nTEACHER_ID = \"bengaliAI/tugstugi_bengaliai-asr_whisper-medium\"\n\nOUT_ROOT = Path(\"/kaggle/working/prep_ts\")\nTRAIN_TS = OUT_ROOT / \"train_vad_timestamps.jsonl\"\nDEV_TS   = OUT_ROOT / \"dev_vad_timestamps.jsonl\"\n\nTRAIN_LBL = OUT_ROOT / \"train_labeled_manifest.jsonl\"\nDEV_LBL   = OUT_ROOT / \"dev_labeled_manifest.jsonl\"\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\ndtype  = torch.float16 if torch.cuda.is_available() else torch.float32\nprint(\"device:\", device, \"| dtype:\", dtype)\n\nprocessor = AutoProcessor.from_pretrained(TEACHER_ID)\nmodel = AutoModelForSpeechSeq2Seq.from_pretrained(\n    TEACHER_ID,\n    torch_dtype=dtype,\n    low_cpu_mem_usage=True,\n    use_safetensors=True,\n).to(device).eval()\n\n# keep generation sane and avoid language-arg issues\nmodel.generation_config.max_length = None\ntry:\n    processor.tokenizer.set_prefix_tokens(language=\"bn\", task=\"transcribe\")\nexcept Exception:\n    pass\n\nTARGET_SR = 16000\nN_SAMPLES = 30 * TARGET_SR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:15:59.494397Z","iopub.execute_input":"2026-02-10T00:15:59.495129Z","iopub.status.idle":"2026-02-10T00:16:16.659944Z","shell.execute_reply.started":"2026-02-10T00:15:59.495089Z","shell.execute_reply":"2026-02-10T00:16:16.659154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_mono(x):\n    return x if x.ndim == 1 else x.mean(axis=1)\n\ndef resample_to(x, sr, target_sr):\n    if sr == target_sr:\n        return x.astype(np.float32), sr\n    g = math.gcd(sr, target_sr)\n    x = resample_poly(x, target_sr // g, sr // g).astype(np.float32)\n    return x, target_sr\n\ndef load_slice(path, start_s, end_s):\n    info = sf.info(path)\n    sr = info.samplerate\n    start = int(round(float(start_s) * sr))\n    frames = int(round((float(end_s) - float(start_s)) * sr))\n    x, sr = sf.read(path, start=start, frames=frames, dtype=\"float32\", always_2d=False)\n    x = to_mono(x)\n    x, sr = resample_to(x, sr, TARGET_SR)\n\n    # pad/trim to 30s (stable)\n    if len(x) >= N_SAMPLES:\n        x = x[:N_SAMPLES]\n    else:\n        x = np.pad(x, (0, N_SAMPLES - len(x)), mode=\"constant\")\n    return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:17:19.380601Z","iopub.execute_input":"2026-02-10T00:17:19.380910Z","iopub.status.idle":"2026-02-10T00:17:19.387603Z","shell.execute_reply.started":"2026-02-10T00:17:19.380881Z","shell.execute_reply":"2026-02-10T00:17:19.386959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH = 16\nMAX_NEW_TOKENS = 128\nparam_dtype = next(model.parameters()).dtype\n\ndef pseudo_label_jsonl(in_jsonl: Path, out_jsonl: Path):\n    out_f = open(out_jsonl, \"w\", encoding=\"utf-8\")\n    buf = []\n\n    def flush(batch_items):\n        audios = [load_slice(r[\"audio\"], r[\"start_s\"], r[\"end_s\"]) for r in batch_items]\n        inp = processor(audios, sampling_rate=TARGET_SR, return_tensors=\"pt\")\n        feats = inp.input_features.to(device)\n        if feats.dtype != param_dtype:\n            feats = feats.to(param_dtype)\n\n        with torch.inference_mode():\n            with torch.cuda.amp.autocast(enabled=torch.cuda.is_available(), dtype=torch.float16):\n                pred = model.generate(feats, max_new_tokens=MAX_NEW_TOKENS, num_beams=1)\n\n        texts = processor.batch_decode(pred, skip_special_tokens=True)\n        for r, t in zip(batch_items, texts):\n            rr = dict(r)\n            rr[\"text\"] = t.strip()\n            out_f.write(json.dumps(rr, ensure_ascii=False) + \"\\n\")\n            if keep_pseudolabel(rr[\"text\"], rr[\"duration_s\"]):\n                out_f.write(json.dumps(rr, ensure_ascii=False) + \"\\n\")\n\n    lines = in_jsonl.read_text(encoding=\"utf-8\").splitlines()\n    for line in tqdm(lines, desc=f\"Pseudo-label -> {out_jsonl.name}\"):\n        if not line.strip():\n            continue\n        buf.append(json.loads(line))\n        if len(buf) >= BATCH:\n            flush(buf)\n            buf = []\n    if buf:\n        flush(buf)\n\n    out_f.close()\n    print(\"Saved:\", out_jsonl)\n\npseudo_label_jsonl(TRAIN_TS, TRAIN_LBL)\npseudo_label_jsonl(DEV_TS, DEV_LBL)\n\nprint(\"DONE. Outputs are in:\", OUT_ROOT)\nprint(\"Files:\", [p.name for p in OUT_ROOT.glob(\"*.jsonl\")])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T00:17:26.581680Z","iopub.execute_input":"2026-02-10T00:17:26.582523Z"}},"outputs":[],"execution_count":null}]}