{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":129276,"databundleVersionId":15506988,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"ce67677c","cell_type":"markdown","source":"# Bengali Long Form Audio ASR Starter\n\nA minimal starter for long-form Bengali Automatic Speech Recognition(ASR) using **`bengaliAI/tugstugi_bengaliai-asr_whisper-medium`**, enabling chunked transcription for lengthy audio files.\n","metadata":{}},{"id":"8dff81cf","cell_type":"markdown","source":"## 1) Install dependencies\n","metadata":{}},{"id":"a4752545","cell_type":"code","source":"!pip -q install -U transformers accelerate\n!pip -q install -U soundfile torchaudio\n!pip -q uninstall -y torchvision\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:28:18.838852Z","iopub.execute_input":"2026-02-10T14:28:18.839352Z","iopub.status.idle":"2026-02-10T14:30:59.005033Z","shell.execute_reply.started":"2026-02-10T14:28:18.839322Z","shell.execute_reply":"2026-02-10T14:30:59.004220Z"}},"outputs":[],"execution_count":null},{"id":"a88bc3d7","cell_type":"markdown","source":"## 2) Imports + device selection","metadata":{}},{"id":"37e2291c-a5ad-4e9f-ac6e-01fb55472114","cell_type":"markdown","source":"### 2.1) Imports","metadata":{}},{"id":"3be2dd54","cell_type":"code","source":"import os, glob, time\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\nimport torch\nimport soundfile as sf\nimport torchaudio\n\nfrom transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:04.702190Z","iopub.execute_input":"2026-02-10T14:32:04.702534Z","iopub.status.idle":"2026-02-10T14:32:20.280806Z","shell.execute_reply.started":"2026-02-10T14:32:04.702498Z","shell.execute_reply":"2026-02-10T14:32:20.280204Z"}},"outputs":[],"execution_count":null},{"id":"0ddb5eb6-96fa-475a-9b05-6444b8f7a09b","cell_type":"markdown","source":"### 2.2) Device Selection","metadata":{}},{"id":"498f2bea-6fdf-4cfb-9252-b4869df74189","cell_type":"code","source":"print(\"torch:\", torch.__version__)\nprint(\"cuda available:\", torch.cuda.is_available())\n\nUSE_CUDA = torch.cuda.is_available()\nDEVICE = \"cuda:0\" if USE_CUDA else \"cpu\"\nDTYPE = torch.float16 if USE_CUDA else torch.float32\nprint(\"DEVICE:\", DEVICE, \"DTYPE:\", DTYPE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:20.282064Z","iopub.execute_input":"2026-02-10T14:32:20.282556Z","iopub.status.idle":"2026-02-10T14:32:20.287128Z","shell.execute_reply.started":"2026-02-10T14:32:20.282530Z","shell.execute_reply":"2026-02-10T14:32:20.286599Z"}},"outputs":[],"execution_count":null},{"id":"39c08466","cell_type":"markdown","source":"## 3) Configuration (model, chunking, paths)","metadata":{}},{"id":"035ab56e","cell_type":"code","source":"MODEL_ID = os.environ.get(\"MODEL_ID\", \"bengaliAI/tugstugi_bengaliai-asr_whisper-medium\")\n\n# HuggingFace pipeline chunking (seconds)\nCHUNK_LENGTH_S = float(os.environ.get(\"CHUNK_LENGTH_S\", 30.0))\nSTRIDE_LENGTH_S = float(os.environ.get(\"STRIDE_LENGTH_S\", 5.0))\n\n# Requested glob\nAUDIO_GLOB = \"/kaggle/input/*/test/*/*.wav\"\n# More robust fallback (recursive)\nAUDIO_GLOB_RECURSIVE = \"/kaggle/input/**/test/**/*.wav\"\n\nOUT_DIR = Path(\"/kaggle/working\")\nOUT_DIR.mkdir(parents=True, exist_ok=True)\n\nSUBMISSION_PATH = OUT_DIR / \"submission.csv\"\nCACHE_PATH = OUT_DIR / \"transcripts_cache.csv\"\n\nTARGET_SR = 16000\n\nprint(\"MODEL_ID:\", MODEL_ID)\nprint(\"CHUNK_LENGTH_S:\", CHUNK_LENGTH_S, \"STRIDE_LENGTH_S:\", STRIDE_LENGTH_S)\nprint(\"AUDIO_GLOB:\", AUDIO_GLOB)\nprint(\"AUDIO_GLOB_RECURSIVE:\", AUDIO_GLOB_RECURSIVE)\nprint(\"SUBMISSION_PATH:\", SUBMISSION_PATH)\nprint(\"CACHE_PATH:\", CACHE_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:20.287881Z","iopub.execute_input":"2026-02-10T14:32:20.288092Z","iopub.status.idle":"2026-02-10T14:32:20.303248Z","shell.execute_reply.started":"2026-02-10T14:32:20.288070Z","shell.execute_reply":"2026-02-10T14:32:20.302604Z"}},"outputs":[],"execution_count":null},{"id":"fa2a1f32","cell_type":"markdown","source":"## 4) Load audio, convert to mono, and resample to 16 kHz\nReads a `.wav` file, averages stereo channels to mono, casts to `float32`, and resamples to the target sample rate for consistent ASR input.\n","metadata":{}},{"id":"e796b105","cell_type":"code","source":"def load_audio_mono_resample(path: str, target_sr: int = 16000) -> tuple[np.ndarray, int]:\n    audio, sr = sf.read(path, always_2d=False)\n\n    # stereo -> mono\n    if getattr(audio, \"ndim\", 1) == 2:\n        audio = audio.mean(axis=1)\n\n    audio = audio.astype(np.float32, copy=False)\n\n    # resample if needed\n    if sr != target_sr:\n        wav = torch.from_numpy(audio).unsqueeze(0)  # [1, T]\n        wav = torchaudio.functional.resample(wav, orig_freq=sr, new_freq=target_sr)\n        audio = wav.squeeze(0).cpu().numpy()\n        sr = target_sr\n\n    return audio, sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:20.304455Z","iopub.execute_input":"2026-02-10T14:32:20.304722Z","iopub.status.idle":"2026-02-10T14:32:20.317512Z","shell.execute_reply.started":"2026-02-10T14:32:20.304687Z","shell.execute_reply":"2026-02-10T14:32:20.316879Z"}},"outputs":[],"execution_count":null},{"id":"adc3cb4a","cell_type":"markdown","source":"## 5) Discover test `.wav` files and derive file IDs\nCollects audio paths using the primary glob (with a recursive fallback), prints a quick sanity check, and creates `file_ids` from each filename stem (no extension) for submission mapping.\n","metadata":{}},{"id":"90411894","cell_type":"code","source":"wav_paths = sorted(glob.glob(AUDIO_GLOB))\nif not wav_paths:\n    # fallback: recursive glob catches more dataset layouts\n    wav_paths = sorted(glob.glob(AUDIO_GLOB_RECURSIVE, recursive=True))\n\nprint(\"Found wav files:\", len(wav_paths))\nprint(\"Example:\", wav_paths[0] if wav_paths else \"None\")\n\ndef stem_no_ext(p: str) -> str:\n    return Path(p).stem\n\nfile_ids = [stem_no_ext(p) for p in wav_paths]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:21.230423Z","iopub.execute_input":"2026-02-10T14:32:21.230982Z","iopub.status.idle":"2026-02-10T14:32:21.619631Z","shell.execute_reply.started":"2026-02-10T14:32:21.230952Z","shell.execute_reply":"2026-02-10T14:32:21.619065Z"}},"outputs":[],"execution_count":null},{"id":"4898f8f1","cell_type":"markdown","source":"## 6) Load the model + processor, build ASR pipeline ","metadata":{}},{"id":"881425da","cell_type":"code","source":"# Load model and processor directly (stable generation config)\nmodel = AutoModelForSpeechSeq2Seq.from_pretrained(\n    MODEL_ID,\n    torch_dtype=DTYPE,\n    low_cpu_mem_usage=True,\n)\nmodel.to(DEVICE)\n\nprocessor = AutoProcessor.from_pretrained(MODEL_ID)\n\nasr = pipeline(\n    task=\"automatic-speech-recognition\",\n    model=model,\n    tokenizer=processor.tokenizer,\n    feature_extractor=processor.feature_extractor,\n    device=0 if DEVICE.startswith(\"cuda\") else -1,  # pipeline expects int device\n    torch_dtype=DTYPE,\n)\n\nprint(\"Pipeline ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-10T14:32:25.810267Z","iopub.execute_input":"2026-02-10T14:32:25.810571Z","iopub.status.idle":"2026-02-10T14:32:42.394184Z","shell.execute_reply.started":"2026-02-10T14:32:25.810543Z","shell.execute_reply":"2026-02-10T14:32:42.393199Z"}},"outputs":[],"execution_count":null},{"id":"f2cf0859","cell_type":"markdown","source":"## 7) Load cached transcripts (resume support)\nIf `transcripts_cache.csv` exists, load it into a `done` lookup to skip already-processed files; otherwise start fresh.\n","metadata":{}},{"id":"0155bf1a","cell_type":"code","source":"done = {}\nif CACHE_PATH.exists():\n    cache_df = pd.read_csv(CACHE_PATH)\n    done = dict(zip(cache_df[\"filename\"].astype(str), cache_df[\"transcript\"].astype(str)))\n    print(\"Loaded cached transcripts:\", len(done))\nelse:\n    print(\"No cache found (fresh run).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-01T19:11:40.087126Z","iopub.execute_input":"2026-02-01T19:11:40.087391Z","iopub.status.idle":"2026-02-01T19:11:40.091908Z","shell.execute_reply.started":"2026-02-01T19:11:40.087366Z","shell.execute_reply":"2026-02-01T19:11:40.091223Z"}},"outputs":[],"execution_count":null},{"id":"a818da78","cell_type":"markdown","source":"## 8) Transcribe one audio file (chunked long-form ASR)\nLoads and resamples the audio, then runs the ASR pipeline with overlapping chunks (`chunk_length_s` + `stride_length_s`) and returns the final cleaned transcript text.\n","metadata":{}},{"id":"17790907","cell_type":"code","source":"def transcribe_file(path: str) -> str:\n    audio, sr = load_audio_mono_resample(path, target_sr=TARGET_SR)\n\n    result = asr(\n        {\"array\": audio, \"sampling_rate\": sr},\n        chunk_length_s=CHUNK_LENGTH_S,\n        stride_length_s=STRIDE_LENGTH_S,\n        return_timestamps=False,\n    )\n\n    return (result.get(\"text\") or \"\").strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-01T19:11:40.092942Z","iopub.execute_input":"2026-02-01T19:11:40.093233Z","iopub.status.idle":"2026-02-01T19:11:40.106840Z","shell.execute_reply.started":"2026-02-01T19:11:40.093201Z","shell.execute_reply":"2026-02-01T19:11:40.106058Z"}},"outputs":[],"execution_count":null},{"id":"db264e93","cell_type":"markdown","source":"## 9) Run transcription over all files (with progress + periodic checkpointing)\nIterates through all test audios, skips items already in cache, transcribes the rest with error handling, and saves progress to `transcripts_cache.csv` every 25 files (and once at the end).\n","metadata":{}},{"id":"6d9cb2cc","cell_type":"code","source":"rows = []\nstart_time = time.time()\n\nfor wav_path, fid in tqdm(list(zip(wav_paths, file_ids)), total=len(wav_paths)):\n    if fid in done:\n        rows.append({\"filename\": fid, \"transcript\": done[fid]})\n        continue\n\n    try:\n        txt = transcribe_file(wav_path)\n    except Exception as e:\n        print(f\"[WARN] Failed on {wav_path}: {e}\")\n        txt = \"\"\n\n    rows.append({\"filename\": fid, \"transcript\": txt})\n\n    # Save incremental cache every 25 items\n    if len(rows) % 25 == 0:\n        pd.DataFrame(rows).to_csv(CACHE_PATH, index=False)\n\nelapsed = time.time() - start_time\nprint(f\"Done. Processed {len(rows)} files in {elapsed/60:.1f} min\")\n\ndf = pd.DataFrame(rows)\ndf.to_csv(CACHE_PATH, index=False)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-01T19:11:40.107851Z","iopub.execute_input":"2026-02-01T19:11:40.108141Z","execution_failed":"2026-02-01T19:17:48.032Z"}},"outputs":[],"execution_count":null},{"id":"391d8e56","cell_type":"markdown","source":"## 10) Write `submission.csv`","metadata":{}},{"id":"e154364d","cell_type":"code","source":"submission_df = df[[\"filename\", \"transcript\"]].copy()\nsubmission_df.to_csv(SUBMISSION_PATH, index=False)\n\nprint(\"Wrote:\", SUBMISSION_PATH)\nprint(\"Rows:\", len(submission_df))\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"execution_failed":"2026-02-01T19:17:48.032Z"}},"outputs":[],"execution_count":null}]}