{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install -U openai-whisper jiwer pandas numpy tqdm librosa soundfile scikit-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T15:43:29.385491Z","iopub.execute_input":"2026-06-14T15:43:29.385772Z","iopub.status.idle":"2026-06-14T15:44:05.336225Z","shell.execute_reply.started":"2026-06-14T15:43:29.385734Z","shell.execute_reply":"2026-06-14T15:44:05.335549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nfrom pathlib import Path\n\nINPUT_ROOT = Path(\"/kaggle/input\")\n\nprint(\"Available input folders:\")\nfor p in INPUT_ROOT.iterdir():\n    print(\"-\", p)\n\nben10_candidates = [\n    p for p in INPUT_ROOT.iterdir()\n    if \"ben10\" in p.name.lower() or \"bhasha\" in p.name.lower() or \"dialect\" in p.name.lower()\n]\n\nBEN10_ROOT = ben10_candidates[0] if ben10_candidates else list(INPUT_ROOT.iterdir())[0]\nprint(\"\\nUsing dataset root:\", BEN10_ROOT)\n\nprint(\"\\nTop-level files/folders:\")\nfor p in BEN10_ROOT.iterdir():\n    print(\"-\", p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T15:57:55.293185Z","iopub.execute_input":"2026-06-14T15:57:55.293598Z","iopub.status.idle":"2026-06-14T15:57:55.301721Z","shell.execute_reply.started":"2026-06-14T15:57:55.293561Z","shell.execute_reply":"2026-06-14T15:57:55.301045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\n\naudio_exts = [\".wav\", \".mp3\", \".flac\", \".ogg\", \".m4a\"]\naudio_files = []\n\nfor ext in audio_exts:\n    audio_files.extend(list(BEN10_ROOT.rglob(f\"*{ext}\")))\n\ntable_files = []\nfor ext in [\".csv\", \".tsv\", \".txt\"]:\n    table_files.extend(list(BEN10_ROOT.rglob(f\"*{ext}\")))\n\nprint(\"Total audio files:\", len(audio_files))\nprint(\"Total table/text files:\", len(table_files))\n\nprint(\"\\nAudio sample:\")\nfor p in audio_files[:10]:\n    print(p)\n\nprint(\"\\nTable files:\")\nfor p in table_files:\n    print(p)\n\ntables = {}\n\nfor p in table_files:\n    try:\n        if p.suffix.lower() == \".tsv\":\n            df_tmp = pd.read_csv(p, sep=\"\\t\")\n        else:\n            df_tmp = pd.read_csv(p)\n        tables[str(p)] = df_tmp\n        print(\"\\nFILE:\", p)\n        print(\"Shape:\", df_tmp.shape)\n        print(\"Columns:\", list(df_tmp.columns))\n        display(df_tmp.head())\n    except Exception as e:\n        print(\"Could not read:\", p, e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T15:58:30.665153Z","iopub.execute_input":"2026-06-14T15:58:30.666132Z","iopub.status.idle":"2026-06-14T15:59:14.951898Z","shell.execute_reply.started":"2026-06-14T15:58:30.666099Z","shell.execute_reply":"2026-06-14T15:59:14.951262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport unicodedata\nimport pandas as pd\nfrom pathlib import Path\n\nBEN10_ROOT = Path(\"/kaggle/input/competitions/ben10\")\nTRAIN_DIR = BEN10_ROOT / \"ben10\" / \"16_kHz_train_audio\"\nTRAIN_CSV = TRAIN_DIR / \"train.csv\"\n\ndef normalize_bangla_text(text):\n    text = str(text)\n    text = unicodedata.normalize(\"NFC\", text)\n\n    # remove special tags like <>\n    text = re.sub(r\"<[^>]*>\", \" \", text)\n\n    # keep Bangla chars and space only\n    text = re.sub(r\"[^\\u0980-\\u09FF\\s]\", \" \", text)\n    text = re.sub(r\"\\s+\", \" \", text).strip()\n    return text\n\ndf = pd.read_csv(TRAIN_CSV)\n\nprint(df.shape)\nprint(df.columns)\ndisplay(df.head())\n\ndf[\"audio_path\"] = df[\"file_name\"].apply(lambda x: str(TRAIN_DIR / x))\ndf[\"transcript\"] = df[\"transcriptions\"].apply(normalize_bangla_text)\ndf[\"dialect\"] = df[\"district\"].astype(str).str.lower().str.strip()\ndf[\"dataset\"] = \"ben10\"\n\n# keep only valid rows\ndf = df[df[\"audio_path\"].apply(lambda x: Path(x).exists())]\ndf = df[df[\"transcript\"].str.len() > 0]\n\nmanifest = df[[\"audio_path\", \"transcript\", \"dialect\", \"dataset\"]].copy()\nmanifest.to_csv(\"/kaggle/working/ben10_manifest_raw.csv\", index=False)\n\nprint(\"Clean manifest size:\", manifest.shape)\nprint(manifest[\"dialect\"].value_counts())\ndisplay(manifest.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T16:08:59.038739Z","iopub.execute_input":"2026-06-14T16:08:59.039803Z","iopub.status.idle":"2026-06-14T16:09:18.259178Z","shell.execute_reply.started":"2026-06-14T16:08:59.039761Z","shell.execute_reply":"2026-06-14T16:09:18.258284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport pandas as pd\n\nmanifest = pd.read_csv(\"/kaggle/working/ben10_manifest_raw.csv\")\n\ntrain_parts = []\ndev_parts = []\ntest_parts = []\n\nfor dialect, g in manifest.groupby(\"dialect\"):\n    train_g, temp_g = train_test_split(\n        g,\n        test_size=0.20,\n        random_state=42,\n        shuffle=True\n    )\n\n    dev_g, test_g = train_test_split(\n        temp_g,\n        test_size=0.50,\n        random_state=42,\n        shuffle=True\n    )\n\n    train_parts.append(train_g)\n    dev_parts.append(dev_g)\n    test_parts.append(test_g)\n\ntrain_df = pd.concat(train_parts).reset_index(drop=True)\ndev_df = pd.concat(dev_parts).reset_index(drop=True)\ntest_df = pd.concat(test_parts).reset_index(drop=True)\n\ntrain_df[\"split\"] = \"train\"\ndev_df[\"split\"] = \"dev\"\ntest_df[\"split\"] = \"test\"\n\nfinal_manifest = pd.concat([train_df, dev_df, test_df]).reset_index(drop=True)\nfinal_manifest.to_csv(\"/kaggle/working/ben10_manifest.csv\", index=False)\n\nprint(final_manifest[\"split\"].value_counts())\nprint(final_manifest.groupby([\"split\", \"dialect\"]).size())\ndisplay(final_manifest.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T16:09:38.464352Z","iopub.execute_input":"2026-06-14T16:09:38.465263Z","iopub.status.idle":"2026-06-14T16:09:38.860606Z","shell.execute_reply.started":"2026-06-14T16:09:38.465223Z","shell.execute_reply":"2026-06-14T16:09:38.859955Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Whisper zero-shot dialect audit","metadata":{}},{"cell_type":"code","source":"!pip -q install -U openai-whisper jiwer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T16:10:15.654006Z","iopub.execute_input":"2026-06-14T16:10:15.654582Z","iopub.status.idle":"2026-06-14T16:10:19.149147Z","shell.execute_reply.started":"2026-06-14T16:10:15.654550Z","shell.execute_reply":"2026-06-14T16:10:19.148168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import whisper\nimport torch\nimport pandas as pd\nfrom tqdm import tqdm\n\nmodel_name = \"small\"\nmodel = whisper.load_model(model_name)\n\ndf = pd.read_csv(\"/kaggle/working/ben10_manifest.csv\")\n\nprint(\"Columns:\", df.columns.tolist())\nprint(df[\"split\"].value_counts())\nprint(df[\"dialect\"].value_counts())\n\neval_df = df[df[\"split\"] == \"test\"].copy()\n\n# safe sampling: প্রতি dialect থেকে 10 sample\nsample_parts = []\n\nfor dialect_name in sorted(eval_df[\"dialect\"].unique()):\n    g = eval_df[eval_df[\"dialect\"] == dialect_name].copy()\n    n = min(len(g), 10)\n    sample_parts.append(g.sample(n=n, random_state=42))\n\neval_df = pd.concat(sample_parts, ignore_index=True)\n\nprint(\"Eval size:\", len(eval_df))\nprint(eval_df[\"dialect\"].value_counts())\ndisplay(eval_df.head())\n\npred_rows = []\n\nfor _, row in tqdm(eval_df.iterrows(), total=len(eval_df)):\n    audio_path = row[\"audio_path\"]\n    ref = normalize_bangla_text(row[\"transcript\"])\n\n    try:\n        out = model.transcribe(\n            audio_path,\n            language=\"bn\",\n            task=\"transcribe\",\n            fp16=torch.cuda.is_available()\n        )\n        pred = normalize_bangla_text(out[\"text\"])\n    except Exception as e:\n        print(\"Error:\", audio_path, e)\n        pred = \"\"\n\n    pred_rows.append({\n        \"audio_path\": audio_path,\n        \"dialect\": row[\"dialect\"],\n        \"reference\": ref,\n        \"prediction\": pred\n    })\n\npred_df = pd.DataFrame(pred_rows)\npred_df.to_csv(\"/kaggle/working/exp1_whisper_small_predictions_quick.csv\", index=False)\n\nprint(pred_df.columns.tolist())\ndisplay(pred_df.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T16:15:14.451325Z","iopub.execute_input":"2026-06-14T16:15:14.452236Z","iopub.status.idle":"2026-06-14T16:46:24.483753Z","shell.execute_reply.started":"2026-06-14T16:15:14.452206Z","shell.execute_reply":"2026-06-14T16:46:24.483079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Per-dialect WER/CER + fairness gap","metadata":{}},{"cell_type":"code","source":"from jiwer import wer, cer\nimport pandas as pd\n\ndef compute_dialect_metrics(pred_df, model_name):\n    rows = []\n\n    for dialect_name in sorted(pred_df[\"dialect\"].unique()):\n        g = pred_df[pred_df[\"dialect\"] == dialect_name]\n\n        refs = g[\"reference\"].astype(str).tolist()\n        hyps = g[\"prediction\"].astype(str).tolist()\n\n        rows.append({\n            \"model\": model_name,\n            \"dialect\": dialect_name,\n            \"samples\": len(g),\n            \"WER\": wer(refs, hyps),\n            \"CER\": cer(refs, hyps)\n        })\n\n    result = pd.DataFrame(rows).sort_values(\"WER\", ascending=False)\n\n    summary = pd.DataFrame([{\n        \"model\": model_name,\n        \"overall_WER\": wer(pred_df[\"reference\"].tolist(), pred_df[\"prediction\"].tolist()),\n        \"overall_CER\": cer(pred_df[\"reference\"].tolist(), pred_df[\"prediction\"].tolist()),\n        \"worst_dialect\": result.iloc[0][\"dialect\"],\n        \"worst_dialect_WER\": result[\"WER\"].max(),\n        \"best_dialect\": result.iloc[-1][\"dialect\"],\n        \"best_dialect_WER\": result[\"WER\"].min(),\n        \"max_min_gap\": result[\"WER\"].max() - result[\"WER\"].min()\n    }])\n\n    return result, summary\n\ndialect_result, summary = compute_dialect_metrics(pred_df, \"whisper-small-zero-shot\")\n\ndialect_result.to_csv(\"/kaggle/working/exp1_whisper_small_dialect_metrics_quick.csv\", index=False)\nsummary.to_csv(\"/kaggle/working/exp1_whisper_small_summary_quick.csv\", index=False)\n\ndisplay(dialect_result)\ndisplay(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:07:22.250472Z","iopub.execute_input":"2026-06-14T17:07:22.251242Z","iopub.status.idle":"2026-06-14T17:07:22.320759Z","shell.execute_reply.started":"2026-06-14T17:07:22.251208Z","shell.execute_reply":"2026-06-14T17:07:22.319884Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Audio duration + raw Whisper debug","metadata":{}},{"cell_type":"code","source":"import librosa\nimport pandas as pd\nfrom IPython.display import Audio, display\n\ndf = pd.read_csv(\"/kaggle/working/ben10_manifest.csv\")\neval_df = df[df[\"split\"] == \"test\"].copy()\n\nsample = eval_df.sample(5, random_state=7).reset_index(drop=True)\n\nfor i, row in sample.iterrows():\n    audio_path = row[\"audio_path\"]\n    y, sr = librosa.load(audio_path, sr=None)\n    duration = len(y) / sr\n    \n    print(\"=\"*80)\n    print(\"Dialect:\", row[\"dialect\"])\n    print(\"Path:\", audio_path)\n    print(\"Duration:\", duration, \"sec\")\n    print(\"Reference:\", row[\"transcript\"])\n    \n    raw = model.transcribe(\n        audio_path,\n        language=\"bn\",\n        task=\"transcribe\",\n        fp16=torch.cuda.is_available(),\n        verbose=False\n    )\n    \n    print(\"RAW Whisper:\", repr(raw[\"text\"]))\n    display(Audio(audio_path))\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:10:51.717646Z","iopub.execute_input":"2026-06-14T17:10:51.718594Z","iopub.status.idle":"2026-06-14T17:12:34.605566Z","shell.execute_reply.started":"2026-06-14T17:10:51.718558Z","shell.execute_reply":"2026-06-14T17:12:34.604733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Whisper direct decode, no-speech skip","metadata":{}},{"cell_type":"code","source":"import whisper\nimport torch\nimport pandas as pd\nfrom tqdm import tqdm\n\ndef whisper_direct_decode(audio_path, model):\n    audio = whisper.load_audio(audio_path)\n    audio = whisper.pad_or_trim(audio)\n\n    mel = whisper.log_mel_spectrogram(audio).to(model.device)\n\n    options = whisper.DecodingOptions(\n        language=\"bn\",\n        task=\"transcribe\",\n        without_timestamps=True,\n        fp16=torch.cuda.is_available(),\n        beam_size=5\n    )\n\n    result = whisper.decode(model, mel, options)\n    return result.text\n\n# quick test\nfor i, row in sample.iterrows():\n    pred_raw = whisper_direct_decode(row[\"audio_path\"], model)\n    print(\"=\"*80)\n    print(\"Dialect:\", row[\"dialect\"])\n    print(\"REF:\", row[\"transcript\"])\n    print(\"RAW DIRECT:\", repr(pred_raw))\n    print(\"NORM DIRECT:\", normalize_bangla_text(pred_raw))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:17:20.660752Z","iopub.execute_input":"2026-06-14T17:17:20.661442Z","iopub.status.idle":"2026-06-14T17:17:30.237912Z","shell.execute_reply.started":"2026-06-14T17:17:20.661408Z","shell.execute_reply":"2026-06-14T17:17:30.237212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eval_df = df[df[\"split\"] == \"test\"].copy()\n\nsample_parts = []\nfor dialect_name in sorted(eval_df[\"dialect\"].unique()):\n    g = eval_df[eval_df[\"dialect\"] == dialect_name].copy()\n    sample_parts.append(g.sample(n=min(len(g), 5), random_state=42))\n\neval_df_quick = pd.concat(sample_parts, ignore_index=True)\n\npred_rows = []\n\nfor _, row in tqdm(eval_df_quick.iterrows(), total=len(eval_df_quick)):\n    try:\n        raw_pred = whisper_direct_decode(row[\"audio_path\"], model)\n    except Exception as e:\n        print(\"Error:\", row[\"audio_path\"], e)\n        raw_pred = \"\"\n\n    pred_rows.append({\n        \"audio_path\": row[\"audio_path\"],\n        \"dialect\": row[\"dialect\"],\n        \"reference\": normalize_bangla_text(row[\"transcript\"]),\n        \"raw_prediction\": raw_pred,\n        \"prediction\": normalize_bangla_text(raw_pred)\n    })\n\npred_df2 = pd.DataFrame(pred_rows)\npred_df2.to_csv(\"/kaggle/working/exp1_whisper_direct_predictions_quick.csv\", index=False)\n\ndisplay(pred_df2[[\"dialect\", \"reference\", \"raw_prediction\", \"prediction\"]].head(30))\n\n\ndialect_result2, summary2 = compute_dialect_metrics(\n    pred_df2[[\"audio_path\", \"dialect\", \"reference\", \"prediction\"]],\n    \"whisper-small-direct-zero-shot\"\n)\n\ndisplay(dialect_result2)\ndisplay(summary2)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:17:51.657704Z","iopub.execute_input":"2026-06-14T17:17:51.657987Z","iopub.status.idle":"2026-06-14T17:18:41.573484Z","shell.execute_reply.started":"2026-06-14T17:17:51.657966Z","shell.execute_reply":"2026-06-14T17:18:41.572739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Bengali Wav2Vec2 model","metadata":{}},{"cell_type":"code","source":"!pip -q install -U transformers accelerate soundfile librosa jiwer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:21:17.973194Z","iopub.execute_input":"2026-06-14T17:21:17.973929Z","iopub.status.idle":"2026-06-14T17:21:32.105009Z","shell.execute_reply.started":"2026-06-14T17:21:17.973898Z","shell.execute_reply":"2026-06-14T17:21:32.104223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport librosa\nimport pandas as pd\nfrom tqdm import tqdm\nfrom transformers import AutoProcessor, AutoModelForCTC\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(\"Device:\", device)\n\ncandidate_models = [\n    \"jonatasgrosman/wav2vec2-large-xlsr-53-bengali\",\n    \"tanim94/wav2vec2-large-xlsr-bengali\",\n    \"arijitx/wav2vec2-large-xlsr-bengali\",\n]\n\nprocessor = None\nasr_model = None\nloaded_model_name = None\n\nfor model_id in candidate_models:\n    try:\n        print(\"Trying:\", model_id)\n        processor = AutoProcessor.from_pretrained(model_id)\n        asr_model = AutoModelForCTC.from_pretrained(model_id).to(device)\n        asr_model.eval()\n        loaded_model_name = model_id\n        print(\"Loaded successfully:\", model_id)\n        break\n    except Exception as e:\n        print(\"Failed:\", model_id)\n        print(e)\n\nif asr_model is None:\n    raise RuntimeError(\"No model loaded. Kaggle Internet ON আছে কিনা check করো.\")\n\nprint(\"Final model:\", loaded_model_name)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:27:51.346560Z","iopub.execute_input":"2026-06-14T17:27:51.347444Z","iopub.status.idle":"2026-06-14T17:28:11.563412Z","shell.execute_reply.started":"2026-06-14T17:27:51.347407Z","shell.execute_reply":"2026-06-14T17:28:11.562364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Single audio prediction test","metadata":{}},{"cell_type":"code","source":"def wav2vec2_transcribe_single(audio_path):\n    speech, sr = librosa.load(audio_path, sr=16000)\n\n    inputs = processor(\n        speech,\n        sampling_rate=16000,\n        return_tensors=\"pt\",\n        padding=True\n    )\n\n    input_values = inputs.input_values.to(device)\n\n    with torch.no_grad():\n        logits = asr_model(input_values).logits\n\n    pred_ids = torch.argmax(logits, dim=-1)\n    pred_text = processor.batch_decode(pred_ids)[0]\n    return pred_text\n\n\ndf = pd.read_csv(\"/kaggle/working/ben10_manifest.csv\")\ntest_df = df[df[\"split\"] == \"test\"].copy()\n\nsample = test_df.sample(5, random_state=42).reset_index(drop=True)\n\nfor i, row in sample.iterrows():\n    raw_pred = wav2vec2_transcribe_single(row[\"audio_path\"])\n    pred = normalize_bangla_text(raw_pred)\n\n    print(\"=\" * 80)\n    print(\"Dialect:\", row[\"dialect\"])\n    print(\"REF :\", row[\"transcript\"])\n    print(\"RAW :\", raw_pred)\n    print(\"PRED:\", pred)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:37:46.900970Z","iopub.execute_input":"2026-06-14T17:37:46.902332Z","iopub.status.idle":"2026-06-14T17:37:48.365033Z","shell.execute_reply.started":"2026-06-14T17:37:46.902266Z","shell.execute_reply":"2026-06-14T17:37:48.364377Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fast batch transcription function","metadata":{}},{"cell_type":"code","source":"def wav2vec2_transcribe_batch(audio_paths, batch_size=4):\n    all_preds = []\n\n    for i in tqdm(range(0, len(audio_paths), batch_size)):\n        batch_paths = audio_paths[i:i + batch_size]\n\n        speeches = []\n        for path in batch_paths:\n            speech, sr = librosa.load(path, sr=16000)\n            speeches.append(speech)\n\n        inputs = processor(\n            speeches,\n            sampling_rate=16000,\n            return_tensors=\"pt\",\n            padding=True\n        )\n\n        input_values = inputs.input_values.to(device)\n\n        with torch.no_grad():\n            logits = asr_model(input_values).logits\n\n        pred_ids = torch.argmax(logits, dim=-1)\n        pred_texts = processor.batch_decode(pred_ids)\n\n        all_preds.extend(pred_texts)\n\n    return all_preds\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:41:10.293022Z","iopub.execute_input":"2026-06-14T17:41:10.294078Z","iopub.status.idle":"2026-06-14T17:41:10.300289Z","shell.execute_reply.started":"2026-06-14T17:41:10.294044Z","shell.execute_reply":"2026-06-14T17:41:10.299572Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quick Wav2Vec2 Exp 1, 10 samples per dialect","metadata":{}},{"cell_type":"code","source":"eval_df = test_df.copy()\n\nsample_parts = []\nfor dialect_name in sorted(eval_df[\"dialect\"].unique()):\n    g = eval_df[eval_df[\"dialect\"] == dialect_name].copy()\n    sample_parts.append(g.sample(n=min(len(g), 10), random_state=42))\n\neval_quick = pd.concat(sample_parts, ignore_index=True)\n\nprint(\"Eval size:\", len(eval_quick))\nprint(eval_quick[\"dialect\"].value_counts())\n\nraw_preds = wav2vec2_transcribe_batch(\n    eval_quick[\"audio_path\"].tolist(),\n    batch_size=4\n)\n\npred_df = eval_quick[[\"audio_path\", \"dialect\", \"transcript\"]].copy()\npred_df[\"reference\"] = pred_df[\"transcript\"].apply(normalize_bangla_text)\npred_df[\"raw_prediction\"] = raw_preds\npred_df[\"prediction\"] = pred_df[\"raw_prediction\"].apply(normalize_bangla_text)\n\npred_df = pred_df[[\"audio_path\", \"dialect\", \"reference\", \"raw_prediction\", \"prediction\"]]\n\npred_df.to_csv(\"/kaggle/working/exp1_wav2vec2_predictions_quick.csv\", index=False)\n\ndisplay(pred_df.head(20))\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:41:14.900355Z","iopub.execute_input":"2026-06-14T17:41:14.901064Z","iopub.status.idle":"2026-06-14T17:41:41.681089Z","shell.execute_reply.started":"2026-06-14T17:41:14.901031Z","shell.execute_reply":"2026-06-14T17:41:41.680465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Wav2Vec2 per-dialect WER/CER","metadata":{}},{"cell_type":"code","source":"from jiwer import wer, cer\n\ndef compute_dialect_metrics(pred_df, model_name):\n    rows = []\n\n    for dialect_name in sorted(pred_df[\"dialect\"].unique()):\n        g = pred_df[pred_df[\"dialect\"] == dialect_name]\n\n        refs = g[\"reference\"].astype(str).tolist()\n        hyps = g[\"prediction\"].astype(str).tolist()\n\n        rows.append({\n            \"model\": model_name,\n            \"dialect\": dialect_name,\n            \"samples\": len(g),\n            \"WER\": wer(refs, hyps),\n            \"CER\": cer(refs, hyps)\n        })\n\n    result = pd.DataFrame(rows).sort_values(\"WER\", ascending=False)\n\n    summary = pd.DataFrame([{\n        \"model\": model_name,\n        \"overall_WER\": wer(pred_df[\"reference\"].tolist(), pred_df[\"prediction\"].tolist()),\n        \"overall_CER\": cer(pred_df[\"reference\"].tolist(), pred_df[\"prediction\"].tolist()),\n        \"worst_dialect\": result.iloc[0][\"dialect\"],\n        \"worst_dialect_WER\": result[\"WER\"].max(),\n        \"best_dialect\": result.iloc[-1][\"dialect\"],\n        \"best_dialect_WER\": result[\"WER\"].min(),\n        \"max_min_gap\": result[\"WER\"].max() - result[\"WER\"].min()\n    }])\n\n    return result, summary\n\nw2v_result, w2v_summary = compute_dialect_metrics(\n    pred_df,\n    f\"{loaded_model_name}-zero-shot\"\n)\n\nw2v_result.to_csv(\"/kaggle/working/exp1_wav2vec2_dialect_metrics_quick.csv\", index=False)\nw2v_summary.to_csv(\"/kaggle/working/exp1_wav2vec2_summary_quick.csv\", index=False)\n\ndisplay(w2v_result)\ndisplay(w2v_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:42:07.230267Z","iopub.execute_input":"2026-06-14T17:42:07.231511Z","iopub.status.idle":"2026-06-14T17:42:07.323971Z","shell.execute_reply.started":"2026-06-14T17:42:07.231463Z","shell.execute_reply":"2026-06-14T17:42:07.323273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Compare Whisper failed baseline vs Wav2Vec2","metadata":{}},{"cell_type":"code","source":"try:\n    whisper_summary = pd.read_csv(\"/kaggle/working/exp1_whisper_direct_predictions_quick.csv\")\n    whisper_summary_metrics, whisper_summary_row = compute_dialect_metrics(\n        whisper_summary[[\"audio_path\", \"dialect\", \"reference\", \"prediction\"]],\n        \"whisper-small-zero-shot\"\n    )\n\n    compare_summary = pd.concat([whisper_summary_row, w2v_summary], ignore_index=True)\n    display(compare_summary)\n\nexcept Exception as e:\n    print(\"Whisper summary comparison skipped:\", e)\n    display(w2v_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:42:33.173958Z","iopub.execute_input":"2026-06-14T17:42:33.174760Z","iopub.status.idle":"2026-06-14T17:42:33.195282Z","shell.execute_reply.started":"2026-06-14T17:42:33.174728Z","shell.execute_reply":"2026-06-14T17:42:33.194442Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Full test prediction with resume","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport unicodedata\nimport torch\nimport librosa\nimport pandas as pd\nfrom tqdm import tqdm\nfrom pathlib import Path\n\ndef normalize_bangla_text(text):\n    text = str(text)\n    text = unicodedata.normalize(\"NFC\", text)\n    text = re.sub(r\"<[^>]*>\", \" \", text)\n    text = re.sub(r\"[^\\u0980-\\u09FF\\s]\", \" \", text)\n    text = re.sub(r\"\\s+\", \" \", text).strip()\n    return text\n\ndef wav2vec2_transcribe_batch(audio_paths, batch_size=4):\n    all_preds = []\n\n    for i in range(0, len(audio_paths), batch_size):\n        batch_paths = audio_paths[i:i + batch_size]\n\n        speeches = []\n        for path in batch_paths:\n            speech, sr = librosa.load(path, sr=16000)\n            speeches.append(speech)\n\n        inputs = processor(\n            speeches,\n            sampling_rate=16000,\n            return_tensors=\"pt\",\n            padding=True\n        )\n\n        input_values = inputs.input_values.to(device)\n\n        with torch.no_grad():\n            logits = asr_model(input_values).logits\n\n        pred_ids = torch.argmax(logits, dim=-1)\n        pred_texts = processor.batch_decode(pred_ids)\n\n        all_preds.extend(pred_texts)\n\n    return all_preds\n\ndf = pd.read_csv(\"/kaggle/working/ben10_manifest.csv\")\nfull_test_df = df[df[\"split\"] == \"test\"].copy().reset_index(drop=True)\n\nprint(\"Full test size:\", len(full_test_df))\nprint(full_test_df[\"dialect\"].value_counts())\n\nout_path = \"/kaggle/working/exp1_wav2vec2_full_predictions.csv\"\n\nif os.path.exists(out_path):\n    done_df = pd.read_csv(out_path)\n    done_paths = set(done_df[\"audio_path\"].tolist())\n    print(\"Existing predictions:\", len(done_df))\nelse:\n    done_df = pd.DataFrame()\n    done_paths = set()\n\nremaining_df = full_test_df[~full_test_df[\"audio_path\"].isin(done_paths)].copy().reset_index(drop=True)\n\nprint(\"Remaining:\", len(remaining_df))\n\nbatch_size = 4\nsave_every_batches = 20\n\nnew_rows = []\n\nfor start in tqdm(range(0, len(remaining_df), batch_size)):\n    batch = remaining_df.iloc[start:start + batch_size].copy()\n    paths = batch[\"audio_path\"].tolist()\n\n    try:\n        raw_preds = wav2vec2_transcribe_batch(paths, batch_size=batch_size)\n    except Exception as e:\n        print(\"Batch error:\", e)\n        raw_preds = [\"\"] * len(paths)\n\n    for (_, row), raw_pred in zip(batch.iterrows(), raw_preds):\n        new_rows.append({\n            \"audio_path\": row[\"audio_path\"],\n            \"dialect\": row[\"dialect\"],\n            \"reference\": normalize_bangla_text(row[\"transcript\"]),\n            \"raw_prediction\": raw_pred,\n            \"prediction\": normalize_bangla_text(raw_pred),\n            \"model\": loaded_model_name\n        })\n\n    if len(new_rows) >= batch_size * save_every_batches:\n        temp_new = pd.DataFrame(new_rows)\n        combined = pd.concat([done_df, temp_new], ignore_index=True)\n        combined = combined.drop_duplicates(\"audio_path\")\n        combined.to_csv(out_path, index=False)\n        done_df = combined.copy()\n        new_rows = []\n        print(\"Saved:\", len(done_df))\n\nif len(new_rows) > 0:\n    temp_new = pd.DataFrame(new_rows)\n    combined = pd.concat([done_df, temp_new], ignore_index=True)\n    combined = combined.drop_duplicates(\"audio_path\")\n    combined.to_csv(out_path, index=False)\n    done_df = combined.copy()\n\nprint(\"Final saved:\", len(done_df))\ndisplay(done_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:45:08.270339Z","iopub.execute_input":"2026-06-14T17:45:08.270892Z","iopub.status.idle":"2026-06-14T17:51:40.529911Z","shell.execute_reply.started":"2026-06-14T17:45:08.270862Z","shell.execute_reply":"2026-06-14T17:51:40.529178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Full per-dialect WER/CER","metadata":{}},{"cell_type":"code","source":"from jiwer import wer, cer\nimport pandas as pd\n\npred_full = pd.read_csv(\"/kaggle/working/exp1_wav2vec2_full_predictions.csv\")\n\npred_full[\"reference\"] = pred_full[\"reference\"].fillna(\"\").astype(str)\npred_full[\"prediction\"] = pred_full[\"prediction\"].fillna(\"\").astype(str)\npred_full[\"dialect\"] = pred_full[\"dialect\"].astype(str)\n\nrows = []\n\nfor dialect_name in sorted(pred_full[\"dialect\"].unique()):\n    g = pred_full[pred_full[\"dialect\"] == dialect_name]\n\n    refs = g[\"reference\"].tolist()\n    hyps = g[\"prediction\"].tolist()\n\n    rows.append({\n        \"model\": loaded_model_name,\n        \"dialect\": dialect_name,\n        \"samples\": len(g),\n        \"WER\": wer(refs, hyps),\n        \"CER\": cer(refs, hyps)\n    })\n\nfull_result = pd.DataFrame(rows).sort_values(\"WER\", ascending=False)\n\nfull_summary = pd.DataFrame([{\n    \"model\": loaded_model_name,\n    \"samples\": len(pred_full),\n    \"overall_WER\": wer(pred_full[\"reference\"].tolist(), pred_full[\"prediction\"].tolist()),\n    \"overall_CER\": cer(pred_full[\"reference\"].tolist(), pred_full[\"prediction\"].tolist()),\n    \"worst_dialect\": full_result.iloc[0][\"dialect\"],\n    \"worst_dialect_WER\": full_result[\"WER\"].max(),\n    \"best_dialect\": full_result.iloc[-1][\"dialect\"],\n    \"best_dialect_WER\": full_result[\"WER\"].min(),\n    \"max_min_gap\": full_result[\"WER\"].max() - full_result[\"WER\"].min()\n}])\n\nfull_result.to_csv(\"/kaggle/working/exp1_wav2vec2_full_dialect_metrics.csv\", index=False)\nfull_summary.to_csv(\"/kaggle/working/exp1_wav2vec2_full_summary.csv\", index=False)\n\ndisplay(full_result)\ndisplay(full_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T17:54:37.725029Z","iopub.execute_input":"2026-06-14T17:54:37.725577Z","iopub.status.idle":"2026-06-14T17:54:38.695124Z","shell.execute_reply.started":"2026-06-14T17:54:37.725547Z","shell.execute_reply":"2026-06-14T17:54:38.694339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport zipfile\nfrom pathlib import Path\n\nfiles_to_save = [\n    \"/kaggle/working/ben10_manifest_raw.csv\",\n    \"/kaggle/working/ben10_manifest.csv\",\n    \"/kaggle/working/exp1_wav2vec2_full_predictions.csv\",\n    \"/kaggle/working/exp1_wav2vec2_full_dialect_metrics.csv\",\n    \"/kaggle/working/exp1_wav2vec2_full_summary.csv\",\n    \"/kaggle/working/exp1_wav2vec2_predictions_quick.csv\",\n    \"/kaggle/working/exp1_wav2vec2_dialect_metrics_quick.csv\",\n    \"/kaggle/working/exp1_wav2vec2_summary_quick.csv\",\n]\n\nzip_path = \"/kaggle/working/exp1_results_ben10.zip\"\n\nwith zipfile.ZipFile(zip_path, \"w\") as zipf:\n    for file in files_to_save:\n        if os.path.exists(file):\n            zipf.write(file, arcname=Path(file).name)\n            print(\"Added:\", file)\n        else:\n            print(\"Missing:\", file)\n\nprint(\"Saved zip:\", zip_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-14T18:00:09.412175Z","iopub.execute_input":"2026-06-14T18:00:09.413015Z","iopub.status.idle":"2026-06-14T18:00:09.462700Z","shell.execute_reply.started":"2026-06-14T18:00:09.412982Z","shell.execute_reply":"2026-06-14T18:00:09.461943Z"}},"outputs":[],"execution_count":null}]}