{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":129276,"databundleVersionId":15506988},{"sourceType":"datasetVersion","sourceId":6707460,"datasetId":3865741,"databundleVersionId":6791840},{"sourceType":"datasetVersion","sourceId":4143520,"datasetId":2447262,"databundleVersionId":4200057}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# =============================================================================\n# Bengali Long-Form ASR — Improved Solution\n# Target: Sub-0.30 WER\n# =============================================================================\n# Key changes from V21:\n# 1. faster-whisper with large-v3 (native long-form, VAD, 4x speed)\n# 2. Whisper's own sequential decoding (not broken HF pipeline chunking)\n# 3. Hallucination detection + temperature fallback\n# 4. Bengali-specific post-processing\n# =============================================================================","metadata":{"_uuid":"c2f80de9-4132-4c00-afa8-4a49bb984a0e","_cell_guid":"e17308e7-4c4e-44bd-a82a-f02f963fea7b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:12:58.174736Z","iopub.execute_input":"2026-02-21T08:12:58.174935Z","iopub.status.idle":"2026-02-21T08:12:58.179132Z","shell.execute_reply.started":"2026-02-21T08:12:58.174915Z","shell.execute_reply":"2026-02-21T08:12:58.178353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 1: Install Dependencies","metadata":{"_uuid":"e733584e-fcea-4c39-860d-56cb7dbe074a","_cell_guid":"3b6a4bad-684b-40af-b0ea-e3ab1c31b8fe","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import subprocess\nimport sys\n\n# Install faster-whisper (CTranslate2 backend - 4x faster than openai-whisper)\nsubprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", \"-q\",\n                       \"faster-whisper>=1.1.0\"])\n\n# Verify\nimport faster_whisper\nprint(f\"faster-whisper version: {faster_whisper.__version__}\")","metadata":{"_uuid":"d9d21e60-677c-4652-9122-cca09c4154e8","_cell_guid":"1487c482-3965-4b48-83cb-edcc3972fb6d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:12:58.180563Z","iopub.execute_input":"2026-02-21T08:12:58.180873Z","iopub.status.idle":"2026-02-21T08:13:18.012039Z","shell.execute_reply.started":"2026-02-21T08:12:58.180839Z","shell.execute_reply":"2026-02-21T08:13:18.011387Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 2: Imports & Setup","metadata":{"_uuid":"6edec4f4-1b04-4fba-b764-3d24091172e5","_cell_guid":"545730a0-f87b-467e-8375-2d8663388eee","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import os\nimport re\nimport gc\nimport unicodedata\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom tqdm.auto import tqdm\nfrom faster_whisper import WhisperModel\n\nwarnings.filterwarnings('ignore')\n\nprint(f\"PyTorch: {torch.__version__}\")\nprint(f\"CUDA: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    # print(f\"VRAM: {torch.cuda.get_device_properties(0).total_mem / 1e9:.1f} GB\")","metadata":{"_uuid":"e371344b-d87e-4dbb-9902-2e6eaded3d2b","_cell_guid":"99d861cd-7d52-4423-aa64-d0b290eac033","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:13:18.012838Z","iopub.execute_input":"2026-02-21T08:13:18.013203Z","iopub.status.idle":"2026-02-21T08:13:18.555129Z","shell.execute_reply.started":"2026-02-21T08:13:18.01318Z","shell.execute_reply":"2026-02-21T08:13:18.55444Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 3: Competition-Mandated Normalization + Bengali Post-Processing","metadata":{"_uuid":"9d0b19b4-1927-4e2f-95b9-0de93750a005","_cell_guid":"490875b0-4e80-41c6-a534-2934e2bcf7d7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"ZW = r\"[\\u200B-\\u200D\\uFEFF]\"\n\ndef normalize_bn_text(s: str) -> str:\n    \"\"\"Exact competition normalization - MUST apply to all outputs.\"\"\"\n    if s is None:\n        return \"\"\n    s = str(s)\n    s = unicodedata.normalize(\"NFC\", s)\n    s = re.sub(ZW, \"\", s)\n    s = s.replace(\"\\u00A0\", \" \")\n    s = \" \".join(s.split())\n    return s\n\n\ndef postprocess_bengali(text: str) -> str:\n    \"\"\"\n    Bengali-specific post-processing for WER optimization.\n    \n    Key insight: The evaluation metric is WER which counts word-level errors.\n    Every non-Bengali token that shouldn't be there is a free error.\n    Punctuation is explicitly NOT required per competition rules.\n    \"\"\"\n    if not text:\n        return \"\"\n    \n    # 1. Remove all punctuation (competition says \"not required\" = hurts WER)\n    # Keep Bengali characters (U+0980-U+09FF), Bengali digits, spaces\n    # Also keep standard digits as they appear in Bengali speech\n    text = re.sub(r'[।,;:!?\\.\\-\\—\\–\\'\\\"\\\"\\\"\\'\\'\\(\\)\\[\\]\\{\\}\\…\\·\\•\\*/\\\\@#$%^&+=<>~`]', ' ', text)\n    \n    # 2. Remove English/Latin characters (Whisper hallucination artifact)\n    text = re.sub(r'[a-zA-Z]', '', text)\n    \n    # 3. Remove Chinese/Japanese/Korean characters (cross-lingual hallucination)\n    text = re.sub(r'[\\u4e00-\\u9fff\\u3040-\\u309f\\u30a0-\\u30ff]', '', text)\n    \n    # 4. Remove Arabic characters\n    text = re.sub(r'[\\u0600-\\u06ff]', '', text)\n    \n    # 5. Remove musical notes and misc symbols (hallucination artifacts)\n    text = re.sub(r'[♪♫♬♩🎵🎶]', '', text)\n    \n    # 6. Detect and remove repeated phrases (Whisper hallucination loops)\n    # Pattern: same phrase repeated 3+ times consecutively\n    text = re.sub(r'(\\b.{4,50}\\b)( \\1){2,}', r'\\1', text)\n    \n    # 7. Apply competition normalization\n    text = normalize_bn_text(text)\n    \n    return text\n\n\n# Test\ntest_cases = [\n    \"আমি ভালো আছি। তুমি কেমন আছো?\",\n    \"Hello আমি world ভালো\",\n    \"আমি আমি আমি আমি ভালো\",\n    \"test \\u200B\\u200C\\u200D case\",\n]\nfor tc in test_cases:\n    print(f\"  '{tc}' -> '{postprocess_bengali(tc)}'\")","metadata":{"_uuid":"47b81808-4c70-4627-a512-5ed002944b22","_cell_guid":"adbcaa2b-dd12-4c45-ab58-69a0a14aaf3e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:13:18.556113Z","iopub.execute_input":"2026-02-21T08:13:18.556499Z","iopub.status.idle":"2026-02-21T08:13:18.566779Z","shell.execute_reply.started":"2026-02-21T08:13:18.556476Z","shell.execute_reply":"2026-02-21T08:13:18.566205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 4: Configuration","metadata":{"_uuid":"67330fb6-c1d5-4772-a70f-67fd64aa5b8e","_cell_guid":"722bcf5f-d1c6-424f-8ce7-9687494cf87f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# =============================================================================\n# MODEL SELECTION STRATEGY\n# =============================================================================\n# Option A: openai/whisper-large-v3 (best multilingual, ~3GB)\n#   - Best general Bengali accuracy\n#   - Fits on T4 16GB with float16\n#   - Use this if you have internet access to download\n#\n# Option B: bengaliAI/tugstugi_bengaliai-asr_whisper-medium (~1.5GB)\n#   - Bengali-fine-tuned medium model\n#   - Worse than large-v3 but pre-cached if offline\n#\n# Option C: openai/whisper-large-v3-turbo (~1.6GB)\n#   - Nearly as accurate as large-v3, 2-3x faster\n#   - Great if inference time is tight\n# =============================================================================\n\n# Try large-v3 first; fall back to the Bengali fine-tuned medium\nPRIMARY_MODEL = \"large-v3\"\nFALLBACK_MODEL = \"medium\"  # faster-whisper will try HuggingFace cache\n\n# Paths\nTEST_AUDIO_DIR = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio\"\nTRAIN_AUDIO_DIR = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/audio\"\nTRAIN_ANNOT_DIR = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/train/annotation\"\nOUTPUT_CSV = \"/kaggle/working/submission.csv\"\n\nprint(\"=\" * 60)\nprint(\"CONFIGURATION\")\nprint(\"=\" * 60)\nprint(f\"Primary model: {PRIMARY_MODEL}\")\nprint(f\"Test audio dir: {TEST_AUDIO_DIR}\")\nprint(f\"Output: {OUTPUT_CSV}\")\nprint(\"=\" * 60)","metadata":{"_uuid":"45bddbb7-d68a-4603-a974-3098df7cfbf1","_cell_guid":"d2dea023-26ad-4c1b-a893-5b98d46d7edb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:13:18.568757Z","iopub.execute_input":"2026-02-21T08:13:18.568976Z","iopub.status.idle":"2026-02-21T08:13:18.58167Z","shell.execute_reply.started":"2026-02-21T08:13:18.568956Z","shell.execute_reply":"2026-02-21T08:13:18.580967Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 5: Load Model with faster-whisper","metadata":{"_uuid":"43a0870e-1fa9-4277-8253-2b3a0c5f8b12","_cell_guid":"b6516cc9-ce7a-4030-a305-1917ad915931","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def load_model(model_size=\"large-v3\"):\n    \"\"\"\n    Load faster-whisper model.\n    \n    faster-whisper uses CTranslate2 backend which is 4x faster than\n    openai-whisper and uses less memory. It also implements Whisper's\n    native long-form transcription algorithm correctly.\n    \"\"\"\n    print(f\"Loading faster-whisper model: {model_size}\")\n    \n    # Determine compute type based on GPU\n    if torch.cuda.is_available():\n        device = \"cuda\"\n        # T4 supports float16 well\n        compute_type = \"float16\"\n    else:\n        device = \"cpu\"\n        compute_type = \"int8\"\n    \n    try:\n        model = WhisperModel(\n            model_size,\n            device=device,\n            compute_type=compute_type,\n            download_root=\"/kaggle/working/whisper_models\",\n            # Number of workers for parallel processing\n            cpu_threads=4,\n        )\n        print(f\"✓ Model loaded: {model_size} on {device} ({compute_type})\")\n        return model\n        \n    except Exception as e:\n        print(f\"✗ Failed to load {model_size}: {e}\")\n        return None\n\n\n# Try primary model, fall back if needed\nmodel = load_model(PRIMARY_MODEL)\nif model is None:\n    print(f\"Falling back to {FALLBACK_MODEL}...\")\n    model = load_model(FALLBACK_MODEL)\n\nassert model is not None, \"Failed to load any model!\"","metadata":{"_uuid":"7fe5ce42-a2a4-478f-8719-be991484fe5e","_cell_guid":"3236d175-a964-4afb-8021-07358b6f5700","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:13:18.582462Z","iopub.execute_input":"2026-02-21T08:13:18.582721Z","iopub.status.idle":"2026-02-21T08:13:28.316914Z","shell.execute_reply.started":"2026-02-21T08:13:18.582695Z","shell.execute_reply":"2026-02-21T08:13:28.31614Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 6: Core Transcription Function\n\nThis is where the magic happens. The key insight is that faster-whisper's\n`transcribe()` method implements Whisper's native long-form algorithm:\n\n1. Process 30-second windows sequentially\n2. Use previous context to condition next window\n3. Apply temperature fallback on failures\n4. Filter hallucinations via compression ratio + log probability\n5. Use VAD to skip silence segments","metadata":{"_uuid":"b24c1cc7-ffad-4df7-be2a-a634cf3f69de","_cell_guid":"556f275d-e903-446b-869e-a2fea8a21aca","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def transcribe_audio(model, audio_path, verbose=False):\n    \"\"\"\n    Transcribe a single long-form audio file using Whisper's native algorithm.\n    \n    This uses faster-whisper which correctly implements the sequential\n    decoding strategy from the Whisper paper (Section 3.8).\n    \n    Key parameters tuned for Bengali long-form:\n    - beam_size=5: Good balance of quality and speed\n    - best_of=5: Sample 5 candidates at each temperature\n    - temperature schedule: 0.0 -> 0.2 -> 0.4 -> 0.6 -> 0.8 -> 1.0\n    - vad_filter=True: Skip silence (critical for long-form)\n    - condition_on_previous_text=True: Maintain context across chunks\n    \"\"\"\n    \n    segments, info = model.transcribe(\n        audio_path,\n        language=\"bn\",                    # Force Bengali\n        beam_size=5,                      # Beam search (5 is Whisper's default)\n        best_of=5,                        # Sample 5 at non-zero temperatures\n        temperature=(0.0, 0.2, 0.4, 0.6, 0.8, 1.0),  # Temperature fallback\n        condition_on_previous_text=True,   # Context across chunks\n        \n        # VAD filtering - CRITICAL for long-form audio\n        vad_filter=True,\n        vad_parameters=dict(\n            min_silence_duration_ms=500,   # Minimum silence to split on\n            speech_pad_ms=400,             # Padding around speech segments\n            threshold=0.5,                 # VAD confidence threshold\n        ),\n        \n        # Hallucination filtering thresholds\n        compression_ratio_threshold=2.4,   # Detect repetitive outputs\n        log_prob_threshold=-1.0,           # Filter low-confidence segments\n        no_speech_threshold=0.6,           # Skip non-speech segments\n        \n        # Output settings\n        word_timestamps=False,             # Don't need word-level timing\n        without_timestamps=True,           # Slightly faster\n    )\n    \n    if verbose:\n        print(f\"  Language: {info.language} (prob: {info.language_probability:.2f})\")\n        print(f\"  Duration: {info.duration:.1f}s\")\n    \n    # Collect all segments with quality filtering\n    texts = []\n    segment_count = 0\n    skipped_count = 0\n    \n    for segment in segments:\n        segment_count += 1\n        \n        # Additional hallucination guards\n        text = segment.text.strip()\n        \n        # Skip empty segments\n        if not text:\n            skipped_count += 1\n            continue\n        \n        # Skip segments that are pure repetition of a very short phrase\n        # (e.g., \"হ্যাঁ হ্যাঁ হ্যাঁ হ্যাঁ হ্যাঁ\")\n        words = text.split()\n        if len(words) >= 5:\n            unique_words = set(words)\n            if len(unique_words) <= 2 and len(words) > 6:\n                skipped_count += 1\n                continue\n        \n        texts.append(text)\n    \n    full_text = \" \".join(texts)\n    \n    if verbose:\n        print(f\"  Segments: {segment_count} (skipped: {skipped_count})\")\n        print(f\"  Output: {len(full_text)} chars, {len(full_text.split())} words\")\n    \n    return full_text\n\n\n# Quick test on first file\ntest_files = sorted([f for f in os.listdir(TEST_AUDIO_DIR) if f.endswith('.wav')])\nprint(f\"Found {len(test_files)} test files\")\n\n# Test on first file\nif test_files:\n    test_path = os.path.join(TEST_AUDIO_DIR, test_files[0])\n    print(f\"\\nTest transcription: {test_files[0]}\")\n    test_text = transcribe_audio(model, test_path, verbose=True)\n    print(f\"  Preview: {test_text[:200]}...\")","metadata":{"_uuid":"0ada0ce7-0b0c-4b84-b7d4-f9e89c37b6c0","_cell_guid":"ec276d4c-e8a2-406a-aceb-12a07f7104c2","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-02-21T08:13:28.318057Z","iopub.execute_input":"2026-02-21T08:13:28.318368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 7: Optional - Validate on Training Data\n\nUse a small subset of training data to sanity-check WER before submission.","metadata":{"_uuid":"c50dbaca-2aec-46c4-96f3-bf5dc38320cb","_cell_guid":"3c6b5b74-91d3-47b2-8729-eb96258e01e6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"def quick_validation(model, n_samples=3):\n    \"\"\"Quick WER check on training data to verify pipeline quality.\"\"\"\n    try:\n        import jiwer\n    except ImportError:\n        subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", \"-q\", \"jiwer\"])\n        import jiwer\n    \n    train_audios = sorted([f for f in os.listdir(TRAIN_AUDIO_DIR) if f.endswith('.wav')])\n    \n    if not train_audios:\n        print(\"No training data found, skipping validation\")\n        return\n    \n    # Sample a few files\n    sample_files = train_audios[:n_samples]\n    \n    wers = []\n    for fname in sample_files:\n        audio_path = os.path.join(TRAIN_AUDIO_DIR, fname)\n        annot_path = os.path.join(TRAIN_ANNOT_DIR, Path(fname).stem + \".txt\")\n        \n        if not os.path.exists(annot_path):\n            continue\n        \n        # Read ground truth\n        with open(annot_path, 'r', encoding='utf-8') as f:\n            reference = normalize_bn_text(f.read().strip())\n        \n        # Transcribe\n        hypothesis = transcribe_audio(model, audio_path)\n        hypothesis = postprocess_bengali(hypothesis)\n        \n        # Compute WER\n        wer = jiwer.wer(reference, hypothesis)\n        wers.append(wer)\n        \n        print(f\"  {fname}: WER = {wer:.4f}\")\n        print(f\"    REF: {reference[:100]}...\")\n        print(f\"    HYP: {hypothesis[:100]}...\")\n        print()\n    \n    if wers:\n        print(f\"  Mean WER on {len(wers)} samples: {np.mean(wers):.4f}\")\n    \n    return wers\n\nprint(\"Quick validation on training data:\")\nprint(\"-\" * 40)\nquick_validation(model, n_samples=2)","metadata":{"_uuid":"d506f8b9-fb52-4e65-ba82-83af8206841e","_cell_guid":"78e28304-a9d6-44aa-80f7-46c8edc7e941","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 8: Full Inference","metadata":{"_uuid":"3f5f8c67-7a7f-4e79-80c0-c71440474fa7","_cell_guid":"103173d7-e613-4df4-876b-4e7fa4341de8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"print(\"\\n\" + \"=\" * 60)\nprint(\"FULL INFERENCE\")\nprint(\"=\" * 60)\n\naudio_files = sorted([f for f in os.listdir(TEST_AUDIO_DIR) if f.endswith(('.wav', '.mp3'))])\nprint(f\"Processing {len(audio_files)} test files\\n\")\n\nresults = []\n\nfor i, fname in enumerate(tqdm(audio_files, desc=\"Transcribing\")):\n    fpath = os.path.join(TEST_AUDIO_DIR, fname)\n    file_id = Path(fname).stem\n    \n    try:\n        # Transcribe using Whisper's native long-form algorithm\n        raw_text = transcribe_audio(model, fpath, verbose=(i == 0))\n        \n        # Apply Bengali post-processing\n        clean_text = postprocess_bengali(raw_text)\n        \n        results.append({\n            \"filename\": file_id,\n            \"transcript\": clean_text,\n        })\n        \n        # Progress logging\n        if (i + 1) % 5 == 0 or i == 0:\n            print(f\"  [{i+1}/{len(audio_files)}] {file_id}: \"\n                  f\"{len(clean_text)} chars, {len(clean_text.split())} words\")\n        \n        # Memory cleanup\n        if torch.cuda.is_available() and (i + 1) % 8 == 0:\n            torch.cuda.empty_cache()\n            gc.collect()\n    \n    except Exception as e:\n        print(f\"\\n  ERROR with {fname}: {e}\")\n        # Emergency fallback: try with simpler settings\n        try:\n            segments, _ = model.transcribe(\n                fpath, language=\"bn\", beam_size=1,\n                vad_filter=True, without_timestamps=True,\n            )\n            fallback_text = \" \".join([s.text.strip() for s in segments])\n            fallback_text = postprocess_bengali(fallback_text)\n            results.append({\"filename\": file_id, \"transcript\": fallback_text})\n            print(f\"  Recovered {file_id} with fallback\")\n        except Exception as e2:\n            print(f\"  FATAL: {e2}\")\n            results.append({\"filename\": file_id, \"transcript\": \"\"})\n\ndf = pd.DataFrame(results)\nprint(f\"\\n✓ Transcription complete: {len(df)} files\")","metadata":{"_uuid":"16ab90a4-9d03-4cc0-af3a-9f66e92d84e5","_cell_guid":"8960a7ee-5f8f-46fe-b7e5-ea301cb43476","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Cell 9: Final Verification & Save","metadata":{"_uuid":"3b384b00-2b07-4851-baa7-102e601dc4f9","_cell_guid":"657f3998-780d-4e49-9b4f-5cb00b52a779","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Final normalization pass (safety net)\ndf['transcript'] = df['transcript'].apply(normalize_bn_text)\n\n# Verify submission integrity\nprint(\"=\" * 60)\nprint(\"SUBMISSION VERIFICATION\")\nprint(\"=\" * 60)\n\n# Check required format\nsample_sub = pd.read_csv(\n    \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/sample_submission .csv\"\n)\nexpected_files = set(sample_sub['filename'].values)\nactual_files = set(df['filename'].values)\n\n# Verify all expected files are present\nmissing = expected_files - actual_files\nextra = actual_files - expected_files\n\nif missing:\n    print(f\"⚠ MISSING files: {missing}\")\n    # Add missing files with empty transcript\n    for f in missing:\n        df = pd.concat([df, pd.DataFrame([{\"filename\": f, \"transcript\": \"\"}])],\n                       ignore_index=True)\nif extra:\n    print(f\"⚠ EXTRA files (will be ignored): {extra}\")\n    df = df[df['filename'].isin(expected_files)]\n\n# Check for empty transcripts\nempty = (df['transcript'].str.strip() == '').sum()\nif empty > 0:\n    print(f\"⚠ WARNING: {empty} empty transcripts!\")\nelse:\n    print(\"✓ All transcripts non-empty\")\n\n# Check column names match expected format\nprint(f\"✓ Columns: {df.columns.tolist()}\")\nprint(f\"✓ Rows: {len(df)}\")\n\n# Stats\nprint(f\"\\nStats:\")\nprint(f\"  Avg chars: {df['transcript'].str.len().mean():.0f}\")\nprint(f\"  Avg words: {df['transcript'].str.split().str.len().mean():.0f}\")\nprint(f\"  Min words: {df['transcript'].str.split().str.len().min()}\")\nprint(f\"  Max words: {df['transcript'].str.split().str.len().max()}\")\n\n# Verify Unicode normalization was applied\nfor _, row in df.iterrows():\n    assert row['transcript'] == normalize_bn_text(row['transcript']), \\\n        f\"NFC normalization failed for {row['filename']}\"\nprint(\"✓ All transcripts NFC-normalized\")\n\n# Save\ndf = df.sort_values('filename').reset_index(drop=True)\ndf.to_csv(OUTPUT_CSV, index=False, encoding='utf-8')\nprint(f\"\\n✓ Saved to {OUTPUT_CSV}\")\n\n# Show sample output\nprint(f\"\\nSample outputs:\")\nfor _, row in df.head(3).iterrows():\n    print(f\"  {row['filename']}: {row['transcript'][:120]}...\")\n\nprint(\"\\n\" + \"=\" * 60)\nprint(\"DONE — Submit submission.csv to Kaggle\")\nprint(\"=\" * 60)","metadata":{"_uuid":"7d4bc532-e73a-436e-b38d-a91e0808f20f","_cell_guid":"12be5caa-53e9-4a3d-af47-e6a1fc79d2da","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}