{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":129276,"databundleVersionId":15506988,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":14908811,"datasetId":9539501,"databundleVersionId":15774299},{"sourceType":"datasetVersion","sourceId":14845919,"datasetId":9495331,"databundleVersionId":15705582}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":null,"end_time":null,"environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-03T06:19:41.795208","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Make sure you click 'Copy & Edit' as I have given view access only, as directly editing this notebook causes kernel restart. Only 'COPY & EDIT' works.**\n###  **This notebook is configured in such a way that a simple Save and Run would suffice after turning off the internet.**\n###  **For an interactive session, make sure you run the dependency installation script first before running all the cells.**","metadata":{}},{"cell_type":"code","source":"# Verify\nprint(\"\\nVerifying installation...\")\nimport torch\nimport whisperx\nprint(f\"✓ PyTorch: {torch.__version__}\")\nprint(f\"✓ CUDA: {torch.cuda.is_available()}\")\nprint(f\"✓ GPUs: {torch.cuda.device_count()}\")\nprint(f\"✓ WhisperX imported successfully\")\n\nprint(\"\\n⚠ IMPORTANT: After running this cell, click 'Restart Runtime' in Kaggle\")\nprint(\"Then skip to the next cell with the main code\")\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":null,"end_time":null,"exception":false,"start_time":"2026-02-03T06:19:44.532711","status":"running"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T17:05:01.616360Z","iopub.execute_input":"2026-02-22T17:05:01.616663Z","iopub.status.idle":"2026-02-22T17:05:08.367574Z","shell.execute_reply.started":"2026-02-22T17:05:01.616634Z","shell.execute_reply":"2026-02-22T17:05:08.366939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Verify installation\nprint(\"\\n✓ Installation complete!\")\nprint(\"\\nVerifying GPU availability...\")\nimport torch\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nprint(f\"Number of GPUs: {torch.cuda.device_count()}\")\nif torch.cuda.is_available():\n    for i in range(torch.cuda.device_count()):\n        print(f\"GPU {i}: {torch.cuda.get_device_name(i)}\")","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T17:05:08.368820Z","iopub.execute_input":"2026-02-22T17:05:08.369139Z","iopub.status.idle":"2026-02-22T17:05:08.386079Z","shell.execute_reply.started":"2026-02-22T17:05:08.369115Z","shell.execute_reply":"2026-02-22T17:05:08.385459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Necessary Imports**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# ============================================================================\n# Main Inference Code\n# ============================================================================\n# ============================================================================\n\nimport os\nimport json\nimport torch\nimport torch.serialization\n\n# Disable weights_only for compatibility with older model formats\nimport torch._weights_only_unpickler as _weights_only_unpickler\n_weights_only_unpickler._get_allowed_globals = lambda: {}\n\n# Monkey patch torch.load to use weights_only=False\n_original_load = torch.load\ndef patched_load(*args, **kwargs):\n    kwargs['weights_only'] = False\n    return _original_load(*args, **kwargs)\ntorch.load = patched_load\n\nimport whisperx\nimport gc\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport pandas as pd\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:05:08.386840Z","iopub.execute_input":"2026-02-22T17:05:08.387030Z","iopub.status.idle":"2026-02-22T17:05:08.657027Z","shell.execute_reply.started":"2026-02-22T17:05:08.387011Z","shell.execute_reply":"2026-02-22T17:05:08.656287Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Loading the model for CT2 conversion**","metadata":{}},{"cell_type":"code","source":"PARENT_DIR = \"/kaggle/input/datasets/nazmussakibnafiz/parquet-sakib-whisper-medium-optimized\"\nCHECKPOINT = f\"{PARENT_DIR}/checkpoint-12000\"\nMERGED_MODEL = \"/kaggle/working/whisper-merged\"\nCONVERTED_MODEL = \"/kaggle/working/whisper-medium-ct2\"\n\nimport os, shutil\nfrom transformers import WhisperForConditionalGeneration, WhisperProcessor\n\n# Load tokenizer/processor from parent (where all tokenizer files live)\nprocessor = WhisperProcessor.from_pretrained(PARENT_DIR)\n\n# Load the best checkpoint weights\nmodel = WhisperForConditionalGeneration.from_pretrained(CHECKPOINT)\n\n# Save together into one clean directory\nos.makedirs(MERGED_MODEL, exist_ok=True)\nmodel.save_pretrained(MERGED_MODEL)\nprocessor.save_pretrained(MERGED_MODEL)\n\nprint(\"Merged model files:\")\n!ls -lh $MERGED_MODEL","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:05:08.658115Z","iopub.execute_input":"2026-02-22T17:05:08.658628Z","iopub.status.idle":"2026-02-22T17:06:14.528083Z","shell.execute_reply.started":"2026-02-22T17:05:08.658595Z","shell.execute_reply":"2026-02-22T17:06:14.527036Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Converting to CT2 for using WhisperX to load our model**","metadata":{}},{"cell_type":"code","source":"MERGED_MODEL = \"/kaggle/working/whisper-merged\"\nCONVERTED_MODEL = \"/kaggle/working/whisper-medium-ct2\"\n\n!ct2-transformers-converter \\\n    --model $MERGED_MODEL \\\n    --output_dir $CONVERTED_MODEL \\\n    --quantization float16 \\\n    --copy_files tokenizer.json preprocessor_config.json \\\n    --force\n\nprint(\"Files in converted model:\")\n!ls -lh $CONVERTED_MODEL\n\nMODEL = CONVERTED_MODEL","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T17:06:14.531745Z","iopub.execute_input":"2026-02-22T17:06:14.533511Z","iopub.status.idle":"2026-02-22T17:06:32.703433Z","shell.execute_reply.started":"2026-02-22T17:06:14.533437Z","shell.execute_reply":"2026-02-22T17:06:32.702728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Defining Parameters**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# CONFIGURATION\n# ============================================================================\nprint(\"\\n\" + \"=\"*80)\nprint(\"WHISPERX DUAL GPU CONFIGURATION\")\nprint(\"=\"*80)\n\n# Model configuration - will be set after conversion\nMODEL = \"/kaggle/working/whisper-medium-ct2\"  # Path to converted model\nLANGUAGE = \"bn\"  # Bengali\nBATCH_SIZE = 32\nCOMPUTE_TYPE = \"float16\"\n\n# Chunking configuration (matching your requirements)\nCHUNK_LENGTH_S = 20.1  # seconds - the length of each chunk\n\n# Generation configuration\nMAX_NEW_TOKENS = 260\nENABLE_BEAM = True\nNUM_BEAMS = 5  # Beam search\n\n\n# Input/Output paths\nINPUT_DIR = \"/kaggle/input/dl-sprint-4-0-bengali-long-form-speech-recognition/transcription/transcription/test/audio\"\nOUTPUT_DIR = \"/kaggle/working\"\nOUTPUT_FILE = \"submission.csv\"\n\nprint(f\"\\nModel: {MODEL}\")\nprint(f\"Language: {LANGUAGE}\")\nprint(f\"Batch Size: {BATCH_SIZE}\")\nprint(f\"Chunk Length: {CHUNK_LENGTH_S}s\")\nprint(f\"Max New Tokens: {MAX_NEW_TOKENS}\")\nprint(f\"Beam Search: {ENABLE_BEAM}\")\nprint(f\"Num Beams: {NUM_BEAMS}\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:34:17.118140Z","iopub.execute_input":"2026-02-22T17:34:17.118691Z","iopub.status.idle":"2026-02-22T17:34:17.124542Z","shell.execute_reply.started":"2026-02-22T17:34:17.118661Z","shell.execute_reply":"2026-02-22T17:34:17.123784Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Helper Functions**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# HELPER FUNCTIONS\n# ============================================================================\n\ndef get_audio_files(directory):\n    \"\"\"Get all audio files from directory\"\"\"\n    audio_extensions = {'.mp3', '.wav', '.flac', '.m4a', '.ogg'}\n    audio_files = []\n    \n    for file_path in Path(directory).rglob('*'):\n        if file_path.suffix.lower() in audio_extensions:\n            audio_files.append(str(file_path))\n    \n    return sorted(audio_files)\n\n\ndef transcribe_audio_whisperx(audio_path, model, device_id):\n    \"\"\"\n    Transcribe a single audio file using WhisperX\n    \n    Args:\n        audio_path: Path to audio file\n        model: WhisperX model\n        device_id: GPU device ID (0 or 1)\n    \n    Returns:\n        Transcribed text\n    \"\"\"\n    try:\n        # Load audio\n        audio = whisperx.load_audio(audio_path)\n        \n        # Transcribe with WhisperX using beam search\n        result = model.transcribe(\n            audio,\n            batch_size=BATCH_SIZE,\n            language=LANGUAGE,\n            chunk_size=CHUNK_LENGTH_S,\n            print_progress=False,\n        )\n        \n        # Extract text from segments\n        if \"segments\" in result:\n            text = \" \".join([segment[\"text\"].strip() for segment in result[\"segments\"]])\n        else:\n            text = result.get(\"text\", \"\")\n\n        print(f\"{audio_path}\")\n        print(f\"{text}\")\n        return text.strip()\n    \n    except Exception as e:\n        print(f\"Error processing {audio_path}: {str(e)}\")\n        return \"\"\n\n\ndef process_chunk_group(audio_files, model, device_id):\n    \"\"\"\n    Process a group of audio files on a specific GPU\n    \n    Args:\n        audio_files: List of audio file paths\n        model: WhisperX model\n        device_id: GPU device ID\n    \n    Returns:\n        List of dictionaries with file_id and transcription\n    \"\"\"\n    results = []\n    \n    print(f\"\\n[GPU {device_id}] Processing {len(audio_files)} files...\")\n    \n    for audio_path in tqdm(audio_files, desc=f\"GPU {device_id}\", position=device_id):\n        # Extract file ID from path\n        file_id = Path(audio_path).stem\n        \n        # Transcribe\n        transcription = transcribe_audio_whisperx(audio_path, model, device_id)\n        \n        results.append({\n            'filename': file_id,\n            'transcript': transcription\n        })\n    \n    return results\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:34:18.306519Z","iopub.execute_input":"2026-02-22T17:34:18.306818Z","iopub.status.idle":"2026-02-22T17:34:18.314954Z","shell.execute_reply.started":"2026-02-22T17:34:18.306788Z","shell.execute_reply":"2026-02-22T17:34:18.314304Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Splitting the test data equally for inferencing them across the two gpus ","metadata":{}},{"cell_type":"code","source":"# PHASE 1: PREPARE DATA\n# ============================================================================\nprint(\"\\n\" + \"=\"*80)\nprint(\"PHASE 1: PREPARING DATA\")\nprint(\"=\"*80)\n\n# Get all audio files\nprint(f\"\\nScanning directory: {INPUT_DIR}\")\nall_audio_files = get_audio_files(INPUT_DIR)\nprint(f\"✓ Found {len(all_audio_files)} audio files\")\n\nif len(all_audio_files) == 0:\n    raise ValueError(f\"No audio files found in {INPUT_DIR}\")\n\n# Split files between two GPUs\nsplit_idx = len(all_audio_files) // 2\ngroup1_files = all_audio_files[:split_idx]\ngroup2_files = all_audio_files[split_idx:]\n\nprint(f\"\\n✓ Group 1 (GPU 0): {len(group1_files)} files\")\nprint(f\"✓ Group 2 (GPU 1): {len(group2_files)} files\")","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:34:20.947206Z","iopub.execute_input":"2026-02-22T17:34:20.947561Z","iopub.status.idle":"2026-02-22T17:34:20.991581Z","shell.execute_reply.started":"2026-02-22T17:34:20.947533Z","shell.execute_reply":"2026-02-22T17:34:20.991007Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\npath = \"/kaggle/working/whisper-medium-ct2\"\nprint(\"Exists:\", os.path.exists(path))\nprint(\"Is dir:\", os.path.isdir(path))\nif os.path.isdir(path):\n    print(\"Contents:\", os.listdir(path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-22T17:34:23.684075Z","iopub.execute_input":"2026-02-22T17:34:23.684472Z","iopub.status.idle":"2026-02-22T17:34:23.689606Z","shell.execute_reply.started":"2026-02-22T17:34:23.684438Z","shell.execute_reply":"2026-02-22T17:34:23.689035Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Loading the model on two gpus**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# PHASE 2: LOAD DUAL GPU MODELS\n# ============================================================================\nprint(\"\\n\" + \"=\"*80)\nprint(\"PHASE 2: LOADING DUAL GPU MODELS\")\nprint(\"=\"*80)\n\nprint(\"\\nLoading WhisperX model on GPU 0...\")\nmodel_gpu0 = whisperx.load_model(\n    MODEL,\n    device=\"cuda\",\n    device_index=0,\n    compute_type=COMPUTE_TYPE,\n    language=LANGUAGE,\n    asr_options={\n        \"beam_size\" : NUM_BEAMS\n    }\n\n)\nprint(\"✓ Model loaded on GPU 0 (cuda:0)\")\n\nprint(\"\\nLoading WhisperX model on GPU 1...\")\nmodel_gpu1 = whisperx.load_model(\n    MODEL,\n    device=\"cuda\",\n    device_index=1,\n    compute_type=COMPUTE_TYPE,\n    language=LANGUAGE,\n    asr_options={\n        \"beam_size\" : NUM_BEAMS\n    }\n\n)\nprint(\"✓ Model loaded on GPU 1 (cuda:1)\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:34:25.559337Z","iopub.execute_input":"2026-02-22T17:34:25.559815Z","iopub.status.idle":"2026-02-22T17:34:27.593214Z","shell.execute_reply.started":"2026-02-22T17:34:25.559785Z","shell.execute_reply":"2026-02-22T17:34:27.592661Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Inferencing using a worker for each gpu**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# PHASE 3: RUN PARALLEL TRANSCRIPTION\n# ============================================================================\nprint(\"\\n\" + \"=\"*80)\nprint(\"PHASE 3: RUNNING PARALLEL TRANSCRIPTION\")\nprint(\"=\"*80)\n\nprint(\"\\nStarting parallel transcription on both GPUs...\\n\")\n\nresults_gpu0 = []\nresults_gpu1 = []\n\nwith ThreadPoolExecutor(max_workers=2) as executor:\n    future_gpu0 = executor.submit(process_chunk_group, group1_files, model_gpu0, 0)\n    future_gpu1 = executor.submit(process_chunk_group, group2_files, model_gpu1, 1)\n    \n    results_gpu0 = future_gpu0.result()\n    results_gpu1 = future_gpu1.result()\n\n# Combine results\nall_results = results_gpu0 + results_gpu1\nprint(f\"\\n✓ Transcription complete! Processed {len(all_results)} files\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T17:34:30.173606Z","iopub.execute_input":"2026-02-22T17:34:30.173877Z","iopub.status.idle":"2026-02-22T18:01:00.254891Z","shell.execute_reply.started":"2026-02-22T17:34:30.173852Z","shell.execute_reply":"2026-02-22T18:01:00.254001Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  **Creating the submission file**","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# PHASE 4: SAVE RESULTS (MULTIPLE VERSIONS)\n# ============================================================================\nimport unicodedata\nimport re\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"PHASE 4: SAVING RESULTS\")\nprint(\"=\"*80)\n\n# -------------------------------\n# Normalization function\n# -------------------------------\nZW = r\"[\\u200B-\\u200D\\uFEFF]\"  # zero-width space/joiners + BOM\n\ndef normalize_bn_text(s: str, strip_arabic: bool = False) -> str:\n    if s is None:\n        return \"\"\n    s = str(s)\n    s = unicodedata.normalize(\"NFC\", s)\n    s = re.sub(ZW, \"\", s)\n    s = re.sub(r\">>\", \"\", s)\n    if strip_arabic:\n        s = re.sub(r\"[\\u0600-\\u06FF\\u0750-\\u077F\\u08A0-\\u08FF\\uFB50-\\uFDFF\\uFE70-\\uFEFF]+\", \" \", s)\n    # remove punctuation (keep Bengali, Hindi, Latin, digits, space)\n    s = re.sub(r\"[^\\u0980-\\u09FF\\u0900-\\u0963\\u0966-\\u097FA-Za-z0-9 ]+\", \" \", s)\n    s = \" \".join(s.split())\n    return s\n\n# -------------------------------\n# Create DataFrame\n# -------------------------------\ndf = pd.DataFrame(all_results)\n\n# Sort by filename to ensure consistent ordering\ndf = df.sort_values('filename').reset_index(drop=True)\n\ndf_normalized = df.copy()\ndf_normalized['transcript'] = df_normalized['transcript'].apply(normalize_bn_text)\n\n# create new filename\nname, ext = os.path.splitext(OUTPUT_FILE)\nnormalized_file = f\"{name}{ext}\"\n\noutput_path_normalized = os.path.join(OUTPUT_DIR, normalized_file)\ndf_normalized.to_csv(output_path_normalized, index=False)\n\nprint(f\"✓ Normalized results saved to: {output_path_normalized}\")\n\n# -------------------------------\n# Display sample results\n# -------------------------------\nprint(\"\\nSample results (Original):\")\nprint(df.head(10))\n\nprint(\"\\nSample results (Normalized):\")\nprint(df_normalized.head(10))\n\n# -------------------------------\n# Statistics\n# -------------------------------\nprint(\"\\nTranscription Statistics (Original):\")\nprint(f\"Total files: {len(df)}\")\nprint(f\"Empty transcriptions: {sum(df['transcript'] == '')}\")\nprint(f\"Average transcription length: {df['transcript'].str.len().mean():.2f} characters\")\n\nprint(\"\\nTranscription Statistics (Normalized):\")\nprint(f\"Empty transcriptions: {sum(df_normalized['transcript'] == '')}\")\nprint(f\"Average transcription length: {df_normalized['transcript'].str.len().mean():.2f} characters\")\n","metadata":{"execution":{"iopub.status.busy":"2026-02-22T18:01:00.256428Z","iopub.execute_input":"2026-02-22T18:01:00.256711Z","iopub.status.idle":"2026-02-22T18:01:00.399104Z","shell.execute_reply.started":"2026-02-22T18:01:00.256676Z","shell.execute_reply":"2026-02-22T18:01:00.398538Z"},"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}