{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":52324,"databundleVersionId":6229904,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":6707460,"datasetId":3865741,"databundleVersionId":6791840},{"sourceType":"kernelVersion","sourceId":297673355,"isSourceIdPinned":false}],"dockerImageVersionId":31400,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\n\n# Trick the datasets library into thinking torchcodec is missing\nsys.modules['torchcodec'] = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T14:34:30.135225Z","iopub.execute_input":"2026-06-22T14:34:30.135539Z","iopub.status.idle":"2026-06-22T14:34:30.143678Z","shell.execute_reply.started":"2026-06-22T14:34:30.135508Z","shell.execute_reply":"2026-06-22T14:34:30.142634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get update && apt-get install -y libsndfile1 -q\n!pip install -U torch torchvision torchaudio datasets soundfile librosa -q --index-url https://download.pytorch.org/whl/cu121","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T14:34:32.460725Z","iopub.execute_input":"2026-06-22T14:34:32.461039Z","iopub.status.idle":"2026-06-22T14:37:47.57065Z","shell.execute_reply.started":"2026-06-22T14:34:32.461013Z","shell.execute_reply":"2026-06-22T14:37:47.569341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\ntorch.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T14:55:42.027607Z","iopub.execute_input":"2026-06-22T14:55:42.028411Z","iopub.status.idle":"2026-06-22T14:55:44.132354Z","shell.execute_reply.started":"2026-06-22T14:55:42.028371Z","shell.execute_reply":"2026-06-22T14:55:44.131383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import load_dataset\n\n# Load the Bengali (bn_in) configuration of the FLEURS dataset\nfleurs = load_dataset(\"google/fleurs\", \"bn_in\")\n\n# Take a look at the splits available (train, validation, test)\nprint(fleurs)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-22T14:55:47.125856Z","iopub.execute_input":"2026-06-22T14:55:47.126767Z","iopub.status.idle":"2026-06-22T14:56:52.33502Z","shell.execute_reply.started":"2026-06-22T14:55:47.126735Z","shell.execute_reply":"2026-06-22T14:56:52.333904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fleurs[\"train\"][\"transcription\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T14:56:56.754231Z","iopub.execute_input":"2026-06-22T14:56:56.755325Z","iopub.status.idle":"2026-06-22T14:56:56.764749Z","shell.execute_reply.started":"2026-06-22T14:56:56.755292Z","shell.execute_reply":"2026-06-22T14:56:56.764036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#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","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import transformers.utils.import_utils\nimport transformers.modeling_utils\n\n# Patch it in both the import utils and modeling utils to be completely safe\ntransformers.utils.import_utils.check_torch_load_is_safe = lambda: None\ntransformers.modeling_utils.check_torch_load_is_safe = lambda: None\n\n\nimport os\nimport torch\nimport pandas as pd\nfrom tqdm import tqdm\nfrom transformers import WhisperProcessor, WhisperForConditionalGeneration\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n\n# Load your local model or fallback to standard repository if needed\nmodel_path = \"/kaggle/input/datasets/bond005/whisper-medium-bengali\"\n\nprocessor = WhisperProcessor.from_pretrained(model_path, task=\"transcribe\")\nmodel = WhisperForConditionalGeneration.from_pretrained(model_path).to(device)\n\nimport io\nimport soundfile as sf\nimport torch\n\nimport io\nimport soundfile as sf\nimport torch\n\ndef transcribe_fleurs_sample(sample):\n    # Safe check to extract audio bytes or load from file path\n    audio_data = sample[\"audio\"]\n    \n    if \"bytes\" in audio_data and audio_data[\"bytes\"] is not None:\n        audio_bytes = audio_data[\"bytes\"]\n        audio_array, sampling_rate = sf.read(io.BytesIO(audio_bytes))\n    elif \"path\" in audio_data and audio_data[\"path\"] is not None:\n        # Fallback to reading from local cache file path\n        audio_array, sampling_rate = sf.read(audio_data[\"path\"])\n    else:\n        raise ValueError(\"Could not find raw bytes or a valid audio file path in the sample.\")\n    \n    # If the audio is stereo, convert to mono by averaging channels\n    if len(audio_array.shape) > 1:\n        audio_array = audio_array.mean(axis=-1)\n        \n    # Whisper requires 16000Hz sampling rate\n    if sampling_rate != 16000:\n        import librosa\n        audio_array = librosa.resample(audio_array, orig_sr=sampling_rate, target_sr=16000)\n        sampling_rate = 16000\n\n    inputs = processor(\n        audio_array,\n        sampling_rate=sampling_rate,\n        return_tensors=\"pt\"\n    )\n\n    with torch.no_grad():\n        predicted_ids = model.generate(\n            inputs.input_features.to(device),\n            language=\"bengali\", \n            task=\"transcribe\",\n        )\n\n    text = processor.batch_decode(\n        predicted_ids,\n        skip_special_tokens=True\n    )[0]\n\n    return text.strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T15:01:58.535494Z","iopub.execute_input":"2026-06-22T15:01:58.536204Z","iopub.status.idle":"2026-06-22T15:02:01.713551Z","shell.execute_reply.started":"2026-06-22T15:01:58.536171Z","shell.execute_reply":"2026-06-22T15:02:01.712765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def transcribe_audio(audio_path):\n#     audio, _ = librosa.load(audio_path, sr=16000)\n\n#     inputs = processor(\n#         audio,\n#         sampling_rate=16000,\n#         return_tensors=\"pt\"\n#     )\n\n#     with torch.no_grad():\n#         predicted_ids = model.generate(\n#             inputs.input_features.to(device),\n#             language=\"bengali\",  # Forces Bengali token & script\n#             task=\"transcribe\",\n#         )\n\n#     text = processor.batch_decode(\n#         predicted_ids,\n#         skip_special_tokens=True\n#     )[0]\n#     print(f\"text: {text}\")\n\n#     return text.strip()\n\n\n# # 1. Load the train metadata once\n# print(\"Loading train.csv for ground-truth mapping...\")\n# train_df = pd.read_csv(\"/kaggle/input/competitions/bengaliai-speech/train.csv\")\n\n# # 2. Build an efficient lookup dictionary mapping 'id' -> 'sentence'\n# id_to_gt = dict(zip(train_df['id'].astype(str), train_df['sentence']))\n# print(f\"Loaded {len(id_to_gt)} ground-truth sentences.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import Audio\n\ndef process_fleurs_range(split_name, start_idx, end_idx, output_csv):\n    # Select the split\n    dataset_split = fleurs[split_name]\n    \n    # Alternative to decode_audio(False): Cast the audio column to NOT decode automatically\n    dataset_split = dataset_split.cast_column(\"audio\", Audio(decode=False))\n    \n    indices = range(start_idx, min(end_idx + 1, len(dataset_split)))\n    results = []\n\n    print(f\"Processing indices {start_idx} to {end_idx} from the '{split_name}' split...\")\n    \n    for idx in tqdm(indices):\n        sample = dataset_split[idx]\n        try:\n            pred_text = transcribe_fleurs_sample(sample)\n            \n            results.append({\n                \"id\": sample[\"id\"],\n                \"filename\": os.path.basename(sample[\"path\"]) if sample.get(\"path\") else f\"sample_{idx}.wav\",\n                \"pred_sentence\": pred_text,\n                \"gt_sentence\": sample[\"transcription\"]\n            })\n        except Exception as e:\n            print(f\"Error processing index {idx}: {e}\")\n\n    df = pd.DataFrame(results)\n    df.to_csv(output_csv, index=False, encoding=\"utf-8-sig\")\n    print(f\"Saved {len(results)} transcriptions to {output_csv}\")\n\n# --- EXECUTE THE RANGE ---\n# Example: Transcribe the first 500 samples of the test split\nprocess_fleurs_range(\n    split_name=\"test\", \n    start_idx=0, \n    end_idx=499, \n    output_csv=\"fleurs_test_predictions.csv\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T15:02:09.277985Z","iopub.execute_input":"2026-06-22T15:02:09.278886Z","iopub.status.idle":"2026-06-22T15:50:55.476509Z","shell.execute_reply.started":"2026-06-22T15:02:09.278809Z","shell.execute_reply":"2026-06-22T15:50:55.475669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}