{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":52324,"databundleVersionId":6229904,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":6707460,"datasetId":3865741,"databundleVersionId":6791840},{"sourceType":"kernelVersion","sourceId":297673355,"isSourceIdPinned":false}],"dockerImageVersionId":31400,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -U torch torchvision torchaudio -q --index-url https://download.pytorch.org/whl/cu121","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:09:20.916175Z","iopub.execute_input":"2026-06-06T15:09:20.916474Z","iopub.status.idle":"2026-06-06T15:09:22.627762Z","shell.execute_reply.started":"2026-06-06T15:09:20.91645Z","shell.execute_reply":"2026-06-06T15:09:22.626675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\ntorch.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:09:01.699201Z","iopub.execute_input":"2026-06-06T15:09:01.69965Z","iopub.status.idle":"2026-06-06T15:09:01.704723Z","shell.execute_reply.started":"2026-06-06T15:09:01.699619Z","shell.execute_reply":"2026-06-06T15:09:01.704115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ndataset = pd.read_csv(\"/kaggle/input/competitions/bengaliai-speech/train.csv\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#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","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.iloc[0]['id']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset.iloc[0]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import WhisperFeatureExtractor\n\nfeature_extractor = WhisperFeatureExtractor.from_pretrained(\"openai/whisper-medium\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import WhisperTokenizer\ntokenizer = WhisperTokenizer.from_pretrained(\"openai/whisper-medium\", task=\"transcribe\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_str = dataset.iloc[0][\"sentence\"]\nlabels = tokenizer(input_str).input_ids\ndecoded_with_special = tokenizer.decode(labels, skip_special_tokens=False)\ndecoded_str = tokenizer.decode(labels, skip_special_tokens=True)\n\nprint(f\"Input:                 {input_str}\")\nprint(f\"Decoded w/ special:    {decoded_with_special}\")\nprint(f\"Decoded w/out special: {decoded_str}\")\nprint(f\"Are equal:             {input_str == decoded_str}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import WhisperProcessor\n\nprocessor = WhisperProcessor.from_pretrained(\"openai/whisper-medium\", task=\"transcribe\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import transformers.modeling_utils\n# Core Fix: Patch the safety checker directly inside the modeling module\ntransformers.modeling_utils.check_torch_load_is_safe = lambda: None\n\nimport os\nimport pandas as pd\nimport torch\nimport librosa\nfrom tqdm import tqdm\nfrom transformers import WhisperProcessor, WhisperForConditionalGeneration\n\nAUDIO_DIR = \"/kaggle/input/competitions/bengaliai-speech/train_mp3s\"\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n# Load processor and model from your local Kaggle dataset path\nprocessor = WhisperProcessor.from_pretrained(\n    \"/kaggle/input/datasets/bond005/whisper-medium-bengali\",\n    task=\"transcribe\"\n)\n\nmodel = WhisperForConditionalGeneration.from_pretrained(\n    \"/kaggle/input/datasets/bond005/whisper-medium-bengali\"\n).to(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:12:06.001892Z","iopub.execute_input":"2026-06-06T15:12:06.002654Z","iopub.status.idle":"2026-06-06T15:12:26.911654Z","shell.execute_reply.started":"2026-06-06T15:12:06.002614Z","shell.execute_reply":"2026-06-06T15:12:26.910988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_files = sorted(\n    f for f in os.listdir(AUDIO_DIR)\n    if f.endswith(\".mp3\")\n)\n\nprint(f\"Total files: {len(all_files)}\")\nprint(all_files[:5])  # sanity check","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T14:17:10.693453Z","iopub.execute_input":"2026-06-06T14:17:10.694102Z","iopub.status.idle":"2026-06-06T14:17:21.138308Z","shell.execute_reply.started":"2026-06-06T14:17:10.694026Z","shell.execute_reply":"2026-06-06T14:17:21.137544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def transcribe_audio(audio_path):\n    audio, _ = librosa.load(audio_path, sr=16000)\n\n    inputs = processor(\n        audio,\n        sampling_rate=16000,\n        return_tensors=\"pt\"\n    )\n\n    with torch.no_grad():\n        predicted_ids = model.generate(\n            inputs.input_features.to(device),\n            language=\"bengali\",  # Forces Bengali token & script\n            task=\"transcribe\",\n        )\n\n    text = processor.batch_decode(\n        predicted_ids,\n        skip_special_tokens=True\n    )[0]\n    print(f\"text: {text}\")\n\n    return text.strip()\n\n\n# 1. Load the train metadata once\nprint(\"Loading train.csv for ground-truth mapping...\")\ntrain_df = pd.read_csv(\"/kaggle/input/competitions/bengaliai-speech/train.csv\")\n\n# 2. Build an efficient lookup dictionary mapping 'id' -> 'sentence'\nid_to_gt = dict(zip(train_df['id'].astype(str), train_df['sentence']))\nprint(f\"Loaded {len(id_to_gt)} ground-truth sentences.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:22:19.510192Z","iopub.execute_input":"2026-06-06T15:22:19.510927Z","iopub.status.idle":"2026-06-06T15:22:23.896765Z","shell.execute_reply.started":"2026-06-06T15:22:19.510896Z","shell.execute_reply":"2026-06-06T15:22:23.895956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef transcribe_range(start_idx, end_idx, output_csv):\n    selected_files = all_files[start_idx:end_idx + 1]\n    results = []\n\n    for fname in tqdm(selected_files):\n        audio_path = os.path.join(AUDIO_DIR, fname)\n\n        try:\n            text = transcribe_audio(audio_path)\n\n            # Extract the raw ID from filename (e.g., \"000005f3362c.mp3\" -> \"000005f3362c\")\n            audio_id = os.path.splitext(fname)[0]\n            \n            # Look up the ground truth sentence (defaults to empty string if missing)\n            gt_text = id_to_gt.get(audio_id, \"\")\n\n            results.append({\n                \"filename\": fname,\n                \"pred_sentence\": text,\n                \"gt_sentence\": gt_text  # Added ground truth under column name 'gt'\n            })\n\n        except Exception as e:\n            print(f\"Error processing {fname}: {e}\")\n\n    pd.DataFrame(results).to_csv(\n        output_csv,\n        index=False,\n        encoding=\"utf-8-sig\"\n    )\n\n    print(f\"Saved {len(results)} transcriptions to {output_csv}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:23:34.528506Z","iopub.execute_input":"2026-06-06T15:23:34.52926Z","iopub.status.idle":"2026-06-06T15:23:34.535024Z","shell.execute_reply.started":"2026-06-06T15:23:34.529228Z","shell.execute_reply":"2026-06-06T15:23:34.534129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntranscribe_range(\n    start_idx=9001,\n    end_idx=16000,\n    output_csv=\"predictions_1001_9000.csv\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T15:33:01.338349Z","iopub.execute_input":"2026-06-06T15:33:01.339102Z","iopub.status.idle":"2026-06-06T16:11:00.041857Z","shell.execute_reply.started":"2026-06-06T15:33:01.33907Z","shell.execute_reply":"2026-06-06T16:11:00.040927Z"}},"outputs":[],"execution_count":null}]}