{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":73047,"databundleVersionId":8823072,"sourceType":"competition"},{"sourceId":6707460,"sourceType":"datasetVersion","datasetId":3865741,"isSourceIdPinned":false},{"sourceId":13660018,"sourceType":"datasetVersion","datasetId":8684865}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n        print(os.path.join(dirname))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:49:50.18982Z","iopub.execute_input":"2025-11-08T22:49:50.190532Z","iopub.status.idle":"2025-11-08T22:50:12.39159Z","shell.execute_reply.started":"2025-11-08T22:49:50.190506Z","shell.execute_reply":"2025-11-08T22:50:12.390908Z"}},"outputs":[{"name":"stdout","text":"/kaggle/input\n/kaggle/input/python-packages2\n/kaggle/input/bengali-ai-asr-submission\n/kaggle/input/bengali-ai-asr-submission/punct-model-11layers\n/kaggle/input/bengali-ai-asr-submission/punct-model-6layers\n/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium\n/kaggle/input/bengali-ai-asr-submission/punct-model-8layers\n/kaggle/input/bengali-ai-asr-submission/punct-model-12layers\n/kaggle/input/ben10\n/kaggle/input/ben10/ben10\n/kaggle/input/ben10/ben10/16_kHz_train_audio\n/kaggle/input/ben10/ben10/16_kHz_valid_audio\n","output_type":"stream"}],"execution_count":1},{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.dataset_download(\"tugstugi/bengali-ai-asr-submission\")\n\nprint(\"Path to dataset files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:50:36.541623Z","iopub.execute_input":"2025-11-08T22:50:36.542479Z","iopub.status.idle":"2025-11-08T22:50:36.821237Z","shell.execute_reply.started":"2025-11-08T22:50:36.54245Z","shell.execute_reply":"2025-11-08T22:50:36.820609Z"}},"outputs":[{"name":"stdout","text":"Path to dataset files: /kaggle/input/bengali-ai-asr-submission\n","output_type":"stream"}],"execution_count":2},{"cell_type":"code","source":"!cp /kaggle/input/bengali-eval-data/predict.py .\n\n!cp -r ../input/python-packages ./\n!tar xvfz ./python-packages/jiwer.tgz\n!pip install ./jiwer/python-Levenshtein-0.12.2.tar.gz -f ./ --no-index\n!pip install ./jiwer/jiwer-2.3.0-py3-none-any.whl -f ./ --no-index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:50:40.368306Z","iopub.execute_input":"2025-11-08T22:50:40.368878Z","iopub.status.idle":"2025-11-08T22:50:43.675587Z","shell.execute_reply.started":"2025-11-08T22:50:40.368852Z","shell.execute_reply":"2025-11-08T22:50:43.674577Z"}},"outputs":[{"name":"stdout","text":"cp: cannot stat '/kaggle/input/bengali-eval-data/predict.py': No such file or directory\ncp: cannot stat '../input/python-packages': No such file or directory\ntar (child): ./python-packages/jiwer.tgz: Cannot open: No such file or directory\ntar (child): Error is not recoverable: exiting now\ntar: Child returned status 2\ntar: Error is not recoverable: exiting now\n\u001b[33mWARNING: Requirement './jiwer/python-Levenshtein-0.12.2.tar.gz' looks like a filename, but the file does not exist\u001b[0m\u001b[33m\n\u001b[0mLooking in links: ./\nProcessing ./jiwer/python-Levenshtein-0.12.2.tar.gz\n\u001b[31mERROR: Could not install packages due to an OSError: [Errno 2] No such file or directory: '/kaggle/working/jiwer/python-Levenshtein-0.12.2.tar.gz'\n\u001b[0m\u001b[31m\n\u001b[0m\u001b[33mWARNING: Requirement './jiwer/jiwer-2.3.0-py3-none-any.whl' looks like a filename, but the file does not exist\u001b[0m\u001b[33m\n\u001b[0mLooking in links: ./\nProcessing ./jiwer/jiwer-2.3.0-py3-none-any.whl\n\u001b[31mERROR: Could not install packages due to an OSError: [Errno 2] No such file or directory: '/kaggle/working/jiwer/jiwer-2.3.0-py3-none-any.whl'\n\u001b[0m\u001b[31m\n\u001b[0m","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"import os\nimport csv\nimport time\nimport glob\n\nMODEL = '/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium/'\nPUNCT_MODELS = [\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-6layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-8layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-11layers/',\n    '/kaggle/input/bengali-ai-asr-submission/punct-model-12layers/'\n]\nCHUNK_LENGTH_S = 20.1\nENABLE_BEAM = True\n# none alone 0.9 0.037964275588318684\n# none alone 0.7 0.03592288063510065\n# none alone 0.4 0.03416501275871846\nPUNCT_WEIGHTS = [[1.0, 1.4, 1.0, 0.8]]\n\nif ENABLE_BEAM:\n    BATCH_SIZE = 4\nelse:\n    BATCH_SIZE = 8\n\nif len(glob.glob(\"/kaggle/input/bengaliai-speech/test_mp3s/*.mp3\")) > 10:\n    EVAL = False\n    DATASET_PATH = '/kaggle/input/bengaliai-speech/test_mp3s/'\nelse:\n    EVAL = True\n    DATASET_PATH = '/kaggle/input/bengaliai-speech/test_mp3s/'\n    \nimport csv\nimport glob\nimport shutil\nimport librosa\nimport argparse\nimport warnings\nfrom pathlib import Path\nimport transformers\nprint(transformers.__version__)\nfrom transformers import pipeline, AutoModelForTokenClassification, AutoTokenizer\n\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n\nfiles = list(glob.glob(DATASET_PATH + '/' + '*.wav'))\nfiles += list(glob.glob(DATASET_PATH + '/' + '*.mp3'))\nfiles.sort()\n\npipe = pipeline(task=\"automatic-speech-recognition\",\n                model=MODEL,\n                tokenizer=MODEL,\n                chunk_length_s=CHUNK_LENGTH_S, device=0, batch_size=BATCH_SIZE)\npipe.model.config.forced_decoder_ids = pipe.tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\nprint(\"model loaded!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:50:57.58215Z","iopub.execute_input":"2025-11-08T22:50:57.582444Z","iopub.status.idle":"2025-11-08T22:51:10.399712Z","shell.execute_reply.started":"2025-11-08T22:50:57.582413Z","shell.execute_reply":"2025-11-08T22:51:10.399105Z"}},"outputs":[{"name":"stdout","text":"4.53.3\n","output_type":"stream"},{"name":"stderr","text":"2025-11-08 22:51:01.129260: E external/local_xla/xla/stream_executor/cuda/cuda_fft.cc:477] Unable to register cuFFT factory: Attempting to register factory for plugin cuFFT when one has already been registered\nWARNING: All log messages before absl::InitializeLog() is called are written to STDERR\nE0000 00:00:1762642261.152306     527 cuda_dnn.cc:8310] Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\nE0000 00:00:1762642261.159280     527 cuda_blas.cc:1418] Unable to register cuBLAS factory: Attempting to register factory for plugin cuBLAS when one has already been registered\n","output_type":"stream"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;31mAttributeError\u001b[0m: 'MessageFactory' object has no attribute 'GetPrototype'"],"ename":"AttributeError","evalue":"'MessageFactory' object has no attribute 'GetPrototype'","output_type":"error"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;31mAttributeError\u001b[0m: 'MessageFactory' object has no attribute 'GetPrototype'"],"ename":"AttributeError","evalue":"'MessageFactory' object has no attribute 'GetPrototype'","output_type":"error"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;31mAttributeError\u001b[0m: 'MessageFactory' object has no attribute 'GetPrototype'"],"ename":"AttributeError","evalue":"'MessageFactory' object has no attribute 'GetPrototype'","output_type":"error"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;31mAttributeError\u001b[0m: 'MessageFactory' object has no attribute 'GetPrototype'"],"ename":"AttributeError","evalue":"'MessageFactory' object has no attribute 'GetPrototype'","output_type":"error"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;31mAttributeError\u001b[0m: 'MessageFactory' object has no attribute 'GetPrototype'"],"ename":"AttributeError","evalue":"'MessageFactory' object has no attribute 'GetPrototype'","output_type":"error"},{"name":"stderr","text":"Device set to use cuda:0\nUsing `chunk_length_s` is very experimental with seq2seq models. The results will not necessarily be entirely accurate and will have caveats. More information: https://github.com/huggingface/transformers/pull/20104. Ignore this warning with pipeline(..., ignore_warning=True). To use Whisper for long-form transcription, use rather the model's `generate` method directly as the model relies on it's own chunking mechanism (cf. Whisper original paper, section 3.8. Long-form Transcription).\n","output_type":"stream"},{"name":"stdout","text":"model loaded!\n","output_type":"stream"}],"execution_count":4},{"cell_type":"code","source":"!pip install audiomentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:51:22.535022Z","iopub.execute_input":"2025-11-08T22:51:22.535633Z","iopub.status.idle":"2025-11-08T22:52:06.154302Z","shell.execute_reply.started":"2025-11-08T22:51:22.535608Z","shell.execute_reply":"2025-11-08T22:52:06.153238Z"}},"outputs":[{"name":"stdout","text":"Collecting audiomentations\n  Using cached audiomentations-0.43.1-py3-none-any.whl.metadata (11 kB)\nRequirement already satisfied: numpy<3,>=1.22.0 in /usr/local/lib/python3.11/dist-packages (from audiomentations) (1.26.4)\nCollecting numpy-minmax<1,>=0.3.0 (from audiomentations)\n  Using cached numpy_minmax-0.5.0-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (4.0 kB)\nCollecting numpy-rms<1,>=0.4.2 (from audiomentations)\n  Using cached numpy_rms-0.6.0-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (3.5 kB)\nRequirement already satisfied: librosa!=0.10.0,<0.12.0,>=0.8.0 in /usr/local/lib/python3.11/dist-packages (from audiomentations) (0.11.0)\nCollecting python-stretch<1,>=0.3.1 (from audiomentations)\n  Using cached python_stretch-0.3.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (3.7 kB)\nRequirement already satisfied: scipy<2,>=1.4 in /usr/local/lib/python3.11/dist-packages (from audiomentations) (1.15.3)\nRequirement already satisfied: soxr<1.0.0,>=0.3.2 in /usr/local/lib/python3.11/dist-packages (from audiomentations) (0.5.0.post1)\nRequirement already satisfied: audioread>=2.1.9 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (3.0.1)\nRequirement already satisfied: numba>=0.51.0 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (0.60.0)\nRequirement already satisfied: scikit-learn>=1.1.0 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (1.2.2)\nRequirement already satisfied: joblib>=1.0 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (1.5.2)\nRequirement already satisfied: decorator>=4.3.0 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (4.4.2)\nRequirement already satisfied: soundfile>=0.12.1 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (0.13.1)\nRequirement already satisfied: pooch>=1.1 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (1.8.2)\nRequirement already satisfied: typing_extensions>=4.1.1 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (4.15.0)\nRequirement already satisfied: lazy_loader>=0.1 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (0.4)\nRequirement already satisfied: msgpack>=1.0 in /usr/local/lib/python3.11/dist-packages (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (1.1.2)\nRequirement already satisfied: mkl_fft in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (1.3.8)\nRequirement already satisfied: mkl_random in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (1.2.4)\nRequirement already satisfied: mkl_umath in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (0.1.1)\nRequirement already satisfied: mkl in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (2025.3.0)\nRequirement already satisfied: tbb4py in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (2022.3.0)\nRequirement already satisfied: mkl-service in /usr/local/lib/python3.11/dist-packages (from numpy<3,>=1.22.0->audiomentations) (2.4.1)\nRequirement already satisfied: cffi>=1.0.0 in /usr/local/lib/python3.11/dist-packages (from numpy-minmax<1,>=0.3.0->audiomentations) (2.0.0)\nCollecting numpy<3,>=1.22.0 (from audiomentations)\n  Using cached numpy-2.3.4-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl.metadata (62 kB)\nRequirement already satisfied: pycparser in /usr/local/lib/python3.11/dist-packages (from cffi>=1.0.0->numpy-minmax<1,>=0.3.0->audiomentations) (2.23)\nRequirement already satisfied: packaging in /usr/local/lib/python3.11/dist-packages (from lazy_loader>=0.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (25.0)\nRequirement already satisfied: llvmlite<0.44,>=0.43.0dev0 in /usr/local/lib/python3.11/dist-packages (from numba>=0.51.0->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (0.43.0)\n  Using cached numpy-2.0.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (60 kB)\nRequirement already satisfied: platformdirs>=2.5.0 in /usr/local/lib/python3.11/dist-packages (from pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (4.5.0)\nRequirement already satisfied: requests>=2.19.0 in /usr/local/lib/python3.11/dist-packages (from pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (2.32.5)\nRequirement already satisfied: threadpoolctl>=2.0.0 in /usr/local/lib/python3.11/dist-packages (from scikit-learn>=1.1.0->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (3.6.0)\nRequirement already satisfied: charset_normalizer<4,>=2 in /usr/local/lib/python3.11/dist-packages (from requests>=2.19.0->pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (3.4.4)\nRequirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.11/dist-packages (from requests>=2.19.0->pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (3.11)\nRequirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.11/dist-packages (from requests>=2.19.0->pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (2.5.0)\nRequirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.11/dist-packages (from requests>=2.19.0->pooch>=1.1->librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations) (2025.10.5)\nRequirement already satisfied: onemkl-license==2025.3.0 in /usr/local/lib/python3.11/dist-packages (from mkl->numpy<3,>=1.22.0->audiomentations) (2025.3.0)\nRequirement already satisfied: intel-openmp<2026,>=2024 in /usr/local/lib/python3.11/dist-packages (from mkl->numpy<3,>=1.22.0->audiomentations) (2024.2.0)\nRequirement already satisfied: tbb==2022.* in /usr/local/lib/python3.11/dist-packages (from mkl->numpy<3,>=1.22.0->audiomentations) (2022.3.0)\nRequirement already satisfied: tcmlib==1.* in /usr/local/lib/python3.11/dist-packages (from tbb==2022.*->mkl->numpy<3,>=1.22.0->audiomentations) (1.4.0)\nRequirement already satisfied: intel-cmplr-lib-ur==2024.2.0 in /usr/local/lib/python3.11/dist-packages (from intel-openmp<2026,>=2024->mkl->numpy<3,>=1.22.0->audiomentations) (2024.2.0)\nINFO: pip is looking at multiple versions of mkl-fft to determine which version is compatible with other requirements. This could take a while.\nCollecting mkl_fft (from numpy<3,>=1.22.0->audiomentations)\n  Using cached mkl_fft-2.1.1-0-cp311-cp311-manylinux_2_28_x86_64.whl.metadata (7.3 kB)\n  Using cached mkl_fft-2.0.0-22-cp311-cp311-manylinux_2_28_x86_64.whl.metadata (7.1 kB)\n  Using cached mkl_fft-1.3.14-0-cp311-cp311-manylinux_2_28_x86_64.whl.metadata (4.8 kB)\n  Using cached mkl_fft-1.3.13-0-cp311-cp311-manylinux_2_28_x86_64.whl.metadata (4.8 kB)\n  Using cached mkl_fft-1.3.11-81-cp311-cp311-manylinux_2_28_x86_64.whl.metadata (4.4 kB)\nCollecting soundfile>=0.12.1 (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations)\n  Using cached soundfile-0.13.1-py2.py3-none-manylinux_2_28_x86_64.whl.metadata (16 kB)\nINFO: pip is still looking at multiple versions of mkl-fft to determine which version is compatible with other requirements. This could take a while.\n  Using cached soundfile-0.13.0-py2.py3-none-manylinux_2_28_x86_64.whl.metadata (16 kB)\nINFO: This is taking longer than usual. You might need to provide the dependency resolver with stricter constraints to reduce runtime. See https://pip.pypa.io/warnings/backtracking for guidance. If you want to abort this run, press Ctrl + C.\n  Using cached soundfile-0.12.1-py2.py3-none-manylinux_2_31_x86_64.whl.metadata (14 kB)\nCollecting scikit-learn>=1.1.0 (from librosa!=0.10.0,<0.12.0,>=0.8.0->audiomentations)\n  Using cached scikit_learn-1.7.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.7.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.7.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (17 kB)\n  Using cached scikit_learn-1.6.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (18 kB)\n  Using cached scikit_learn-1.6.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (18 kB)\n  Using cached scikit_learn-1.5.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (13 kB)\n  Using cached scikit_learn-1.5.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (12 kB)\n  Using cached scikit_learn-1.5.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.4.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.4.1.post1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\nINFO: pip is looking at multiple versions of scikit-learn to determine which version is compatible with other requirements. This could take a while.\n  Using cached scikit_learn-1.4.0-1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.3.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.3.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.3.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\nINFO: pip is still looking at multiple versions of scikit-learn to determine which version is compatible with other requirements. This could take a while.\n  Using cached scikit_learn-1.2.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.2.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\nINFO: This is taking longer than usual. You might need to provide the dependency resolver with stricter constraints to reduce runtime. See https://pip.pypa.io/warnings/backtracking for guidance. If you want to abort this run, press Ctrl + C.\n  Using cached scikit_learn-1.2.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (11 kB)\n  Using cached scikit_learn-1.1.3-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (10 kB)\n  Using cached scikit-learn-1.1.2.tar.gz (7.0 MB)\n  Installing build dependencies ... \u001b[?25l\u001b[?25hdone\n  Getting requirements to build wheel ... \u001b[?25l\u001b[?25hdone\n  \u001b[1;31merror\u001b[0m: \u001b[1msubprocess-exited-with-error\u001b[0m\n  \n  \u001b[31m×\u001b[0m \u001b[32mPreparing metadata \u001b[0m\u001b[1;32m(\u001b[0m\u001b[32mpyproject.toml\u001b[0m\u001b[1;32m)\u001b[0m did not run successfully.\n  \u001b[31m│\u001b[0m exit code: \u001b[1;36m1\u001b[0m\n  \u001b[31m╰─>\u001b[0m See above for output.\n  \n  \u001b[1;35mnote\u001b[0m: This error originates from a subprocess, and is likely not a problem with pip.\n  Preparing metadata (pyproject.toml) ... \u001b[?25l\u001b[?25herror\n\u001b[1;31merror\u001b[0m: \u001b[1mmetadata-generation-failed\u001b[0m\n\n\u001b[31m×\u001b[0m Encountered error while generating package metadata.\n\u001b[31m╰─>\u001b[0m See above for output.\n\n\u001b[1;35mnote\u001b[0m: This is an issue with the package mentioned above, not pip.\n\u001b[1;36mhint\u001b[0m: See above for details.\n","output_type":"stream"}],"execution_count":5},{"cell_type":"code","source":"!pip install noisereduce soundfile","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:23.029827Z","iopub.execute_input":"2025-11-08T22:52:23.030684Z","iopub.status.idle":"2025-11-08T22:52:26.302504Z","shell.execute_reply.started":"2025-11-08T22:52:23.030647Z","shell.execute_reply":"2025-11-08T22:52:26.301784Z"}},"outputs":[{"name":"stdout","text":"Requirement already satisfied: noisereduce in /usr/local/lib/python3.11/dist-packages (3.0.3)\nRequirement already satisfied: soundfile in /usr/local/lib/python3.11/dist-packages (0.13.1)\nRequirement already satisfied: scipy in /usr/local/lib/python3.11/dist-packages (from noisereduce) (1.15.3)\nRequirement already satisfied: matplotlib in /usr/local/lib/python3.11/dist-packages (from noisereduce) (3.7.2)\nRequirement already satisfied: numpy in /usr/local/lib/python3.11/dist-packages (from noisereduce) (1.26.4)\nRequirement already satisfied: tqdm in /usr/local/lib/python3.11/dist-packages (from noisereduce) (4.67.1)\nRequirement already satisfied: joblib in /usr/local/lib/python3.11/dist-packages (from noisereduce) (1.5.2)\nRequirement already satisfied: cffi>=1.0 in /usr/local/lib/python3.11/dist-packages (from soundfile) (2.0.0)\nRequirement already satisfied: pycparser in /usr/local/lib/python3.11/dist-packages (from cffi>=1.0->soundfile) (2.23)\nRequirement already satisfied: contourpy>=1.0.1 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (1.3.2)\nRequirement already satisfied: cycler>=0.10 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (0.12.1)\nRequirement already satisfied: fonttools>=4.22.0 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (4.59.0)\nRequirement already satisfied: kiwisolver>=1.0.1 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (1.4.8)\nRequirement already satisfied: packaging>=20.0 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (25.0)\nRequirement already satisfied: pillow>=6.2.0 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (11.3.0)\nRequirement already satisfied: pyparsing<3.1,>=2.3.1 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (3.0.9)\nRequirement already satisfied: python-dateutil>=2.7 in /usr/local/lib/python3.11/dist-packages (from matplotlib->noisereduce) (2.9.0.post0)\nRequirement already satisfied: mkl_fft in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (1.3.8)\nRequirement already satisfied: mkl_random in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (1.2.4)\nRequirement already satisfied: mkl_umath in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (0.1.1)\nRequirement already satisfied: mkl in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (2025.3.0)\nRequirement already satisfied: tbb4py in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (2022.3.0)\nRequirement already satisfied: mkl-service in /usr/local/lib/python3.11/dist-packages (from numpy->noisereduce) (2.4.1)\nRequirement already satisfied: six>=1.5 in /usr/local/lib/python3.11/dist-packages (from python-dateutil>=2.7->matplotlib->noisereduce) (1.17.0)\nRequirement already satisfied: onemkl-license==2025.3.0 in /usr/local/lib/python3.11/dist-packages (from mkl->numpy->noisereduce) (2025.3.0)\nRequirement already satisfied: intel-openmp<2026,>=2024 in /usr/local/lib/python3.11/dist-packages (from mkl->numpy->noisereduce) (2024.2.0)\nRequirement already satisfied: tbb==2022.* in /usr/local/lib/python3.11/dist-packages (from mkl->numpy->noisereduce) (2022.3.0)\nRequirement already satisfied: tcmlib==1.* in /usr/local/lib/python3.11/dist-packages (from tbb==2022.*->mkl->numpy->noisereduce) (1.4.0)\nRequirement already satisfied: intel-cmplr-lib-rt in /usr/local/lib/python3.11/dist-packages (from mkl_umath->numpy->noisereduce) (2024.2.0)\nRequirement already satisfied: intel-cmplr-lib-ur==2024.2.0 in /usr/local/lib/python3.11/dist-packages (from intel-openmp<2026,>=2024->mkl->numpy->noisereduce) (2024.2.0)\n","output_type":"stream"}],"execution_count":6},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport glob\nimport librosa\nimport noisereduce as nr\nimport soundfile as sf\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nprint(\"Libraries imported for preprocessing.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:30.151953Z","iopub.execute_input":"2025-11-08T22:52:30.152782Z","iopub.status.idle":"2025-11-08T22:52:30.181534Z","shell.execute_reply.started":"2025-11-08T22:52:30.152718Z","shell.execute_reply":"2025-11-08T22:52:30.180778Z"}},"outputs":[{"name":"stdout","text":"Libraries imported for preprocessing.\n","output_type":"stream"}],"execution_count":7},{"cell_type":"code","source":"# --- Input Paths ---\n# This is your new audio folder\nINPUT_AUDIO_PATH = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/\"\n# This is your new annotations file\nTRAIN_CSV_PATH = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/train.csv\"\n\n# --- Output Path (for this 30-sample test) ---\nOUTPUT_PATH_TEST = \"test_processed/\"\nos.makedirs(OUTPUT_PATH_TEST, exist_ok=True)\n\n# --- Amplification Parameter ---\nTARGET_PEAK_AMP = 0.9  # Normalize to 90% peak (loud but safe)\n\nprint(f\"Reading from: {INPUT_AUDIO_PATH}\")\nprint(f\"Reading CSV from: {TRAIN_CSV_PATH}\")\nprint(f\"Saving 30 test files to: {OUTPUT_PATH_TEST}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:33.235823Z","iopub.execute_input":"2025-11-08T22:52:33.237223Z","iopub.status.idle":"2025-11-08T22:52:33.24234Z","shell.execute_reply.started":"2025-11-08T22:52:33.237193Z","shell.execute_reply":"2025-11-08T22:52:33.24149Z"}},"outputs":[{"name":"stdout","text":"Reading from: /kaggle/input/ben10/ben10/16_kHz_train_audio/\nReading CSV from: /kaggle/input/ben10/ben10/16_kHz_train_audio/train.csv\nSaving 30 test files to: test_processed/\n","output_type":"stream"}],"execution_count":8},{"cell_type":"code","source":"# Load the full CSV\ndf_full = pd.read_csv(TRAIN_CSV_PATH)\n\n# Get 30 random samples\ndf_sample = df_full.sample(n=30, random_state=42)\n\nprint(f\"Loaded {len(df_full)} total samples.\")\nprint(f\"Processing a random sample of {len(df_sample)} files.\")\ndisplay(df_sample.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:37.885298Z","iopub.execute_input":"2025-11-08T22:52:37.885572Z","iopub.status.idle":"2025-11-08T22:52:38.034321Z","shell.execute_reply.started":"2025-11-08T22:52:37.885551Z","shell.execute_reply":"2025-11-08T22:52:38.03346Z"}},"outputs":[{"name":"stdout","text":"Loaded 13342 total samples.\nProcessing a random sample of 30 files.\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"                          file_name  \\\n5340         train_narail (155).wav   \n1368     train_chittagong (240).wav   \n10105         train_sylhet (16).wav   \n3545   train_kishoreganj (1363).wav   \n5466          train_narail (27).wav   \n\n                                          transcriptions     district  \n5340   যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি। আর আমাগে এ...       narail  \n1368   খরলা ইবে খত? ইবে খয়দে একশো বিশ টেঁয়া। ইয়্যত তফ...   chittagong  \n10105  ফরিছিত ও, তোর ব্যাছমেট বে <> আও <> ক্লাসো আও <...       sylhet  \n3545   তে আল্লার রহমতে গর করছি কত বছর! অহনো গিয়া দেহো...  kishoreganj  \n5466   এহন যদি না হয়। তালি তো জিনিসটা আসলেই মর্মান্তি...       narail  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>file_name</th>\n      <th>transcriptions</th>\n      <th>district</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>5340</th>\n      <td>train_narail (155).wav</td>\n      <td>যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি। আর আমাগে এ...</td>\n      <td>narail</td>\n    </tr>\n    <tr>\n      <th>1368</th>\n      <td>train_chittagong (240).wav</td>\n      <td>খরলা ইবে খত? ইবে খয়দে একশো বিশ টেঁয়া। ইয়্যত তফ...</td>\n      <td>chittagong</td>\n    </tr>\n    <tr>\n      <th>10105</th>\n      <td>train_sylhet (16).wav</td>\n      <td>ফরিছিত ও, তোর ব্যাছমেট বে &lt;&gt; আও &lt;&gt; ক্লাসো আও &lt;...</td>\n      <td>sylhet</td>\n    </tr>\n    <tr>\n      <th>3545</th>\n      <td>train_kishoreganj (1363).wav</td>\n      <td>তে আল্লার রহমতে গর করছি কত বছর! অহনো গিয়া দেহো...</td>\n      <td>kishoreganj</td>\n    </tr>\n    <tr>\n      <th>5466</th>\n      <td>train_narail (27).wav</td>\n      <td>এহন যদি না হয়। তালি তো জিনিসটা আসলেই মর্মান্তি...</td>\n      <td>narail</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":9},{"cell_type":"code","source":"import string\nimport re \n\n# Bangla punctuation marks to remove\nbangla_punctuation = \"<.>!।()[]{}\\,?_~\"\n\n# Function to remove punctuation marks\ndef remove_punctuation(text):\n    translator = str.maketrans('', '', bangla_punctuation)\n    text = text.translate(translator)\n    # Remove single dots not part of a sequence of continuous dots\n    #text = re.sub(r'(?<!\\.)\\.(?!\\.)', '', text)\n    return text\n\n# Apply the function to remove punctuation marks\ndf_sample[\"transcriptions\"] = df_sample[\"transcriptions\"].apply(remove_punctuation)\n\ndf_sample.head(20)\ndf_sample['transcriptions'].to_csv(\"dots.csv\",index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:40.714876Z","iopub.execute_input":"2025-11-08T22:52:40.715164Z","iopub.status.idle":"2025-11-08T22:52:40.724296Z","shell.execute_reply.started":"2025-11-08T22:52:40.715143Z","shell.execute_reply":"2025-11-08T22:52:40.723697Z"}},"outputs":[],"execution_count":10},{"cell_type":"code","source":"df_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:43.815292Z","iopub.execute_input":"2025-11-08T22:52:43.81557Z","iopub.status.idle":"2025-11-08T22:52:43.823912Z","shell.execute_reply.started":"2025-11-08T22:52:43.815549Z","shell.execute_reply":"2025-11-08T22:52:43.823121Z"}},"outputs":[{"execution_count":11,"output_type":"execute_result","data":{"text/plain":"                          file_name  \\\n5340         train_narail (155).wav   \n1368     train_chittagong (240).wav   \n10105         train_sylhet (16).wav   \n3545   train_kishoreganj (1363).wav   \n5466          train_narail (27).wav   \n\n                                          transcriptions     district  \n5340   যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...       narail  \n1368   খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...   chittagong  \n10105  ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...       sylhet  \n3545   তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...  kishoreganj  \n5466   এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...       narail  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>file_name</th>\n      <th>transcriptions</th>\n      <th>district</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>5340</th>\n      <td>train_narail (155).wav</td>\n      <td>যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...</td>\n      <td>narail</td>\n    </tr>\n    <tr>\n      <th>1368</th>\n      <td>train_chittagong (240).wav</td>\n      <td>খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...</td>\n      <td>chittagong</td>\n    </tr>\n    <tr>\n      <th>10105</th>\n      <td>train_sylhet (16).wav</td>\n      <td>ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...</td>\n      <td>sylhet</td>\n    </tr>\n    <tr>\n      <th>3545</th>\n      <td>train_kishoreganj (1363).wav</td>\n      <td>তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...</td>\n      <td>kishoreganj</td>\n    </tr>\n    <tr>\n      <th>5466</th>\n      <td>train_narail (27).wav</td>\n      <td>এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...</td>\n      <td>narail</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":11},{"cell_type":"code","source":"import pandas as pd\nimport re\n\n# Assuming 'data' is your DataFrame with the 'transcripts' column\n# Replace 'data' with the actual name of your DataFrame\n\n# Define a function to remove English characters from a string using regular expressions\ndef remove_english_characters(text):\n    # Regular expression to match English characters (both lowercase and uppercase)\n    pattern = re.compile(\"[a-zA-Z]\")\n    # Replace English characters with an empty string\n    return re.sub(pattern, \"\", text)\n\n# Apply the function to the 'transcripts' column\ndf_sample['transcriptions'] = df_sample['transcriptions'].apply(remove_english_characters)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:46.255372Z","iopub.execute_input":"2025-11-08T22:52:46.255892Z","iopub.status.idle":"2025-11-08T22:52:46.260991Z","shell.execute_reply.started":"2025-11-08T22:52:46.255867Z","shell.execute_reply":"2025-11-08T22:52:46.260289Z"}},"outputs":[],"execution_count":12},{"cell_type":"code","source":"df_sample[\"transcriptions\"] = df_sample[\"transcriptions\"].str.strip()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:50.679404Z","iopub.execute_input":"2025-11-08T22:52:50.680057Z","iopub.status.idle":"2025-11-08T22:52:50.684543Z","shell.execute_reply.started":"2025-11-08T22:52:50.680031Z","shell.execute_reply":"2025-11-08T22:52:50.683663Z"}},"outputs":[],"execution_count":13},{"cell_type":"code","source":"TASK = \"transcribe\"\nMODEL_NAME = \"/kaggle/input/bengali-ai-asr-submission/bengali-whisper-medium\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:52.622582Z","iopub.execute_input":"2025-11-08T22:52:52.623283Z","iopub.status.idle":"2025-11-08T22:52:52.626858Z","shell.execute_reply.started":"2025-11-08T22:52:52.623258Z","shell.execute_reply":"2025-11-08T22:52:52.625987Z"}},"outputs":[],"execution_count":14},{"cell_type":"code","source":"import os\n\nimport pandas as pd\n\nimport librosa\nimport librosa.display\n\nimport numpy as np\n\nimport IPython.display as ipd\n\nimport matplotlib.pyplot as plt\n\nimport random\n\nfrom collections import Counter\n\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torchaudio\n\nfrom dataclasses import dataclass\nfrom typing import Any, Dict, List, Union\nfrom datasets import DatasetDict\nfrom datasets import Dataset as DS\n\nfrom transformers import (\n    WhisperFeatureExtractor,\n    WhisperTokenizer,\n    WhisperProcessor,\n    WhisperForConditionalGeneration,\n    Seq2SeqTrainingArguments,\n    Seq2SeqTrainer,\n    TrainerCallback,\n    TrainingArguments,\n    TrainerState,\n    TrainerControl,\n    EarlyStoppingCallback,\n    pipeline\n)\n\nfrom torchmetrics.text import WordErrorRate, CharErrorRate","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:54.706696Z","iopub.execute_input":"2025-11-08T22:52:54.707026Z","iopub.status.idle":"2025-11-08T22:52:55.373198Z","shell.execute_reply.started":"2025-11-08T22:52:54.707002Z","shell.execute_reply":"2025-11-08T22:52:55.372596Z"}},"outputs":[],"execution_count":15},{"cell_type":"code","source":"feature_extractor = WhisperFeatureExtractor.from_pretrained(MODEL_NAME)\ntokenizer = WhisperTokenizer.from_pretrained(MODEL_NAME, language='bn', task=TASK)\nprocessor = WhisperProcessor.from_pretrained(MODEL_NAME, language='bn', task=TASK)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:52:59.753869Z","iopub.execute_input":"2025-11-08T22:52:59.755032Z","iopub.status.idle":"2025-11-08T22:53:00.107983Z","shell.execute_reply.started":"2025-11-08T22:52:59.755006Z","shell.execute_reply":"2025-11-08T22:53:00.107134Z"}},"outputs":[],"execution_count":16},{"cell_type":"code","source":"@dataclass\nclass DataCollatorSpeechSeq2SeqWithPadding:\n    processor: Any\n\n    def __call__(self, features: List[Dict[str, Union[List[int], torch.Tensor]]]) -> Dict[str, torch.Tensor]:\n        # split inputs and labels since they have to be of different lengths and need different padding methods\n        # first treat the audio inputs by simply returning torch tensors\n        input_features = [{\"input_features\": feature[\"input_features\"]} for feature in features]\n        batch = self.processor.feature_extractor.pad(input_features, return_tensors=\"pt\")\n\n        # get the tokenized label sequences\n        label_features = [{\"input_ids\": feature[\"labels\"]} for feature in features]\n        # pad the labels to max length\n        labels_batch = self.processor.tokenizer.pad(label_features, return_tensors=\"pt\")\n\n        # replace padding with -100 to ignore loss correctly\n        labels = labels_batch[\"input_ids\"].masked_fill(labels_batch.attention_mask.ne(1), -100)\n\n        # if bos token is appended in previous tokenization step,\n        # cut bos token here as it's append later anyways\n        if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():\n            labels = labels[:, 1:]\n\n        batch[\"labels\"] = labels\n        \n       # torch.cuda.empty_cache()\n\n        return batch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:04.156602Z","iopub.execute_input":"2025-11-08T22:53:04.157427Z","iopub.status.idle":"2025-11-08T22:53:04.164004Z","shell.execute_reply.started":"2025-11-08T22:53:04.157399Z","shell.execute_reply":"2025-11-08T22:53:04.163002Z"}},"outputs":[],"execution_count":17},{"cell_type":"code","source":"data_collator = DataCollatorSpeechSeq2SeqWithPadding(processor=processor)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:08.814089Z","iopub.execute_input":"2025-11-08T22:53:08.814366Z","iopub.status.idle":"2025-11-08T22:53:08.818547Z","shell.execute_reply.started":"2025-11-08T22:53:08.814345Z","shell.execute_reply":"2025-11-08T22:53:08.817671Z"}},"outputs":[],"execution_count":18},{"cell_type":"code","source":"import librosa\nimport os\nimport noisereduce as nr\nimport numpy as np\n\n# Define a target amplitude for peak normalization\n# 1.0 represents the maximum possible amplitude.\nTARGET_PEAK_AMP = 1.0\n\n# --- Corrected Preprocessing Function ---\n\ndef prepare_dataset(example):\n    Folder_path = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/\"\n    \n    # 1. Create the full audio path\n    audio_path = os.path.join(Folder_path, example[\"file_name\"])\n    \n    # 2. Load the audio\n    audio, sr = librosa.load(audio_path, sr=16_000)\n    \n    # 3. Denoise the audio\n    denoised_audio = nr.reduce_noise(y=audio, sr=sr)\n    \n    # 4. Amplify (using your provided peak normalization logic)\n    max_amp = np.max(np.abs(denoised_audio))\n    \n    amplified_audio = denoised_audio  # Default to denoised audio\n    if max_amp > 0.001:  # Avoid division by zero or near-zero on silent files\n        gain = TARGET_PEAK_AMP / max_amp\n        amplified_audio = denoised_audio * gain\n    \n    # 5. Extract features from the clean, amplified audio\n    example[\"input_features\"] = feature_extractor(amplified_audio, sampling_rate=sr).input_features[0]\n    \n    # 6. Process text\n    example[\"labels\"] = tokenizer(f\"{example['transcriptions']}\", max_length=448, padding=True, truncation=True).input_ids\n    \n    return example\n\n# --- The filter functions remain the same ---\n\ndef filter_inputs(input_audio):\n    \"\"\"filter inputs with zero input length\"\"\"\n    return 0 < len(input_audio)\n\ndef filter_labels(input_labels):\n    \"\"\"filter empty label sequences\"\"\"\n    return 0 < len(input_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:12.240535Z","iopub.execute_input":"2025-11-08T22:53:12.240942Z","iopub.status.idle":"2025-11-08T22:53:12.247541Z","shell.execute_reply.started":"2025-11-08T22:53:12.240918Z","shell.execute_reply":"2025-11-08T22:53:12.246663Z"}},"outputs":[],"execution_count":19},{"cell_type":"code","source":"df_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:16.480021Z","iopub.execute_input":"2025-11-08T22:53:16.48069Z","iopub.status.idle":"2025-11-08T22:53:16.4895Z","shell.execute_reply.started":"2025-11-08T22:53:16.480655Z","shell.execute_reply":"2025-11-08T22:53:16.488593Z"}},"outputs":[{"execution_count":20,"output_type":"execute_result","data":{"text/plain":"                          file_name  \\\n5340         train_narail (155).wav   \n1368     train_chittagong (240).wav   \n10105         train_sylhet (16).wav   \n3545   train_kishoreganj (1363).wav   \n5466          train_narail (27).wav   \n\n                                          transcriptions     district  \n5340   যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...       narail  \n1368   খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...   chittagong  \n10105  ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...       sylhet  \n3545   তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...  kishoreganj  \n5466   এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...       narail  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>file_name</th>\n      <th>transcriptions</th>\n      <th>district</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>5340</th>\n      <td>train_narail (155).wav</td>\n      <td>যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...</td>\n      <td>narail</td>\n    </tr>\n    <tr>\n      <th>1368</th>\n      <td>train_chittagong (240).wav</td>\n      <td>খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...</td>\n      <td>chittagong</td>\n    </tr>\n    <tr>\n      <th>10105</th>\n      <td>train_sylhet (16).wav</td>\n      <td>ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...</td>\n      <td>sylhet</td>\n    </tr>\n    <tr>\n      <th>3545</th>\n      <td>train_kishoreganj (1363).wav</td>\n      <td>তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...</td>\n      <td>kishoreganj</td>\n    </tr>\n    <tr>\n      <th>5466</th>\n      <td>train_narail (27).wav</td>\n      <td>এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...</td>\n      <td>narail</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":20},{"cell_type":"code","source":"# 1. Load the model\nmodel = WhisperForConditionalGeneration.from_pretrained(MODEL_NAME, device_map=\"auto\")\n\n# 2. ❗ FIX: Disable 'use_cache'\n# This is required to fix the RuntimeError when using gradient_checkpointing\nmodel.config.use_cache = False\n\n# 3. Get the special tokens for Bengali transcription\nprompt_ids = tokenizer.get_decoder_prompt_ids(language=\"bn\", task=\"transcribe\")\n\n# 4. Set config for TRAINING\n# The model's internal 'forward()' method uses this during training\nmodel.config.forced_decoder_ids = prompt_ids\nmodel.config.suppress_tokens = []  # This is good, it forces text-only output\n\n# 5. Set config for GENERATION (used for evaluation/inference)\n# Make this consistent with the training config.\n# DO NOT set forced_decoder_ids to None.\nmodel.generation_config.language = \"bn\"\nmodel.generation_config.task = \"transcribe\"\nmodel.generation_config.forced_decoder_ids = prompt_ids\nmodel.generation_config.suppress_tokens = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:19.42179Z","iopub.execute_input":"2025-11-08T22:53:19.422092Z","iopub.status.idle":"2025-11-08T22:53:22.62846Z","shell.execute_reply.started":"2025-11-08T22:53:19.422069Z","shell.execute_reply":"2025-11-08T22:53:22.627699Z"}},"outputs":[],"execution_count":21},{"cell_type":"code","source":"print(model.generation_config)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:26.185753Z","iopub.execute_input":"2025-11-08T22:53:26.186063Z","iopub.status.idle":"2025-11-08T22:53:26.19066Z","shell.execute_reply.started":"2025-11-08T22:53:26.186043Z","shell.execute_reply":"2025-11-08T22:53:26.189802Z"}},"outputs":[{"name":"stdout","text":"GenerationConfig {\n  \"begin_suppress_tokens\": [\n    220,\n    50257\n  ],\n  \"bos_token_id\": 50257,\n  \"decoder_start_token_id\": 50258,\n  \"eos_token_id\": 50257,\n  \"forced_decoder_ids\": [\n    [\n      1,\n      50302\n    ],\n    [\n      2,\n      50359\n    ],\n    [\n      3,\n      50363\n    ]\n  ],\n  \"language\": \"bn\",\n  \"max_length\": 448,\n  \"pad_token_id\": 50257,\n  \"suppress_tokens\": [],\n  \"task\": \"transcribe\"\n}\n\n","output_type":"stream"}],"execution_count":22},{"cell_type":"code","source":"model_id = f\"whisper-medium-reg-ben/train_1_no_pretrained_25_epochs\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:35.663377Z","iopub.execute_input":"2025-11-08T22:53:35.663672Z","iopub.status.idle":"2025-11-08T22:53:35.667501Z","shell.execute_reply.started":"2025-11-08T22:53:35.663645Z","shell.execute_reply":"2025-11-08T22:53:35.666633Z"}},"outputs":[],"execution_count":23},{"cell_type":"code","source":"training_args = Seq2SeqTrainingArguments(\n    output_dir=model_id,\n    per_device_train_batch_size=16,\n    per_device_eval_batch_size=8,\n    gradient_accumulation_steps=1,\n    gradient_checkpointing=True,\n    fp16=True,\n    learning_rate=1e-5, \n    weight_decay=0.0025,\n    warmup_steps=200,\n    num_train_epochs=25,\n    eval_strategy=\"steps\", # or \"epochs\"\n    predict_with_generate=True,\n#     generation_max_length=448,\n    save_steps=1000,\n    eval_steps=1000,\n    logging_steps=100,\n    save_total_limit=5,\n    load_best_model_at_end=True,\n    metric_for_best_model=\"wer\",\n    greater_is_better=False,\n    push_to_hub=False,\n    report_to=\"none\",\n    remove_unused_columns=False,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:39.4509Z","iopub.execute_input":"2025-11-08T22:53:39.451208Z","iopub.status.idle":"2025-11-08T22:53:39.488709Z","shell.execute_reply.started":"2025-11-08T22:53:39.451186Z","shell.execute_reply":"2025-11-08T22:53:39.487984Z"}},"outputs":[],"execution_count":24},{"cell_type":"code","source":"# Assuming your DataFrame is named df_sample\ndf_sample['split'] = df_sample['file_name'].str.split('_').str[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:42.401537Z","iopub.execute_input":"2025-11-08T22:53:42.401851Z","iopub.status.idle":"2025-11-08T22:53:42.40731Z","shell.execute_reply.started":"2025-11-08T22:53:42.401829Z","shell.execute_reply":"2025-11-08T22:53:42.406472Z"}},"outputs":[],"execution_count":25},{"cell_type":"code","source":"df_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:44.441439Z","iopub.execute_input":"2025-11-08T22:53:44.442207Z","iopub.status.idle":"2025-11-08T22:53:44.453129Z","shell.execute_reply.started":"2025-11-08T22:53:44.442173Z","shell.execute_reply":"2025-11-08T22:53:44.452379Z"}},"outputs":[{"execution_count":26,"output_type":"execute_result","data":{"text/plain":"                          file_name  \\\n5340         train_narail (155).wav   \n1368     train_chittagong (240).wav   \n10105         train_sylhet (16).wav   \n3545   train_kishoreganj (1363).wav   \n5466          train_narail (27).wav   \n\n                                          transcriptions     district  split  \n5340   যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...       narail  train  \n1368   খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...   chittagong  train  \n10105  ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...       sylhet  train  \n3545   তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...  kishoreganj  train  \n5466   এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...       narail  train  ","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>file_name</th>\n      <th>transcriptions</th>\n      <th>district</th>\n      <th>split</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>5340</th>\n      <td>train_narail (155).wav</td>\n      <td>যেগুলো এওনো বিড়ি-মিড়ি খাওয়া শিখিনি আর আমাগে এল...</td>\n      <td>narail</td>\n      <td>train</td>\n    </tr>\n    <tr>\n      <th>1368</th>\n      <td>train_chittagong (240).wav</td>\n      <td>খরলা ইবে খত ইবে খয়দে একশো বিশ টেঁয়া ইয়্যত তফাত...</td>\n      <td>chittagong</td>\n      <td>train</td>\n    </tr>\n    <tr>\n      <th>10105</th>\n      <td>train_sylhet (16).wav</td>\n      <td>ফরিছিত ও তোর ব্যাছমেট বে  আও  ক্লাসো আও  তে আফ...</td>\n      <td>sylhet</td>\n      <td>train</td>\n    </tr>\n    <tr>\n      <th>3545</th>\n      <td>train_kishoreganj (1363).wav</td>\n      <td>তে আল্লার রহমতে গর করছি কত বছর অহনো গিয়া দেহো ...</td>\n      <td>kishoreganj</td>\n      <td>train</td>\n    </tr>\n    <tr>\n      <th>5466</th>\n      <td>train_narail (27).wav</td>\n      <td>এহন যদি না হয় তালি তো জিনিসটা আসলেই মর্মান্তিক...</td>\n      <td>narail</td>\n      <td>train</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}],"execution_count":26},{"cell_type":"code","source":"\"\"\"\n    adjust test size accordingly.\n\"\"\"\ntrain_df, eval_df = train_test_split(df_sample, test_size=0.00001, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:47.256162Z","iopub.execute_input":"2025-11-08T22:53:47.256947Z","iopub.status.idle":"2025-11-08T22:53:47.263093Z","shell.execute_reply.started":"2025-11-08T22:53:47.25691Z","shell.execute_reply":"2025-11-08T22:53:47.262233Z"}},"outputs":[],"execution_count":27},{"cell_type":"code","source":"len(train_df), len(eval_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:49.960841Z","iopub.execute_input":"2025-11-08T22:53:49.961562Z","iopub.status.idle":"2025-11-08T22:53:49.968524Z","shell.execute_reply.started":"2025-11-08T22:53:49.961528Z","shell.execute_reply":"2025-11-08T22:53:49.967717Z"}},"outputs":[{"execution_count":28,"output_type":"execute_result","data":{"text/plain":"(29, 1)"},"metadata":{}}],"execution_count":28},{"cell_type":"code","source":"ben_reg_voice_ds = DatasetDict()\n\ntrain_split = DS.from_pandas(train_df)\neval_split = DS.from_pandas(eval_df)\n\nds_splits = DatasetDict({\n    'train': train_split,\n    'eval': eval_split\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:54.509718Z","iopub.execute_input":"2025-11-08T22:53:54.510317Z","iopub.status.idle":"2025-11-08T22:53:54.534721Z","shell.execute_reply.started":"2025-11-08T22:53:54.510292Z","shell.execute_reply":"2025-11-08T22:53:54.533928Z"}},"outputs":[],"execution_count":29},{"cell_type":"code","source":"s_splits = ds_splits.remove_columns([\"split\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:57.584544Z","iopub.execute_input":"2025-11-08T22:53:57.584963Z","iopub.status.idle":"2025-11-08T22:53:57.591871Z","shell.execute_reply.started":"2025-11-08T22:53:57.584939Z","shell.execute_reply":"2025-11-08T22:53:57.591018Z"}},"outputs":[],"execution_count":30},{"cell_type":"code","source":"ds_splits = ds_splits.map(prepare_dataset, remove_columns=ds_splits.column_names[\"train\"], num_proc=1 \n                        # open for multithreadding\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:53:59.786835Z","iopub.execute_input":"2025-11-08T22:53:59.787625Z","iopub.status.idle":"2025-11-08T22:54:09.857671Z","shell.execute_reply.started":"2025-11-08T22:53:59.787597Z","shell.execute_reply":"2025-11-08T22:54:09.85669Z"}},"outputs":[{"output_type":"display_data","data":{"text/plain":"Map (num_proc=1):   0%|          | 0/29 [00:00<?, ? examples/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"2dea64914ea44e76b3e53c0f1f1147c4"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"Map (num_proc=1):   0%|          | 0/1 [00:00<?, ? examples/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"a354a4397150477aa7abacb3bf3a070e"}},"metadata":{}}],"execution_count":31},{"cell_type":"code","source":"cer = CharErrorRate()\nwer = WordErrorRate()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:54:15.415843Z","iopub.execute_input":"2025-11-08T22:54:15.416179Z","iopub.status.idle":"2025-11-08T22:54:15.425747Z","shell.execute_reply.started":"2025-11-08T22:54:15.416148Z","shell.execute_reply":"2025-11-08T22:54:15.425001Z"}},"outputs":[],"execution_count":32},{"cell_type":"code","source":"def compute_metrics(pred):\n    pred_ids = pred.predictions\n    label_ids = pred.label_ids\n\n    # ✅ Ensure everything is detached and on CPU\n    if isinstance(pred_ids, torch.Tensor):\n        pred_ids = pred_ids.detach().cpu().numpy()\n    if isinstance(label_ids, torch.Tensor):\n        label_ids = label_ids.detach().cpu().numpy()\n\n    label_ids[label_ids == -100] = tokenizer.pad_token_id\n\n    pred_str = tokenizer.batch_decode(pred_ids, skip_special_tokens=True)\n    label_str = tokenizer.batch_decode(label_ids, skip_special_tokens=True)\n\n    wer_res = wer(pred_str, label_str)\n    cer_res = cer(pred_str, label_str)\n\n    print(\"WER:\", wer_res, \"| CER:\", cer_res)\n    print(\"Pred:\", pred_str[0])\n    print(\"Label:\", label_str[0])\n\n    return {\"wer\": wer_res, \"cer\": cer_res}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:54:18.026495Z","iopub.execute_input":"2025-11-08T22:54:18.027041Z","iopub.status.idle":"2025-11-08T22:54:18.032691Z","shell.execute_reply.started":"2025-11-08T22:54:18.027014Z","shell.execute_reply":"2025-11-08T22:54:18.032011Z"}},"outputs":[],"execution_count":33},{"cell_type":"code","source":"trainer = Seq2SeqTrainer(\n    args=training_args,\n    model=model,\n    train_dataset=ds_splits[\"train\"],\n    eval_dataset=ds_splits[\"eval\"],\n    data_collator=data_collator,\n    tokenizer=processor,\n    compute_metrics=compute_metrics,\n#     callbacks=[EarlyStoppingCallback(2, 1.0)]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T23:06:56.404158Z","iopub.execute_input":"2025-11-08T23:06:56.404798Z","iopub.status.idle":"2025-11-08T23:06:56.415307Z","shell.execute_reply.started":"2025-11-08T23:06:56.404764Z","shell.execute_reply":"2025-11-08T23:06:56.414539Z"}},"outputs":[],"execution_count":40},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T22:55:09.68607Z","iopub.execute_input":"2025-11-08T22:55:09.686796Z","iopub.status.idle":"2025-11-08T22:55:09.72145Z","shell.execute_reply.started":"2025-11-08T22:55:09.686755Z","shell.execute_reply":"2025-11-08T22:55:09.720696Z"}},"outputs":[],"execution_count":36},{"cell_type":"code","source":"trainer.train()\n\n# to use the high-level pipeline, ensure both the processor outputs and model outputs exist in the same dir\ntrainer.save_model(training_args.output_dir)\nprocessor.save_pretrained(training_args.output_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T23:07:01.933298Z","iopub.execute_input":"2025-11-08T23:07:01.933579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}