{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install -q jiwer\n! pip  install -q datasets -U\n! pip  install -q bnunicodenormalizer","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:25:55.771459Z","iopub.execute_input":"2023-10-06T02:25:55.771824Z","iopub.status.idle":"2023-10-06T02:26:33.241123Z","shell.execute_reply.started":"2023-10-06T02:25:55.771792Z","shell.execute_reply":"2023-10-06T02:26:33.239581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport json\nfrom datasets import Audio\nfrom datasets import Dataset\nfrom bnunicodenormalizer import Normalizer \nbnorm=Normalizer()\nfrom datasets import concatenate_datasets","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:33.244450Z","iopub.execute_input":"2023-10-06T02:26:33.244962Z","iopub.status.idle":"2023-10-06T02:26:34.298422Z","shell.execute_reply.started":"2023-10-06T02:26:33.244914Z","shell.execute_reply":"2023-10-06T02:26:34.297375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/common-voice-13-bengali-normalized/train.tsv\", delimiter=\"\\t\")\ntest = pd.read_csv(\"/kaggle/input/common-voice-13-bengali-normalized/test.tsv\", delimiter=\"\\t\")\ndisplay(train.head())\ndisplay(test.head())\nprint(\"Train shape : \",train.shape)\nprint(\"Validation set shape : \",test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.299681Z","iopub.execute_input":"2023-10-06T02:26:34.300257Z","iopub.status.idle":"2023-10-06T02:26:34.577093Z","shell.execute_reply.started":"2023-10-06T02:26:34.300225Z","shell.execute_reply":"2023-10-06T02:26:34.575863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Read the dataframe\n# df = pd.read_csv(\"/kaggle/input/bengaliai-speech/train.csv\")\n# train = df[df.split==\"train\"]\n# val = df[df.split==\"valid\"]\n# display(train.head())\n# display(val.head())\n# print(\"Train shape : \",train.shape)\n# print(\"Validation set shape : \",val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.579880Z","iopub.execute_input":"2023-10-06T02:26:34.580384Z","iopub.status.idle":"2023-10-06T02:26:34.586032Z","shell.execute_reply.started":"2023-10-06T02:26:34.580339Z","shell.execute_reply":"2023-10-06T02:26:34.584744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.sample(frac=1,random_state=42)\ntest = test.sample(frac=1,random_state=42)\nprint(\"Train shape : \",train.shape)\nprint(\"Validation set shape : \",test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.590095Z","iopub.execute_input":"2023-10-06T02:26:34.590739Z","iopub.status.idle":"2023-10-06T02:26:34.610583Z","shell.execute_reply.started":"2023-10-06T02:26:34.590642Z","shell.execute_reply":"2023-10-06T02:26:34.608932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train = train.iloc[90000:100000]\ntest = test.iloc[:4000]\nprint(\"Train shape : \",train.shape)\nprint(\"Validation set shape : \",test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.612265Z","iopub.execute_input":"2023-10-06T02:26:34.613193Z","iopub.status.idle":"2023-10-06T02:26:34.620298Z","shell.execute_reply.started":"2023-10-06T02:26:34.613158Z","shell.execute_reply":"2023-10-06T02:26:34.618978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audio_dir = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\n# train_paths = train['id'].apply(lambda x:audio_dir+x+\".mp3\")\n# train_ds = Dataset.from_dict({\"audio\":train_paths ,\"sentence\":train['sentence'].tolist()}).cast_column(\"audio\", Audio(sampling_rate=16000))","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.621845Z","iopub.execute_input":"2023-10-06T02:26:34.622312Z","iopub.status.idle":"2023-10-06T02:26:34.631367Z","shell.execute_reply.started":"2023-10-06T02:26:34.622279Z","shell.execute_reply":"2023-10-06T02:26:34.630422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_dir = \"/kaggle/input/common-voice-13-bengali-normalized/train/\"\naudio_test = \"/kaggle/input/common-voice-13-bengali-normalized/test/\"\ntrain_paths = train['path'].apply(lambda x:audio_dir+x)\ntrain_ds = Dataset.from_dict({\"audio\":train_paths ,\"sentence\":train['sentence'].tolist()}).cast_column(\"audio\", Audio(sampling_rate=16000))","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.632371Z","iopub.execute_input":"2023-10-06T02:26:34.632684Z","iopub.status.idle":"2023-10-06T02:26:34.778136Z","shell.execute_reply.started":"2023-10-06T02:26:34.632657Z","shell.execute_reply":"2023-10-06T02:26:34.776874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nchars_to_ignore_regex = '[\\,\\?\\.\\!\\-\\;\\:\\\"\\—\\‘\\'\\‚\\“\\”\\…]'\n\ndef remove_special_characters(batch):\n    batch[\"sentence\"] = re.sub(chars_to_ignore_regex, '', batch[\"sentence\"]) + \" \"\n    return batch\n\ndef normalize(batch):\n    _words = [bnorm(word)['normalized']  for word in batch[\"sentence\"].split()]\n    batch[\"sentence\"] =  \" \".join([word for word in _words if word is not None])\n    return batch","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.779811Z","iopub.execute_input":"2023-10-06T02:26:34.781101Z","iopub.status.idle":"2023-10-06T02:26:34.788323Z","shell.execute_reply.started":"2023-10-06T02:26:34.781031Z","shell.execute_reply":"2023-10-06T02:26:34.786937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.map(remove_special_characters)\ntrain_ds = train_ds.map(normalize)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:26:34.790123Z","iopub.execute_input":"2023-10-06T02:26:34.790589Z","iopub.status.idle":"2023-10-06T02:27:47.902119Z","shell.execute_reply.started":"2023-10-06T02:26:34.790552Z","shell.execute_reply":"2023-10-06T02:27:47.901348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import Wav2Vec2Processor, Wav2Vec2CTCTokenizer, Wav2Vec2FeatureExtractor\n\nmodel_name = \"arijitx/wav2vec2-xls-r-300m-bengali\"\ntask = \"transcribe\"\n\nfeature_extractor = Wav2Vec2FeatureExtractor.from_pretrained(model_name)\ntokenizer = Wav2Vec2CTCTokenizer.from_pretrained(model_name, language = \"bn\", task=task)\nprocessor = Wav2Vec2Processor.from_pretrained(model_name, language = \"bn\", task=task)\n#processor = Wav2Vec2Processor.from_pretrained(\"facebook/wav2vec2-large-960h\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:27:47.903205Z","iopub.execute_input":"2023-10-06T02:27:47.903918Z","iopub.status.idle":"2023-10-06T02:27:52.485107Z","shell.execute_reply.started":"2023-10-06T02:27:47.903887Z","shell.execute_reply":"2023-10-06T02:27:52.483843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_str = train_ds[0][\"sentence\"]\nlabels = tokenizer(input_str).input_ids\ndecoded_with_special = tokenizer.decode(labels, skip_special_tokens=False)\ndecoded_str = tokenizer.decode(labels, skip_special_tokens=True)\n\nprint(f\"Input:                 {input_str}\")\nprint(f\"Decoded w/ special:    {decoded_with_special}\")\nprint(f\"Decoded w/out special: {decoded_str}\")\nprint(f\"Are equal:             {input_str == decoded_str}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:27:52.487010Z","iopub.execute_input":"2023-10-06T02:27:52.487949Z","iopub.status.idle":"2023-10-06T02:28:15.482666Z","shell.execute_reply.started":"2023-10-06T02:27:52.487846Z","shell.execute_reply":"2023-10-06T02:28:15.481121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(batch):\n\n    batch[\"audio\"][\"array\"] = np.trim_zeros(batch[\"audio\"][\"array\"], 'fb')\n    audio = batch[\"audio\"]\n    \n\n    # batched output is \"un-batched\" to ensure mapping is correct\n    batch[\"input_values\"] = processor(audio[\"array\"], sampling_rate=16000).input_values[0]\n    batch[\"input_length\"] = len(batch[\"input_values\"])\n    \n    with processor.as_target_processor():\n        batch[\"labels\"] = processor(batch[\"sentence\"]).input_ids\n    return batch","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:28:15.484867Z","iopub.execute_input":"2023-10-06T02:28:15.486463Z","iopub.status.idle":"2023-10-06T02:28:15.492909Z","shell.execute_reply.started":"2023-10-06T02:28:15.486423Z","shell.execute_reply":"2023-10-06T02:28:15.491580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.map(prepare_dataset, remove_columns=train_ds.column_names)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:28:15.496579Z","iopub.execute_input":"2023-10-06T02:28:15.497095Z","iopub.status.idle":"2023-10-06T02:34:58.433389Z","shell.execute_reply.started":"2023-10-06T02:28:15.497034Z","shell.execute_reply":"2023-10-06T02:34:58.432004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def trim_silence(batch):\n    arr = batch['input_values']\n    \n    try:\n        _max = max(max(arr), -min(arr))\n        old_length = len(arr)\n        \n        threshold = 30\n\n        for i,e in enumerate(arr):\n            if threshold*e>_max:\n                break\n\n        for j,e in enumerate(reversed(arr)):\n            if threshold*e>_max:\n                break\n\n        batch['input_values'] = arr[i:old_length-j]\n        batch['input_length'] = old_length -i -j\n    except:\n        print(batch['input_length'])\n    return batch","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:34:58.434899Z","iopub.execute_input":"2023-10-06T02:34:58.435273Z","iopub.status.idle":"2023-10-06T02:34:58.442215Z","shell.execute_reply.started":"2023-10-06T02:34:58.435244Z","shell.execute_reply":"2023-10-06T02:34:58.441372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.map(trim_silence)","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:34:58.443578Z","iopub.execute_input":"2023-10-06T02:34:58.444142Z","iopub.status.idle":"2023-10-06T02:54:06.086099Z","shell.execute_reply.started":"2023-10-06T02:34:58.444109Z","shell.execute_reply":"2023-10-06T02:54:06.084128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_input_length_in_sec = 10.0\nmin_input_length_in_sec = 1\n\ntrain_ds = train_ds.filter(lambda x: x < max_input_length_in_sec * 16000, input_columns=[\"input_length\"])\ntrain_ds = train_ds.filter(lambda x: x > min_input_length_in_sec * 16000, input_columns=[\"input_length\"])","metadata":{"execution":{"iopub.status.busy":"2023-10-06T02:54:06.088671Z","iopub.execute_input":"2023-10-06T02:54:06.089165Z","iopub.status.idle":"2023-10-06T02:54:06.431438Z","shell.execute_reply.started":"2023-10-06T02:54:06.089126Z","shell.execute_reply":"2023-10-06T02:54:06.430121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.to_parquet(\"/kaggle/working/train.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-10-06T03:02:57.963116Z","iopub.execute_input":"2023-10-06T03:02:57.963573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(df):\n    paths = df['path'].apply(lambda x:audio_test+x)\n    dataset = Dataset.from_dict({\"audio\":paths ,\"sentence\":df['sentence'].tolist()}).cast_column(\"audio\", Audio(sampling_rate=16000))\n    dataset=dataset.map(remove_special_characters)\n    dataset = dataset.map(normalize)\n    dataset_train = dataset.map(prepare_dataset, remove_columns=dataset.column_names)\n    dataset = dataset_train.map(trim_silence)\n\n    dataset = dataset.filter(lambda x: x < max_input_length_in_sec * 16000, input_columns=[\"input_length\"])\n    dataset = dataset.filter(lambda x: x > min_input_length_in_sec * 16000, input_columns=[\"input_length\"])\n    \n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = create_dataset(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds.to_parquet(\"/kaggle/working/test.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nshutil.make_archive(\"output\", 'zip', \"/kaggle/working/output\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}