{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":52324,"databundleVersionId":6229904,"sourceType":"competition"},{"sourceId":4120850,"sourceType":"datasetVersion","datasetId":2435696},{"sourceId":7408220,"sourceType":"datasetVersion","datasetId":4308645}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"> In this Notebook, we'll see how to preprocess the dataset for training Wav2Vec2. \n> The notebook is mostly modified from YellowKing's data preprocessing [Notebook](https://www.kaggle.com/code/sameen53/yellowking-dlsprint-datapreprocessingv1)","metadata":{"execution":{"iopub.status.busy":"2023-08-03T18:11:26.651706Z","iopub.execute_input":"2023-08-03T18:11:26.653104Z","iopub.status.idle":"2023-08-03T18:12:03.067322Z","shell.execute_reply.started":"2023-08-03T18:11:26.653032Z","shell.execute_reply":"2023-08-03T18:12:03.065501Z"}}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\" style=\"padding:10px; line-height: 1.7em; font-family: Verdana;\">\n    <center style=\"font-family: consolas; font-size: 32px; font-weight: bold; color:Black\"> Preprocessing Steps</center></div>\n","metadata":{}},{"cell_type":"markdown","source":"**Preprocessing steps** :\n> * [Create Dataset and Resample audios at 16000 sampling rate](#1)\n> * [Remove special characters and Normalize using bnunicodenormalizer](#2)\n> * [Tokenize using Wav2Vec2Processor](#3)\n> * [Trim silences](#4)\n> * [Filter audios of duration 1-10s](#5)","metadata":{}},{"cell_type":"markdown","source":"We'll preprocess ```30k audios for training``` and ```5k audios for validation```. If we want to preprocess more audios we'll have to do that in chunks because kaggle won't allow us to have more than 19GB data in the disk.","metadata":{}},{"cell_type":"markdown","source":"# Install Dependencies","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install transformers\n!pip install jiwer\n!apt install git-lfs","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:05:07.987196Z","iopub.execute_input":"2024-01-17T18:05:07.987702Z","iopub.status.idle":"2024-01-17T18:05:41.317489Z","shell.execute_reply.started":"2024-01-17T18:05:07.987645Z","shell.execute_reply":"2024-01-17T18:05:41.315555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install bnunicodenormalizer","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:05:41.320069Z","iopub.execute_input":"2024-01-17T18:05:41.320488Z","iopub.status.idle":"2024-01-17T18:05:57.839993Z","shell.execute_reply.started":"2024-01-17T18:05:41.320450Z","shell.execute_reply":"2024-01-17T18:05:57.838643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Imports**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport json\nfrom datasets import Audio\nfrom datasets import Dataset\nfrom bnunicodenormalizer import Normalizer \nbnorm=Normalizer()\nfrom datasets import concatenate_datasets","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:05:57.841529Z","iopub.execute_input":"2024-01-17T18:05:57.841912Z","iopub.status.idle":"2024-01-17T18:05:59.091134Z","shell.execute_reply.started":"2024-01-17T18:05:57.841878Z","shell.execute_reply":"2024-01-17T18:05:59.089892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n<div class=\"alert alert-block alert-info\" style=\"padding:25px; line-height: 1.7em; font-family: Verdana;\">\n    <center style=\"font-family: consolas; font-size: 24px; font-weight: bold; color:Black\"> Create Dataset and Resample audios at 16000 sampling rate</center>","metadata":{}},{"cell_type":"code","source":"all_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input/dataset-435b394286b7'):\n    for filename in filenames:\n        all_files.append(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:05:59.094592Z","iopub.execute_input":"2024-01-17T18:05:59.095471Z","iopub.status.idle":"2024-01-17T18:06:22.400358Z","shell.execute_reply.started":"2024-01-17T18:05:59.095421Z","shell.execute_reply":"2024-01-17T18:06:22.398739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_files_2 = [i.split('.') for i in all_files]\nall_files_2 = [i for i in all_files_2 if len(i)>1]\nfolder_contents_df = pd.DataFrame(all_files_2, columns=['path','type'])","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:22.402260Z","iopub.execute_input":"2024-01-17T18:06:22.403282Z","iopub.status.idle":"2024-01-17T18:06:22.995969Z","shell.execute_reply.started":"2024-01-17T18:06:22.403241Z","shell.execute_reply":"2024-01-17T18:06:22.994304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transcript_files_df = folder_contents_df[folder_contents_df['type']=='tsv']\ntranscripts = []\nfor idx, row in transcript_files_df.iterrows():\n    ts = pd.read_csv(row['path'] + '.' + row['type'], header=None, sep=\"\\t\", names=['x', 'y', 't'])\n    transcripts.append(ts)\n    \ntranscripts_df = pd.concat(transcripts)\ntranscripts_df = transcripts_df.sort_values(by=['x'])\ntranscripts_df.drop_duplicates(inplace=True)\ntranscripts_df","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:22.997659Z","iopub.execute_input":"2024-01-17T18:06:22.998069Z","iopub.status.idle":"2024-01-17T18:06:37.983444Z","shell.execute_reply.started":"2024-01-17T18:06:22.998019Z","shell.execute_reply":"2024-01-17T18:06:37.982294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_files_df = folder_contents_df[folder_contents_df['type']=='flac']\naudio_files_df['filename'] = [i.split('/')[-1] for i in audio_files_df['path']]\naudio_files_df = audio_files_df.sort_values(by=['filename'])\naudio_files_df","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:37.985082Z","iopub.execute_input":"2024-01-17T18:06:37.986323Z","iopub.status.idle":"2024-01-17T18:06:38.635084Z","shell.execute_reply.started":"2024-01-17T18:06:37.986277Z","shell.execute_reply":"2024-01-17T18:06:38.633758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = pd.merge(\n    left=audio_files_df,\n    right=transcripts_df,\n    left_on='filename',\n    right_on='x',\n    how='inner')\ndata_df","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:38.636549Z","iopub.execute_input":"2024-01-17T18:06:38.636924Z","iopub.status.idle":"2024-01-17T18:06:38.891877Z","shell.execute_reply.started":"2024-01-17T18:06:38.636891Z","shell.execute_reply":"2024-01-17T18:06:38.890626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df = data_df[['path','t']]\ndata_df.columns = ['audio_file_path','label']","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:38.893633Z","iopub.execute_input":"2024-01-17T18:06:38.894198Z","iopub.status.idle":"2024-01-17T18:06:38.943795Z","shell.execute_reply.started":"2024-01-17T18:06:38.894157Z","shell.execute_reply":"2024-01-17T18:06:38.942002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_df['audio_file_path'] = [i + '.flac' for i in data_df['audio_file_path']]","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:38.947518Z","iopub.execute_input":"2024-01-17T18:06:38.947873Z","iopub.status.idle":"2024-01-17T18:06:39.024241Z","shell.execute_reply.started":"2024-01-17T18:06:38.947844Z","shell.execute_reply":"2024-01-17T18:06:39.023081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll take a subset of training and validation set for this notebook. Let's shuffle the dataset and take 30k training samples and 10k validation samples randomly","metadata":{}},{"cell_type":"code","source":"train = data_df.sample(frac=0.9, random_state=42).reset_index(drop=True)\nval = data_df.drop(train.index).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:39.025883Z","iopub.execute_input":"2024-01-17T18:06:39.026296Z","iopub.status.idle":"2024-01-17T18:06:39.079768Z","shell.execute_reply.started":"2024-01-17T18:06:39.026265Z","shell.execute_reply":"2024-01-17T18:06:39.078457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[70000:]\n# train2 = train[70000:]","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:39.082146Z","iopub.execute_input":"2024-01-17T18:06:39.082514Z","iopub.status.idle":"2024-01-17T18:06:39.088949Z","shell.execute_reply.started":"2024-01-17T18:06:39.082483Z","shell.execute_reply":"2024-01-17T18:06:39.087511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at what we have here","metadata":{}},{"cell_type":"code","source":"train_ds = Dataset.from_dict({\"audio\":train['audio_file_path'].tolist() ,\n            \"sentence\":train['label'].astype(str).tolist()}).cast_column(\"audio\", Audio(sampling_rate=16000))","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:39.091882Z","iopub.execute_input":"2024-01-17T18:06:39.092824Z","iopub.status.idle":"2024-01-17T18:06:39.265737Z","shell.execute_reply.started":"2024-01-17T18:06:39.092781Z","shell.execute_reply":"2024-01-17T18:06:39.264414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's listen the audio","metadata":{}},{"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(train_ds[0]['audio']['array'],rate = 16000)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:39.267369Z","iopub.execute_input":"2024-01-17T18:06:39.267818Z","iopub.status.idle":"2024-01-17T18:06:52.709228Z","shell.execute_reply.started":"2024-01-17T18:06:39.267776Z","shell.execute_reply":"2024-01-17T18:06:52.707913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n<div class=\"alert alert-block alert-info\" style=\"padding:25px; line-height: 1.7em; font-family: Verdana;\">\n    <center style=\"font-family: consolas; font-size: 24px; font-weight: bold; color:Black\"> Remove special characters and Normalize using bnunicodenormalizer</center>","metadata":{}},{"cell_type":"markdown","source":"We'll remove the punctuations from the sentences and also normalize the sentences using bnunicodenormalizer. ","metadata":{}},{"cell_type":"code","source":"import re\nchars_to_ignore_regex = '[\\,\\?\\.\\!\\-\\;\\:\\\"\\—\\‘\\'\\‚\\“\\”\\…]'\n\ndef remove_special_characters(batch):\n    batch[\"sentence\"] = re.sub(chars_to_ignore_regex, '', batch[\"sentence\"]) + \" \"\n    return batch\n\ndef normalize(batch):\n    _words = [bnorm(word)['normalized']  for word in batch[\"sentence\"].split()]\n    batch[\"sentence\"] =  \" \".join([word for word in _words if word is not None])\n    return batch\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:52.711131Z","iopub.execute_input":"2024-01-17T18:06:52.712334Z","iopub.status.idle":"2024-01-17T18:06:52.721250Z","shell.execute_reply.started":"2024-01-17T18:06:52.712284Z","shell.execute_reply":"2024-01-17T18:06:52.719962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n<div class=\"alert alert-block alert-info\" style=\"padding:25px; line-height: 1.7em; font-family: Verdana;\">\n    <center style=\"font-family: consolas; font-size: 24px; font-weight: bold; color:Black\"> Tokenize using pretrained processor</center>","metadata":{}},{"cell_type":"markdown","source":"In this step we'll tokenize using the pretrained processor from the [publicly available model ](https://huggingface.co/arijitx/wav2vec2-xls-r-300m-bengali) using Wav2Vec2Processor.from_pretrained(\"arijitx/wav2vec2-xls-r-300m-bengali\")","metadata":{}},{"cell_type":"code","source":"from transformers import Wav2Vec2Processor\n\nprocessor = Wav2Vec2Processor.from_pretrained(\"arijitx/wav2vec2-xls-r-300m-bengali\")","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:52.722969Z","iopub.execute_input":"2024-01-17T18:06:52.723439Z","iopub.status.idle":"2024-01-17T18:06:56.660195Z","shell.execute_reply.started":"2024-01-17T18:06:52.723397Z","shell.execute_reply":"2024-01-17T18:06:56.658845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_dataset(batch):\n\n    batch[\"audio\"][\"array\"] = np.trim_zeros(batch[\"audio\"][\"array\"], 'fb')\n    audio = batch[\"audio\"]\n    \n\n    # batched output is \"un-batched\" to ensure mapping is correct\n    batch[\"input_values\"] = processor(audio[\"array\"], sampling_rate=16000).input_values[0]\n    batch[\"input_length\"] = len(batch[\"input_values\"])\n    \n    #\n    arr = batch['input_values']\n    \n    try:\n        _max = max(max(arr), -min(arr))\n        old_length = len(arr)\n        \n        threshold = 25\n\n        for i,e in enumerate(arr):\n            if threshold*e>_max:\n                break\n\n        for j,e in enumerate(reversed(arr)):\n            if threshold*e>_max:\n                break\n\n        batch['input_values'] = arr[i:old_length-j]\n        batch['input_length'] = old_length -i -j\n        \n    except:\n        print(batch['input_length'])\n    \n    with processor.as_target_processor():\n        batch[\"labels\"] = processor(batch[\"sentence\"]).input_ids\n    return batch\n","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:56.662045Z","iopub.execute_input":"2024-01-17T18:06:56.662875Z","iopub.status.idle":"2024-01-17T18:06:56.676701Z","shell.execute_reply.started":"2024-01-17T18:06:56.662829Z","shell.execute_reply":"2024-01-17T18:06:56.675136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.map(remove_special_characters)\ntrain_ds = train_ds.map(normalize)\ntrain_ds = train_ds.map(prepare_dataset, remove_columns=train_ds.column_names)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T18:06:56.678500Z","iopub.execute_input":"2024-01-17T18:06:56.678888Z","iopub.status.idle":"2024-01-17T19:15:51.190108Z","shell.execute_reply.started":"2024-01-17T18:06:56.678854Z","shell.execute_reply":"2024-01-17T19:15:51.188338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.save_to_disk(\"val\")","metadata":{"execution":{"iopub.status.busy":"2024-01-17T19:24:22.582931Z","iopub.execute_input":"2024-01-17T19:24:22.583381Z","iopub.status.idle":"2024-01-17T19:24:36.841424Z","shell.execute_reply.started":"2024-01-17T19:24:22.583345Z","shell.execute_reply":"2024-01-17T19:24:36.840076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**A complete function with all the steps**","metadata":{}}]}