{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Sample files","metadata":{}},{"cell_type":"code","source":"from pydub import AudioSegment\npath1 = '/kaggle/input/ben10/ben10/16_kHz_train_audio/train_barishal (1).wav'\ndisplay(AudioSegment.from_file(path1))\npath2 = '/kaggle/input/ben10/ben10/16_kHz_valid_audio/valid_chittagong (110).wav'\ndisplay(AudioSegment.from_file(path2))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:45.435184Z","iopub.execute_input":"2024-04-08T20:27:45.435591Z","iopub.status.idle":"2024-04-08T20:27:46.527612Z","shell.execute_reply.started":"2024-04-08T20:27:45.435561Z","shell.execute_reply":"2024-04-08T20:27:46.526394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initial comments on the dataset\n\nThe recordings are **16kHz *.wav** audio files. The data are coming from 373 different speakers from 10 regions in Bangladesh **(rangpur, kishoreganj, narail, chittagong, narsingdi, tangail, habiganj, barishal, sylhet, sandwip)** There are in total 13610 samples in the train set (63:08:42 hrs) and 1703 samples in the phase 1 test set (8:01:11 hrs).\n\n**<> notation** is present throughout the train and test sets where the there exists a word, but not intelligible (due to noise, slurred speech, overlap etc). **Newline operators (\\n)** are present only in the training data but not in the test ground truth files.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport soundfile as sf\nfrom pydub import AudioSegment\nimport IPython.display as ipd\nfrom collections import Counter\nimport os\nimport librosa\nimport time\nfrom multiprocessing import Pool\nimport seaborn as sns\nimport random\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:46.529770Z","iopub.execute_input":"2024-04-08T20:27:46.530160Z","iopub.status.idle":"2024-04-08T20:27:49.703620Z","shell.execute_reply.started":"2024-04-08T20:27:46.530121Z","shell.execute_reply":"2024-04-08T20:27:49.702292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset Observation","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/ben10/ben10/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:49.705175Z","iopub.execute_input":"2024-04-08T20:27:49.705919Z","iopub.status.idle":"2024-04-08T20:27:50.070596Z","shell.execute_reply.started":"2024-04-08T20:27:49.705869Z","shell.execute_reply":"2024-04-08T20:27:50.069508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.073231Z","iopub.execute_input":"2024-04-08T20:27:50.074112Z","iopub.status.idle":"2024-04-08T20:27:50.098117Z","shell.execute_reply.started":"2024-04-08T20:27:50.074077Z","shell.execute_reply":"2024-04-08T20:27:50.096659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/ben10/sample_submission.csv')\ntest.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.099774Z","iopub.execute_input":"2024-04-08T20:27:50.100148Z","iopub.status.idle":"2024-04-08T20:27:50.128095Z","shell.execute_reply.started":"2024-04-08T20:27:50.100114Z","shell.execute_reply":"2024-04-08T20:27:50.126912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"regions = ['rangpur', 'kishoreganj', 'narail', 'chittagong', 'narsingdi', 'tangail', 'habiganj', 'barishal', 'sylhet', 'sandwip']\nsplits = ['train', 'valid']","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.130255Z","iopub.execute_input":"2024-04-08T20:27:50.130648Z","iopub.status.idle":"2024-04-08T20:27:50.138523Z","shell.execute_reply.started":"2024-04-08T20:27:50.130618Z","shell.execute_reply":"2024-04-08T20:27:50.136612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def label_file_region(file_path):\n    for region in regions:\n        if region in file_path:\n            return str.capitalize(region)\n        \ndef label_file_split(file_path):\n    for split in splits:\n        if split in file_path:\n            return split","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.140095Z","iopub.execute_input":"2024-04-08T20:27:50.140437Z","iopub.status.idle":"2024-04-08T20:27:50.151767Z","shell.execute_reply.started":"2024-04-08T20:27:50.140410Z","shell.execute_reply":"2024-04-08T20:27:50.150361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['region'] = df['file_name'].copy().map(label_file_region)\ndf['split'] = df['file_name'].copy().map(label_file_split)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.152950Z","iopub.execute_input":"2024-04-08T20:27:50.153271Z","iopub.status.idle":"2024-04-08T20:27:50.194030Z","shell.execute_reply.started":"2024-04-08T20:27:50.153245Z","shell.execute_reply":"2024-04-08T20:27:50.192911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['region'] = test['id'].copy().map(label_file_region)\ntest['split'] = test['id'].copy().map(label_file_split)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.196025Z","iopub.execute_input":"2024-04-08T20:27:50.196504Z","iopub.status.idle":"2024-04-08T20:27:50.209059Z","shell.execute_reply.started":"2024-04-08T20:27:50.196462Z","shell.execute_reply":"2024-04-08T20:27:50.207499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.split.unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.215884Z","iopub.execute_input":"2024-04-08T20:27:50.216484Z","iopub.status.idle":"2024-04-08T20:27:50.234966Z","shell.execute_reply.started":"2024-04-08T20:27:50.216442Z","shell.execute_reply":"2024-04-08T20:27:50.233247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.split.unique()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.237539Z","iopub.execute_input":"2024-04-08T20:27:50.238617Z","iopub.status.idle":"2024-04-08T20:27:50.248531Z","shell.execute_reply.started":"2024-04-08T20:27:50.238579Z","shell.execute_reply":"2024-04-08T20:27:50.246901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Obsevations\nThe train.csv only has files for train audios. Sample submission csv file has a sample answer for each validation audio files. So we have to split the train set for training and testing.","metadata":{}},{"cell_type":"markdown","source":"## Listening some audios with their corresponding transcripts from train set\n\nLet's hear some of the audios from different regions","metadata":{}},{"cell_type":"code","source":"df.sample(10).index","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.251024Z","iopub.execute_input":"2024-04-08T20:27:50.251417Z","iopub.status.idle":"2024-04-08T20:27:50.263780Z","shell.execute_reply.started":"2024-04-08T20:27:50.251389Z","shell.execute_reply":"2024-04-08T20:27:50.262581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = \"/kaggle/input/ben10/ben10/16_kHz_train_audio\"\n\nfor idx in df.sample(10).index:\n    \n    file_path = os.path.join(root_path, df['file_name'].iloc[idx])\n    text = df['transcripts'].iloc[idx]\n    region = df['region'].iloc[idx]\n    print(f'Region: {str.capitalize(region)}')\n    display(AudioSegment.from_file(file_path))\n    print(f\"Original transcription : {text}\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:50.266399Z","iopub.execute_input":"2024-04-08T20:27:50.266863Z","iopub.status.idle":"2024-04-08T20:27:52.349227Z","shell.execute_reply.started":"2024-04-08T20:27:50.266831Z","shell.execute_reply":"2024-04-08T20:27:52.347974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\nMultiple persons speaking in a single audio. This might be a problem.","metadata":{}},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:52.350638Z","iopub.execute_input":"2024-04-08T20:27:52.351082Z","iopub.status.idle":"2024-04-08T20:27:52.370081Z","shell.execute_reply.started":"2024-04-08T20:27:52.351047Z","shell.execute_reply":"2024-04-08T20:27:52.368570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This part has been inspired from [this](https://www.kaggle.com/code/mbmmurad/eda-resources-and-submission) notebook.\n### 1. Distribution accross regions","metadata":{}},{"cell_type":"code","source":"# Plot the distributions\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# df.region.value_counts().sort_values().plot(kind='barh', ax=axes[1])\nsns.countplot(data=df, y='region', ax = axes[0], order=df['region'].value_counts().index)\naxes[0].set_title('Distribution of Regions in training set')\n\n# test.region.value_counts().sort_values().plot(kind='barh', ax=axes[1])\nsns.countplot(data=test, y='region', ax = axes[1], order=test['region'].value_counts().index)\naxes[1].set_title('Distribution of Regions in test')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:52.371681Z","iopub.execute_input":"2024-04-08T20:27:52.372171Z","iopub.status.idle":"2024-04-08T20:27:53.149392Z","shell.execute_reply.started":"2024-04-08T20:27:52.372127Z","shell.execute_reply":"2024-04-08T20:27:53.148108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\n- The distribution for train set and test set is quite similar\n- Sylhet region has the most samples and barishal has the least","metadata":{}},{"cell_type":"markdown","source":"### 2. Histogram of durations for files","metadata":{}},{"cell_type":"code","source":"train_dir = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/\"\npaths = [train_dir+path for path in os.listdir(train_dir)]\n\ndef get_duration(file):\n    try:\n        # Load audio file\n        y, sr = librosa.load(file, sr=None)\n        # Calculate duration\n        duration = librosa.get_duration(y=y, sr=sr)\n        return duration\n    except Exception as e:\n        return file, None\n\ndef get_durations_parallel(files):\n    with Pool() as pool:\n        results = pool.map(get_duration, files)\n    return results\n\n\n# Getting durations here. Also checking how much time does it take to load all the audios\nstart = time.time()\ndurations_train = get_durations_parallel(paths)\nprint(\"Total time taken in seconds : \",time.time()-start)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:27:53.151093Z","iopub.execute_input":"2024-04-08T20:27:53.151554Z","iopub.status.idle":"2024-04-08T20:29:24.653291Z","shell.execute_reply.started":"2024-04-08T20:27:53.151515Z","shell.execute_reply":"2024-04-08T20:29:24.648687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (12, 6))\nsns.histplot(data=durations_train, bins = [i for i in range(0,31,1)])\nplt.xticks(np.arange(0, 35, step=1))\nplt.xlabel('Duration (s)')\nplt.ylabel('Frequency')\nplt.title('Training audio file durations')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:24.660564Z","iopub.execute_input":"2024-04-08T20:29:24.662828Z","iopub.status.idle":"2024-04-08T20:29:25.414504Z","shell.execute_reply.started":"2024-04-08T20:29:24.662742Z","shell.execute_reply":"2024-04-08T20:29:25.412736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\n- Most of the audios are 15-18s clips. Quite large tbh!\n- There are some very large clips too(>20s). Might be a good idea to split them in shorter clips!","metadata":{}},{"cell_type":"markdown","source":"### 3. Length distribution of transcripts","metadata":{}},{"cell_type":"code","source":"df['transcripts'][0]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:25.420974Z","iopub.execute_input":"2024-04-08T20:29:25.421480Z","iopub.status.idle":"2024-04-08T20:29:25.434013Z","shell.execute_reply.started":"2024-04-08T20:29:25.421420Z","shell.execute_reply":"2024-04-08T20:29:25.432637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['transcripts'][0].split()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:25.436810Z","iopub.execute_input":"2024-04-08T20:29:25.437609Z","iopub.status.idle":"2024-04-08T20:29:25.453567Z","shell.execute_reply.started":"2024-04-08T20:29:25.437562Z","shell.execute_reply":"2024-04-08T20:29:25.452018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['sen_len'] = df['transcripts'].apply(lambda x: len(x.split()))\n\nplt.figure(figsize=(8, 6))\nsns.histplot(data = df, x = 'sen_len' ,bins=[i for i in range(0,151,10)])\n# sns.histplot(data = df, x = 'sen_len', bins=10)\nplt.xticks(np.arange(0, 150, step=10))\nplt.xlabel('Sentence Length')\nplt.title('Train sentence length distributions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:25.455730Z","iopub.execute_input":"2024-04-08T20:29:25.457132Z","iopub.status.idle":"2024-04-08T20:29:26.001579Z","shell.execute_reply.started":"2024-04-08T20:29:25.457064Z","shell.execute_reply":"2024-04-08T20:29:26.000263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['sen_len']==0]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:26.003493Z","iopub.execute_input":"2024-04-08T20:29:26.003916Z","iopub.status.idle":"2024-04-08T20:29:26.023442Z","shell.execute_reply.started":"2024-04-08T20:29:26.003883Z","shell.execute_reply":"2024-04-08T20:29:26.022037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['sen_len']<5]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:26.025379Z","iopub.execute_input":"2024-04-08T20:29:26.025791Z","iopub.status.idle":"2024-04-08T20:29:26.048958Z","shell.execute_reply.started":"2024-04-08T20:29:26.025760Z","shell.execute_reply":"2024-04-08T20:29:26.047080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = df[df['sen_len']>60]\nidx = [random.randint(0,len(sample)) for _ in range(5)]\nfor i in sample.transcripts.iloc[idx].tolist():\n    print(\"Sample sentence : \",i,\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:29:26.051216Z","iopub.execute_input":"2024-04-08T20:29:26.051945Z","iopub.status.idle":"2024-04-08T20:29:26.070194Z","shell.execute_reply.started":"2024-04-08T20:29:26.051898Z","shell.execute_reply":"2024-04-08T20:29:26.068374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\n- There are some really large sentences\n- There are some audios without transcriptions. Some has garbage transcripts\n- These are really very large clips. Usually in ASR it's better to use shorter sentences. These longer sentences basically contains a large dialogue","metadata":{}},{"cell_type":"markdown","source":"### 4. Checking the vocabulary list and character list","metadata":{}},{"cell_type":"code","source":"vocab = {}\nfor sen in tqdm(df.transcripts):\n    for j in sen.split(\" \"):\n        try:\n            vocab[j] += 1\n        except:\n            vocab[j] = 1\nprint(\"Total words in vocabulary : \",len(vocab))\n\nsorted_vocab = sorted(vocab.items(), key = lambda kv: kv[1], reverse=True)\nsorted_vocab[:30]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T21:07:22.553527Z","iopub.execute_input":"2024-04-08T21:07:22.553973Z","iopub.status.idle":"2024-04-08T21:07:22.898376Z","shell.execute_reply.started":"2024-04-08T21:07:22.553938Z","shell.execute_reply":"2024-04-08T21:07:22.897167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_vocab[-50:]","metadata":{"execution":{"iopub.status.busy":"2024-04-08T21:07:22.901276Z","iopub.execute_input":"2024-04-08T21:07:22.902231Z","iopub.status.idle":"2024-04-08T21:07:22.915842Z","shell.execute_reply.started":"2024-04-08T21:07:22.902176Z","shell.execute_reply":"2024-04-08T21:07:22.914565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chars = {}\nfor sen in tqdm(df.transcripts):\n    for j in sen:\n        try:\n            chars[j] += 1\n        except:\n            chars[j] = 1","metadata":{"execution":{"iopub.status.busy":"2024-04-08T20:31:13.743570Z","iopub.execute_input":"2024-04-08T20:31:13.744018Z","iopub.status.idle":"2024-04-08T20:31:14.594066Z","shell.execute_reply.started":"2024-04-08T20:31:13.743985Z","shell.execute_reply":"2024-04-08T20:31:14.593055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chars.items()","metadata":{"execution":{"iopub.status.busy":"2024-04-08T21:07:32.682502Z","iopub.execute_input":"2024-04-08T21:07:32.683599Z","iopub.status.idle":"2024-04-08T21:07:32.692023Z","shell.execute_reply.started":"2024-04-08T21:07:32.683563Z","shell.execute_reply":"2024-04-08T21:07:32.690755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total characters :\", len(chars))","metadata":{"execution":{"iopub.status.busy":"2024-04-08T21:08:29.393044Z","iopub.execute_input":"2024-04-08T21:08:29.393468Z","iopub.status.idle":"2024-04-08T21:08:29.399507Z","shell.execute_reply.started":"2024-04-08T21:08:29.393438Z","shell.execute_reply":"2024-04-08T21:08:29.398508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Observations\n\n- There are a lot of unnecessary characters. Sentences will need to be normalized before training","metadata":{}},{"cell_type":"markdown","source":"## Text Normalization\n\nThe necessity of text normalization for bangla can be found in this [notebook](https://www.kaggle.com/code/mbmmurad/detailed-eda-normalizer-and-wer?scriptVersionId=141093405&cellId=61). \n\nA compact answer for the usecase of the normalizer is that it takes series of steps to preprocess the text, including:\n\n- Unicode normalization \n- Fixing quotes \n- Replacing URLs, punctuation, emojis, and specific characters\n- Cleaning up whitespace. \n\nThe specific replacements and normalization steps can be customized based on the parameters provided to the normalize function\n","metadata":{}},{"cell_type":"markdown","source":"### Installation","metadata":{}},{"cell_type":"code","source":"!pip install git+https://github.com/csebuetnlp/normalizer","metadata":{"execution":{"iopub.status.busy":"2024-04-08T22:40:23.658834Z","iopub.execute_input":"2024-04-08T22:40:23.659161Z","iopub.status.idle":"2024-04-08T22:40:48.992416Z","shell.execute_reply.started":"2024-04-08T22:40:23.659136Z","shell.execute_reply":"2024-04-08T22:40:48.991271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from normalizer import normalize\n\n# Applying on the whole sentence\n\nsentence = \"এমনকি উকুন ঘরবাড়ি ও খাদ্য-সম্ভারের উপর ছড়িয়ে পড়তে লাগল।\"\nnormalized = normalize(sentence)\nprint(normalized)\nprint(normalize==sentence)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T22:41:22.977394Z","iopub.execute_input":"2024-04-08T22:41:22.978304Z","iopub.status.idle":"2024-04-08T22:41:23.203112Z","shell.execute_reply.started":"2024-04-08T22:41:22.978263Z","shell.execute_reply":"2024-04-08T22:41:23.202019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use case for word level\nnormalize(\"নিয়ে\")==normalize(\"নিয়ে\")","metadata":{"execution":{"iopub.status.busy":"2024-04-08T22:42:05.033776Z","iopub.execute_input":"2024-04-08T22:42:05.034473Z","iopub.status.idle":"2024-04-08T22:42:05.043631Z","shell.execute_reply.started":"2024-04-08T22:42:05.034430Z","shell.execute_reply":"2024-04-08T22:42:05.042946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}