{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hugging Face🤗 Wav2Vec2.0 Preprocess Notebook\n\n## [This code was very helpful in creating this notebook.](https://colab.research.google.com/github/patrickvonplaten/notebooks/blob/master/Fine_Tune_XLSR_Wav2Vec2_on_Turkish_ASR_with_%F0%9F%A4%97_Transformers.ipynb)","metadata":{}},{"cell_type":"markdown","source":"## Import Libralies","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport yaml\nfrom tqdm import tqdm\nimport json\nfrom transformers import Wav2Vec2CTCTokenizer","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:11:36.201609Z","iopub.execute_input":"2023-07-23T06:11:36.202025Z","iopub.status.idle":"2023-07-23T06:11:38.540071Z","shell.execute_reply.started":"2023-07-23T06:11:36.201990Z","shell.execute_reply":"2023-07-23T06:11:38.538219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load csv","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/bengaliai-speech/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:11:38.542682Z","iopub.execute_input":"2023-07-23T06:11:38.543406Z","iopub.status.idle":"2023-07-23T06:11:44.194082Z","shell.execute_reply.started":"2023-07-23T06:11:38.543363Z","shell.execute_reply":"2023-07-23T06:11:44.193110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:11:44.195273Z","iopub.execute_input":"2023-07-23T06:11:44.195589Z","iopub.status.idle":"2023-07-23T06:11:44.202714Z","shell.execute_reply.started":"2023-07-23T06:11:44.195562Z","shell.execute_reply":"2023-07-23T06:11:44.201562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = df[df['split']!='train']\ntrain = df[df['split']=='train']\nlen(train), len(val)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:11:44.206096Z","iopub.execute_input":"2023-07-23T06:11:44.206554Z","iopub.status.idle":"2023-07-23T06:11:44.660708Z","shell.execute_reply.started":"2023-07-23T06:11:44.206475Z","shell.execute_reply":"2023-07-23T06:11:44.659679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gather vocabulary in the sentences","metadata":{}},{"cell_type":"code","source":"vocab_val = []\nfor i in tqdm(range(len(val))):\n    for j in range(len(val['sentence'].iloc[i])):\n        vocab_val.append(val['sentence'].iloc[i][j:j+1])\n        \nvocab_train = []\nfor i in tqdm(range(len(train))):\n    for j in range(len(train['sentence'].iloc[i])):\n        vocab_train.append(train['sentence'].iloc[i][j:j+1])\n        \n\nvocab_list = sorted(list(set(set(vocab_train) | set(vocab_val))))\nvocab_dict = {v: k for k, v in enumerate(vocab_list)}\nvocab_dict","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:11:44.662192Z","iopub.execute_input":"2023-07-23T06:11:44.662511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add special token","metadata":{}},{"cell_type":"code","source":"vocab_dict[\"|\"] = vocab_dict[\" \"]\ndel vocab_dict[\" \"]\nvocab_dict[\"[UNK]\"] = len(vocab_dict)\nvocab_dict[\"[PAD]\"] = len(vocab_dict)\nlen(vocab_dict)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Seve Vocablary dict","metadata":{}},{"cell_type":"code","source":"with open(f'vocab_bengali.json', 'w') as vocab_file:\n    json.dump(vocab_dict, vocab_file)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"tokenizer = Wav2Vec2CTCTokenizer(f\"./vocab_bengali.json\", unk_token=\"[UNK]\", pad_token=\"[PAD]\", word_delimiter_token=\"|\")\ntokenizer","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save splitted dataframes","metadata":{}},{"cell_type":"code","source":"train.to_csv(f'train_bengali.csv')\nval.to_csv(f'val_bengali.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}