{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![TPU](https://diamond-thumbnails.s3.us-west-2.amazonaws.com/thinkbigcms/Product/logo/6e276569-90f0-4491-a0ed-45b07b8b05eb.png?hash=f54001f5e45cdaa749504dafc8d86bc4) \n\n\nTPU is an accelerator available on **colab** and **kaggle** which provides a way to train **tensorflow** models way faster than on gpus. To train any data on TPU the dataset has to be converted into **tfrecords** format. In this notebook we demonstarte how to create general purpose tfrecords and how to parse them. Once tfrecord generation is done a separate notebook will be provided on how to use TPU's with tfrecords to accelarate training process. \n\n\n**TPU for (~20x)faster training** (ph-ad)\n* [what is TPU and why do we need them](https://www.quora.com/What-is-TPU-and-GPU-Why-and-when-do-we-need-them)\n* [Kaggle TPU a-z](https://www.kaggle.com/docs/tpu)\n\n**TFRecords**\n* [Official Tensorflow Doc](https://www.tensorflow.org/tutorials/load_data/tfrecord)\n* [basics](https://www.kaggle.com/code/ryanholbrook/tfrecords-basics/notebook)","metadata":{}},{"cell_type":"code","source":"!pip install -U bnunicodenormalizer","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#-------------------------------\n# imports\n#-------------------------------\nimport os\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport tensorflow as tf\n\nimport pandas as pd \nimport warnings\nimport librosa\nimport numpy as np \n\nfrom tqdm.auto import tqdm\nfrom pandarallel import pandarallel\nfrom bnunicodenormalizer import Normalizer \nfrom multiprocessing import Process\n\npandarallel.initialize(progress_bar=True,nb_workers=32)\ntqdm.pandas()\nwarnings.filterwarnings('ignore')\nbnorm=Normalizer()","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:29:02.779545Z","iopub.status.busy":"2022-07-07T12:29:02.779039Z","iopub.status.idle":"2022-07-07T12:29:05.106428Z","shell.execute_reply":"2022-07-07T12:29:05.105114Z","shell.execute_reply.started":"2022-07-07T12:29:02.779499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV Data Loading","metadata":{}},{"cell_type":"code","source":"errors=[\"common_voice_bn_31727562\",\n        'common_voice_bn_30998934',\n        'common_voice_bn_31595526',\n        'common_voice_bn_31534853',\n        'common_voice_bn_31518061',\n        'common_voice_bn_31518373',\n        'common_voice_bn_31613621',\n        'common_voice_bn_31555333',\n        'common_voice_bn_31772113',\n        'common_voice_bn_31605391',\n        'common_voice_bn_31631175',\n        'common_voice_bn_31563901',\n        'common_voice_bn_31691690',\n        'common_voice_bn_31692010',\n        'common_voice_bn_31683653',\n        'common_voice_bn_31692182',\n        'common_voice_bn_31519976',\n        'common_voice_bn_31675793',\n        'common_voice_bn_31019914',\n        'common_voice_bn_31660287',\n        'common_voice_bn_31660384',\n        'common_voice_bn_31557261',\n        'common_voice_bn_31633101',\n        'common_voice_bn_31599243',\n        'common_voice_bn_31521515',\n        'common_voice_bn_31777802',\n        'common_voice_bn_31777848',\n        'common_voice_bn_31669646',\n        'common_voice_bn_31566083',\n        'common_voice_bn_31530331',\n        'common_voice_bn_31727697',\n        'common_voice_bn_31513270',\n        'common_voice_bn_31686295',\n        'common_voice_bn_31753693',\n        'common_voice_bn_31686334',\n        'common_voice_bn_31765546',\n        'common_voice_bn_31765548',\n        'common_voice_bn_31662742',\n        'common_voice_bn_31704856',\n        'common_voice_bn_31635344',\n        'common_voice_bn_31618327',\n        'common_voice_bn_31743074',\n        'common_voice_bn_31678862',\n        'common_voice_bn_31626674',\n        'common_voice_bn_31626677',\n        'common_voice_bn_31523889',\n        'common_voice_bn_31610804',\n        'common_voice_bn_31769538',\n        'common_voice_bn_31533273',\n        'common_voice_bn_31445621',\n        'common_voice_bn_31620650']\n#---------------\n# data filtering\n#---------------\ndef filter_votes(x):\n    p=x[\"path\"]\n    # avoid error data\n    for pe in errors:\n        if pe in p:\n            return None\n    # now process votes\n    up=x[\"up_votes\"]\n    down=x[\"down_votes\"]\n    if up-down<=0:\n        return \"unv\"\n    elif up==0:\n        return \"unv\"\n    else:\n        return up\n# ------------------------- train data----------------------------------------\ntrain_path=\"/backup2/cv/kaggle/train_wavs\"\n\ndf=pd.read_csv(\"/backup2/cv/kaggle/train.csv\")\ndf[\"path\"]=df[\"path\"].progress_apply(lambda x:os.path.join(train_path,x).replace(\".mp3\",\".wav\"))\n\nprint(\"Total Data before filtering:\",len(df))\ndf[\"up_votes\"]=df.progress_apply(lambda x:filter_votes(x),axis=1)\ndf.dropna(subset=[\"up_votes\"],inplace=True)\ntrain_df=df.loc[df.up_votes!=\"unv\"]\ntrain_df.reset_index(drop=True,inplace=True)\nprint(\"Total Data after filtering:\",len(train_df))\nunv_df=df.loc[df.up_votes==\"unv\"]\nunv_df.reset_index(drop=True,inplace=True)\nprint(\"Total unverified Data after filtering:\",len(unv_df))\ntrain_df=train_df[[\"path\",\"sentence\"]]\nunv_df=unv_df[[\"path\",\"sentence\"]]\n# ------------------------- eval data----------------------------------------\nval_path=\"/backup2/cv/kaggle/validation_files_wav\"\nval_df=pd.read_csv(\"/backup2/cv/kaggle/validation.csv\")\nval_df=val_df[[\"path\",\"sentence\"]]\nval_df[\"path\"]=val_df[\"path\"].progress_apply(lambda x:os.path.join(val_path,x).replace(\".mp3\",\".wav\"))\nprint(\"Total validation Data :\",len(val_df))","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:15:15.691076Z","iopub.status.busy":"2022-07-07T12:15:15.690452Z","iopub.status.idle":"2022-07-07T12:15:22.990290Z","shell.execute_reply":"2022-07-07T12:15:22.989206Z","shell.execute_reply.started":"2022-07-07T12:15:15.691032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text processing\n* Normalize\n* create vocab\n* fix-missing vocab\n    * numbers\n    * no space charecter\n    * sep: special token to indicate both start and end\n    * pad: pad token to make all labels the same length\n* find max_label_len\n","metadata":{}},{"cell_type":"code","source":"def normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None]) \n\nval_df[\"sentence\"]=val_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\ntrain_df[\"sentence\"]=train_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\nunv_df[\"sentence\"]=unv_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\n","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:18:45.250617Z","iopub.status.busy":"2022-07-07T12:18:45.250161Z","iopub.status.idle":"2022-07-07T12:20:24.320607Z","shell.execute_reply":"2022-07-07T12:20:24.319581Z","shell.execute_reply.started":"2022-07-07T12:18:45.250581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab=[ 'pad','start','end','\\u200d',\n        ' ','!',\"'\",',','-','.',':',';','=','?','।',\n        'ঁ','ং','ঃ',\n        'অ','আ','ই','ঈ','উ','ঊ','ঋ','এ','ঐ','ও','ঔ',\n        'ক','খ','গ','ঘ','ঙ',\n        'চ','ছ','জ','ঝ','ঞ',\n        'ট','ঠ','ড','ঢ','ণ',\n        'ত','থ','দ','ধ','ন',\n        'প','ফ','ব','ভ','ম',\n        'য','র','ল',\n        'শ','ষ','স','হ',\n        'া','ি','ী','ু','ূ','ৃ','ে','ৈ','ো','ৌ','্',\n        'ৎ','ড়','ঢ়','য়',\n        '০','১','২','৩','৪','৫','৬','৭','৮','৯']","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:20:24.324715Z","iopub.status.busy":"2022-07-07T12:20:24.323951Z","iopub.status.idle":"2022-07-07T12:20:25.821947Z","shell.execute_reply":"2022-07-07T12:20:25.821038Z","shell.execute_reply.started":"2022-07-07T12:20:24.324644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label Encoding","metadata":{}},{"cell_type":"code","source":"def encode_label(sen):\n    sen=[c for c in sen if c in vocab]\n    sen=[\"start\"]+sen+[\"end\"]\n    label=[vocab.index(c) for c in sen]\n    label=\" \".join([str(x) for x in label])\n    label=bytes(label, \"utf-8\")\n    return label\nsen=train_df.iloc[0,1]\nprint(\"sentence:\",sen)\nprint(\"encoded label:\",encode_label(sen))","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:47:06.974554Z","iopub.status.busy":"2022-07-07T12:47:06.973682Z","iopub.status.idle":"2022-07-07T12:47:06.982733Z","shell.execute_reply":"2022-07-07T12:47:06.981963Z","shell.execute_reply.started":"2022-07-07T12:47:06.974508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TFRecord creation","metadata":{}},{"cell_type":"code","source":"SAMPLE_RATE  = 16000   # the sample rate at which wav were created\n#---------------------------------------------------------------\n# helpers functions\n#---------------------------------------------------------------\ndef create_dir(base,ext):\n    '''\n        creates a directory extending base\n        args:\n            base    =   base path \n            ext     =   the folder to create\n    '''\n    _path=os.path.join(base,ext)\n    if not os.path.exists(_path):\n        os.mkdir(_path)\n    return _path\n\ndef load_data(path):\n    \"\"\"loads a wav\"\"\"\n    wave,_= librosa.load(path, sr=SAMPLE_RATE, mono=True)\n    wave=np.trim_zeros(wave)\n    return tf.audio.encode_wav(tf.expand_dims(wave, axis=-1), sample_rate=SAMPLE_RATE).numpy()\n\n#---------------------------------------------------------------\n# data functions\n#---------------------------------------------------------------\n# feature fuctions\ndef _bytes_feature(list_of_bytestrings):\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=list_of_bytestrings))\n\ndef toTfrecord(df,rnum,rec_path,sidx):\n    '''\n        args:\n            df      :   the dataframe that contains the information to store\n            rnum    :   record number\n            rec_path:   save_path\n            mask_dim:   the dimension of the mask\n    '''\n    tfrecord_name=f'{sidx}_{rnum}.tfrecord'\n    tfrecord_path=os.path.join(rec_path,tfrecord_name) \n    with tf.io.TFRecordWriter(tfrecord_path) as writer:    \n        \n        for idx in tqdm(range(len(df))):\n            try:\n                _path=df.iloc[idx,0]\n                sen=df.iloc[idx,1]\n\n                audio=load_data(_path)\n                label=encode_label(sen)\n                # feature desc\n                data ={ 'audio':_bytes_feature([audio]),\n                        'label':_bytes_feature([label])}\n\n                features=tf.train.Features(feature=data)\n                example= tf.train.Example(features=features)\n                serialized=example.SerializeToString()\n                writer.write(serialized)  \n            except Exception as e:\n                print(_path)\n                print(sen)\n                print(e)\n\n\ndef createRecords(data,save_path,sidx,tf_size=1024):\n    print(f\"Creating TFRECORDS:{save_path}\")\n    for idx in tqdm(range(0,len(data),tf_size)):\n        df        =   data.iloc[idx:idx+tf_size] \n        df.reset_index(drop=True,inplace=True) \n        rnum      =   idx//tf_size\n        toTfrecord(df,rnum,save_path,sidx)\n\n#---------------------------------------------------------------\n# parallel processing functions\n#---------------------------------------------------------------\n    \ndef process_df(df,save_path,split=10240,num_proc=16,tf_size=1024):\n    dfs=[df[idx:idx+split] for idx in range(0,len(df),split)]\n    max_end=len(dfs)\n\n\n    def run(idx):\n        if idx <len(dfs):\n            tf_path=create_dir(save_path,str(idx))\n            createRecords(dfs[idx],tf_path,idx,tf_size)\n\n\n    def execute(start,end):\n        process_list=[]\n        for idx in range(start,end):\n            p =  Process(target= run, args = [idx])\n            p.start()\n            process_list.append(p)\n        for process in process_list:\n            process.join()\n\n\n    if max_end==1:\n        dfs=[df]\n        run(0)\n    else:\n        num_proc=min(max_end,num_proc)\n        for i in range(0,max_end,num_proc):\n            start=i\n            end=start+num_proc\n            if end>max_end:end=max_end-1\n            execute(start,end) \n    ","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:49:20.884102Z","iopub.status.busy":"2022-07-07T12:49:20.882057Z","iopub.status.idle":"2022-07-07T12:49:20.926115Z","shell.execute_reply":"2022-07-07T12:49:20.924194Z","shell.execute_reply.started":"2022-07-07T12:49:20.884023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_save=create_dir(os.getcwd(),\"eval\")\nunv_save=create_dir(os.getcwd(),\"unverified\")\ntrain_save=create_dir(os.getcwd(),\"voted\")\n","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:49:21.770288Z","iopub.status.busy":"2022-07-07T12:49:21.769454Z","iopub.status.idle":"2022-07-07T12:49:21.775202Z","shell.execute_reply":"2022-07-07T12:49:21.774354Z","shell.execute_reply.started":"2022-07-07T12:49:21.770246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_df(val_df,eval_save,split=1024,tf_size=256,num_proc=32)","metadata":{"execution":{"iopub.execute_input":"2022-07-07T12:49:23.160244Z","iopub.status.busy":"2022-07-07T12:49:23.158919Z","iopub.status.idle":"2022-07-07T12:50:51.456161Z","shell.execute_reply":"2022-07-07T12:50:51.454513Z","shell.execute_reply.started":"2022-07-07T12:49:23.160203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_df(train_df,train_save,split=1024,tf_size=256,num_proc=32)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_df(unv_df,unv_save,split=1024,tf_size=256,num_proc=32)","metadata":{},"execution_count":null,"outputs":[]}]}