{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![TPU](https://diamond-thumbnails.s3.us-west-2.amazonaws.com/thinkbigcms/Product/logo/6e276569-90f0-4491-a0ed-45b07b8b05eb.png?hash=f54001f5e45cdaa749504dafc8d86bc4) \n\n\nTPU is an accelerator available on **colab** and **kaggle** which provides a way to train **tensorflow** models way faster than on gpus. To train any data on TPU the dataset has to be converted into **tfrecords** format. In this notebook we demonstarte how to create general purpose tfrecords and how to parse them. Once tfrecord generation is done a separate notebook will be provided on how to use TPU's with tfrecords to accelarate training process. \n\n\n**TPU for (~20x)faster training** (ph-ad)\n* [what is TPU and why do we need them](https://www.quora.com/What-is-TPU-and-GPU-Why-and-when-do-we-need-them)\n* [Kaggle TPU a-z](https://www.kaggle.com/docs/tpu)\n\n**TFRecords**\n* [Official Tensorflow Doc](https://www.tensorflow.org/tutorials/load_data/tfrecord)\n* [basics](https://www.kaggle.com/code/ryanholbrook/tfrecords-basics/notebook)","metadata":{}},{"cell_type":"code","source":"!pip install -U bnunicodenormalizer","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#-------------------------------\n# imports\n#-------------------------------\nimport os\nos.environ[\"CUDA_VISIBLE_DEVICES\"] = \"-1\"\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport tensorflow as tf\n\nimport pandas as pd \nimport warnings\nimport librosa\nimport numpy as np \n\nfrom tqdm.auto import tqdm\nfrom pandarallel import pandarallel\nfrom bnunicodenormalizer import Normalizer \nfrom multiprocessing import Process\n\npandarallel.initialize(progress_bar=True,nb_workers=32)\ntqdm.pandas()\nwarnings.filterwarnings('ignore')\nbnorm=Normalizer()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV Data Loading","metadata":{}},{"cell_type":"code","source":"errors=[\"common_voice_bn_31727562\",\n        'common_voice_bn_30998934',\n        'common_voice_bn_31595526',\n        'common_voice_bn_31534853',\n        'common_voice_bn_31518061',\n        'common_voice_bn_31518373',\n        'common_voice_bn_31613621',\n        'common_voice_bn_31555333',\n        'common_voice_bn_31772113',\n        'common_voice_bn_31605391',\n        'common_voice_bn_31631175',\n        'common_voice_bn_31563901',\n        'common_voice_bn_31691690',\n        'common_voice_bn_31692010',\n        'common_voice_bn_31683653',\n        'common_voice_bn_31692182',\n        'common_voice_bn_31519976',\n        'common_voice_bn_31675793',\n        'common_voice_bn_31019914',\n        'common_voice_bn_31660287',\n        'common_voice_bn_31660384',\n        'common_voice_bn_31557261',\n        'common_voice_bn_31633101',\n        'common_voice_bn_31599243',\n        'common_voice_bn_31521515',\n        'common_voice_bn_31777802',\n        'common_voice_bn_31777848',\n        'common_voice_bn_31669646',\n        'common_voice_bn_31566083',\n        'common_voice_bn_31530331',\n        'common_voice_bn_31727697',\n        'common_voice_bn_31513270',\n        'common_voice_bn_31686295',\n        'common_voice_bn_31753693',\n        'common_voice_bn_31686334',\n        'common_voice_bn_31765546',\n        'common_voice_bn_31765548',\n        'common_voice_bn_31662742',\n        'common_voice_bn_31704856',\n        'common_voice_bn_31635344',\n        'common_voice_bn_31618327',\n        'common_voice_bn_31743074',\n        'common_voice_bn_31678862',\n        'common_voice_bn_31626674',\n        'common_voice_bn_31626677',\n        'common_voice_bn_31523889',\n        'common_voice_bn_31610804',\n        'common_voice_bn_31769538',\n        'common_voice_bn_31533273',\n        'common_voice_bn_31445621',\n        'common_voice_bn_31620650']\n#---------------\n# data filtering\n#---------------\ndef filter_votes(x):\n    p=x[\"path\"]\n    # avoid error data\n    for pe in errors:\n        if pe in p:\n            return None\n    # now process votes\n    up=x[\"up_votes\"]\n    down=x[\"down_votes\"]\n    if up-down<=0:\n        return \"unv\"\n    elif up==0:\n        return \"unv\"\n    else:\n        return up\n# ------------------------- train data----------------------------------------\ntrain_path=\"../input/train-wavs-voted-dl-sprint/train_files_wav\"\n\ndf=pd.read_csv(\"../input/dlsprint/train.csv\")\ndf[\"path\"]=df[\"path\"].progress_apply(lambda x:os.path.join(train_path,x).replace(\".mp3\",\".wav\"))\n\nprint(\"Total Data before filtering:\",len(df))\ndf[\"up_votes\"]=df.progress_apply(lambda x:filter_votes(x),axis=1)\ndf.dropna(subset=[\"up_votes\"],inplace=True)\ntrain_df=df.loc[df.up_votes!=\"unv\"]\ntrain_df.reset_index(drop=True,inplace=True)\nprint(\"Total Data after filtering:\",len(train_df))\ntrain_df=train_df[[\"path\",\"sentence\"]]\n# ------------------------- eval data----------------------------------------\nval_path=\"../input/validation-fileswav-format/validation_files_wav\"\nval_df=pd.read_csv(\"../input/dlsprint/validation.csv\")\nval_df=val_df[[\"path\",\"sentence\"]]\nval_df[\"path\"]=val_df[\"path\"].progress_apply(lambda x:os.path.join(val_path,x).replace(\".mp3\",\".wav\"))\nprint(\"Total validation Data :\",len(val_df))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text processing\n* Normalize\n* create vocab\n* fix-missing vocab\n    * numbers\n    * no space charecter\n    * sep: special token to indicate both start and end\n    * pad: pad token to make all labels the same length\n* find max_label_len\n","metadata":{}},{"cell_type":"code","source":"def normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None]) \n\nval_df[\"sentence\"]=val_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\ntrain_df[\"sentence\"]=train_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\n#unv_df[\"sentence\"]=unv_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Mask non norm vocabs while encoding","metadata":{}},{"cell_type":"code","source":"vocab_norm=['\\u200d',' ','!',\"'\",',','-','.',':',';','=','?','।',\n            'ঁ','ং','ঃ',\n            'অ','আ','ই','ঈ','উ','ঊ','ঋ','এ','ঐ','ও','ঔ',\n            'ক','খ','গ','ঘ','ঙ',\n            'চ','ছ','জ','ঝ','ঞ',\n            'ট','ঠ','ড','ঢ','ণ',\n            'ত','থ','দ','ধ','ন',\n            'প','ফ','ব','ভ','ম',\n            'য','র','ল',\n            'শ','ষ','স','হ',\n            'া','ি','ী','ু','ূ','ৃ','ে','ৈ','ো','ৌ','্',\n            'ৎ','ড়','ঢ়','য়',\n            '০','১','২','৩','৪','৫','৬','৭','৮','৯']\nvocab_transfer=[\" \", \"_\", \"a\", \"b\", \"c\", \"d\", \"e\", \"f\", \"g\", \"h\", \n                \"i\", \"j\", \"k\", \"l\", \"m\", \"n\", \"o\", \"p\", \"r\", \"s\", \n                \"t\", \"u\", \"v\", \"w\", \"x\", \"y\", \"z\", \"\", \"\", \"œ\", \n                \"।\", \"ঁ\", \"ং\", \"ঃ\", \"অ\", \"আ\", \"ই\", \"ঈ\", \"উ\", \"ঊ\", \n                \"ঋ\", \"এ\", \"ঐ\", \"ও\", \"ঔ\", \"ক\", \"খ\", \"গ\", \"ঘ\", \"ঙ\", \n                \"চ\", \"ছ\", \"জ\", \"ঝ\", \"ঞ\", \"ট\", \"ঠ\", \"ড\", \"ঢ\", \"ণ\", \n                \"ত\", \"থ\", \"দ\", \"ধ\", \"ন\", \"প\", \"ফ\", \"ব\", \"ভ\", \"ম\", \n                \"য\", \"র\", \"ল\", \"শ\", \"ষ\", \"স\", \"হ\", \"়\", \"া\", \"ি\",\n                \"ী\", \"ু\", \"ূ\", \"ৃ\", \"ে\", \"ৈ\", \"ো\", \"ৌ\", \"্\", \"ৎ\",\n                \"ৗ\", \"ড়\", \"ঢ়\", \"য়\", \"০\", \"১\", \"২\", \"৩\", \"৪\", \"৫\", \n                \"৬\", \"৭\", \"৮\", \"৯\", \"ৰ\", \"‌\", \"‍\", \"‎\", \"⁇\"]  #[\"\", \"<s>\", \"</s>\"]\nfor c in vocab_transfer:\n    if c not in vocab_norm:\n        idx=vocab_transfer.index(c)\n        vocab_transfer[idx]=\"<empty>\"\n\nvocab=vocab_transfer+[\"\", \"<s>\", \"</s>\"]        \nprint(\"New Vocab:\")\nprint(vocab)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Label Encoding","metadata":{}},{"cell_type":"code","source":"def encode_label(sen):\n    sen=[c for c in sen if c in vocab]\n    label=[vocab.index(c) for c in sen]\n    label=\" \".join([str(x) for x in label])\n    label=bytes(label, \"utf-8\")\n    return label\nsen=train_df.iloc[0,1]\nprint(\"sentence:\",sen)\nprint(\"encoded label:\",encode_label(sen))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TFRecord creation","metadata":{}},{"cell_type":"code","source":"SAMPLE_RATE  = 16000   # the sample rate at which wav were created\n#---------------------------------------------------------------\n# helpers functions\n#---------------------------------------------------------------\ndef create_dir(base,ext):\n    '''\n        creates a directory extending base\n        args:\n            base    =   base path \n            ext     =   the folder to create\n    '''\n    _path=os.path.join(base,ext)\n    if not os.path.exists(_path):\n        os.mkdir(_path)\n    return _path\n\ndef load_data(path):\n    \"\"\"loads a wav\"\"\"\n    wave,_= librosa.load(path, sr=SAMPLE_RATE, mono=True)\n    wave=np.trim_zeros(wave)\n    return tf.audio.encode_wav(tf.expand_dims(wave, axis=-1), sample_rate=SAMPLE_RATE).numpy()\n\n#---------------------------------------------------------------\n# data functions\n#---------------------------------------------------------------\n# feature fuctions\ndef _bytes_feature(list_of_bytestrings):\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=list_of_bytestrings))\n\ndef toTfrecord(df,rnum,rec_path,sidx):\n    '''\n        args:\n            df      :   the dataframe that contains the information to store\n            rnum    :   record number\n            rec_path:   save_path\n            mask_dim:   the dimension of the mask\n    '''\n    tfrecord_name=f'{sidx}_{rnum}.tfrecord'\n    tfrecord_path=os.path.join(rec_path,tfrecord_name) \n    with tf.io.TFRecordWriter(tfrecord_path) as writer:    \n        \n        for idx in tqdm(range(len(df))):\n            try:\n                _path=df.iloc[idx,0]\n                sen=df.iloc[idx,1]\n\n                audio=load_data(_path)\n                label=encode_label(sen)\n                # feature desc\n                data ={ 'audio':_bytes_feature([audio]),\n                        'label':_bytes_feature([label])}\n\n                features=tf.train.Features(feature=data)\n                example= tf.train.Example(features=features)\n                serialized=example.SerializeToString()\n                writer.write(serialized)  \n            except Exception as e:\n                print(_path)\n                print(sen)\n                print(e)\n\n\ndef createRecords(data,save_path,sidx,tf_size=1024):\n    print(f\"Creating TFRECORDS:{save_path}\")\n    for idx in tqdm(range(0,len(data),tf_size)):\n        df        =   data.iloc[idx:idx+tf_size] \n        df.reset_index(drop=True,inplace=True) \n        rnum      =   idx//tf_size\n        toTfrecord(df,rnum,save_path,sidx)\n\n#---------------------------------------------------------------\n# parallel processing functions\n#---------------------------------------------------------------\n    \ndef process_df(df,save_path,split=10240,num_proc=16,tf_size=1024):\n    dfs=[df[idx:idx+split] for idx in range(0,len(df),split)]\n    max_end=len(dfs)\n\n\n    def run(idx):\n        if idx <len(dfs):\n            tf_path=create_dir(save_path,str(idx))\n            createRecords(dfs[idx],tf_path,idx,tf_size)\n\n\n    def execute(start,end):\n        process_list=[]\n        for idx in range(start,end):\n            p =  Process(target= run, args = [idx])\n            p.start()\n            process_list.append(p)\n        for process in process_list:\n            process.join()\n\n\n    if max_end==1:\n        dfs=[df]\n        run(0)\n    else:\n        num_proc=min(max_end,num_proc)\n        for i in range(0,max_end,num_proc):\n            start=i\n            end=start+num_proc\n            if end>max_end:end=max_end-1\n            execute(start,end) \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eval_save=create_dir(os.getcwd(),\"eval\")\ntrain_save=create_dir(os.getcwd(),\"voted\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_df(val_df,eval_save,split=1024,tf_size=256,num_proc=32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"process_df(train_df,train_save,split=1024,tf_size=256,num_proc=32)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}