{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install bnunicodenormalizer\n!pip install image-classifiers","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:06:31.710278Z","iopub.execute_input":"2022-07-18T10:06:31.711054Z","iopub.status.idle":"2022-07-18T10:06:49.811183Z","shell.execute_reply.started":"2022-07-18T10:06:31.710961Z","shell.execute_reply":"2022-07-18T10:06:49.810021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#-------------------------------\n# imports\n#-------------------------------\nimport os\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \nimport tensorflow as tf\n\nimport pandas as pd \nimport warnings\nimport random\nimport matplotlib.pyplot as plt\nimport json\n\nfrom tqdm.auto import tqdm\nfrom pandarallel import pandarallel\nfrom IPython.display import display,Audio\nfrom bnunicodenormalizer import Normalizer \nfrom pprint import pprint\nfrom multiprocessing import Process\nfrom classification_models.tfkeras import Classifiers\n\npandarallel.initialize(progress_bar=True,nb_workers=8)\ntqdm.pandas()\nwarnings.filterwarnings('ignore')\nbnorm=Normalizer()\n\ngpus = tf.config.experimental.list_physical_devices('GPU')\nprint(gpus)\nfor gpu in gpus:\n    tf.config.experimental.set_memory_growth(gpu, True)\nstrategy = tf.distribute.get_strategy()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:06:49.814375Z","iopub.execute_input":"2022-07-18T10:06:49.814753Z","iopub.status.idle":"2022-07-18T10:06:51.511321Z","shell.execute_reply.started":"2022-07-18T10:06:49.814717Z","shell.execute_reply":"2022-07-18T10:06:51.510255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CSV Data Loading","metadata":{}},{"cell_type":"code","source":"errors=[\"common_voice_bn_31727562\",\n        'common_voice_bn_30998934',\n        'common_voice_bn_31595526',\n        'common_voice_bn_31534853',\n        'common_voice_bn_31518061',\n        'common_voice_bn_31518373',\n        'common_voice_bn_31613621',\n        'common_voice_bn_31555333',\n        'common_voice_bn_31772113',\n        'common_voice_bn_31605391',\n        'common_voice_bn_31631175',\n        'common_voice_bn_31563901',\n        'common_voice_bn_31691690',\n        'common_voice_bn_31692010',\n        'common_voice_bn_31683653',\n        'common_voice_bn_31692182',\n        'common_voice_bn_31519976',\n        'common_voice_bn_31675793',\n        'common_voice_bn_31019914',\n        'common_voice_bn_31660287',\n        'common_voice_bn_31660384',\n        'common_voice_bn_31557261',\n        'common_voice_bn_31633101',\n        'common_voice_bn_31599243',\n        'common_voice_bn_31521515',\n        'common_voice_bn_31777802',\n        'common_voice_bn_31777848',\n        'common_voice_bn_31669646',\n        'common_voice_bn_31566083',\n        'common_voice_bn_31530331',\n        'common_voice_bn_31727697',\n        'common_voice_bn_31513270',\n        'common_voice_bn_31686295',\n        'common_voice_bn_31753693',\n        'common_voice_bn_31686334',\n        'common_voice_bn_31765546',\n        'common_voice_bn_31765548',\n        'common_voice_bn_31662742',\n        'common_voice_bn_31704856',\n        'common_voice_bn_31635344',\n        'common_voice_bn_31618327',\n        'common_voice_bn_31743074',\n        'common_voice_bn_31678862',\n        'common_voice_bn_31626674',\n        'common_voice_bn_31626677',\n        'common_voice_bn_31523889',\n        'common_voice_bn_31610804',\n        'common_voice_bn_31769538',\n        'common_voice_bn_31533273',\n        'common_voice_bn_31445621',\n        'common_voice_bn_31620650']\n\n#---------------\n# data filtering\n#---------------\ndef filter_votes(x):\n    p=x[\"path\"]\n    # avoid error data\n    for pe in errors:\n        if pe in p:\n            return None\n        \n    up=x[\"up_votes\"]\n    down=x[\"down_votes\"]\n    if up-down<=0:\n        return None\n    elif up==0:\n        return None\n    else:\n        return up\n# ------------------------- train data----------------------------------------\ntrain_path=\"../input/train-wavs-voted-dl-sprint/train_files_wav\"\ntrain_df=pd.read_csv(\"../input/dlsprint/train.csv\")\nprint(\"Total Data before filtering:\",len(train_df))\ntrain_df[\"up_votes\"]=train_df.progress_apply(lambda x:filter_votes(x),axis=1)\ntrain_df.dropna(subset = ['up_votes'],inplace=True)\nprint(\"Total Data after filtering:\",len(train_df))\ntrain_df[\"path\"]=train_df[\"path\"].progress_apply(lambda x:os.path.join(train_path,x).replace(\".mp3\",\".wav\"))\ntrain_df=train_df[[\"path\",\"sentence\"]]\n\n\n# ------------------------- eval data----------------------------------------\nval_path=\"../input/validation-fileswav-format/validation_files_wav/\"\nval_df=pd.read_csv(\"../input/dlsprint/validation.csv\")\nval_df=val_df[[\"path\",\"sentence\"]]\nval_df[\"path\"]=val_df[\"path\"].progress_apply(lambda x:os.path.join(val_path,x).replace(\".mp3\",\".wav\"))\nprint(\"Total validation Data :\",len(val_df))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:07:21.748131Z","iopub.execute_input":"2022-07-18T10:07:21.748889Z","iopub.status.idle":"2022-07-18T10:07:29.016832Z","shell.execute_reply.started":"2022-07-18T10:07:21.748821Z","shell.execute_reply":"2022-07-18T10:07:29.015872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Log Mel Spectogram \n(ph-ad)\n* resources:\n    * https://www.youtube.com/watch?v=9GHCiiDLHQ4\n    * https://www.youtube.com/watch?v=hF72sY70_IQ\n    * https://ieeexplore.ieee.org/document/9383491\n\n### **This class will be used for inference as well**\n* required imports for the class to work : ```librosa```,```numpy``` \n\n\n\n## padding log mel spectogram images\n\n** ```max_width=1088``` in class initialization. Why? \n\n1. lms.frame_step= 160 # in our default setting\n   \n2. features.shape=(64,x,1)--> here x depends on len of audio . and according to [this analysis](https://www.kaggle.com/code/nazmuddhohaansary/wav-trimming-and-length-analysis) , max audio 1d len= 169982\n\n3. so MAX_WIDTH has to be >= ```169982/160===1063``` \n4. The standard input to output dimension ratio for a model is ```32```. So to make the data be useable for standard cnn models we have to keep this factor in mind and set the width and height that can be downfactored by 32. \n\n**we can select any SOTA model without top (except Inception models) and check the input and output- height and width ratio with randomly given height and width to verify this**\n\n```python \n#-------\nmodels=[tf.keras.applications.resnet50.ResNet50(include_top=False,weights=None,input_shape=(64,512,3)),\n        tf.keras.applications.mobilenet.MobileNet(include_top=False,weights=None,input_shape=(256,256,3)),\n        tf.keras.applications.vgg19.VGG19(include_top=False,weights=None,input_shape=(224,128,3))]\n\nfor model in models:\n    i=model.input\n    o=model.output\n    print(\"height ratio:\",i.shape[1]//o.shape[1])\n    print(\"width ration:\",i.shape[2]//o.shape[2])\n    print(\"-----------------------------------------------------------\")\n    \n```\n\n> the output \n\n```\nheight ratio: 32\nwidth ration: 32\n-----------------------------------------------------------\nheight ratio: 32\nwidth ration: 32\n-----------------------------------------------------------\nheight ratio: 32\nwidth ration: 32\n```\n\n5. since 32x33=1056 < 1063 we choose- 32x34=1088 as the max_width\n","metadata":{}},{"cell_type":"code","source":"#-------------------------------------------------------\n# Log Mel Spectogram feature processor\n#-------------------------------------------------------\nimport librosa\nimport numpy as np  \n\nclass LogMelSpectrogramProcesor(object):\n    def __init__(self,\n                 sample_rate=16000,\n                 preemphasis_coeff=0.97,\n                 frame_ms=25,\n                 stride_ms=10,\n                 center=True,\n                 num_feature_bins=64,\n                 max_width=1088):\n        \"\"\"\n            class to process and extract log mel spectrogram features from audio\n        \"\"\"\n        \n        self.sample_rate          =  sample_rate\n        self.frame_length         =  int(self.sample_rate * (frame_ms / 1000))\n        self.frame_step           =  int(self.sample_rate * (stride_ms/ 1000))\n        self.preemphasis_coeff    =  preemphasis_coeff\n        self.center               =  center\n        self.num_feature_bins     =  num_feature_bins\n        self.nfft                 =  2 ** (self.frame_length - 1).bit_length()\n        self.max_audio_len        =  self.frame_step*max_width -1  \n        \n    #------------------------------------------------------------\n    # generally useable functions for audio processing\n    #------------------------------------------------------------\n    def load_data(self,path):\n        \"\"\"loads a wav\"\"\"\n        wave,_= librosa.load(path, sr=self.sample_rate, mono=True)\n        wave=np.trim_zeros(wave)\n        return wave\n    \n    def normalize_signal(self,signal):\n        \"\"\"Normailize signal to [-1, 1] range\"\"\"\n        gain = 1.0 / (np.max(np.abs(signal)) + 1e-9)\n        return signal * gain\n\n    def normalize_audio_feature(self,audio_feature):\n        \"\"\"Mean and variance normalization\"\"\"\n        mean = np.mean(audio_feature)\n        std_dev = np.sqrt(np.var(audio_feature) + 1e-9)\n        normalized = (audio_feature - mean) / std_dev\n        return normalized\n\n    def preemphasis(self,signal):\n        \"\"\"\n        Apply Pre-emphasis with the defined preemphasis coefficient \n        ** preemphasis: the intentional alteration of the relative strengths \n        of signals at different frequencies (as in radio and in disc recording) \n        to reduce adverse effects (as noise) in the following parts of the system.\n\n        \"\"\"\n        return np.append(signal[0], signal[1:] - self.preemphasis_coeff * signal[:-1])\n    \n    def pad_signal(self,signal):\n        '''\n            pads a 1d array with 0\n        '''\n        # shape\n        _len=signal.shape[0]\n        # pads\n        pad =np.array([0.0 for _ in range(self.max_audio_len-_len)])\n        # pad\n        signal =np.concatenate([signal,pad])\n        return signal\n    #------------------------------------------------------------\n    # log mel spectrogram specific\n    #------------------------------------------------------------\n    \n    def power_to_db(self,S,ref=1.0,amin=1e-10,top_db=80.0):\n        \"\"\"wrapper for libroba power to db\"\"\"\n        return librosa.power_to_db(S, ref=ref, amin=amin, top_db=top_db)\n    \n    def stft(self,signal):\n        return np.square(np.abs(librosa.stft( signal,\n                                              n_fft=self.nfft,\n                                              hop_length=self.frame_step,\n                                              win_length=self.frame_length,\n                                              center=self.center,\n                                              window=\"hann\")))\n\n    def compute_log_mel_spectrogram(self,signal):\n        S = self.stft(signal)\n        mel = librosa.filters.mel(self.sample_rate, \n                                  self.nfft, \n                                  n_mels=self.num_feature_bins, \n                                  fmin=0.0, \n                                  fmax=int(self.sample_rate / 2))\n        mel_spectrogram = np.dot(S.T, mel.T)\n        return self.power_to_db(mel_spectrogram)\n    #------------------------------------------------------------\n    def __call__(self,path):\n        \"\"\"\n        Extract speech features from signals \n        * normalizes audio signal\n        * preemphasis on audio\n        * computes log mel spectrogram\n        * normalizes the features\n        \"\"\"\n        # signal\n        signal=self.load_data(path)\n        signal = np.asfortranarray(signal)\n        # normalize\n        signal = self.normalize_signal(signal)\n        # preemphasis\n        signal = self.preemphasis(signal)\n        # pad data\n        signal = self.pad_signal(signal)\n        # log mel spectrogram features\n        features = self.compute_log_mel_spectrogram(signal)\n        features = features.T\n        # expand axis\n        features = np.expand_dims(features, axis=-1)\n        # normalize feats\n        features = self.normalize_audio_feature(features)\n        return features\n\n#------------------------------------------------------- \nlms=LogMelSpectrogramProcesor()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:07:37.092213Z","iopub.execute_input":"2022-07-18T10:07:37.092573Z","iopub.status.idle":"2022-07-18T10:07:37.829955Z","shell.execute_reply.started":"2022-07-18T10:07:37.092545Z","shell.execute_reply":"2022-07-18T10:07:37.828808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What is happening when the processor is being called? \n1. The data is being loaded from the provided path\n2. The data is being normalized from -1 to 1\n3. Preemphasis is being applied\n\n```\npreemphasis: the intentional alteration of the relative strengths \nof signals at different frequencies (as in radio and in disc recording) \nto reduce adverse effects (as noise) in the following parts of the system.\n```\n4. this audio data is then transformed into a 2d array that contains log-mel-spectrogram features \n\nLets inspect step by step.\n\n\nSelect a random data to inspect","metadata":{}},{"cell_type":"code","source":"idx=random.randint(0,len(train_df)-1)\n_path=train_df.iloc[idx,0]\n# loading\nsignal=lms.load_data(_path)\nsignal = np.asfortranarray(signal)\nprint(\"shape of file:\",signal.shape)\nprint(\"Trimmed raw audio\")\ndisplay(Audio(data=signal, rate=16000))\n\n# normalize\nsignal = lms.normalize_signal(signal)\nprint(\"normalized signal\")\ndisplay(Audio(data=signal, rate=16000))\n\n# preemphasis\nsignal = lms.preemphasis(signal)\nprint(\"applying preemphasis\")\ndisplay(Audio(data=signal, rate=16000))\n\n# padding\nsignal = lms.pad_signal(signal)\nprint(\"padded signal\")\ndisplay(Audio(data=signal, rate=16000))\n\n# log mel spectrogram (2d feature)\nfeatures = lms.compute_log_mel_spectrogram(signal)\nfeatures = features.T\nfeatures = np.expand_dims(features, axis=-1)\nfeatures = lms.normalize_audio_feature(features)\nprint(\"2D log mel spectrogram transposed\")\nprint(\"shape of feature:\",features.shape)\nplt.figure(figsize=(20,20))\nplt.imshow(features)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:07:41.496463Z","iopub.execute_input":"2022-07-18T10:07:41.497113Z","iopub.status.idle":"2022-07-18T10:07:41.902888Z","shell.execute_reply.started":"2022-07-18T10:07:41.497061Z","shell.execute_reply":"2022-07-18T10:07:41.901905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text processing\n* Normalize\n* create vocab\n* fix-missing vocab\n    * numbers\n    * sep: special token to indicate both start and end\n    * pad: pad token to make all labels the same length\n* find max_label_len\n","metadata":{}},{"cell_type":"code","source":"def normalize(sen):\n    _words = [bnorm(word)['normalized']  for word in sen.split()]\n    return \" \".join([word for word in _words if word is not None]) \n\nval_df[\"sentence\"]=val_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\ntrain_df[\"sentence\"]=train_df[\"sentence\"].parallel_apply(lambda x:normalize(x))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:07:43.010812Z","iopub.execute_input":"2022-07-18T10:07:43.011272Z","iopub.status.idle":"2022-07-18T10:10:40.831285Z","shell.execute_reply.started":"2022-07-18T10:07:43.011221Z","shell.execute_reply":"2022-07-18T10:10:40.830152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sens=train_df[\"sentence\"].tolist()+val_df[\"sentence\"].tolist()\nprint(\"number of total sentences:\",len(sens))\nvocab=[]\nfor sen in tqdm(sens):\n    for c in sen:\n        if c not in vocab:\n            vocab.append(c)\nvocab=sorted(vocab)\nprint(\"vocab(unicodes):\")\nfor idx,c in enumerate(vocab):\n    if idx%20==0:print()\n    print(c,end=\",\")        ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:10:40.834099Z","iopub.execute_input":"2022-07-18T10:10:40.834508Z","iopub.status.idle":"2022-07-18T10:10:42.186530Z","shell.execute_reply.started":"2022-07-18T10:10:40.834472Z","shell.execute_reply":"2022-07-18T10:10:42.185544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nvocab=['pad','sep']+sorted(vocab+['০', '১', '২', '৩', '৪', '৫', '৬', '৭', '৮', '৯'])\nprint(\"vocab(unicodes) with numbers <fixed missing>:\")\nfor idx,c in enumerate(vocab):\n    if idx%20==0:print()\n    print(c,end=\",\")\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:10:42.188033Z","iopub.execute_input":"2022-07-18T10:10:42.188622Z","iopub.status.idle":"2022-07-18T10:10:42.196354Z","shell.execute_reply.started":"2022-07-18T10:10:42.188585Z","shell.execute_reply":"2022-07-18T10:10:42.195422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_df[\"label_len\"]=val_df[\"sentence\"].progress_apply(lambda x:len(x))\ntrain_df[\"label_len\"]=train_df[\"sentence\"].progress_apply(lambda x:len(x))\nprint(\"max train label len:\",max(train_df[\"label_len\"].tolist()))\nprint(\"max val label len:\",max(val_df[\"label_len\"].tolist()))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:10:42.198494Z","iopub.execute_input":"2022-07-18T10:10:42.199076Z","iopub.status.idle":"2022-07-18T10:10:42.393107Z","shell.execute_reply.started":"2022-07-18T10:10:42.199039Z","shell.execute_reply":"2022-07-18T10:10:42.392019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Label Encoding\n* This is fairly straight forward. We take ```POS_MAX=200``` to cover sentence lengths upto 200 charecters","metadata":{}},{"cell_type":"code","source":"POS_MAX=200\n\ndef encode_label(sen):\n    sen=[c for c in sen if c in vocab]\n    sen=[\"sep\"]+sen+[\"sep\"]\n    sen=sen+[\"pad\" for _ in range(POS_MAX-len(sen))]\n    label=[vocab.index(c) for c in sen]\n    return label\nsen=val_df.iloc[0,1]\nprint(\"sentence:\",sen)\nprint(\"encoded label:\",encode_label(sen))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:28:56.657664Z","iopub.execute_input":"2022-07-18T10:28:56.658043Z","iopub.status.idle":"2022-07-18T10:28:56.667123Z","shell.execute_reply.started":"2022-07-18T10:28:56.658011Z","shell.execute_reply":"2022-07-18T10:28:56.665887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TF Data Api : Dataloader","metadata":{}},{"cell_type":"code","source":"#--------------------------------------------------------\n# gen vars\n#--------------------------------------------------------\nimg_height,img_width,nb_channels=img_dim=features.shape\nlabel_dim=np.array(encode_label(sen)).shape\nBATCH_SIZE=8\n#--------------------------------------------------------\n# base generators\n#--------------------------------------------------------\ndef train_gen():\n    for idx in range(len(train_df)):\n        _path=train_df.iloc[idx,0]\n        sen=train_df.iloc[idx,1]\n        yield lms(_path),np.array(encode_label(sen),dtype=np.float32)\n\ndef eval_gen():\n    for idx in range(len(val_df)):\n        _path=train_df.iloc[idx,0]\n        sen=train_df.iloc[idx,1]\n        yield lms(_path),np.array(encode_label(sen),dtype=np.float32)\n\n        \n#--------------------------------------------------------\n# tf data api dataset\n#--------------------------------------------------------\ndef get_dataset(gen):\n    dataset= tf.data.Dataset.from_generator(gen,\n                                            output_signature=(tf.TensorSpec(shape=img_dim, dtype=tf.float32),\n                                                              tf.TensorSpec(shape=label_dim, dtype=tf.float32)))\n    dataset = dataset.shuffle(32,reshuffle_each_iteration=True)\n    dataset = dataset.repeat()\n    dataset = dataset.batch(BATCH_SIZE,drop_remainder=True)\n    dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n    return dataset\n\ntrain_ds=get_dataset(train_gen)\neval_ds =get_dataset(eval_gen)\n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:10.817051Z","iopub.execute_input":"2022-07-18T10:29:10.817543Z","iopub.status.idle":"2022-07-18T10:29:13.323078Z","shell.execute_reply.started":"2022-07-18T10:29:10.817511Z","shell.execute_reply":"2022-07-18T10:29:13.322128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#--------------------------------------------------------\n# visualize\n#--------------------------------------------------------\nfor x,y in eval_ds.take(1):\n    print(\"feature shape:\",x.shape)\n    print(\"label shape:\",y.shape)\n    \n    plt.figure(figsize=(20,20))\n    plt.imshow(x[0])\n    plt.show()\n    \n    print(\"label:\",y[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:13.325176Z","iopub.execute_input":"2022-07-18T10:29:13.325534Z","iopub.status.idle":"2022-07-18T10:29:16.184484Z","shell.execute_reply.started":"2022-07-18T10:29:13.325499Z","shell.execute_reply":"2022-07-18T10:29:16.183379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"#-----------------------------------\n#for creating Embedding Weights\n#-----------------------------------\nimport torch\nimport torch.nn as nn\n#--------------------------------------------------------------------\n# attention modules\n#--------------------------------------------------------------------\n\nclass DotAttention(tf.keras.layers.Layer):\n    '''\n            Calculate the attention weights.\n\n            args:\n                q   : query shape == (..., seq_len_q, depth)\n                k   : key shape == (..., seq_len_k, depth)\n                v   : value shape == (..., seq_len_v, depth_v)\n                mask: Float tensor with shape broadcastable to (..., seq_len_q, seq_len_k). Defaults to None.\n            returns:\n                output, attention_weights\n            NOTES:\n            * q, k, v must have matching leading dimensions.\n            * k, v must have matching penultimate dimension, i.e.: seq_len_k = seq_len_v.\n            * The mask has different shapes depending on its type(padding or look ahead) but it must be broadcastable for addition.\n\n    '''\n    def __init__(self):\n        super().__init__()\n        self.inf_val=-1e9\n        \n    def call(self,q, k, v, mask):\n        \n        matmul_qk = tf.matmul(q, k, transpose_b=True)  # (..., seq_len_q, seq_len_k)\n       \n        # scale matmul_qk\n        dk = tf.cast(tf.shape(k)[-1], tf.float32)\n        scaled_attention_logits = matmul_qk / tf.math.sqrt(dk)\n\n        # add the mask to the scaled tensor.\n        if mask is not None:\n            scaled_attention_logits += (mask * self.inf_val)\n\n        # softmax is normalized on the last axis (seq_len_k) so that the scores\n        # add up to 1.\n        attention_weights = tf.nn.softmax(scaled_attention_logits, axis=-1)  # (..., seq_len_q, seq_len_k)\n\n        output = tf.matmul(attention_weights, v)  # (..., seq_len_q, depth_v)\n\n        return output\n    \n#--------------------------------------------------------------------\n\nclass PositionalEncoding(tf.keras.layers.Layer):\n    '''\n    tensorflow wrapper for positional encoding layer\n    args:\n      num_seq  :   incoming sequence length\n      projection_dim  :   projection_dim\n      use_torch_weights : torch weights help converge faster for basic dot attention\n    '''\n    def __init__(self,num_seq,projection_dim,use_torch_weights=False):\n        super(PositionalEncoding, self).__init__()\n        self.use_torch_weights=use_torch_weights\n        self.projection_dim=projection_dim\n        self.num_seq = num_seq\n        self.projection = tf.keras.layers.Dense(units=projection_dim)\n        if use_torch_weights:\n            pos_emb              = nn.Embedding(num_seq+1,projection_dim)\n            pos_emb_weight       = pos_emb.weight.data.numpy()\n            self.position_embedding = tf.keras.layers.Embedding(input_dim=num_seq+1, output_dim=projection_dim,weights=[pos_emb_weight])\n        else:\n            \n            self.position_embedding = tf.keras.layers.Embedding(input_dim=num_seq, output_dim=projection_dim)\n\n    def call(self, x):\n        positions = tf.range(start=0, limit=self.num_seq, delta=1)\n        if x is None:\n            return self.position_embedding(positions)\n        encoded = self.projection(x) + self.position_embedding(positions)\n        return encoded\n    \n    def get_config(self):\n        config = super().get_config().copy()\n        config.update({'num_seq': self.num_seq,'projection_dim':self.projection_dim,\"use_torch_weights\":use_torch_weights})\n        return config\n\n#--------------------------------------------------------------------\n \nclass TransformerBlock(tf.keras.layers.Layer):\n    '''\n        transformer encoder block based on multihead self-attention\n    '''\n    def __init__(self, embed_dim, num_heads, ff_dim, rate=0.1):\n        super(TransformerBlock, self).__init__()\n        self.embed_dim=embed_dim\n        self.num_heads=num_heads\n        self.ff_dim   =ff_dim\n\n        self.att = tf.keras.layers.MultiHeadAttention(num_heads=num_heads, key_dim=embed_dim)\n        self.ffn = tf.keras.Sequential(\n            [tf.keras.layers.Dense(ff_dim, activation=\"relu\"), tf.keras.layers.Dense(embed_dim),]\n        )\n        self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)\n        self.dropout1 = tf.keras.layers.Dropout(rate)\n        self.dropout2 = tf.keras.layers.Dropout(rate)\n\n    def call(self, inputs, training):\n        attn_output = self.att(inputs, inputs)\n        attn_output = self.dropout1(attn_output, training=training)\n        out1 = self.layernorm1(inputs + attn_output)\n        ffn_output = self.ffn(out1)\n        ffn_output = self.dropout2(ffn_output, training=training)\n        return self.layernorm2(out1 + ffn_output)\n    def get_config(self):\n        config = super().get_config().copy()\n        config.update({'embed_dim': self.embed_dim,\n                       'num_heads': self.num_heads,\n                       'ff_dim':self.ff_dim})\n        return config\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:20.290741Z","iopub.execute_input":"2022-07-18T10:29:20.291279Z","iopub.status.idle":"2022-07-18T10:29:22.187205Z","shell.execute_reply.started":"2022-07-18T10:29:20.291243Z","shell.execute_reply":"2022-07-18T10:29:22.186155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def attend(x,num_heads,num_blocks,reshape=True):\n    '''\n        basic self-attention wrapper\n        args:\n            x : input tensor\n            num_heads : heads to use in multihead attention\n            num_blocks: how many attention blocks to use \n    '''\n    bs,h,w,nc=x.shape\n    x = tf.keras.layers.Reshape((h*w,nc))(x)\n    x = PositionalEncoding(h*w,nc)(x)\n    for _ in range(num_blocks):\n        x=TransformerBlock(embed_dim=nc, num_heads=num_heads,ff_dim=4*nc)(x)\n    if reshape:\n        x = tf.keras.layers.Reshape((h,w,nc))(x)\n    return x\n\n\ndef create_basic_model(cfg):\n    '''\n        creates a basic cnn-seftattention-positional-attention based model\n        **flow**\n        cnn_feat(input_image)---> h//f,w//f,c shaped tensor = feat\n        attend(feat)---> self attention applied on the features= enc\n        pos_attention(enc)--> align encoded features with positional data=logits\n    '''\n    #-----------cnn feature extractor------------------\n    cnn,_ = Classifiers.get(cfg.backbone)\n    cnn = cnn(cfg.img_dim,weights=None,include_top=False)\n    inp = cnn.input\n    x   = cnn.output\n    bs,h,w,fc=x.shape\n    if fc!=cfg.embed_dim:\n        x=tf.keras.layers.Conv2D(cfg.embed_dim,3,padding='same')(x)\n    print(\"model input:\",inp.shape)\n    print(\"cnn feat:\",x.shape)\n    #-----------self attention------------------\n    x=attend(x,cfg.num_blocks,cfg.num_heads,reshape=False)\n    print(\"feat attention (seq,embed_dim):\",x.shape)\n    #-----------positional attention------------------\n    pos=PositionalEncoding(cfg.pos_max,cfg.embed_dim,use_torch_weights=True)(None)\n    attn=DotAttention()(pos,x,x,None)\n    print(\"positional attention (pos_max,embed_dim):\",attn.shape)\n    x=tf.keras.layers.Dense(cfg.logits_len)(attn)\n    print(\"logits(pos_max,logits_len):\",x.shape)\n    model = tf.keras.Model(inputs=inp,outputs=x)\n    return model\n    \n\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:22.190925Z","iopub.execute_input":"2022-07-18T10:29:22.191651Z","iopub.status.idle":"2022-07-18T10:29:22.204104Z","shell.execute_reply.started":"2022-07-18T10:29:22.191621Z","shell.execute_reply":"2022-07-18T10:29:22.203067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN backbone model selection\nThe following models are available in this script **but any custom cnn backbone can be used according to input/output downsample ratio**\n* [source](https://github.com/qubvel/classification_models/blob/master/README.md)\n* The scores are for ```imagenet``` image classification challange\n\n| Model           |Acc@1|Acc@5|Time*|Source|\n|-----------------|:---:|:---:|:---:|------|\n|vgg16            |70.79|89.74|24.95|[keras](https://github.com/keras-team/keras-applications)|\n|vgg19            |70.89|89.69|24.95|[keras](https://github.com/keras-team/keras-applications)|\n|resnet18         |68.24|88.49|16.07|[mxnet](https://github.com/Microsoft/MMdnn)|\n|resnet34         |72.17|90.74|17.37|[mxnet](https://github.com/Microsoft/MMdnn)|\n|resnet50         |74.81|92.38|22.62|[mxnet](https://github.com/Microsoft/MMdnn)|\n|resnet101        |76.58|93.10|33.03|[mxnet](https://github.com/Microsoft/MMdnn)|\n|resnet152        |76.66|93.08|42.37|[mxnet](https://github.com/Microsoft/MMdnn)|\n|resnet50v2       |69.73|89.31|19.56|[keras](https://github.com/keras-team/keras-applications)|\n|resnet101v2      |71.93|90.41|28.80|[keras](https://github.com/keras-team/keras-applications)|\n|resnet152v2      |72.29|90.61|41.09|[keras](https://github.com/keras-team/keras-applications)|\n|resnext50        |77.36|93.48|37.57|[keras](https://github.com/keras-team/keras-applications)|\n|resnext101       |78.48|94.00|60.07|[keras](https://github.com/keras-team/keras-applications)|\n|densenet121      |74.67|92.04|27.66|[keras](https://github.com/keras-team/keras-applications)|\n|densenet169      |75.85|92.93|33.71|[keras](https://github.com/keras-team/keras-applications)|\n|densenet201      |77.13|93.43|42.40|[keras](https://github.com/keras-team/keras-applications)|\n|inceptionv3      |77.55|93.48|38.94|[keras](https://github.com/keras-team/keras-applications)|\n|xception         |78.87|94.20|42.18|[keras](https://github.com/keras-team/keras-applications)|\n|inceptionresnetv2|80.03|94.89|54.77|[keras](https://github.com/keras-team/keras-applications)|\n|seresnet18       |69.41|88.84|20.19|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnet34       |72.60|90.91|22.20|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnet50       |76.44|93.02|23.64|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnet101      |77.92|94.00|32.55|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnet152      |78.34|94.08|47.88|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnext50      |78.74|94.30|38.29|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|seresnext101     |79.88|94.87|62.80|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|senet154         |81.06|95.24|137.36|[pytorch](https://github.com/Cadene/pretrained-models.pytorch)|\n|nasnetlarge      |**82.12**|**95.72**|116.53|[keras](https://github.com/keras-team/keras-applications)|\n|nasnetmobile     |74.04|91.54|27.73|[keras](https://github.com/keras-team/keras-applications)|\n|mobilenet        |70.36|89.39|15.50|[keras](https://github.com/keras-team/keras-applications)|\n|mobilenetv2      |71.63|90.35|18.31|[keras](https://github.com/keras-team/keras-applications)|","metadata":{}},{"cell_type":"code","source":"class cfg:\n    backbone='resnet18'\n    img_dim =(64,1088,1)     #features.shape\n    pos_max =200             #np.array(encode_label(sen)).shape\n    num_heads=8              # number of self-attention heads\n    num_blocks=4             # number of transformer blocks\n    embed_dim =256           # reduced channel for sequencing\n    logits_len =85           # len(vocab)+1\n \n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:22.934537Z","iopub.execute_input":"2022-07-18T10:29:22.935211Z","iopub.status.idle":"2022-07-18T10:29:22.940822Z","shell.execute_reply.started":"2022-07-18T10:29:22.935173Z","shell.execute_reply":"2022-07-18T10:29:22.939681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model=create_basic_model(cfg)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:23.579306Z","iopub.execute_input":"2022-07-18T10:29:23.579999Z","iopub.status.idle":"2022-07-18T10:29:27.783624Z","shell.execute_reply.started":"2022-07-18T10:29:23.579951Z","shell.execute_reply":"2022-07-18T10:29:27.782632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# losses and metrics","metadata":{}},{"cell_type":"code","source":"#------------------metrics---------------------------\ndef C_acc(y_true, y_pred):\n    '''\n        calculates how many charecters are predicted correctly\n    '''\n    pad_value=0 # label pad index\n    accuracies = tf.equal(tf.cast(y_true,tf.int64), tf.argmax(y_pred, axis=2))\n    mask = tf.math.logical_not(tf.math.equal(y_true,pad_value))\n    accuracies = tf.math.logical_and(mask, accuracies)\n    accuracies = tf.cast(accuracies, dtype=tf.float32)\n    mask = tf.cast(mask, dtype=tf.float32)\n    return tf.reduce_sum(accuracies)/tf.reduce_sum(mask)\n#------------------loss--------------------------\nclass CharLoss(tf.keras.losses.Loss):\n    \"\"\"\n        a loss function to estimate charecter accuracy loss ignoring pad value\n    \"\"\"\n    def __init__(self,pad_value):\n        super(CharLoss, self).__init__(name=\"char_loss\")\n        self.pad_value=pad_value\n        self.loss_object = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True, reduction='none')\n    def call(self, y_true, y_pred):\n        mask = tf.math.logical_not(tf.math.equal(y_true, self.pad_value))\n        loss_ = self.loss_object(y_true, y_pred)\n        mask = tf.cast(mask, dtype=loss_.dtype)\n        loss_ *= mask\n        return tf.reduce_sum(loss_)/tf.reduce_sum(mask)\n\nclass CTCLoss(tf.keras.losses.Loss):\n    \"\"\" A class that wraps the function of tf.nn.ctc_loss. \n    \n    Attributes:\n        logits_time_major: If False (default) , shape is [batch, time, logits], \n            If True, logits is shaped [time, batch, logits]. \n        blank_index: Set the class index to use for the blank label. default is\n            -1 (num_classes - 1). \n    \"\"\"\n\n    def __init__(self, logits_time_major=False, name='ctc_loss'):\n        super().__init__(name=name)\n        self.logits_time_major = logits_time_major\n\n    def call(self, y_true, y_pred):\n        \"\"\" \n            Computes CTC (Connectionist Temporal Classification) loss. \n        \"\"\"\n        y_true = tf.cast(y_true, tf.int32)\n        logit_length = tf.fill([tf.shape(y_pred)[0]], tf.shape(y_pred)[1])\n        label_length = tf.fill([tf.shape(y_true)[0]], tf.shape(y_true)[1])\n        loss = tf.nn.ctc_loss(\n            labels=y_true,\n            logits=y_pred,\n            label_length=label_length,\n            logit_length=logit_length,\n            logits_time_major=self.logits_time_major,\n            blank_index=0)\n        return tf.math.reduce_mean(loss)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:27.786451Z","iopub.execute_input":"2022-07-18T10:29:27.786814Z","iopub.status.idle":"2022-07-18T10:29:27.801292Z","shell.execute_reply.started":"2022-07-18T10:29:27.786776Z","shell.execute_reply":"2022-07-18T10:29:27.800164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Callbacks","metadata":{}},{"cell_type":"code","source":"# early stopping\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=5, \n                                                  verbose=1, \n                                                  mode = 'auto') \ncallbacks = [tf.keras.callbacks.ModelCheckpoint(\"model.h5\",\n                                                save_best_only=True,\n                                                save_weights_only=True,\n                                                verbose=1),\n             early_stopping]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:27.803069Z","iopub.execute_input":"2022-07-18T10:29:27.803690Z","iopub.status.idle":"2022-07-18T10:29:27.817736Z","shell.execute_reply.started":"2022-07-18T10:29:27.803652Z","shell.execute_reply":"2022-07-18T10:29:27.816750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# compile","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n\n    lr_schedule = tf.keras.experimental.CosineDecay(initial_learning_rate=0.0001,\n                                                             decay_steps=600000,\n                                                             alpha= 0.01)\n\n    model.compile(optimizer=tf.keras.optimizers.Adam(lr_schedule),\n                  loss=CharLoss(vocab.index(\"pad\")),\n                  metrics=[C_acc])","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:28.840331Z","iopub.execute_input":"2022-07-18T10:29:28.840678Z","iopub.status.idle":"2022-07-18T10:29:28.861612Z","shell.execute_reply.started":"2022-07-18T10:29:28.840649Z","shell.execute_reply":"2022-07-18T10:29:28.860682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"EPOCHS=100\nSTEPS_PER_EPOCH=len(train_df)//BATCH_SIZE\nEVAL_STEPS=len(val_df)//BATCH_SIZE\n\n\nhistory=model.fit(train_ds,\n                  epochs=EPOCHS,\n                  steps_per_epoch=STEPS_PER_EPOCH,\n                  verbose=1,\n                  validation_data=eval_ds,\n                  validation_steps=EVAL_STEPS, \n                  callbacks=callbacks)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:29:39.658936Z","iopub.execute_input":"2022-07-18T10:29:39.659578Z","iopub.status.idle":"2022-07-18T10:30:38.636149Z","shell.execute_reply.started":"2022-07-18T10:29:39.659542Z","shell.execute_reply":"2022-07-18T10:30:38.632756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"curves={}\nfor key in history.history.keys():\n    curves[key]=history.history[key]\ncurves=pd.DataFrame(curves)\ncurves.to_csv(f\"history.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"markdown","source":"* **train as needed**\n* you can only access curves after training is done","metadata":{}},{"cell_type":"code","source":"sub=pd.read_csv(\"../input/dlsprint/sample_submission.csv\")\nsub","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:31:50.202433Z","iopub.execute_input":"2022-07-18T10:31:50.203038Z","iopub.status.idle":"2022-07-18T10:31:50.262051Z","shell.execute_reply.started":"2022-07-18T10:31:50.202996Z","shell.execute_reply":"2022-07-18T10:31:50.260968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Batch Inference**","metadata":{}},{"cell_type":"code","source":"TEST_WAVS=\"../input/test-wav-files-dl-sprint/test_files_wav\"\nPREDS=[]\nfor idx in tqdm(range(0,len(sub),BATCH_SIZE)):\n    batch=[]\n    for bi in range(idx,idx+BATCH_SIZE):\n        _path=sub.iloc[bi,0]\n        _path=os.path.join(TEST_WAVS,_path).replace(\".mp3\",\".wav\")\n        signal=lms(_path)\n        batch.append(np.expand_dims(signal,axis=0))\n    batch=np.vstack(batch)\n    preds=model(batch,training=False)\n    for pred in preds:\n        out=np.argmax(pred,axis=-1)\n        text=[vocab[i] for i in out]\n        text=\"\".join(text)\n        PREDS.append(text)\n    \n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:41:07.391331Z","iopub.execute_input":"2022-07-18T10:41:07.391938Z","iopub.status.idle":"2022-07-18T10:42:59.046561Z","shell.execute_reply.started":"2022-07-18T10:41:07.391901Z","shell.execute_reply":"2022-07-18T10:42:59.045042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"sentence\"]=PREDS\nsub","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TPU version coming soon .  \n** issues: zero valued arrays occur sometimes \n** create tfrecords to solve this issue permanantly","metadata":{}}]}