{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Contents\n\nIn this Notebook I will start with the very Basics of RNN's and Build all the way to latest deep learning architectures to solve NLP problems. It will cover the Following:\n* Simple RNN's\n* Word Embeddings : Definition and How to get them\n* LSTM's\n* GRU's\n\nI will divide every Topic into four subsections:\n* Basic Overview\n* In-Depth Understanding : In this I will attach links of articles and videos to learn about the topic in depth\n* Code-Implementation\n* Code Explanation\n\nThis is a comprehensive kernel and if you follow along till the end , I promise you would learn all the techniques completely\n\nNote that the aim of this notebook is not to have a High LB score but to present a beginner guide to understand Deep Learning techniques used for NLP. Also after discussing all of these ideas , I will present a starter solution for this competiton","metadata":{}},{"cell_type":"code","source":"!pip install seaborn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install plotly","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:00.155178Z","iopub.execute_input":"2023-07-06T01:39:00.155614Z","iopub.status.idle":"2023-07-06T01:39:16.268583Z","shell.execute_reply.started":"2023-07-06T01:39:00.155581Z","shell.execute_reply":"2023-07-06T01:39:16.267366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers import Embedding,BatchNormalization\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-07-06T01:39:20.803603Z","iopub.execute_input":"2023-07-06T01:39:20.804123Z","iopub.status.idle":"2023-07-06T01:39:21.349503Z","shell.execute_reply.started":"2023-07-06T01:39:20.804079Z","shell.execute_reply":"2023-07-06T01:39:21.348301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuring TPU's\n\nFor this version of Notebook we will be using TPU's as we have to built a BERT Model","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:25.627698Z","iopub.execute_input":"2023-07-06T01:39:25.628136Z","iopub.status.idle":"2023-07-06T01:39:34.391262Z","shell.execute_reply.started":"2023-07-06T01:39:25.628104Z","shell.execute_reply":"2023-07-06T01:39:34.390307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-07-06T01:39:42.811781Z","iopub.execute_input":"2023-07-06T01:39:42.812757Z","iopub.status.idle":"2023-07-06T01:39:46.174857Z","shell.execute_reply.started":"2023-07-06T01:39:42.81271Z","shell.execute_reply":"2023-07-06T01:39:46.173644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will drop the other columns and approach this problem as a Binary Classification Problem and also we will have our exercise done on a smaller subsection of the dataset(only 12000 data points) to make it easier to train the models","metadata":{}},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:48.592466Z","iopub.execute_input":"2023-07-06T01:39:48.593526Z","iopub.status.idle":"2023-07-06T01:39:48.612706Z","shell.execute_reply.started":"2023-07-06T01:39:48.593476Z","shell.execute_reply":"2023-07-06T01:39:48.611687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000,:]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:52.295361Z","iopub.execute_input":"2023-07-06T01:39:52.296309Z","iopub.status.idle":"2023-07-06T01:39:52.304204Z","shell.execute_reply.started":"2023-07-06T01:39:52.296263Z","shell.execute_reply":"2023-07-06T01:39:52.303109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will check the maximum number of words that can be present in a comment , this will help us in padding later","metadata":{}},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:55.827932Z","iopub.execute_input":"2023-07-06T01:39:55.8289Z","iopub.status.idle":"2023-07-06T01:39:55.910697Z","shell.execute_reply.started":"2023-07-06T01:39:55.828856Z","shell.execute_reply":"2023-07-06T01:39:55.909765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Writing a function for getting auc score for validation","metadata":{}},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:59.865638Z","iopub.execute_input":"2023-07-06T01:39:59.866293Z","iopub.status.idle":"2023-07-06T01:39:59.873173Z","shell.execute_reply.started":"2023-07-06T01:39:59.866254Z","shell.execute_reply":"2023-07-06T01:39:59.871964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preparation","metadata":{}},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:40:03.989784Z","iopub.execute_input":"2023-07-06T01:40:03.990225Z","iopub.status.idle":"2023-07-06T01:40:04.002334Z","shell.execute_reply.started":"2023-07-06T01:40:03.990192Z","shell.execute_reply":"2023-07-06T01:40:04.00122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple RNN\n\n## Basic Overview\n\nWhat is a RNN?\n\nRecurrent Neural Network(RNN) are a type of Neural Network where the output from previous step are fed as input to the current step. In traditional neural networks, all the inputs and outputs are independent of each other, but in cases like when it is required to predict the next word of a sentence, the previous words are required and hence there is a need to remember the previous words. Thus RNN came into existence, which solved this issue with the help of a Hidden Layer.\n\nWhy RNN's?\n\nhttps://www.quora.com/Why-do-we-use-an-RNN-instead-of-a-simple-neural-network\n\n## In-Depth Understanding\n\n* https://medium.com/mindorks/understanding-the-recurrent-neural-network-44d593f112a2\n* https://www.youtube.com/watch?v=2E65LDnM2cA&list=PL1F3ABbhcqa3BBWo170U4Ev2wfsF7FN8l\n* https://www.d2l.ai/chapter_recurrent-neural-networks/rnn.html\n\n## Code Implementation\n\nSo first I will implement the and then I will explain the code step by step","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n# using keras tokenizer here\ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n#zero pad the sequences\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:41:10.211644Z","iopub.execute_input":"2023-07-06T01:41:10.212679Z","iopub.status.idle":"2023-07-06T01:41:12.017262Z","shell.execute_reply.started":"2023-07-06T01:41:10.212632Z","shell.execute_reply":"2023-07-06T01:41:12.016164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # A simpleRNN without any pretrained embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     input_length=max_len))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:41:15.826916Z","iopub.execute_input":"2023-07-06T01:41:15.827375Z","iopub.status.idle":"2023-07-06T01:41:18.427694Z","shell.execute_reply.started":"2023-07-06T01:41:15.82734Z","shell.execute_reply":"2023-07-06T01:41:18.426814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:03.33997Z","iopub.execute_input":"2023-07-06T01:42:03.340416Z","iopub.status.idle":"2023-07-06T01:42:24.179874Z","shell.execute_reply.started":"2023-07-06T01:42:03.340385Z","shell.execute_reply":"2023-07-06T01:42:24.17874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:30.121382Z","iopub.execute_input":"2023-07-06T01:42:30.121844Z","iopub.status.idle":"2023-07-06T01:42:33.385173Z","shell.execute_reply.started":"2023-07-06T01:42:30.12181Z","shell.execute_reply":"2023-07-06T01:42:33.383942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:37.695008Z","iopub.execute_input":"2023-07-06T01:42:37.695426Z","iopub.status.idle":"2023-07-06T01:42:37.703304Z","shell.execute_reply.started":"2023-07-06T01:42:37.695396Z","shell.execute_reply":"2023-07-06T01:42:37.702125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Code Explanantion\n* Tokenization<br><br>\n So if you have watched the videos and referred to the links, you would know that in an RNN we input a sentence word by word. We represent every word as one hot vectors of dimensions : Numbers of words in Vocab +1. <br>\n  What keras Tokenizer does is , it takes all the unique words in the corpus,forms a dictionary with words as keys and their number of occurences as values,it then sorts the dictionary in descending order of counts. It then assigns the first value 1 , second value 2 and so on. So let's suppose word 'the' occured the most in the corpus then it will assigned index 1 and vector representing 'the' would be a one-hot vector with value 1 at position 1 and rest zereos.<br>\n  Try printing first 2 elements of xtrain_seq you will see every word is represented as a digit now","metadata":{}},{"cell_type":"code","source":"xtrain_seq[:1]","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:44.6728Z","iopub.execute_input":"2023-07-06T01:42:44.673755Z","iopub.status.idle":"2023-07-06T01:42:44.680455Z","shell.execute_reply.started":"2023-07-06T01:42:44.673705Z","shell.execute_reply":"2023-07-06T01:42:44.679361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<b>Now you might be wondering What is padding? Why its done</b><br><br>\n\nHere is the answer :\n* https://www.quora.com/Which-effect-does-sequence-padding-have-on-the-training-of-a-neural-network\n* https://machinelearningmastery.com/data-preparation-variable-length-input-sequences-sequence-prediction/\n* https://www.coursera.org/lecture/natural-language-processing-tensorflow/padding-2Cyzs\n\nAlso sometimes people might use special tokens while tokenizing like EOS(end of string) and BOS(Begining of string). Here is the reason why it's done\n* https://stackoverflow.com/questions/44579161/why-do-we-do-padding-in-nlp-tasks\n\n\nThe code token.word_index simply gives the dictionary of vocab that keras created for us","metadata":{}},{"cell_type":"markdown","source":"* Building the Neural Network\n\nTo understand the Dimensions of input and output given to RNN in keras her is a beautiful article : https://medium.com/@shivajbd/understanding-input-and-output-shape-in-lstm-keras-c501ee95c65e\n\nThe first line model.Sequential() tells keras that we will be building our network sequentially . Then we first add the Embedding layer.\nEmbedding layer is also a layer of neurons which takes in as input the nth dimensional one hot vector of every word and converts it into 300 dimensional vector , it gives us word embeddings similar to word2vec. We could have used word2vec but the embeddings layer learns during training to enhance the embeddings.\nNext we add an 100 LSTM units without any dropout or regularization\nAt last we add a single neuron with sigmoid function which takes output from 100 LSTM cells (Please note we have 100 LSTM cells not layers) to predict the results and then we compile the model using adam optimizer \n\n* Comments on the model<br><br>\nWe can see our model achieves an accuracy of 1 which is just insane , we are clearly overfitting I know , but this was the simplest model of all ,we can tune a lot of hyperparameters like RNN units, we can do batch normalization , dropouts etc to get better result. The point is we got an AUC score of 0.82 without much efforts and we know have learnt about RNN's .Deep learning is really revolutionary","metadata":{}},{"cell_type":"markdown","source":"# Word Embeddings\n\nWhile building our simple RNN models we talked about using word-embeddings , So what is word-embeddings and how do we get word-embeddings?\nHere is the answer :\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/6Oq70/word-representation\n* https://machinelearningmastery.com/what-are-word-embeddings/\n<br> <br>\nThe latest approach to getting word Embeddings is using pretained GLoVe or using Fasttext. Without going into too much details, I would explain how to create sentence vectors and how can we use them to create a machine learning model on top of it and since I am a fan of GloVe vectors, word2vec and fasttext. In this Notebook, I'll be using the GloVe vectors. You can download the GloVe vectors from here http://www-nlp.stanford.edu/data/glove.840B.300d.zip or you can search for GloVe in datasets on Kaggle and add the file","metadata":{}},{"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:51.587318Z","iopub.execute_input":"2023-07-06T01:42:51.58774Z","iopub.status.idle":"2023-07-06T01:47:14.033898Z","shell.execute_reply.started":"2023-07-06T01:42:51.58771Z","shell.execute_reply":"2023-07-06T01:47:14.032452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM's\n\n## Basic Overview\n\nSimple RNN's were certainly better than classical ML algorithms and gave state of the art results, but it failed to capture long term dependencies that is present in sentences . So in 1998-99 LSTM's were introduced to counter to these drawbacks.\n\n## In Depth Understanding\n\nWhy LSTM's?\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/PKMRR/vanishing-gradients-with-rnns\n* https://www.analyticsvidhya.com/blog/2017/12/fundamentals-of-deep-learning-introduction-to-lstm/\n\nWhat are LSTM's?\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/KXoay/long-short-term-memory-lstm\n* https://distill.pub/2019/memorization-in-rnns/\n* https://towardsdatascience.com/illustrated-guide-to-lstms-and-gru-s-a-step-by-step-explanation-44e9eb85bf21\n\n# Code Implementation\n\nWe have already tokenized and paded our text for input to LSTM's","metadata":{}},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:47:19.357747Z","iopub.execute_input":"2023-07-06T01:47:19.358977Z","iopub.status.idle":"2023-07-06T01:47:19.566112Z","shell.execute_reply.started":"2023-07-06T01:47:19.358942Z","shell.execute_reply":"2023-07-06T01:47:19.564766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    \n    # A simple LSTM with glove embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n\n    model.add(LSTM(100, dropout=0.3, recurrent_dropout=0.3))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:47:23.019466Z","iopub.execute_input":"2023-07-06T01:47:23.020434Z","iopub.status.idle":"2023-07-06T01:47:26.4265Z","shell.execute_reply.started":"2023-07-06T01:47:23.020401Z","shell.execute_reply":"2023-07-06T01:47:26.425339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:47:53.04121Z","iopub.execute_input":"2023-07-06T01:47:53.042018Z","iopub.status.idle":"2023-07-06T01:48:22.186707Z","shell.execute_reply.started":"2023-07-06T01:47:53.041961Z","shell.execute_reply":"2023-07-06T01:48:22.185415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:48:24.335813Z","iopub.execute_input":"2023-07-06T01:48:24.336424Z","iopub.status.idle":"2023-07-06T01:48:30.235751Z","shell.execute_reply.started":"2023-07-06T01:48:24.336389Z","shell.execute_reply":"2023-07-06T01:48:30.234396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:48:34.808764Z","iopub.execute_input":"2023-07-06T01:48:34.810088Z","iopub.status.idle":"2023-07-06T01:48:34.816704Z","shell.execute_reply.started":"2023-07-06T01:48:34.81005Z","shell.execute_reply":"2023-07-06T01:48:34.815577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Code Explanation\n\nAs a first step we calculate embedding matrix for our vocabulary from the pretrained GLoVe vectors . Then while building the embedding layer we pass Embedding Matrix as weights to the layer instead of training it over Vocabulary and thus we pass trainable = False.\nRest of the model is same as before except we have replaced the SimpleRNN By LSTM Units\n\n* Comments on the Model\n\nWe now see that the model is not overfitting and achieves an auc score of 0.96 which is quite commendable , also we close in on the gap between accuracy and auc .\nWe see that in this case we used dropout and prevented overfitting the data","metadata":{}},{"cell_type":"markdown","source":"# GRU's\n\n## Basic  Overview\n\nIntroduced by Cho, et al. in 2014, GRU (Gated Recurrent Unit) aims to solve the vanishing gradient problem which comes with a standard recurrent neural network. GRU's are a variation on the LSTM because both are designed similarly and, in some cases, produce equally excellent results . GRU's were designed to be simpler and faster than LSTM's and in most cases produce equally good results and thus there is no clear winner.\n\n## In Depth Explanation\n\n* https://towardsdatascience.com/understanding-gru-networks-2ef37df6c9be\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/agZiL/gated-recurrent-unit-gru\n* https://www.geeksforgeeks.org/gated-recurrent-unit-networks/\n\n## Code Implementation","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # GRU with glove embeddings and two dense layers\n     model = Sequential()\n     model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n     model.add(SpatialDropout1D(0.3))\n     model.add(GRU(300))\n     model.add(Dense(1, activation='sigmoid'))\n\n     model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:48:39.629674Z","iopub.execute_input":"2023-07-06T01:48:39.630801Z","iopub.status.idle":"2023-07-06T01:48:43.823029Z","shell.execute_reply.started":"2023-07-06T01:48:39.630761Z","shell.execute_reply":"2023-07-06T01:48:43.821863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:48:52.189954Z","iopub.execute_input":"2023-07-06T01:48:52.191155Z","iopub.status.idle":"2023-07-06T01:49:18.364628Z","shell.execute_reply.started":"2023-07-06T01:48:52.191104Z","shell.execute_reply":"2023-07-06T01:49:18.363296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:49:20.923577Z","iopub.execute_input":"2023-07-06T01:49:20.924075Z","iopub.status.idle":"2023-07-06T01:49:26.674258Z","shell.execute_reply.started":"2023-07-06T01:49:20.924036Z","shell.execute_reply":"2023-07-06T01:49:26.672959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'GRU','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:49:28.449535Z","iopub.execute_input":"2023-07-06T01:49:28.449951Z","iopub.status.idle":"2023-07-06T01:49:28.457203Z","shell.execute_reply.started":"2023-07-06T01:49:28.449921Z","shell.execute_reply":"2023-07-06T01:49:28.456146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:49:31.163932Z","iopub.execute_input":"2023-07-06T01:49:31.165112Z","iopub.status.idle":"2023-07-06T01:49:31.171594Z","shell.execute_reply.started":"2023-07-06T01:49:31.165072Z","shell.execute_reply":"2023-07-06T01:49:31.170517Z"},"trusted":true},"execution_count":null,"outputs":[]}]}