{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# GRU's\n\n## Basic  Overview\n\nIntroduced by Cho, et al. in 2014, GRU (Gated Recurrent Unit) aims to solve the vanishing gradient problem which comes with a standard recurrent neural network. GRU's are a variation on the LSTM because both are designed similarly and, in some cases, produce equally excellent results . GRU's were designed to be simpler and faster than LSTM's and in most cases produce equally good results and thus there is no clear winner.\n\n## In Depth Explanation\n\n* https://towardsdatascience.com/understanding-gru-networks-2ef37df6c9be\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/agZiL/gated-recurrent-unit-gru\n* https://www.geeksforgeeks.org/gated-recurrent-unit-networks/\n\n## Code Implementation","metadata":{}},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:51:38.597227Z","iopub.execute_input":"2023-07-06T19:51:38.597516Z","iopub.status.idle":"2023-07-06T19:51:44.586216Z","shell.execute_reply.started":"2023-07-06T19:51:38.597489Z","shell.execute_reply":"2023-07-06T19:51:44.585075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install plotly","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:51:44.588138Z","iopub.execute_input":"2023-07-06T19:51:44.588426Z","iopub.status.idle":"2023-07-06T19:52:00.724037Z","shell.execute_reply.started":"2023-07-06T19:51:44.588398Z","shell.execute_reply":"2023-07-06T19:52:00.722899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers import Embedding,BatchNormalization\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:00.725637Z","iopub.execute_input":"2023-07-06T19:52:00.725976Z","iopub.status.idle":"2023-07-06T19:52:42.481111Z","shell.execute_reply.started":"2023-07-06T19:52:00.725942Z","shell.execute_reply":"2023-07-06T19:52:42.479939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:42.483558Z","iopub.execute_input":"2023-07-06T19:52:42.484310Z","iopub.status.idle":"2023-07-06T19:52:51.322062Z","shell.execute_reply.started":"2023-07-06T19:52:42.484280Z","shell.execute_reply":"2023-07-06T19:52:51.321157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:51.323225Z","iopub.execute_input":"2023-07-06T19:52:51.323504Z","iopub.status.idle":"2023-07-06T19:52:54.541021Z","shell.execute_reply.started":"2023-07-06T19:52:51.323478Z","shell.execute_reply":"2023-07-06T19:52:54.539908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.542387Z","iopub.execute_input":"2023-07-06T19:52:54.542752Z","iopub.status.idle":"2023-07-06T19:52:54.561640Z","shell.execute_reply.started":"2023-07-06T19:52:54.542711Z","shell.execute_reply":"2023-07-06T19:52:54.560436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000,:]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.563003Z","iopub.execute_input":"2023-07-06T19:52:54.563330Z","iopub.status.idle":"2023-07-06T19:52:54.571937Z","shell.execute_reply.started":"2023-07-06T19:52:54.563300Z","shell.execute_reply":"2023-07-06T19:52:54.570876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.573217Z","iopub.execute_input":"2023-07-06T19:52:54.573522Z","iopub.status.idle":"2023-07-06T19:52:54.656447Z","shell.execute_reply.started":"2023-07-06T19:52:54.573494Z","shell.execute_reply":"2023-07-06T19:52:54.655478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.657696Z","iopub.execute_input":"2023-07-06T19:52:54.658056Z","iopub.status.idle":"2023-07-06T19:52:54.664219Z","shell.execute_reply.started":"2023-07-06T19:52:54.658023Z","shell.execute_reply":"2023-07-06T19:52:54.663218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.667512Z","iopub.execute_input":"2023-07-06T19:52:54.667846Z","iopub.status.idle":"2023-07-06T19:52:54.682158Z","shell.execute_reply.started":"2023-07-06T19:52:54.667818Z","shell.execute_reply":"2023-07-06T19:52:54.681205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n# using keras tokenizer here\ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n#zero pad the sequences\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:54.683156Z","iopub.execute_input":"2023-07-06T19:52:54.683410Z","iopub.status.idle":"2023-07-06T19:52:56.440437Z","shell.execute_reply.started":"2023-07-06T19:52:54.683386Z","shell.execute_reply":"2023-07-06T19:52:56.439340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:56.441667Z","iopub.execute_input":"2023-07-06T19:52:56.441980Z","iopub.status.idle":"2023-07-06T19:52:56.445885Z","shell.execute_reply.started":"2023-07-06T19:52:56.441952Z","shell.execute_reply":"2023-07-06T19:52:56.445056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:52:56.446879Z","iopub.execute_input":"2023-07-06T19:52:56.447138Z","iopub.status.idle":"2023-07-06T19:57:07.578140Z","shell.execute_reply.started":"2023-07-06T19:52:56.447115Z","shell.execute_reply":"2023-07-06T19:57:07.577028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:07.579291Z","iopub.execute_input":"2023-07-06T19:57:07.579599Z","iopub.status.idle":"2023-07-06T19:57:07.760725Z","shell.execute_reply.started":"2023-07-06T19:57:07.579556Z","shell.execute_reply":"2023-07-06T19:57:07.759788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # GRU with glove embeddings and two dense layers\n     model = Sequential()\n     model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n     model.add(SpatialDropout1D(0.3))\n     model.add(GRU(300))\n     model.add(Dense(1, activation='sigmoid'))\n\n     model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:07.761918Z","iopub.execute_input":"2023-07-06T19:57:07.762205Z","iopub.status.idle":"2023-07-06T19:57:13.552785Z","shell.execute_reply.started":"2023-07-06T19:57:07.762179Z","shell.execute_reply":"2023-07-06T19:57:13.551944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:13.553961Z","iopub.execute_input":"2023-07-06T19:57:13.554257Z","iopub.status.idle":"2023-07-06T19:57:39.864959Z","shell.execute_reply.started":"2023-07-06T19:57:13.554228Z","shell.execute_reply":"2023-07-06T19:57:39.863537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:39.866374Z","iopub.execute_input":"2023-07-06T19:57:39.866742Z","iopub.status.idle":"2023-07-06T19:57:45.401380Z","shell.execute_reply.started":"2023-07-06T19:57:39.866709Z","shell.execute_reply":"2023-07-06T19:57:45.400320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'GRU','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:45.402557Z","iopub.execute_input":"2023-07-06T19:57:45.402869Z","iopub.status.idle":"2023-07-06T19:57:45.409034Z","shell.execute_reply.started":"2023-07-06T19:57:45.402842Z","shell.execute_reply":"2023-07-06T19:57:45.408119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-07-06T19:57:45.410178Z","iopub.execute_input":"2023-07-06T19:57:45.410488Z","iopub.status.idle":"2023-07-06T19:57:45.423357Z","shell.execute_reply.started":"2023-07-06T19:57:45.410461Z","shell.execute_reply":"2023-07-06T19:57:45.422448Z"},"trusted":true},"execution_count":null,"outputs":[]}]}