{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Contents\n\nIn this Notebook I will start with the very Basics of RNN's and Build all the way to latest deep learning architectures to solve NLP problems. It will cover the Following:\n* Simple RNN's\n* Word Embeddings : Definition and How to get them\n* LSTM's\n* GRU's\n\nI will divide every Topic into four subsections:\n* Basic Overview\n* In-Depth Understanding : In this I will attach links of articles and videos to learn about the topic in depth\n* Code-Implementation\n* Code Explanation\n\nThis is a comprehensive kernel and if you follow along till the end , I promise you would learn all the techniques completely\n\nNote that the aim of this notebook is not to have a High LB score but to present a beginner guide to understand Deep Learning techniques used for NLP. Also after discussing all of these ideas , I will present a starter solution for this competiton","metadata":{}},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:05:28.933015Z","iopub.execute_input":"2023-07-06T20:05:28.933604Z","iopub.status.idle":"2023-07-06T20:05:34.884607Z","shell.execute_reply.started":"2023-07-06T20:05:28.933569Z","shell.execute_reply":"2023-07-06T20:05:34.883646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install plotly","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:05:34.886555Z","iopub.execute_input":"2023-07-06T20:05:34.886855Z","iopub.status.idle":"2023-07-06T20:05:51.467682Z","shell.execute_reply.started":"2023-07-06T20:05:34.886828Z","shell.execute_reply":"2023-07-06T20:05:51.466768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers import Embedding,BatchNormalization\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:05:51.468968Z","iopub.execute_input":"2023-07-06T20:05:51.469273Z","iopub.status.idle":"2023-07-06T20:06:32.983345Z","shell.execute_reply.started":"2023-07-06T20:05:51.469247Z","shell.execute_reply":"2023-07-06T20:06:32.982224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:32.984659Z","iopub.execute_input":"2023-07-06T20:06:32.985459Z","iopub.status.idle":"2023-07-06T20:06:41.961893Z","shell.execute_reply.started":"2023-07-06T20:06:32.985427Z","shell.execute_reply":"2023-07-06T20:06:41.960876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:41.964629Z","iopub.execute_input":"2023-07-06T20:06:41.964936Z","iopub.status.idle":"2023-07-06T20:06:45.163684Z","shell.execute_reply.started":"2023-07-06T20:06:41.964907Z","shell.execute_reply":"2023-07-06T20:06:45.162725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.164804Z","iopub.execute_input":"2023-07-06T20:06:45.165098Z","iopub.status.idle":"2023-07-06T20:06:45.182914Z","shell.execute_reply.started":"2023-07-06T20:06:45.165073Z","shell.execute_reply":"2023-07-06T20:06:45.182055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000,:]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.184040Z","iopub.execute_input":"2023-07-06T20:06:45.184352Z","iopub.status.idle":"2023-07-06T20:06:45.207106Z","shell.execute_reply.started":"2023-07-06T20:06:45.184326Z","shell.execute_reply":"2023-07-06T20:06:45.206146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.208311Z","iopub.execute_input":"2023-07-06T20:06:45.208629Z","iopub.status.idle":"2023-07-06T20:06:45.292508Z","shell.execute_reply.started":"2023-07-06T20:06:45.208602Z","shell.execute_reply":"2023-07-06T20:06:45.291600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.293692Z","iopub.execute_input":"2023-07-06T20:06:45.294018Z","iopub.status.idle":"2023-07-06T20:06:45.299692Z","shell.execute_reply.started":"2023-07-06T20:06:45.293989Z","shell.execute_reply":"2023-07-06T20:06:45.298815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.300819Z","iopub.execute_input":"2023-07-06T20:06:45.301304Z","iopub.status.idle":"2023-07-06T20:06:45.330496Z","shell.execute_reply.started":"2023-07-06T20:06:45.301274Z","shell.execute_reply":"2023-07-06T20:06:45.329562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple RNN\n\n## Basic Overview\n\nWhat is a RNN?\n\nRecurrent Neural Network(RNN) are a type of Neural Network where the output from previous step are fed as input to the current step. In traditional neural networks, all the inputs and outputs are independent of each other, but in cases like when it is required to predict the next word of a sentence, the previous words are required and hence there is a need to remember the previous words. Thus RNN came into existence, which solved this issue with the help of a Hidden Layer.\n\nWhy RNN's?\n\nhttps://www.quora.com/Why-do-we-use-an-RNN-instead-of-a-simple-neural-network\n\n## In-Depth Understanding\n\n* https://medium.com/mindorks/understanding-the-recurrent-neural-network-44d593f112a2\n* https://www.youtube.com/watch?v=2E65LDnM2cA&list=PL1F3ABbhcqa3BBWo170U4Ev2wfsF7FN8l\n* https://www.d2l.ai/chapter_recurrent-neural-networks/rnn.html\n\n## Code Implementation\n\nSo first I will implement the and then I will explain the code step by step","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n# using keras tokenizer here\ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n#zero pad the sequences\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:45.331511Z","iopub.execute_input":"2023-07-06T20:06:45.331819Z","iopub.status.idle":"2023-07-06T20:06:47.154514Z","shell.execute_reply.started":"2023-07-06T20:06:45.331774Z","shell.execute_reply":"2023-07-06T20:06:47.153495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # A simpleRNN without any pretrained embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     input_length=max_len))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:47.155835Z","iopub.execute_input":"2023-07-06T20:06:47.156141Z","iopub.status.idle":"2023-07-06T20:06:49.711242Z","shell.execute_reply.started":"2023-07-06T20:06:47.156116Z","shell.execute_reply":"2023-07-06T20:06:49.710349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:06:49.712358Z","iopub.execute_input":"2023-07-06T20:06:49.712731Z","iopub.status.idle":"2023-07-06T20:07:10.934384Z","shell.execute_reply.started":"2023-07-06T20:06:49.712705Z","shell.execute_reply":"2023-07-06T20:07:10.933069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:07:10.938536Z","iopub.execute_input":"2023-07-06T20:07:10.938869Z","iopub.status.idle":"2023-07-06T20:07:14.213805Z","shell.execute_reply.started":"2023-07-06T20:07:10.938840Z","shell.execute_reply":"2023-07-06T20:07:14.212697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:07:14.215060Z","iopub.execute_input":"2023-07-06T20:07:14.215438Z","iopub.status.idle":"2023-07-06T20:07:14.222058Z","shell.execute_reply.started":"2023-07-06T20:07:14.215400Z","shell.execute_reply":"2023-07-06T20:07:14.221060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain_seq[:1]","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:07:14.223090Z","iopub.execute_input":"2023-07-06T20:07:14.223444Z","iopub.status.idle":"2023-07-06T20:07:14.237038Z","shell.execute_reply.started":"2023-07-06T20:07:14.223410Z","shell.execute_reply":"2023-07-06T20:07:14.236206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:07:14.238328Z","iopub.execute_input":"2023-07-06T20:07:14.238635Z","iopub.status.idle":"2023-07-06T20:11:32.015487Z","shell.execute_reply.started":"2023-07-06T20:07:14.238609Z","shell.execute_reply":"2023-07-06T20:11:32.014079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM's\n\n## Basic Overview\n\nSimple RNN's were certainly better than classical ML algorithms and gave state of the art results, but it failed to capture long term dependencies that is present in sentences . So in 1998-99 LSTM's were introduced to counter to these drawbacks.\n\n## In Depth Understanding\n\nWhy LSTM's?\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/PKMRR/vanishing-gradients-with-rnns\n* https://www.analyticsvidhya.com/blog/2017/12/fundamentals-of-deep-learning-introduction-to-lstm/\n\nWhat are LSTM's?\n* https://www.coursera.org/learn/nlp-sequence-models/lecture/KXoay/long-short-term-memory-lstm\n* https://distill.pub/2019/memorization-in-rnns/\n* https://towardsdatascience.com/illustrated-guide-to-lstms-and-gru-s-a-step-by-step-explanation-44e9eb85bf21\n\n# Code Implementation\n\nWe have already tokenized and paded our text for input to LSTM's","metadata":{}},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:11:32.016871Z","iopub.execute_input":"2023-07-06T20:11:32.017352Z","iopub.status.idle":"2023-07-06T20:11:32.199737Z","shell.execute_reply.started":"2023-07-06T20:11:32.017315Z","shell.execute_reply":"2023-07-06T20:11:32.198595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    \n    # A simple LSTM with glove embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n\n    model.add(LSTM(100, dropout=0.3, recurrent_dropout=0.3))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:11:32.200840Z","iopub.execute_input":"2023-07-06T20:11:32.201119Z","iopub.status.idle":"2023-07-06T20:11:35.586302Z","shell.execute_reply.started":"2023-07-06T20:11:32.201095Z","shell.execute_reply":"2023-07-06T20:11:35.584696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:11:35.587434Z","iopub.execute_input":"2023-07-06T20:11:35.587705Z","iopub.status.idle":"2023-07-06T20:12:04.412016Z","shell.execute_reply.started":"2023-07-06T20:11:35.587682Z","shell.execute_reply":"2023-07-06T20:12:04.410629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:12:04.413572Z","iopub.execute_input":"2023-07-06T20:12:04.413939Z","iopub.status.idle":"2023-07-06T20:12:10.250997Z","shell.execute_reply.started":"2023-07-06T20:12:04.413909Z","shell.execute_reply":"2023-07-06T20:12:10.249664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:12:10.252309Z","iopub.execute_input":"2023-07-06T20:12:10.252601Z","iopub.status.idle":"2023-07-06T20:12:10.258862Z","shell.execute_reply.started":"2023-07-06T20:12:10.252574Z","shell.execute_reply":"2023-07-06T20:12:10.257922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-07-06T20:12:10.260017Z","iopub.execute_input":"2023-07-06T20:12:10.260417Z","iopub.status.idle":"2023-07-06T20:12:10.273071Z","shell.execute_reply.started":"2023-07-06T20:12:10.260393Z","shell.execute_reply":"2023-07-06T20:12:10.272209Z"},"trusted":true},"execution_count":null,"outputs":[]}]}