{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers.recurrent import LSTM, GRU,SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers.embeddings import Embedding\nfrom keras.layers import BatchNormalization\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-14T11:00:07.533129Z","iopub.execute_input":"2023-07-14T11:00:07.534364Z","iopub.status.idle":"2023-07-14T11:00:17.632621Z","shell.execute_reply.started":"2023-07-14T11:00:07.534243Z","shell.execute_reply":"2023-07-14T11:00:17.631575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-07-14T11:00:17.634884Z","iopub.execute_input":"2023-07-14T11:00:17.635304Z","iopub.status.idle":"2023-07-14T11:00:21.795983Z","shell.execute_reply.started":"2023-07-14T11:00:17.635265Z","shell.execute_reply":"2023-07-14T11:00:21.794939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am going to solve this problem as a binary classification and not a multi-label classification task","metadata":{}},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.798000Z","iopub.execute_input":"2023-07-14T11:00:21.798435Z","iopub.status.idle":"2023-07-14T11:00:21.828059Z","shell.execute_reply.started":"2023-07-14T11:00:21.798396Z","shell.execute_reply":"2023-07-14T11:00:21.827075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.830923Z","iopub.execute_input":"2023-07-14T11:00:21.831846Z","iopub.status.idle":"2023-07-14T11:00:21.842886Z","shell.execute_reply.started":"2023-07-14T11:00:21.831813Z","shell.execute_reply":"2023-07-14T11:00:21.841677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using the 5000 rows for training\n\ntrain = train.sample(5000)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.844286Z","iopub.execute_input":"2023-07-14T11:00:21.844595Z","iopub.status.idle":"2023-07-14T11:00:21.883828Z","shell.execute_reply.started":"2023-07-14T11:00:21.844567Z","shell.execute_reply":"2023-07-14T11:00:21.882695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#checking the maximum length of strings for the texts","metadata":{}},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.885659Z","iopub.execute_input":"2023-07-14T11:00:21.886050Z","iopub.status.idle":"2023-07-14T11:00:21.926144Z","shell.execute_reply.started":"2023-07-14T11:00:21.886010Z","shell.execute_reply":"2023-07-14T11:00:21.925193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Writing a function for getting auc score for validation","metadata":{}},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc_score = metrics.auc(fpr, tpr)\n    return roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.927888Z","iopub.execute_input":"2023-07-14T11:00:21.928218Z","iopub.status.idle":"2023-07-14T11:00:21.934998Z","shell.execute_reply.started":"2023-07-14T11:00:21.928182Z","shell.execute_reply":"2023-07-14T11:00:21.933598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preparation","metadata":{}},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.25, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.936601Z","iopub.execute_input":"2023-07-14T11:00:21.937258Z","iopub.status.idle":"2023-07-14T11:00:21.951815Z","shell.execute_reply.started":"2023-07-14T11:00:21.937221Z","shell.execute_reply":"2023-07-14T11:00:21.950596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using keras tokenizer \ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\n\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\n\n#converting texts to sequences\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n\n#zero paddingthe sequences\nxtrain_pad = sequence.pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = sequence.pad_sequences(xvalid_seq, maxlen=max_len)\n\n#checking the vocabs\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:21.953559Z","iopub.execute_input":"2023-07-14T11:00:21.954002Z","iopub.status.idle":"2023-07-14T11:00:22.686558Z","shell.execute_reply.started":"2023-07-14T11:00:21.953964Z","shell.execute_reply":"2023-07-14T11:00:22.685487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# A simpleRNN without any pretrained embeddings and one dense layer\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,300,input_length=max_len))\nmodel.add(SimpleRNN(50))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:00:22.691254Z","iopub.execute_input":"2023-07-14T11:00:22.691584Z","iopub.status.idle":"2023-07-14T11:00:25.657790Z","shell.execute_reply.started":"2023-07-14T11:00:22.691554Z","shell.execute_reply":"2023-07-14T11:00:25.655739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, validation_data = (xvalid_pad,yvalid), epochs=10, batch_size=128) ","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:06:45.636523Z","iopub.execute_input":"2023-07-14T11:06:45.637076Z","iopub.status.idle":"2023-07-14T11:14:18.367250Z","shell.execute_reply.started":"2023-07-14T11:06:45.637041Z","shell.execute_reply":"2023-07-14T11:14:18.366199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(f'rouc_auc_score: {roc_auc(scores,yvalid)}')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:14:18.369454Z","iopub.execute_input":"2023-07-14T11:14:18.369836Z","iopub.status.idle":"2023-07-14T11:14:22.471502Z","shell.execute_reply.started":"2023-07-14T11:14:18.369797Z","shell.execute_reply":"2023-07-14T11:14:22.470414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:14:22.473026Z","iopub.execute_input":"2023-07-14T11:14:22.473404Z","iopub.status.idle":"2023-07-14T11:14:22.483053Z","shell.execute_reply.started":"2023-07-14T11:14:22.473373Z","shell.execute_reply":"2023-07-14T11:14:22.481883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Word Embeddings\n\n","metadata":{}},{"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:14:22.486293Z","iopub.execute_input":"2023-07-14T11:14:22.486783Z","iopub.status.idle":"2023-07-14T11:19:31.125570Z","shell.execute_reply.started":"2023-07-14T11:14:22.486744Z","shell.execute_reply":"2023-07-14T11:19:31.124527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:19:31.127117Z","iopub.execute_input":"2023-07-14T11:19:31.127995Z","iopub.status.idle":"2023-07-14T11:19:31.254264Z","shell.execute_reply.started":"2023-07-14T11:19:31.127955Z","shell.execute_reply":"2023-07-14T11:19:31.253238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    \n# A simple LSTM with glove embeddings and one dense layer\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,300,weights=[embedding_matrix],input_length=max_len,trainable=False))\n#always set trainable as false when using glove with the neural network model in the embedding layer.\n\nmodel.add(LSTM(50, dropout=0.3, recurrent_dropout=0.3))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:19:31.256036Z","iopub.execute_input":"2023-07-14T11:19:31.256749Z","iopub.status.idle":"2023-07-14T11:19:31.468194Z","shell.execute_reply.started":"2023-07-14T11:19:31.256701Z","shell.execute_reply":"2023-07-14T11:19:31.467108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, validation_data = (xvalid_pad,yvalid), epochs=3, batch_size=128)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:19:31.469788Z","iopub.execute_input":"2023-07-14T11:19:31.470447Z","iopub.status.idle":"2023-07-14T11:33:20.552204Z","shell.execute_reply.started":"2023-07-14T11:19:31.470406Z","shell.execute_reply":"2023-07-14T11:33:20.551117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(f'roc_auc_score,{roc_auc(scores,yvalid)}')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:34:10.230751Z","iopub.execute_input":"2023-07-14T11:34:10.231233Z","iopub.status.idle":"2023-07-14T11:34:10.239492Z","shell.execute_reply.started":"2023-07-14T11:34:10.231191Z","shell.execute_reply":"2023-07-14T11:34:10.238406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:34:18.988586Z","iopub.execute_input":"2023-07-14T11:34:18.988972Z","iopub.status.idle":"2023-07-14T11:34:18.996127Z","shell.execute_reply.started":"2023-07-14T11:34:18.988940Z","shell.execute_reply":"2023-07-14T11:34:18.994912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using GRU","metadata":{}},{"cell_type":"code","source":"\n    # GRU with glove embeddings and two dense layers\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,300,weights=[embedding_matrix],input_length=max_len,trainable=False))\nmodel.add(SpatialDropout1D(0.25))\nmodel.add(GRU(50))\nmodel.add(Dense(1, activation='sigmoid'))\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:34:25.582417Z","iopub.execute_input":"2023-07-14T11:34:25.582809Z","iopub.status.idle":"2023-07-14T11:34:25.755671Z","shell.execute_reply.started":"2023-07-14T11:34:25.582777Z","shell.execute_reply":"2023-07-14T11:34:25.754582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, batch_size=64,validation_data = (xvalid_pad,yvalid), epochs=3)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:34:27.089954Z","iopub.execute_input":"2023-07-14T11:34:27.090691Z","iopub.status.idle":"2023-07-14T11:53:12.035648Z","shell.execute_reply.started":"2023-07-14T11:34:27.090646Z","shell.execute_reply":"2023-07-14T11:53:12.034445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(f'roc_auc_score: {roc_auc(scores,yvalid)}')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:53:12.037889Z","iopub.execute_input":"2023-07-14T11:53:12.038383Z","iopub.status.idle":"2023-07-14T11:53:28.224889Z","shell.execute_reply.started":"2023-07-14T11:53:12.038346Z","shell.execute_reply":"2023-07-14T11:53:28.223872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'GRU','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:53:28.227042Z","iopub.execute_input":"2023-07-14T11:53:28.228146Z","iopub.status.idle":"2023-07-14T11:53:28.234959Z","shell.execute_reply.started":"2023-07-14T11:53:28.228106Z","shell.execute_reply":"2023-07-14T11:53:28.233956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:53:28.238192Z","iopub.execute_input":"2023-07-14T11:53:28.239113Z","iopub.status.idle":"2023-07-14T11:53:28.248069Z","shell.execute_reply.started":"2023-07-14T11:53:28.239076Z","shell.execute_reply":"2023-07-14T11:53:28.247001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using Bi-Directional RNN's\n\n","metadata":{}},{"cell_type":"code","source":"\n# A simple bidirectional LSTM with glove embeddings and one dense layer\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,300,weights=[embedding_matrix],input_length=max_len,trainable=False))\nmodel.add(Bidirectional(LSTM(50, dropout=0.3, recurrent_dropout=0.3)))\n\nmodel.add(Dense(1,activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:53:28.249698Z","iopub.execute_input":"2023-07-14T11:53:28.250090Z","iopub.status.idle":"2023-07-14T11:53:28.565114Z","shell.execute_reply.started":"2023-07-14T11:53:28.250052Z","shell.execute_reply":"2023-07-14T11:53:28.564059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, batch_size=64,validation_data = (xvalid_pad,yvalid), epochs=3)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T11:53:28.566880Z","iopub.execute_input":"2023-07-14T11:53:28.567621Z","iopub.status.idle":"2023-07-14T12:47:09.705187Z","shell.execute_reply.started":"2023-07-14T11:53:28.567581Z","shell.execute_reply":"2023-07-14T12:47:09.704190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(f'roc_auc_score: {roc_auc(scores,yvalid)}')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T12:51:51.789832Z","iopub.execute_input":"2023-07-14T12:51:51.790264Z","iopub.status.idle":"2023-07-14T12:52:29.741495Z","shell.execute_reply.started":"2023-07-14T12:51:51.790230Z","shell.execute_reply":"2023-07-14T12:52:29.740378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'Bi-directional LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-14T12:52:29.743958Z","iopub.execute_input":"2023-07-14T12:52:29.744726Z","iopub.status.idle":"2023-07-14T12:52:29.751665Z","shell.execute_reply.started":"2023-07-14T12:52:29.744686Z","shell.execute_reply":"2023-07-14T12:52:29.750607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-07-14T12:52:29.754208Z","iopub.execute_input":"2023-07-14T12:52:29.755145Z","iopub.status.idle":"2023-07-14T12:52:29.765672Z","shell.execute_reply.started":"2023-07-14T12:52:29.755051Z","shell.execute_reply":"2023-07-14T12:52:29.764629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model = []\n# AUC_Score= []\n# for i in range(len(scores_model)):\n#     AUC_Score.append(scores_model[i]['AUC_Score'])\n#     Model.append(scores_model[i]['Model'])\n\npd.DataFrame(scores_model)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T13:22:54.802248Z","iopub.execute_input":"2023-07-14T13:22:54.803237Z","iopub.status.idle":"2023-07-14T13:22:54.814631Z","shell.execute_reply.started":"2023-07-14T13:22:54.803184Z","shell.execute_reply":"2023-07-14T13:22:54.813532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Awesome! We can see that the Bi-directional LSTM Model was the best with an auc_score of 0.967556, next is the GRU(Gated Reccurent Units) which also got a similar score like the birectional lstm of 0.964404. LSTM also perfomed well with auc score of 0.946055. \nSimple rnn was the least with 0.77861, the model was too simple and usually considered as a naive baseline for the more sophisticated models like LSTM. \n\nThe results could become better if I had allowed it to run for more epochs but because of time and unavailability of a GPU right now, i didn't train for long. If you are using this notebook. Try to train for 100 epochs and add drop out layers to prevent overfitting. Check kaggle for more interesting problems to solve.","metadata":{}},{"cell_type":"code","source":"# Visualization of Results obtained from various Deep learning models\nresults = pd.DataFrame(scores_model).sort_values(by='AUC_Score',ascending=False)\nresults.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T13:22:27.142534Z","iopub.execute_input":"2023-07-14T13:22:27.142937Z","iopub.status.idle":"2023-07-14T13:22:27.225287Z","shell.execute_reply.started":"2023-07-14T13:22:27.142906Z","shell.execute_reply":"2023-07-14T13:22:27.224100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Funnelarea(\n    text =results.Model,\n    values = results.AUC_Score,\n    title = {\"position\": \"top center\", \"text\": \"Funnel-Chart of Sentiment Distribution\"}\n    ))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T13:23:18.021222Z","iopub.execute_input":"2023-07-14T13:23:18.021608Z","iopub.status.idle":"2023-07-14T13:23:18.098306Z","shell.execute_reply.started":"2023-07-14T13:23:18.021577Z","shell.execute_reply":"2023-07-14T13:23:18.097127Z"},"trusted":true},"execution_count":null,"outputs":[]}]}