{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-17T15:11:18.258635Z","iopub.execute_input":"2023-07-17T15:11:18.258964Z","iopub.status.idle":"2023-07-17T15:11:19.512389Z","shell.execute_reply.started":"2023-07-17T15:11:18.258938Z","shell.execute_reply":"2023-07-17T15:11:19.511411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing import text,sequence\nfrom keras.utils import pad_sequences\nfrom keras.models import Sequential\nfrom keras.layers import Embedding,SimpleRNN,LSTM,SpatialDropout1D,GRU,Bidirectional,Input\nfrom keras.layers.core import Dense,Activation,Dropout","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:02.479242Z","iopub.execute_input":"2023-07-17T15:12:02.479652Z","iopub.status.idle":"2023-07-17T15:12:02.486287Z","shell.execute_reply.started":"2023-07-17T15:12:02.479622Z","shell.execute_reply":"2023-07-17T15:12:02.485199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:14:03.358320Z","iopub.execute_input":"2023-07-17T15:14:03.359573Z","iopub.status.idle":"2023-07-17T15:14:03.375709Z","shell.execute_reply.started":"2023-07-17T15:14:03.359527Z","shell.execute_reply":"2023-07-17T15:14:03.374545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#configurint TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU', tpu.master())\nexcept ValueError:\n    tpu = None","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:05.726544Z","iopub.execute_input":"2023-07-17T15:12:05.726989Z","iopub.status.idle":"2023-07-17T15:12:05.732797Z","shell.execute_reply.started":"2023-07-17T15:12:05.726955Z","shell.execute_reply":"2023-07-17T15:12:05.731827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    #default distribution strategy in tensorflow, Works on CPU and single GPU\n    strategy = tf.distribute.TPUStrategy()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:09.092280Z","iopub.execute_input":"2023-07-17T15:12:09.093128Z","iopub.status.idle":"2023-07-17T15:12:17.255720Z","shell.execute_reply.started":"2023-07-17T15:12:09.093089Z","shell.execute_reply":"2023-07-17T15:12:17.254532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:23.960017Z","iopub.execute_input":"2023-07-17T15:12:23.960429Z","iopub.status.idle":"2023-07-17T15:12:27.290047Z","shell.execute_reply.started":"2023-07-17T15:12:23.960398Z","shell.execute_reply":"2023-07-17T15:12:27.288621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:27.291878Z","iopub.execute_input":"2023-07-17T15:12:27.292187Z","iopub.status.idle":"2023-07-17T15:12:27.302370Z","shell.execute_reply.started":"2023-07-17T15:12:27.292160Z","shell.execute_reply":"2023-07-17T15:12:27.301404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:27.303532Z","iopub.execute_input":"2023-07-17T15:12:27.303803Z","iopub.status.idle":"2023-07-17T15:12:27.327550Z","shell.execute_reply.started":"2023-07-17T15:12:27.303780Z","shell.execute_reply":"2023-07-17T15:12:27.326526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:30.273512Z","iopub.execute_input":"2023-07-17T15:12:30.273926Z","iopub.status.idle":"2023-07-17T15:12:30.288220Z","shell.execute_reply.started":"2023-07-17T15:12:30.273895Z","shell.execute_reply":"2023-07-17T15:12:30.286901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check max len of comment_text column to use this for padding in future\npad_len = train['comment_text'].apply(lambda x:len(str(x).split())).max()\nprint('max len of comment_text column',pad_len)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:30.996433Z","iopub.execute_input":"2023-07-17T15:12:30.996912Z","iopub.status.idle":"2023-07-17T15:12:32.234261Z","shell.execute_reply.started":"2023-07-17T15:12:30.996876Z","shell.execute_reply":"2023-07-17T15:12:32.232972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preperation","metadata":{}},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, stratify = train.toxic.values, random_state = 42,test_size = 0.2,shuffle = True)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:34.234373Z","iopub.execute_input":"2023-07-17T15:12:34.234805Z","iopub.status.idle":"2023-07-17T15:12:34.311956Z","shell.execute_reply.started":"2023-07-17T15:12:34.234773Z","shell.execute_reply":"2023-07-17T15:12:34.310682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(xtrain),len(xvalid)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:34.893723Z","iopub.execute_input":"2023-07-17T15:12:34.894075Z","iopub.status.idle":"2023-07-17T15:12:34.900985Z","shell.execute_reply.started":"2023-07-17T15:12:34.894047Z","shell.execute_reply":"2023-07-17T15:12:34.899989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tokenisation and Padding with max len of words in curpus","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:37.169870Z","iopub.execute_input":"2023-07-17T15:12:37.170284Z","iopub.status.idle":"2023-07-17T15:12:37.182009Z","shell.execute_reply.started":"2023-07-17T15:12:37.170251Z","shell.execute_reply":"2023-07-17T15:12:37.180672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using keras tokenizer \ntoken = text.Tokenizer(num_words = None)\nmax_len = 2400\nxtest = test.content.values\ntoken.fit_on_texts(list(xtrain) + list(xvalid) + list(xtest))\n\nx_train_seq = token.texts_to_sequences(xtrain)\nx_valid_seq = token.texts_to_sequences(xvalid)\nx_test_seq = token.texts_to_sequences(xtest)\n\n#zero pad the sequences\nx_train_pad = pad_sequences(x_train_seq,maxlen = max_len)\nx_valid_pad = pad_sequences(x_valid_seq,maxlen = max_len)\nx_test_pad = pad_sequences(x_test_seq,maxlen = max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:12:37.868873Z","iopub.execute_input":"2023-07-17T15:12:37.869366Z","iopub.status.idle":"2023-07-17T15:13:26.872175Z","shell.execute_reply.started":"2023-07-17T15:12:37.869328Z","shell.execute_reply":"2023-07-17T15:13:26.870915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1: classification on basic RNN Network","metadata":{}},{"cell_type":"code","source":"len(word_index) + 1","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:05:22.644641Z","iopub.execute_input":"2023-07-17T14:05:22.644934Z","iopub.status.idle":"2023-07-17T14:05:22.650447Z","shell.execute_reply.started":"2023-07-17T14:05:22.644909Z","shell.execute_reply":"2023-07-17T14:05:22.649591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    \n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     input_length=max_len))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1,activation = 'sigmoid'))\n    model.compile(loss = 'binary_crossentropy',optimizer = 'adam',metrics = ['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:05:22.651511Z","iopub.execute_input":"2023-07-17T14:05:22.651774Z","iopub.status.idle":"2023-07-17T14:05:27.911978Z","shell.execute_reply.started":"2023-07-17T14:05:22.651745Z","shell.execute_reply":"2023-07-17T14:05:27.910765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using strategy to run the TPU\nmodel.fit(x_train_pad,ytrain,epochs = 5,batch_size = 64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:05:27.913280Z","iopub.execute_input":"2023-07-17T14:05:27.913604Z","iopub.status.idle":"2023-07-17T14:10:12.360201Z","shell.execute_reply.started":"2023-07-17T14:05:27.913575Z","shell.execute_reply":"2023-07-17T14:10:12.359073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nfrom tqdm import tqdm\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:10:12.361591Z","iopub.execute_input":"2023-07-17T14:10:12.361907Z","iopub.status.idle":"2023-07-17T14:10:12.378776Z","shell.execute_reply.started":"2023-07-17T14:10:12.361882Z","shell.execute_reply":"2023-07-17T14:10:12.377852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_val = model.predict(x_valid_pad)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:10:12.379890Z","iopub.execute_input":"2023-07-17T14:10:12.380208Z","iopub.status.idle":"2023-07-17T14:10:49.272412Z","shell.execute_reply.started":"2023-07-17T14:10:12.380180Z","shell.execute_reply":"2023-07-17T14:10:49.271037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy = roc_auc_score(yvalid,pred_val)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:10:49.274294Z","iopub.execute_input":"2023-07-17T14:10:49.274619Z","iopub.status.idle":"2023-07-17T14:10:49.295083Z","shell.execute_reply.started":"2023-07-17T14:10:49.274592Z","shell.execute_reply":"2023-07-17T14:10:49.294086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy_ls = []\nmodel_accuracy_ls.append({'model':'simpleRNN','AUC_SCORE':model_accuracy})","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:10:49.296284Z","iopub.execute_input":"2023-07-17T14:10:49.296597Z","iopub.status.idle":"2023-07-17T14:10:49.301325Z","shell.execute_reply.started":"2023-07-17T14:10:49.296568Z","shell.execute_reply":"2023-07-17T14:10:49.300407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy_ls","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:10:49.302389Z","iopub.execute_input":"2023-07-17T14:10:49.302664Z","iopub.status.idle":"2023-07-17T14:10:49.313356Z","shell.execute_reply.started":"2023-07-17T14:10:49.302640Z","shell.execute_reply":"2023-07-17T14:10:49.312508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 : classification with pretrained glove word Embedding with Basic LSTM Model","metadata":{}},{"cell_type":"code","source":"# load glove vector in a dictionary\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(value) for value in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found {} word vectors'.format(len(embeddings_index)))","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:14:21.621818Z","iopub.execute_input":"2023-07-17T15:14:21.622897Z","iopub.status.idle":"2023-07-17T15:18:37.608853Z","shell.execute_reply.started":"2023-07-17T15:14:21.622858Z","shell.execute_reply":"2023-07-17T15:18:37.607814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create an embedding metrics for the words which are part of our datasets\nembedding_metrics = np.zeros((len(word_index) + 1,300))\nfor word,i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_metrics[i] = embedding_vector\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:18:38.997871Z","iopub.execute_input":"2023-07-17T15:18:38.998214Z","iopub.status.idle":"2023-07-17T15:18:40.424730Z","shell.execute_reply.started":"2023-07-17T15:18:38.998176Z","shell.execute_reply":"2023-07-17T15:18:40.423634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_metrics.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:18:40.426888Z","iopub.execute_input":"2023-07-17T15:18:40.427231Z","iopub.status.idle":"2023-07-17T15:18:40.433659Z","shell.execute_reply.started":"2023-07-17T15:18:40.427195Z","shell.execute_reply":"2023-07-17T15:18:40.432726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    \n    #simple LSTM Model\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n             300, weights = [embedding_metrics],input_length = max_len,trainable = False))\n    model.add(LSTM(100,dropout = 0.3, recurrent_dropout = 0.3))\n    model.add(Dense(1, activation = 'sigmoid'))\n    model.compile(loss = 'binary_crossentropy',optimizer = 'adam',metrics = ['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:15:06.976853Z","iopub.execute_input":"2023-07-17T14:15:06.977173Z","iopub.status.idle":"2023-07-17T14:15:18.506909Z","shell.execute_reply.started":"2023-07-17T14:15:06.977131Z","shell.execute_reply":"2023-07-17T14:15:18.505986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train_pad,ytrain,epochs = 5, batch_size = 64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:15:18.508160Z","iopub.execute_input":"2023-07-17T14:15:18.508470Z","iopub.status.idle":"2023-07-17T14:23:08.076265Z","shell.execute_reply.started":"2023-07-17T14:15:18.508442Z","shell.execute_reply":"2023-07-17T14:23:08.074927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lstm_pred = model.predict(x_valid_pad)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:23:08.077628Z","iopub.execute_input":"2023-07-17T14:23:08.077936Z","iopub.status.idle":"2023-07-17T14:24:41.460708Z","shell.execute_reply.started":"2023-07-17T14:23:08.077910Z","shell.execute_reply":"2023-07-17T14:24:41.459272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy = roc_auc_score(yvalid,lstm_pred)\nmodel_accuracy_ls.append({'model':'LSTM','AUC_SCORE':model_accuracy})","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:24:41.463755Z","iopub.execute_input":"2023-07-17T14:24:41.464154Z","iopub.status.idle":"2023-07-17T14:24:41.485740Z","shell.execute_reply.started":"2023-07-17T14:24:41.464085Z","shell.execute_reply":"2023-07-17T14:24:41.484519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy_ls","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:24:41.486931Z","iopub.execute_input":"2023-07-17T14:24:41.487260Z","iopub.status.idle":"2023-07-17T14:24:41.493212Z","shell.execute_reply.started":"2023-07-17T14:24:41.487232Z","shell.execute_reply":"2023-07-17T14:24:41.492322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification on GRU(gated Recurrent Unit) Netwrok","metadata":{}},{"cell_type":"code","source":"%%time\n\nwith strategy.scope():\n    \n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                       300,\n                       weights = [embedding_metrics],\n                       input_length = max_len,\n                       trainable = False))\n    model.add(SpatialDropout1D(0.3))\n    model.add(GRU(300))\n    model.add(Dense(1, activation = 'sigmoid'))\n    \n    model.compile(loss = 'binary_crossentropy', optimizer = 'adam', metrics = ['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:24:41.494216Z","iopub.execute_input":"2023-07-17T14:24:41.494510Z","iopub.status.idle":"2023-07-17T14:24:54.247046Z","shell.execute_reply.started":"2023-07-17T14:24:41.494486Z","shell.execute_reply":"2023-07-17T14:24:54.246074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train_pad,ytrain, epochs = 5,batch_size = 64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:24:54.248980Z","iopub.execute_input":"2023-07-17T14:24:54.249344Z","iopub.status.idle":"2023-07-17T14:32:19.144340Z","shell.execute_reply.started":"2023-07-17T14:24:54.249313Z","shell.execute_reply":"2023-07-17T14:32:19.143147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gru_pred = model.predict(x_valid_pad)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:32:19.151179Z","iopub.execute_input":"2023-07-17T14:32:19.151486Z","iopub.status.idle":"2023-07-17T14:33:43.357880Z","shell.execute_reply.started":"2023-07-17T14:32:19.151460Z","shell.execute_reply":"2023-07-17T14:33:43.356455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy = roc_auc_score(yvalid,gru_pred)\nmodel_accuracy_ls.append({'model':'GRU','AUC_SCORE':model_accuracy})","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:33:43.361138Z","iopub.execute_input":"2023-07-17T14:33:43.361488Z","iopub.status.idle":"2023-07-17T14:33:43.382938Z","shell.execute_reply.started":"2023-07-17T14:33:43.361458Z","shell.execute_reply":"2023-07-17T14:33:43.381639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy_ls","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:33:43.384186Z","iopub.execute_input":"2023-07-17T14:33:43.384515Z","iopub.status.idle":"2023-07-17T14:33:43.392171Z","shell.execute_reply.started":"2023-07-17T14:33:43.384487Z","shell.execute_reply":"2023-07-17T14:33:43.391133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification with BiDirectional RNN's","metadata":{}},{"cell_type":"code","source":"%%time\n\nwith strategy.scope():\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                       300,\n                       weights = [embedding_metrics],\n                       input_length = max_len,\n                       trainable = False))\n    model.add(Bidirectional(LSTM(300, dropout = 0.3, recurrent_dropout = 0.3)))\n    model.add(Dense(1,activation = 'sigmoid'))\n    model.compile(loss = 'binary_crossentropy', optimizer = 'adam',metrics = ['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:18:40.434761Z","iopub.execute_input":"2023-07-17T15:18:40.435033Z","iopub.status.idle":"2023-07-17T15:18:54.792312Z","shell.execute_reply.started":"2023-07-17T15:18:40.435009Z","shell.execute_reply":"2023-07-17T15:18:54.791323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(x_train_pad, ytrain, epochs = 5, batch_size = 64* strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T15:18:54.793621Z","iopub.execute_input":"2023-07-17T15:18:54.793932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bidirectional_score = model.predict(x_valid_pad)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_accuracy = roc_auc_score(yvalid,bidirectional_score)\nmodel_accuracy_ls.append({'model':'Bidircetional_RNN','AUC_SCORE':model_accuracy})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization of results obtained from various Deep learning Model\nresults_df = pd.DataFrame(model_accuracy_ls).sort_values(by = 'AUC_SCORE', ascending = False)\nresults_df.style.background_gradient(cmap = 'Blues')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BERT Model","metadata":{}},{"cell_type":"code","source":"!pip install transformers","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.190426Z","iopub.status.idle":"2023-07-17T14:34:16.190815Z","shell.execute_reply.started":"2023-07-17T14:34:16.190613Z","shell.execute_reply":"2023-07-17T14:34:16.190631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers\nfrom tokenizers import BertWordPieceTokenizer\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.191821Z","iopub.status.idle":"2023-07-17T14:34:16.192232Z","shell.execute_reply.started":"2023-07-17T14:34:16.192010Z","shell.execute_reply":"2023-07-17T14:34:16.192029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.193775Z","iopub.status.idle":"2023-07-17T14:34:16.194187Z","shell.execute_reply.started":"2023-07-17T14:34:16.193966Z","shell.execute_reply":"2023-07-17T14:34:16.193984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tokenization","metadata":{}},{"cell_type":"code","source":"# load the real tokenizer\ntokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n\n#save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n\n#reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt',lowercase = False)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.195285Z","iopub.status.idle":"2023-07-17T14:34:16.195667Z","shell.execute_reply.started":"2023-07-17T14:34:16.195472Z","shell.execute_reply":"2023-07-17T14:34:16.195491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.196663Z","iopub.status.idle":"2023-07-17T14:34:16.197058Z","shell.execute_reply.started":"2023-07-17T14:34:16.196847Z","shell.execute_reply":"2023-07-17T14:34:16.196865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = fast_encode(train.comment_text.astype(str), fast_tokenizer, maxlen=192)\nx_valid = fast_encode(validation.comment_text.astype(str), fast_tokenizer, maxlen=192)\nx_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=192)\n\ny_train = train.toxic.values\ny_valid = validation.toxic.values","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.198311Z","iopub.status.idle":"2023-07-17T14:34:16.198699Z","shell.execute_reply.started":"2023-07-17T14:34:16.198504Z","shell.execute_reply":"2023-07-17T14:34:16.198523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 512\nAUTO = tf.data.experimental.AUTOTUNE\n\ntrain_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.200421Z","iopub.status.idle":"2023-07-17T14:34:16.200868Z","shell.execute_reply.started":"2023-07-17T14:34:16.200613Z","shell.execute_reply":"2023-07-17T14:34:16.200631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    funtion for training Bert Model\n    \"\"\"\n    \n    input_word_ids = Input(shape = (max_len,), dtype = tf.int32, name = \"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:,0,:]\n    out = Dense(1, activation = 'sigmoid')(cls_token)\n    model = Model(inputs = input_word_ids, outputs = out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.202389Z","iopub.status.idle":"2023-07-17T14:34:16.202780Z","shell.execute_reply.started":"2023-07-17T14:34:16.202581Z","shell.execute_reply":"2023-07-17T14:34:16.202600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# start model training \nwith strategy.scope():\n    transformer_layer = (\n    transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer,max_len = 192)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.203830Z","iopub.status.idle":"2023-07-17T14:34:16.204248Z","shell.execute_reply.started":"2023-07-17T14:34:16.204010Z","shell.execute_reply":"2023-07-17T14:34:16.204028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] # BATCH_SIZE\ntrain_history = model.fit(train_dataset,\n                         steps_per_epoch = n_steps,\n                         validation_data=valid_dataset,\n                         epochs=2\n                         )","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.205157Z","iopub.status.idle":"2023-07-17T14:34:16.205534Z","shell.execute_reply.started":"2023-07-17T14:34:16.205340Z","shell.execute_reply":"2023-07-17T14:34:16.205358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_1 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=2*2\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.206913Z","iopub.status.idle":"2023-07-17T14:34:16.207340Z","shell.execute_reply.started":"2023-07-17T14:34:16.207118Z","shell.execute_reply":"2023-07-17T14:34:16.207137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['toxic'] = model.predict(test_dataset, verbose=1)\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-17T14:34:16.208236Z","iopub.status.idle":"2023-07-17T14:34:16.208619Z","shell.execute_reply.started":"2023-07-17T14:34:16.208424Z","shell.execute_reply":"2023-07-17T14:34:16.208443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}