{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tqdm import tqdm\n\n\n# Keras \nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, SimpleRNN\nfrom keras.layers import Dense, Activation, Dropout\nfrom keras.layers import Embedding\nfrom keras.layers import BatchNormalization\nfrom keras.utils import np_utils, pad_sequences\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, Flatten, MaxPooling1D, Bidirectional, SpatialDropout1D\nfrom keras.callbacks import EarlyStopping\n\n# Sklearn\nfrom keras.preprocessing import sequence, text\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-04T23:09:55.323477Z","iopub.execute_input":"2023-04-04T23:09:55.323906Z","iopub.status.idle":"2023-04-04T23:10:10.830327Z","shell.execute_reply.started":"2023-04-04T23:09:55.323866Z","shell.execute_reply":"2023-04-04T23:10:10.828931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## TPU Configuration\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:15.908553Z","iopub.execute_input":"2023-04-04T23:10:15.909007Z","iopub.status.idle":"2023-04-04T23:10:20.544730Z","shell.execute_reply.started":"2023-04-04T23:10:15.908964Z","shell.execute_reply":"2023-04-04T23:10:20.543272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\ndisplay(train.head())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:24.401354Z","iopub.execute_input":"2023-04-04T23:10:24.402347Z","iopub.status.idle":"2023-04-04T23:10:30.425671Z","shell.execute_reply.started":"2023-04-04T23:10:24.402272Z","shell.execute_reply":"2023-04-04T23:10:30.424471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate'], axis =1 , inplace = True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:33.801449Z","iopub.execute_input":"2023-04-04T23:10:33.801856Z","iopub.status.idle":"2023-04-04T23:10:33.836883Z","shell.execute_reply.started":"2023-04-04T23:10:33.801818Z","shell.execute_reply":"2023-04-04T23:10:33.835562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000, :]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:34.811331Z","iopub.execute_input":"2023-04-04T23:10:34.812366Z","iopub.status.idle":"2023-04-04T23:10:34.823422Z","shell.execute_reply.started":"2023-04-04T23:10:34.812317Z","shell.execute_reply":"2023-04-04T23:10:34.821695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.comment_text.apply(lambda x: len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:36.324075Z","iopub.execute_input":"2023-04-04T23:10:36.324589Z","iopub.status.idle":"2023-04-04T23:10:36.400480Z","shell.execute_reply.started":"2023-04-04T23:10:36.324550Z","shell.execute_reply":"2023-04-04T23:10:36.398995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_auc(predictions, target):\n    \n    fpr, tpr, thresholds= metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:37.283246Z","iopub.execute_input":"2023-04-04T23:10:37.283801Z","iopub.status.idle":"2023-04-04T23:10:37.291194Z","shell.execute_reply.started":"2023-04-04T23:10:37.283759Z","shell.execute_reply":"2023-04-04T23:10:37.289603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = model_selection.train_test_split(train.comment_text.values, train.toxic.values, stratify=train.toxic.values, shuffle = True, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:38.686882Z","iopub.execute_input":"2023-04-04T23:10:38.687646Z","iopub.status.idle":"2023-04-04T23:10:38.705854Z","shell.execute_reply.started":"2023-04-04T23:10:38.687564Z","shell.execute_reply":"2023-04-04T23:10:38.703753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = text.Tokenizer(num_words = None)\nmax_len = 1500\n\ntokenizer.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = tokenizer.texts_to_sequences(xtrain)\nxvalid_seq = tokenizer.texts_to_sequences(xvalid)\n\n# Padding\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\n\nword_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:39.719846Z","iopub.execute_input":"2023-04-04T23:10:39.720369Z","iopub.status.idle":"2023-04-04T23:10:41.659141Z","shell.execute_reply.started":"2023-04-04T23:10:39.720326Z","shell.execute_reply":"2023-04-04T23:10:41.657338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# word_index","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:41.661827Z","iopub.execute_input":"2023-04-04T23:10:41.662294Z","iopub.status.idle":"2023-04-04T23:10:41.669238Z","shell.execute_reply.started":"2023-04-04T23:10:41.662256Z","shell.execute_reply":"2023-04-04T23:10:41.667151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n## SimpleRNN\n\nwith strategy.scope():\n    # A simpleRNN without any pretrained embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index)+1, 300, input_length=max_len))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1, activation='sigmoid'))\n    \n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:10:42.336385Z","iopub.execute_input":"2023-04-04T23:10:42.338057Z","iopub.status.idle":"2023-04-04T23:10:44.631173Z","shell.execute_reply.started":"2023-04-04T23:10:42.337985Z","shell.execute_reply":"2023-04-04T23:10:44.629817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:11:02.295838Z","iopub.execute_input":"2023-04-04T23:11:02.296283Z","iopub.status.idle":"2023-04-04T23:11:21.643836Z","shell.execute_reply.started":"2023-04-04T23:11:02.296242Z","shell.execute_reply":"2023-04-04T23:11:21.642221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"AUC: %.2f%%\" %roc_auc(scores, yvalid))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:11:30.482110Z","iopub.execute_input":"2023-04-04T23:11:30.482658Z","iopub.status.idle":"2023-04-04T23:11:33.667511Z","shell.execute_reply.started":"2023-04-04T23:11:30.482621Z","shell.execute_reply":"2023-04-04T23:11:33.665779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN', 'AUC': roc_auc(scores, yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:11:38.189350Z","iopub.execute_input":"2023-04-04T23:11:38.189785Z","iopub.status.idle":"2023-04-04T23:11:38.197812Z","shell.execute_reply.started":"2023-04-04T23:11:38.189749Z","shell.execute_reply":"2023-04-04T23:11:38.196164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing GLOVE Embeddings\n# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:11:46.420068Z","iopub.execute_input":"2023-04-04T23:11:46.420622Z","iopub.status.idle":"2023-04-04T23:17:00.487678Z","shell.execute_reply.started":"2023-04-04T23:11:46.420566Z","shell.execute_reply":"2023-04-04T23:17:00.484919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:17:09.464778Z","iopub.execute_input":"2023-04-04T23:17:09.465383Z","iopub.status.idle":"2023-04-04T23:17:09.918620Z","shell.execute_reply.started":"2023-04-04T23:17:09.465329Z","shell.execute_reply":"2023-04-04T23:17:09.916510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    \n    # A simple LSTM with glove embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n\n    model.add(LSTM(100, dropout=0.3, recurrent_dropout=0.3))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:17:13.726862Z","iopub.execute_input":"2023-04-04T23:17:13.727486Z","iopub.status.idle":"2023-04-04T23:17:16.499225Z","shell.execute_reply.started":"2023-04-04T23:17:13.727435Z","shell.execute_reply":"2023-04-04T23:17:16.497559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:17:22.042114Z","iopub.execute_input":"2023-04-04T23:17:22.042593Z","iopub.status.idle":"2023-04-04T23:17:50.583691Z","shell.execute_reply.started":"2023-04-04T23:17:22.042553Z","shell.execute_reply":"2023-04-04T23:17:50.581598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:18:25.384342Z","iopub.execute_input":"2023-04-04T23:18:25.384826Z","iopub.status.idle":"2023-04-04T23:18:31.006915Z","shell.execute_reply.started":"2023-04-04T23:18:25.384786Z","shell.execute_reply":"2023-04-04T23:18:31.004968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:18:34.378745Z","iopub.execute_input":"2023-04-04T23:18:34.379176Z","iopub.status.idle":"2023-04-04T23:18:34.388146Z","shell.execute_reply.started":"2023-04-04T23:18:34.379139Z","shell.execute_reply":"2023-04-04T23:18:34.386632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GRU","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # GRU with glove embeddings and two dense layers\n     model = Sequential()\n     model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n     model.add(SpatialDropout1D(0.3))\n     model.add(GRU(300))\n     model.add(Dense(1, activation='sigmoid'))\n\n     model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:18:46.576462Z","iopub.execute_input":"2023-04-04T23:18:46.576959Z","iopub.status.idle":"2023-04-04T23:18:50.007686Z","shell.execute_reply.started":"2023-04-04T23:18:46.576919Z","shell.execute_reply":"2023-04-04T23:18:50.005770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:18:55.347855Z","iopub.execute_input":"2023-04-04T23:18:55.348377Z","iopub.status.idle":"2023-04-04T23:19:20.949463Z","shell.execute_reply.started":"2023-04-04T23:18:55.348335Z","shell.execute_reply":"2023-04-04T23:19:20.947847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:19:37.201225Z","iopub.execute_input":"2023-04-04T23:19:37.201752Z","iopub.status.idle":"2023-04-04T23:19:42.484623Z","shell.execute_reply.started":"2023-04-04T23:19:37.201712Z","shell.execute_reply":"2023-04-04T23:19:42.483345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model.append({'Model': 'GRU','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:19:44.785437Z","iopub.execute_input":"2023-04-04T23:19:44.785842Z","iopub.status.idle":"2023-04-04T23:19:44.796889Z","shell.execute_reply.started":"2023-04-04T23:19:44.785802Z","shell.execute_reply":"2023-04-04T23:19:44.794635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:19:46.752690Z","iopub.execute_input":"2023-04-04T23:19:46.753200Z","iopub.status.idle":"2023-04-04T23:19:46.763230Z","shell.execute_reply.started":"2023-04-04T23:19:46.753144Z","shell.execute_reply":"2023-04-04T23:19:46.761871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bidirectional LSTM","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # A simple bidirectional LSTM with glove embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n    model.add(Bidirectional(LSTM(300, dropout=0.3, recurrent_dropout=0.3)))\n\n    model.add(Dense(1,activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:19:54.701580Z","iopub.execute_input":"2023-04-04T23:19:54.701995Z","iopub.status.idle":"2023-04-04T23:19:58.248399Z","shell.execute_reply.started":"2023-04-04T23:19:54.701958Z","shell.execute_reply":"2023-04-04T23:19:58.246514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:20:01.878801Z","iopub.execute_input":"2023-04-04T23:20:01.879240Z","iopub.status.idle":"2023-04-04T23:21:17.286811Z","shell.execute_reply.started":"2023-04-04T23:20:01.879175Z","shell.execute_reply":"2023-04-04T23:21:17.284354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))\n# Auc: 0.97%\nscores_model.append({'Model': 'Bi-directional LSTM','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:21:31.518486Z","iopub.execute_input":"2023-04-04T23:21:31.518974Z","iopub.status.idle":"2023-04-04T23:21:42.304225Z","shell.execute_reply.started":"2023-04-04T23:21:31.518933Z","shell.execute_reply":"2023-04-04T23:21:42.302618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization of Results obtained from various Deep learning models\nresults = pd.DataFrame(scores_model).sort_values(by='AUC_Score',ascending=False)\nresults.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:21:48.523030Z","iopub.execute_input":"2023-04-04T23:21:48.523482Z","iopub.status.idle":"2023-04-04T23:21:48.642570Z","shell.execute_reply.started":"2023-04-04T23:21:48.523446Z","shell.execute_reply":"2023-04-04T23:21:48.641054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Funnelarea(\n    text =results.Model,\n    values = results.AUC_Score,\n    title = {\"position\": \"top center\", \"text\": \"Funnel-Chart of Sentiment Distribution\"}\n    ))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:21:56.921177Z","iopub.execute_input":"2023-04-04T23:21:56.922446Z","iopub.status.idle":"2023-04-04T23:21:57.076469Z","shell.execute_reply.started":"2023-04-04T23:21:56.922375Z","shell.execute_reply":"2023-04-04T23:21:57.074659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BERT","metadata":{}},{"cell_type":"code","source":"# Loading Dependencies\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\n\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:22:04.135324Z","iopub.execute_input":"2023-04-04T23:22:04.136273Z","iopub.status.idle":"2023-04-04T23:22:04.492534Z","shell.execute_reply.started":"2023-04-04T23:22:04.136146Z","shell.execute_reply":"2023-04-04T23:22:04.490419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOADING THE DATA\n\ntrain1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:22:06.922433Z","iopub.execute_input":"2023-04-04T23:22:06.923291Z","iopub.status.idle":"2023-04-04T23:22:11.093766Z","shell.execute_reply.started":"2023-04-04T23:22:06.923156Z","shell.execute_reply":"2023-04-04T23:22:11.091767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:27:34.730829Z","iopub.execute_input":"2023-04-04T23:27:34.731366Z","iopub.status.idle":"2023-04-04T23:27:34.741768Z","shell.execute_reply.started":"2023-04-04T23:27:34.731321Z","shell.execute_reply":"2023-04-04T23:27:34.739344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#IMP DATA FOR CONFIG\n\nAUTO = tf.data.experimental.AUTOTUNE\n\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:23:34.201175Z","iopub.execute_input":"2023-04-04T23:23:34.201631Z","iopub.status.idle":"2023-04-04T23:23:34.209546Z","shell.execute_reply.started":"2023-04-04T23:23:34.201593Z","shell.execute_reply":"2023-04-04T23:23:34.207831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:23:41.371143Z","iopub.execute_input":"2023-04-04T23:23:41.373018Z","iopub.status.idle":"2023-04-04T23:23:43.920446Z","shell.execute_reply.started":"2023-04-04T23:23:41.372950Z","shell.execute_reply":"2023-04-04T23:23:43.919471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = fast_encode(train1.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_valid = fast_encode(valid.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=MAX_LEN)\n\ny_train = train1.toxic.values\ny_valid = valid.toxic.values","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:27:39.602428Z","iopub.execute_input":"2023-04-04T23:27:39.603697Z","iopub.status.idle":"2023-04-04T23:28:40.106074Z","shell.execute_reply.started":"2023-04-04T23:27:39.603635Z","shell.execute_reply":"2023-04-04T23:28:40.103901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:29:54.144944Z","iopub.execute_input":"2023-04-04T23:29:54.145802Z","iopub.status.idle":"2023-04-04T23:29:56.981238Z","shell.execute_reply.started":"2023-04-04T23:29:54.145716Z","shell.execute_reply":"2023-04-04T23:29:56.979301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    function for training the BERT model\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:30:02.852318Z","iopub.execute_input":"2023-04-04T23:30:02.853067Z","iopub.status.idle":"2023-04-04T23:30:02.863859Z","shell.execute_reply.started":"2023-04-04T23:30:02.852959Z","shell.execute_reply":"2023-04-04T23:30:02.862119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:30:05.717499Z","iopub.execute_input":"2023-04-04T23:30:05.717932Z","iopub.status.idle":"2023-04-04T23:30:46.588735Z","shell.execute_reply.started":"2023-04-04T23:30:05.717896Z","shell.execute_reply":"2023-04-04T23:30:46.586637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:30:53.856821Z","iopub.execute_input":"2023-04-04T23:30:53.857437Z","iopub.status.idle":"2023-04-04T23:38:50.249523Z","shell.execute_reply.started":"2023-04-04T23:30:53.857391Z","shell.execute_reply":"2023-04-04T23:38:50.248266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS*2\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:39:17.520143Z","iopub.execute_input":"2023-04-04T23:39:17.520784Z","iopub.status.idle":"2023-04-04T23:40:14.169356Z","shell.execute_reply.started":"2023-04-04T23:39:17.520737Z","shell.execute_reply":"2023-04-04T23:40:14.167903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['toxic'] = model.predict(test_dataset, verbose=1)\nsub","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:40:20.879610Z","iopub.execute_input":"2023-04-04T23:40:20.880065Z","iopub.status.idle":"2023-04-04T23:40:41.266916Z","shell.execute_reply.started":"2023-04-04T23:40:20.880025Z","shell.execute_reply":"2023-04-04T23:40:41.265571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-04T23:40:50.459749Z","iopub.execute_input":"2023-04-04T23:40:50.460220Z","iopub.status.idle":"2023-04-04T23:40:50.591411Z","shell.execute_reply.started":"2023-04-04T23:40:50.460160Z","shell.execute_reply":"2023-04-04T23:40:50.590044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}