{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install seaborn\n!pip install plotly","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-29T21:22:05.644658Z","iopub.execute_input":"2023-07-29T21:22:05.645477Z","iopub.status.idle":"2023-07-29T21:22:28.526647Z","shell.execute_reply.started":"2023-07-29T21:22:05.645437Z","shell.execute_reply":"2023-07-29T21:22:28.525404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU,SimpleRNN,Embedding,BatchNormalization\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.utils import np_utils,pad_sequences\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\n\n\n# import matplotlib.pyplot as plt\n# %matplotlib inline\n# import seaborn as sns\n# from plotly import graph_objs as go\n# import plotly.express as px\n# import plotly.figure_factory as ff","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:24:36.954773Z","iopub.execute_input":"2023-07-30T15:24:36.955133Z","iopub.status.idle":"2023-07-30T15:24:36.962882Z","shell.execute_reply.started":"2023-07-30T15:24:36.955103Z","shell.execute_reply":"2023-07-30T15:24:36.962072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up tpus \n# detect and init the TPU\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.TPUStrategy(tpu)\ntpu_cores=strategy.num_replicas_in_sync\n\nprint(\"REPLICAS: \", tpu_cores)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:24:38.954805Z","iopub.execute_input":"2023-07-30T15:24:38.955170Z","iopub.status.idle":"2023-07-30T15:24:52.177364Z","shell.execute_reply.started":"2023-07-30T15:24:38.955129Z","shell.execute_reply":"2023-07-30T15:24:52.176403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract Glove embedding weights\nembeddings_index = {}\nf = open('../input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:25:01.546667Z","iopub.execute_input":"2023-07-30T15:25:01.547065Z","iopub.status.idle":"2023-07-30T15:29:16.923674Z","shell.execute_reply.started":"2023-07-30T15:25:01.547034Z","shell.execute_reply":"2023-07-30T15:29:16.922529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare datasets\ntrain = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/test.csv')\ntrain=train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:30:31.358835Z","iopub.execute_input":"2023-07-30T15:30:31.359851Z","iopub.status.idle":"2023-07-30T15:30:34.738498Z","shell.execute_reply.started":"2023-07-30T15:30:31.359811Z","shell.execute_reply":"2023-07-30T15:30:34.737509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Auc metric function\ndef roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc\n\n# Split the data\n# train = train.loc[:12000,:]\nxtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:39:32.853403Z","iopub.execute_input":"2023-07-30T15:39:32.854163Z","iopub.status.idle":"2023-07-30T15:39:32.867988Z","shell.execute_reply.started":"2023-07-30T15:39:32.854113Z","shell.execute_reply":"2023-07-30T15:39:32.866853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenizing and padding\n\n# Define keras'tokenizer model\ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\n\n# Fit the model on the data\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n\n\n# zero pad the sequences\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:39:39.008001Z","iopub.execute_input":"2023-07-30T15:39:39.008511Z","iopub.status.idle":"2023-07-30T15:39:40.863432Z","shell.execute_reply.started":"2023-07-30T15:39:39.008470Z","shell.execute_reply":"2023-07-30T15:39:40.861977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in tqdm(word_index.items()):\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:39:47.702362Z","iopub.execute_input":"2023-07-30T15:39:47.702807Z","iopub.status.idle":"2023-07-30T15:39:47.891653Z","shell.execute_reply.started":"2023-07-30T15:39:47.702763Z","shell.execute_reply":"2023-07-30T15:39:47.890340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # GRU with glove embeddings and two dense layers\n     model = Sequential()\n     model.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n     model.add(SpatialDropout1D(0.3))\n     model.add(Bidirectional(LSTM(300)))\n     model.add(Dense(1, activation='sigmoid'))\n\n     model.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel.summary()\nmodel.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*tpu_cores)","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:39:50.772788Z","iopub.execute_input":"2023-07-30T15:39:50.774197Z","iopub.status.idle":"2023-07-30T15:40:36.281806Z","shell.execute_reply.started":"2023-07-30T15:39:50.774155Z","shell.execute_reply":"2023-07-30T15:40:36.280621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict and evaluate\nscores = model.predict(xvalid_pad)\nprint(f'{roc_auc(scores,yvalid):.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-07-30T15:41:03.926014Z","iopub.execute_input":"2023-07-30T15:41:03.926508Z","iopub.status.idle":"2023-07-30T15:41:12.701606Z","shell.execute_reply.started":"2023-07-30T15:41:03.926472Z","shell.execute_reply":"2023-07-30T15:41:12.700203Z"},"trusted":true},"execution_count":null,"outputs":[]}]}