{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:38:25.896438Z","iopub.execute_input":"2023-07-06T01:38:25.897132Z","iopub.status.idle":"2023-07-06T01:38:31.836664Z","shell.execute_reply.started":"2023-07-06T01:38:25.897094Z","shell.execute_reply":"2023-07-06T01:38:31.835614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install plotly","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:00.155178Z","iopub.execute_input":"2023-07-06T01:39:00.155614Z","iopub.status.idle":"2023-07-06T01:39:16.268583Z","shell.execute_reply.started":"2023-07-06T01:39:00.155581Z","shell.execute_reply":"2023-07-06T01:39:16.267366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.models import Sequential\nfrom keras.layers import LSTM, GRU, SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers import Embedding,BatchNormalization\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-07-06T01:39:20.803603Z","iopub.execute_input":"2023-07-06T01:39:20.804123Z","iopub.status.idle":"2023-07-06T01:39:21.349503Z","shell.execute_reply.started":"2023-07-06T01:39:20.804079Z","shell.execute_reply":"2023-07-06T01:39:21.348301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:25.627698Z","iopub.execute_input":"2023-07-06T01:39:25.628136Z","iopub.status.idle":"2023-07-06T01:39:34.391262Z","shell.execute_reply.started":"2023-07-06T01:39:25.628104Z","shell.execute_reply":"2023-07-06T01:39:34.390307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-07-06T01:39:42.811781Z","iopub.execute_input":"2023-07-06T01:39:42.812757Z","iopub.status.idle":"2023-07-06T01:39:46.174857Z","shell.execute_reply.started":"2023-07-06T01:39:42.81271Z","shell.execute_reply":"2023-07-06T01:39:46.173644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:48.592466Z","iopub.execute_input":"2023-07-06T01:39:48.593526Z","iopub.status.idle":"2023-07-06T01:39:48.612706Z","shell.execute_reply.started":"2023-07-06T01:39:48.593476Z","shell.execute_reply":"2023-07-06T01:39:48.611687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000,:]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:52.295361Z","iopub.execute_input":"2023-07-06T01:39:52.296309Z","iopub.status.idle":"2023-07-06T01:39:52.304204Z","shell.execute_reply.started":"2023-07-06T01:39:52.296263Z","shell.execute_reply":"2023-07-06T01:39:52.303109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:55.827932Z","iopub.execute_input":"2023-07-06T01:39:55.8289Z","iopub.status.idle":"2023-07-06T01:39:55.910697Z","shell.execute_reply.started":"2023-07-06T01:39:55.828856Z","shell.execute_reply":"2023-07-06T01:39:55.909765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:39:59.865638Z","iopub.execute_input":"2023-07-06T01:39:59.866293Z","iopub.status.idle":"2023-07-06T01:39:59.873173Z","shell.execute_reply.started":"2023-07-06T01:39:59.866254Z","shell.execute_reply":"2023-07-06T01:39:59.871964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:40:03.989784Z","iopub.execute_input":"2023-07-06T01:40:03.990225Z","iopub.status.idle":"2023-07-06T01:40:04.002334Z","shell.execute_reply.started":"2023-07-06T01:40:03.990192Z","shell.execute_reply":"2023-07-06T01:40:04.00122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n# using keras tokenizer here\ntoken = text.Tokenizer(num_words=None)\nmax_len = 1500\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n#zero pad the sequences\nxtrain_pad = pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = pad_sequences(xvalid_seq, maxlen=max_len)\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:41:10.211644Z","iopub.execute_input":"2023-07-06T01:41:10.212679Z","iopub.status.idle":"2023-07-06T01:41:12.017262Z","shell.execute_reply.started":"2023-07-06T01:41:10.212632Z","shell.execute_reply":"2023-07-06T01:41:12.016164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # A simpleRNN without any pretrained embeddings and one dense layer\n    model = Sequential()\n    model.add(Embedding(len(word_index) + 1,\n                     300,\n                     input_length=max_len))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:41:15.826916Z","iopub.execute_input":"2023-07-06T01:41:15.827375Z","iopub.status.idle":"2023-07-06T01:41:18.427694Z","shell.execute_reply.started":"2023-07-06T01:41:15.82734Z","shell.execute_reply":"2023-07-06T01:41:18.426814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:03.33997Z","iopub.execute_input":"2023-07-06T01:42:03.340416Z","iopub.status.idle":"2023-07-06T01:42:24.179874Z","shell.execute_reply.started":"2023-07-06T01:42:03.340385Z","shell.execute_reply":"2023-07-06T01:42:24.17874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:30.121382Z","iopub.execute_input":"2023-07-06T01:42:30.121844Z","iopub.status.idle":"2023-07-06T01:42:33.385173Z","shell.execute_reply.started":"2023-07-06T01:42:30.12181Z","shell.execute_reply":"2023-07-06T01:42:33.383942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN','AUC_Score': roc_auc(scores,yvalid)})","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:37.695008Z","iopub.execute_input":"2023-07-06T01:42:37.695426Z","iopub.status.idle":"2023-07-06T01:42:37.703304Z","shell.execute_reply.started":"2023-07-06T01:42:37.695396Z","shell.execute_reply":"2023-07-06T01:42:37.702125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xtrain_seq[:1]","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:44.6728Z","iopub.execute_input":"2023-07-06T01:42:44.673755Z","iopub.status.idle":"2023-07-06T01:42:44.680455Z","shell.execute_reply.started":"2023-07-06T01:42:44.673705Z","shell.execute_reply":"2023-07-06T01:42:44.679361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/glove840b300dtxt/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in tqdm(f):\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-07-06T01:42:51.587318Z","iopub.execute_input":"2023-07-06T01:42:51.58774Z","iopub.status.idle":"2023-07-06T01:47:14.033898Z","shell.execute_reply.started":"2023-07-06T01:42:51.58771Z","shell.execute_reply":"2023-07-06T01:47:14.032452Z"},"trusted":true},"execution_count":null,"outputs":[]}]}