{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-29T11:47:51.366622Z","iopub.execute_input":"2021-10-29T11:47:51.367120Z","iopub.status.idle":"2021-10-29T11:47:51.395678Z","shell.execute_reply.started":"2021-10-29T11:47:51.367022Z","shell.execute_reply":"2021-10-29T11:47:51.394948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import RNN,LSTM,GRU,SimpleRNN\nfrom tensorflow.keras.layers import Embedding,Dense,GlobalAveragePooling1D\n\nimport matplotlib.pyplot as plt\nimport tqdm.notebook as tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:47:52.067100Z","iopub.execute_input":"2021-10-29T11:47:52.067635Z","iopub.status.idle":"2021-10-29T11:47:57.508910Z","shell.execute_reply.started":"2021-10-29T11:47:52.067594Z","shell.execute_reply":"2021-10-29T11:47:57.508069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RANDOM_STATE = 12","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:47:57.510526Z","iopub.execute_input":"2021-10-29T11:47:57.511479Z","iopub.status.idle":"2021-10-29T11:47:57.517460Z","shell.execute_reply.started":"2021-10-29T11:47:57.511449Z","shell.execute_reply":"2021-10-29T11:47:57.516302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Reading","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:47:57.519344Z","iopub.execute_input":"2021-10-29T11:47:57.520673Z","iopub.status.idle":"2021-10-29T11:48:00.968685Z","shell.execute_reply.started":"2021-10-29T11:47:57.520632Z","shell.execute_reply":"2021-10-29T11:48:00.967937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:00.971081Z","iopub.execute_input":"2021-10-29T11:48:00.971623Z","iopub.status.idle":"2021-10-29T11:48:00.994753Z","shell.execute_reply.started":"2021-10-29T11:48:00.971582Z","shell.execute_reply":"2021-10-29T11:48:00.994108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)\nprint(validation.shape)\nprint(test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:00.996080Z","iopub.execute_input":"2021-10-29T11:48:00.996336Z","iopub.status.idle":"2021-10-29T11:48:01.001886Z","shell.execute_reply.started":"2021-10-29T11:48:00.996302Z","shell.execute_reply":"2021-10-29T11:48:01.001061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.003393Z","iopub.execute_input":"2021-10-29T11:48:01.003895Z","iopub.status.idle":"2021-10-29T11:48:01.160088Z","shell.execute_reply.started":"2021-10-29T11:48:01.003854Z","shell.execute_reply":"2021-10-29T11:48:01.159272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# first try binary classification\ntrain.drop(columns = ['severe_toxic','obscene',\n                      'threat','insult',\n                      'identity_hate','id'],\n          inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.161396Z","iopub.execute_input":"2021-10-29T11:48:01.161668Z","iopub.status.idle":"2021-10-29T11:48:01.176344Z","shell.execute_reply.started":"2021-10-29T11:48:01.161633Z","shell.execute_reply":"2021-10-29T11:48:01.174835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.177585Z","iopub.execute_input":"2021-10-29T11:48:01.178029Z","iopub.status.idle":"2021-10-29T11:48:01.191279Z","shell.execute_reply.started":"2021-10-29T11:48:01.177980Z","shell.execute_reply":"2021-10-29T11:48:01.190460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.loc[:12000,:]","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.192627Z","iopub.execute_input":"2021-10-29T11:48:01.193001Z","iopub.status.idle":"2021-10-29T11:48:01.198130Z","shell.execute_reply.started":"2021-10-29T11:48:01.192958Z","shell.execute_reply":"2021-10-29T11:48:01.197365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.202154Z","iopub.execute_input":"2021-10-29T11:48:01.202426Z","iopub.status.idle":"2021-10-29T11:48:01.208992Z","shell.execute_reply.started":"2021-10-29T11:48:01.202377Z","shell.execute_reply":"2021-10-29T11:48:01.208188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply basically applies a function on all datasamples\n# lambda x: starts a function with x as the input\n# apply(lambda x: f(x)) applies the lambda function on all elements\\\npadding_len = train['comment_text'].apply(lambda x: len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.210313Z","iopub.execute_input":"2021-10-29T11:48:01.210717Z","iopub.status.idle":"2021-10-29T11:48:01.278245Z","shell.execute_reply.started":"2021-10-29T11:48:01.210681Z","shell.execute_reply":"2021-10-29T11:48:01.277611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training and Validation split\nX_train, X_valid, Y_train, Y_valid = train_test_split(train['comment_text'].values,\n                                                      train['toxic'].values,\n                                                      random_state = RANDOM_STATE,\n                                                      test_size = 0.2,\n                                                      shuffle= True)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.280119Z","iopub.execute_input":"2021-10-29T11:48:01.280587Z","iopub.status.idle":"2021-10-29T11:48:01.287540Z","shell.execute_reply.started":"2021-10-29T11:48:01.280552Z","shell.execute_reply":"2021-10-29T11:48:01.286857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ndel validation","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.288731Z","iopub.execute_input":"2021-10-29T11:48:01.289046Z","iopub.status.idle":"2021-10-29T11:48:01.294640Z","shell.execute_reply.started":"2021-10-29T11:48:01.288960Z","shell.execute_reply":"2021-10-29T11:48:01.293858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check if pre-processing is needed","metadata":{}},{"cell_type":"markdown","source":"## Model Building","metadata":{}},{"cell_type":"code","source":"def pre_process(train,\n                valid,\n                number_of_words,\n                padding_type,\n                max_len):\n    tokenizer = Tokenizer(num_words = number_of_words,\n                          filters='!\"#$%&()*+,-./:;<=>?@[\\\\]^_`{|}~\\t\\n',\n                          lower=False,\n                          split=' ',\n                          oov_token=\"<OOV>\")\n    tokenizer.fit_on_texts(list(train)+list(valid))\n    print('TOKENIZED')\n    train_sequence = tokenizer.texts_to_sequences(train)\n    print('SAVED VARIABLE 1')\n    valid_sequence = tokenizer.texts_to_sequences(valid)\n    print('SAVED VARIABLE 2')\n    padded_train = pad_sequences(train_sequence,\n                                maxlen=max_len,\n                                padding=padding_type,\n                                truncating=\"post\")\n    print('PADDED 1')\n    padded_valid = pad_sequences(valid_sequence,\n                                maxlen=max_len,\n                                padding=padding_type,\n                                truncating=\"post\")\n    print('PADDED 2')\n    word_index = tokenizer.word_index\n    return word_index,padded_train,padded_valid","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.295738Z","iopub.execute_input":"2021-10-29T11:48:01.296436Z","iopub.status.idle":"2021-10-29T11:48:01.304909Z","shell.execute_reply.started":"2021-10-29T11:48:01.296377Z","shell.execute_reply":"2021-10-29T11:48:01.304135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(type(X_train))","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.305989Z","iopub.execute_input":"2021-10-29T11:48:01.306334Z","iopub.status.idle":"2021-10-29T11:48:01.318179Z","shell.execute_reply.started":"2021-10-29T11:48:01.306299Z","shell.execute_reply":"2021-10-29T11:48:01.317371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_index,padded_train,padded_valid = pre_process(X_train.astype(str),\n                                                   X_valid.astype(str),\n                                                   None,\n                                                   \"post\",\n                                                   padding_len)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:01.319754Z","iopub.execute_input":"2021-10-29T11:48:01.320599Z","iopub.status.idle":"2021-10-29T11:48:03.433753Z","shell.execute_reply.started":"2021-10-29T11:48:01.320561Z","shell.execute_reply":"2021-10-29T11:48:03.432973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Building - RNN Simple","metadata":{}},{"cell_type":"code","source":"vocab_size = len(word_index.keys())\nembedding_size = 300\ninput_length = padded_train.shape[1]","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:03.435164Z","iopub.execute_input":"2021-10-29T11:48:03.435433Z","iopub.status.idle":"2021-10-29T11:48:03.441374Z","shell.execute_reply.started":"2021-10-29T11:48:03.435382Z","shell.execute_reply":"2021-10-29T11:48:03.440699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    Embedding(vocab_size,\n              embedding_size,\n              input_length = input_length),\n    SimpleRNN(100),\n    Dense(1,activation='sigmoid')\n])","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:03.443621Z","iopub.execute_input":"2021-10-29T11:48:03.444077Z","iopub.status.idle":"2021-10-29T11:48:05.910835Z","shell.execute_reply.started":"2021-10-29T11:48:03.444037Z","shell.execute_reply":"2021-10-29T11:48:05.909972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:05.912294Z","iopub.execute_input":"2021-10-29T11:48:05.912787Z","iopub.status.idle":"2021-10-29T11:48:05.924233Z","shell.execute_reply.started":"2021-10-29T11:48:05.912747Z","shell.execute_reply":"2021-10-29T11:48:05.922549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy',\n              optimizer = 'Adam',\n              metrics=['Accuracy','AUC'])","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:05.925375Z","iopub.execute_input":"2021-10-29T11:48:05.925653Z","iopub.status.idle":"2021-10-29T11:48:05.938052Z","shell.execute_reply.started":"2021-10-29T11:48:05.925617Z","shell.execute_reply":"2021-10-29T11:48:05.937013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(padded_train,Y_train,validation_data=(padded_valid, Y_valid),epochs = 5,batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T11:48:05.940483Z","iopub.execute_input":"2021-10-29T11:48:05.940993Z","iopub.status.idle":"2021-10-29T12:02:29.483104Z","shell.execute_reply.started":"2021-10-29T11:48:05.940958Z","shell.execute_reply":"2021-10-29T12:02:29.482335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['Accuracy']\nval_acc = history.history['val_Accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nplt.figure(figsize=(8, 8))\nplt.subplot(2, 1, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.ylabel('Accuracy')\n# plt.ylim([min(plt.ylim()),1])\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(2, 1, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.ylabel('Cross Entropy')\n# plt.ylim([0,1.0])\nplt.title('Training and Validation Loss')\nplt.xlabel('epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-29T12:02:29.484843Z","iopub.execute_input":"2021-10-29T12:02:29.485106Z","iopub.status.idle":"2021-10-29T12:02:29.869602Z","shell.execute_reply.started":"2021-10-29T12:02:29.485071Z","shell.execute_reply":"2021-10-29T12:02:29.868925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2021-10-29T12:02:29.870624Z","iopub.execute_input":"2021-10-29T12:02:29.871033Z","iopub.status.idle":"2021-10-29T12:02:29.877385Z","shell.execute_reply.started":"2021-10-29T12:02:29.870947Z","shell.execute_reply":"2021-10-29T12:02:29.876658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(padded_valid)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,Y_valid)))","metadata":{"execution":{"iopub.status.busy":"2021-10-29T12:02:29.878538Z","iopub.execute_input":"2021-10-29T12:02:29.878984Z","iopub.status.idle":"2021-10-29T12:02:37.527101Z","shell.execute_reply.started":"2021-10-29T12:02:29.878926Z","shell.execute_reply":"2021-10-29T12:02:37.526068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.evaluate(padded_valid, Y_valid)\nprint(\"test loss, test acc, test auc:\", results)","metadata":{"execution":{"iopub.status.busy":"2021-10-29T12:02:37.528734Z","iopub.execute_input":"2021-10-29T12:02:37.529035Z","iopub.status.idle":"2021-10-29T12:02:45.170440Z","shell.execute_reply.started":"2021-10-29T12:02:37.528993Z","shell.execute_reply":"2021-10-29T12:02:45.169471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Embedding Algorithm ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}