{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T15:39:28.038073Z","iopub.execute_input":"2022-07-15T15:39:28.038683Z","iopub.status.idle":"2022-07-15T15:39:28.046469Z","shell.execute_reply.started":"2022-07-15T15:39:28.038645Z","shell.execute_reply":"2022-07-15T15:39:28.045461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\nimport nltk\nfrom nltk.stem.porter import PorterStemmer\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\n\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')\nstop_words = stopwords.words('english')\nimport html\nimport unicodedata\n\nimport spacy\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport re\nimport string\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n\nfrom tensorflow.keras.preprocessing.text import text_to_word_sequence\nfrom tensorflow.keras.preprocessing.text import Tokenizer  \nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras import models\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import losses\nfrom tensorflow.keras import metrics\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras.utils import plot_model\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:28.076453Z","iopub.execute_input":"2022-07-15T15:39:28.078331Z","iopub.status.idle":"2022-07-15T15:39:28.089898Z","shell.execute_reply.started":"2022-07-15T15:39:28.078303Z","shell.execute_reply":"2022-07-15T15:39:28.089034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data\ntrain = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/train.csv.zip')\ntest = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/test.csv.zip')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:28.106608Z","iopub.execute_input":"2022-07-15T15:39:28.107179Z","iopub.status.idle":"2022-07-15T15:39:30.533527Z","shell.execute_reply.started":"2022-07-15T15:39:28.107144Z","shell.execute_reply":"2022-07-15T15:39:30.532558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sample text to visualize\ntrain.sample(1)['comment_text'].values[0]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:30.536997Z","iopub.execute_input":"2022-07-15T15:39:30.537356Z","iopub.status.idle":"2022-07-15T15:39:30.549705Z","shell.execute_reply.started":"2022-07-15T15:39:30.537329Z","shell.execute_reply":"2022-07-15T15:39:30.548868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentence_lengths = [len(sentence) for sentence in train['comment_text']]\nplt.hist(sentence_lengths,500)\nplt.xlabel('Length of comments')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:30.551160Z","iopub.execute_input":"2022-07-15T15:39:30.551707Z","iopub.status.idle":"2022-07-15T15:39:32.170376Z","shell.execute_reply.started":"2022-07-15T15:39:30.551671Z","shell.execute_reply":"2022-07-15T15:39:32.169489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = train.drop(['id', 'comment_text'], axis=1)     ### Removed unnecessary columns - id and comment_text\ncounts = []                                               ### A list that contains tuple which consists of class label and number of comments for that particular class \ncategories = list(feature.columns.values)\nfor i in categories:\n    counts.append((i, feature[i].sum()))\n    \ndf_1 = pd.DataFrame(counts, columns=['Feature Labels', 'Total Comments'])   ### Dataframe made up of category and total number of comments\ndf_1.plot(x='Feature Labels', y='Total Comments', kind='bar',figsize=(8,8))\nplt.title(\"Comments per category\")\nplt.ylabel('Total comments', fontsize=12)\nplt.xlabel('Feature Labels', fontsize=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:32.173559Z","iopub.execute_input":"2022-07-15T15:39:32.173941Z","iopub.status.idle":"2022-07-15T15:39:32.387072Z","shell.execute_reply.started":"2022-07-15T15:39:32.173902Z","shell.execute_reply":"2022-07-15T15:39:32.386103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Text preprocessing\n","metadata":{}},{"cell_type":"code","source":"import string\n\ndef remove_special_chars(text):\n    re1 = re.compile(r'  +')\n    x1 = text.lower().replace('#39;', \"'\").replace('amp;', '&').replace('#146;', \"'\").replace(\n        'nbsp;', ' ').replace('#36;', '$').replace('\\\\n', \"\\n\").replace('quot;', \"'\").replace(\n        '<br />', \"\\n\").replace('\\\\\"', '\"').replace('<unk>', 'u_n').replace(' @.@ ', '.').replace(\n        ' @-@ ', '-').replace('\\\\', ' \\\\ ')\n    return re1.sub(' ', html.unescape(x1))\n\n\ndef to_lowercase(text):\n    return text.lower()\n\n\n\ndef remove_punctuation(text):\n    \"\"\"Remove punctuation from list of tokenized words\"\"\"\n    translator = str.maketrans('', '', string.punctuation)\n    return text.translate(translator)\n\n\ndef replace_numbers(text):\n    \"\"\"Replace all interger occurrences in list of tokenized words with textual representation\"\"\"\n    return re.sub(r'\\d+', '', text)\n\n\ndef remove_whitespaces(text):\n    return text.strip()\n\n\ndef remove_stopwords(words, stop_words):\n    \"\"\"\n    :param words:\n    :type words:\n    :param stop_words: from sklearn.feature_extraction.stop_words import ENGLISH_STOP_WORDS\n    or\n    from spacy.lang.en.stop_words import STOP_WORDS\n    :type stop_words:\n    :return:\n    :rtype:\n    \"\"\"\n    return [word for word in words if word not in stop_words]\n\n\ndef stem_words(words):\n    \"\"\"Stem words in text\"\"\"\n    stemmer = PorterStemmer()\n    return [stemmer.stem(word) for word in words]\n\ndef lemmatize_words(words):\n    \"\"\"Lemmatize words in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return [lemmatizer.lemmatize(word) for word in words]\n\ndef lemmatize_verbs(words):\n    \"\"\"Lemmatize verbs in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return ' '.join([lemmatizer.lemmatize(word, pos='v') for word in words])\n\ndef text2words(text):\n    return word_tokenize(text)\n\ndef clean_text( text):\n    text = remove_special_chars(text)\n    text = remove_punctuation(text)\n    text = to_lowercase(text)\n    text = replace_numbers(text)\n    words = text2words(text)\n    words = remove_stopwords(words, stop_words)\n    #words = stem_words(words)# Either stem ovocar lemmatize\n    words = lemmatize_words(words)\n    words = lemmatize_verbs(words)\n\n    return ''.join(words)\n\ntrain['comment_text'] = train['comment_text'].apply(lambda x: clean_text(x))\ntrain.sample(1)['comment_text'].values[0]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:39:32.388870Z","iopub.execute_input":"2022-07-15T15:39:32.389281Z","iopub.status.idle":"2022-07-15T15:42:19.852652Z","shell.execute_reply.started":"2022-07-15T15:39:32.389241Z","shell.execute_reply":"2022-07-15T15:42:19.851572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['comment_text'] = test['comment_text'].apply(lambda x: clean_text(x))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:42:19.854052Z","iopub.execute_input":"2022-07-15T15:42:19.854520Z","iopub.status.idle":"2022-07-15T15:44:48.546031Z","shell.execute_reply.started":"2022-07-15T15:42:19.854483Z","shell.execute_reply":"2022-07-15T15:44:48.545109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## BoW ","metadata":{}},{"cell_type":"code","source":"tok = Tokenizer(num_words=1000, oov_token='UNK')\ntok.fit_on_texts(train['comment_text'] )\n# Extract binary BoW features\nx_train = tok.texts_to_sequences(train['comment_text'])\nx_test = tok.texts_to_sequences(test['comment_text'])\n\nvocab_size = len(tok.word_index) + 1\nvocab_size","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:44:48.547903Z","iopub.execute_input":"2022-07-15T15:44:48.548281Z","iopub.status.idle":"2022-07-15T15:45:04.480403Z","shell.execute_reply.started":"2022-07-15T15:44:48.548247Z","shell.execute_reply":"2022-07-15T15:45:04.479326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM","metadata":{}},{"cell_type":"code","source":"maxlen = max([len(t) for t in x_train])\nmaxlen\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:45:04.482016Z","iopub.execute_input":"2022-07-15T15:45:04.482400Z","iopub.status.idle":"2022-07-15T15:45:04.506971Z","shell.execute_reply.started":"2022-07-15T15:45:04.482365Z","shell.execute_reply":"2022-07-15T15:45:04.506122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_padded = pad_sequences(x_train,\n                                maxlen=50, \n                                truncating='post', \n                                padding='post'\n                               )\ntest_padded = pad_sequences(x_test,\n                            maxlen=50, \n                            truncating='post', \n                            padding='post'\n                               )\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:45:04.508233Z","iopub.execute_input":"2022-07-15T15:45:04.508665Z","iopub.status.idle":"2022-07-15T15:45:05.998529Z","shell.execute_reply.started":"2022-07-15T15:45:04.508629Z","shell.execute_reply":"2022-07-15T15:45:05.997539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nmodel = models.Sequential()\nmodel.add(layers.Embedding(vocab_size, 128, input_length=50))\nmodel.add(layers.LSTM(512, dropout=0.2, recurrent_dropout=0.2, return_sequences=True))\nmodel.add(layers.LSTM(128, dropout=0.2,recurrent_dropout=0.2))\nmodel.add(layers.Dense(16, activation='relu'))\nmodel.add(layers.Dense(6, activation='sigmoid'))\n\n\n\nmodel.compile(\n    loss='binary_crossentropy',\n    optimizer='Adamax',\n    metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:45:06.002698Z","iopub.execute_input":"2022-07-15T15:45:06.002982Z","iopub.status.idle":"2022-07-15T15:45:06.353451Z","shell.execute_reply.started":"2022-07-15T15:45:06.002957Z","shell.execute_reply":"2022-07-15T15:45:06.352457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(training_padded,\n                     train[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=5,\n                    batch_size=512,\n                   validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:45:06.354712Z","iopub.execute_input":"2022-07-15T15:45:06.356435Z","iopub.status.idle":"2022-07-15T15:54:32.236910Z","shell.execute_reply.started":"2022-07-15T15:45:06.356395Z","shell.execute_reply":"2022-07-15T15:54:32.235634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\nhistory_dict.keys()\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:54:32.238451Z","iopub.execute_input":"2022-07-15T15:54:32.239343Z","iopub.status.idle":"2022-07-15T15:54:32.245630Z","shell.execute_reply.started":"2022-07-15T15:54:32.239296Z","shell.execute_reply":"2022-07-15T15:54:32.244576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:54:32.247518Z","iopub.execute_input":"2022-07-15T15:54:32.248404Z","iopub.status.idle":"2022-07-15T15:54:32.456817Z","shell.execute_reply.started":"2022-07-15T15:54:32.248367Z","shell.execute_reply":"2022-07-15T15:54:32.455917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.plot(epochs, acc, 'bo', label='Training accuracy')\nplt.plot(epochs, val_acc, 'b', label='Validation accuracy')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('roc_auc')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:54:32.458018Z","iopub.execute_input":"2022-07-15T15:54:32.458376Z","iopub.status.idle":"2022-07-15T15:54:32.656187Z","shell.execute_reply.started":"2022-07-15T15:54:32.458332Z","shell.execute_reply":"2022-07-15T15:54:32.655183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bidirectional","metadata":{}},{"cell_type":"code","source":"\nlstm_dim = 32\nmodel_bilstm = models.Sequential()\nmodel_bilstm.add(layers.Embedding(vocab_size, 512, input_length=50))\nmodel_bilstm.add(layers.Bidirectional(layers.LSTM(128, dropout=0.2, recurrent_dropout=0.2, return_sequences=True)))\nmodel_bilstm.add(layers.Flatten())\nmodel_bilstm.add(layers.Dense(16, activation='relu'))\nmodel_bilstm.add(layers.Dense(6, activation='sigmoid'))\n\n\nmodel_bilstm.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\n\n# Print the model summary\nmodel_bilstm.summary()\n\nhistory = model_bilstm.fit(training_padded,\n                     train[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=5,\n                    batch_size=512,\n                   validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T15:54:32.657489Z","iopub.execute_input":"2022-07-15T15:54:32.657819Z","iopub.status.idle":"2022-07-15T16:04:59.153977Z","shell.execute_reply.started":"2022-07-15T15:54:32.657784Z","shell.execute_reply":"2022-07-15T16:04:59.152961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\nhistory_dict.keys()\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:04:59.155682Z","iopub.execute_input":"2022-07-15T16:04:59.156531Z","iopub.status.idle":"2022-07-15T16:04:59.163920Z","shell.execute_reply.started":"2022-07-15T16:04:59.156489Z","shell.execute_reply":"2022-07-15T16:04:59.162900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:04:59.165181Z","iopub.execute_input":"2022-07-15T16:04:59.165619Z","iopub.status.idle":"2022-07-15T16:04:59.371971Z","shell.execute_reply.started":"2022-07-15T16:04:59.165584Z","shell.execute_reply":"2022-07-15T16:04:59.371075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(epochs, acc, 'bo', label='Training auc')\nplt.plot(epochs, val_acc, 'b', label='Validation auc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('roc_auc')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:04:59.373225Z","iopub.execute_input":"2022-07-15T16:04:59.374166Z","iopub.status.idle":"2022-07-15T16:04:59.565289Z","shell.execute_reply.started":"2022-07-15T16:04:59.374128Z","shell.execute_reply":"2022-07-15T16:04:59.564379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GRU","metadata":{}},{"cell_type":"code","source":"\nmodel_gru = models.Sequential()\nmodel_gru.add(layers.Embedding(1000, 20, input_length=maxlen))\nmodel_gru.add(layers.Bidirectional(layers.GRU(64)))\nmodel_gru.add(layers.Flatten())\nmodel_gru.add(layers.Dense(6, activation='sigmoid'))\n\n# Set the training parameters\nmodel_gru.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\n\n# Print the model summary\nmodel_gru.summary()\n\n\nhistory = model_gru.fit(training_padded,\n                     train[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=5,\n                    batch_size=128,\n                   validation_split=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:04:59.567217Z","iopub.execute_input":"2022-07-15T16:04:59.567873Z","iopub.status.idle":"2022-07-15T16:05:46.729144Z","shell.execute_reply.started":"2022-07-15T16:04:59.567836Z","shell.execute_reply":"2022-07-15T16:05:46.728253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\nhistory_dict.keys()\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:05:46.730906Z","iopub.execute_input":"2022-07-15T16:05:46.731267Z","iopub.status.idle":"2022-07-15T16:05:46.737168Z","shell.execute_reply.started":"2022-07-15T16:05:46.731232Z","shell.execute_reply":"2022-07-15T16:05:46.736133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:05:46.738756Z","iopub.execute_input":"2022-07-15T16:05:46.739237Z","iopub.status.idle":"2022-07-15T16:05:46.945285Z","shell.execute_reply.started":"2022-07-15T16:05:46.739185Z","shell.execute_reply":"2022-07-15T16:05:46.944394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(epochs, acc, 'bo', label='Training auc')\nplt.plot(epochs, val_acc, 'b', label='Validation auc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('roc_auc')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T16:05:46.946675Z","iopub.execute_input":"2022-07-15T16:05:46.947036Z","iopub.status.idle":"2022-07-15T16:05:47.157038Z","shell.execute_reply.started":"2022-07-15T16:05:46.947000Z","shell.execute_reply":"2022-07-15T16:05:47.156019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}