{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Cleaning and removing misspells from texts","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"Sources:  \n- https://www.kaggle.com/shonenkov/hack-with-parallel-corpus\n- https://www.kaggle.com/c/jigsaw-multilingual-toxic-comment-classification/discussion/147417","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Importing dependencies","execution_count":null},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!pip install -q pandarallel\n!pip install -q spacy \n!pip install -q spacy_cld\n!pip install -q pyspellchecker\n!python -m spacy download xx_ent_wiki_sm > /dev/null","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport gc\n\nimport spacy\nfrom spacy_cld import LanguageDetector\nimport xx_ent_wiki_sm\n\nfrom spellchecker import SpellChecker\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport time\nimport random\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nimport re\nimport nltk\n\nfrom pandarallel import pandarallel\npandarallel.initialize(progress_bar=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom sklearn.utils import shuffle\n\nfrom wordcloud import WordCloud, STOPWORDS\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, CSVLogger\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tokenizers import BertWordPieceTokenizer\nfrom colorama import Fore, Back, Style, init\nimport plotly.graph_objects as go\n\nfrom tensorflow.keras.layers import (Dense, Input, LSTM, Bidirectional, Activation, Conv1D, GRU,\n                          Embedding, Flatten, Dropout, Add, concatenate, MaxPooling1D,\n                         GlobalAveragePooling1D,  GlobalMaxPooling1D, GlobalMaxPool1D,\n                        SpatialDropout1D)\n\nfrom tensorflow.keras import (initializers, regularizers, constraints, \n                              optimizers, layers, callbacks)\nimport seaborn as sns\nsns.set(style=\"darkgrid\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('jigsaw-multilingual-toxic-comment-classification')\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 256\nMODEL = 'jplu/tf-xlm-roberta-large'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Loading Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain1['lang'] = 'en'\n\ntrain_es = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-es-cleaned.csv')\ntrain_es['lang'] = 'es'\n\ntrain_fr = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-fr-cleaned.csv')\ntrain_fr['lang'] = 'fr'\n\ntrain_pt = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-pt-cleaned.csv')\ntrain_pt['lang'] = 'pt'\n\ntrain_ru = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-ru-cleaned.csv')\ntrain_ru['lang'] = 'ru'\n\ntrain_it = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-it-cleaned.csv')\ntrain_it['lang'] = 'it'\n\ntrain_tr = pd.read_csv('/kaggle/input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-tr-cleaned.csv')\ntrain_tr['lang'] = 'tr'\n\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\ntrain2.toxic = train2.toxic.round().astype(int)\ntrain2['lang'] = 'en'\n\n\n# Valid set\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(valid['lang'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(test['lang'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train1.toxic.value_counts())\nprint(train2.toxic.value_counts())\nprint(train_es.toxic.value_counts())\nprint(train_fr.toxic.value_counts())\nprint(train_pt.toxic.value_counts())\nprint(train_ru.toxic.value_counts())\nprint(train_it.toxic.value_counts())\nprint(train_tr.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Cleaning The Data","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def decontracted(phrase):\n\n    # specific\n    phrase = re.sub(r\"won't\", \"will not\", phrase)\n    phrase = re.sub(r\"can\\'t\", \"can not\", phrase)\n    phrase = re.sub(r\"I\\'m\", \"I am\", phrase)\n    phrase = re.sub(r\"i\\'m\", \"i am\", phrase)\n\n    # general\n    phrase = re.sub(r\"n\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'re\", \" are\", phrase)\n    phrase = re.sub(r\"\\'s\", \" is\", phrase)\n    phrase = re.sub(r\"\\'d\", \" would\", phrase)\n    phrase = re.sub(r\"\\'ll\", \" will\", phrase)\n    phrase = re.sub(r\"\\'t\", \" not\", phrase)\n    phrase = re.sub(r\"\\'ve\", \" have\", phrase)\n    phrase = re.sub(r'[0-9\"]', '', phrase)\n    return phrase","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def sen_len(txt):\n    return len(txt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def spell_correction(txt):\n    txt = re.sub(r\"AAA\", \"\", txt)\n    txt = re.sub(r\"BBB\", \"\", txt)\n    txt = re.sub(r\"CCC\", \"\", txt)\n    txt = re.sub(r\"DDD\", \"\", txt)\n    txt = re.sub(r\"EEE\", \"\", txt)\n    txt = re.sub(r\"FFF\", \"\", txt)\n    txt = re.sub(r\"GGG\", \"\", txt)\n    txt = re.sub(r\"HHH\", \"\", txt)\n    txt = re.sub(r\"III\", \"\", txt)\n    txt = re.sub(r\"JJJ\", \"\", txt)\n    txt = re.sub(r\"KKK\", \"\", txt)\n    txt = re.sub(r\"LLL\", \"\", txt)\n    txt = re.sub(r\"MMM\", \"\", txt)\n    txt = re.sub(r\"NNN\", \"\", txt)\n    txt = re.sub(r\"OOO\", \"\", txt)\n    txt = re.sub(r\"PPP\", \"\", txt)\n    txt = re.sub(r\"QQQ\", \"\", txt)\n    txt = re.sub(r\"RRR\", \"\", txt)\n    txt = re.sub(r\"SSS\", \"\", txt)\n    txt = re.sub(r\"TTT\", \"\", txt)\n    txt = re.sub(r\"UUU\", \"\", txt)\n    txt = re.sub(r\"VVV\", \"\", txt)\n    txt = re.sub(r\"WWW\", \"\", txt)\n    txt = re.sub(r\"XXX\", \"\", txt)\n    txt = re.sub(r\"YYY\", \"\", txt)\n    txt = re.sub(r\"ZZZ\", \"\", txt)\n    txt = re.sub(r\"aaa\", \"\", txt)\n    txt = re.sub(r\"bbb\", \"\", txt)\n    txt = re.sub(r\"ccc\", \"\", txt)\n    txt = re.sub(r\"ddd\", \"\", txt)\n    txt = re.sub(r\"eee\", \"\", txt)\n    txt = re.sub(r\"fff\", \"\", txt)\n    txt = re.sub(r\"ggg\", \"\", txt)\n    txt = re.sub(r\"hhh\", \"\", txt)\n    txt = re.sub(r\"iii\", \"\", txt)\n    txt = re.sub(r\"jjj\", \"\", txt)\n    txt = re.sub(r\"kkk\", \"\", txt)\n    txt = re.sub(r\"lll\", \"\", txt)\n    txt = re.sub(r\"mmm\", \"\", txt)\n    txt = re.sub(r\"nnn\", \"\", txt)\n    txt = re.sub(r\"ooo\", \"\", txt)\n    txt = re.sub(r\"ppp\", \"\", txt)\n    txt = re.sub(r\"qqq\", \"\", txt)\n    txt = re.sub(r\"rrr\", \"\", txt)\n    txt = re.sub(r\"sss\", \"\", txt)\n    txt = re.sub(r\"ttt\", \"\", txt)\n    txt = re.sub(r\"uuu\", \"\", txt)\n    txt = re.sub(r\"vvv\", \"\", txt)\n    txt = re.sub(r\"www\", \"\", txt)\n    txt = re.sub(r\"xxx\", \"\", txt)\n    txt = re.sub(r\"yyy\", \"\", txt)\n    txt = re.sub(r\"zzz\", \"\", txt)\n    return txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"symbols = [',', '..', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", \" '\",\"' \", '$', '&', '/', \n                  '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£',\n                  '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←',\n                  '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', '\\xa0', '\\t','“', '★', '”', '–', '●', \n                  'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥',\n                  '▓', '—', '‹', '─', '\\u3000', '\\u202f', '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’',\n                  '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', '«',\n                 '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗',\n                  '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n\ndef remove_symbols(text):\n    for sym in symbols:\n        text = text.replace(sym,\"\")\n    return text","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Quick Example**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Quick Example\n\nsentences1 = [\"\\n I ate dinner 908\",\n            \"We had a three-course meal 5630....Hoz yor dey\",\n            \"@Brad came to dinner with us 4434.\",\n            \"@xyz loves fish 334  █ tacos. https://www.kaggle.com/maunish/nlp-cleaning/\",\n            \"@xyz loves fish 334  } tacos. www.kaggle.com/maunish/nlp-cleaning/\",\n            \"                    How are you. http://www.its.caltech.edu/~atomic/snowcrystals/myths/myths.htm#perfection\",\n            \"In the end, we 3434 all felt like we ate #too much                    \",\n             \"AAAAAAWWWWWWWWWWWW Howwwww are yoUUUUUUUUUUUU\"]\n\ndf1 = pd.DataFrame({\"sentences\":sentences1})\n\n\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: decontracted(x))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: remove_symbols(x))\ndf1['sentences'] = df1['sentences'].progress_apply(lambda x: spell_correction(x))\n\ndf1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.concat([\n    train1[['comment_text', 'lang', 'toxic']],\n    train2[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=150000, random_state=0)\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Applying Cleaning Methods**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train['comment_text'] = train['comment_text'].progress_apply(lambda x: decontracted(x))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: remove_symbols(x))\ntrain['comment_text'] = train['comment_text'].progress_apply(lambda x: spell_correction(x))\n\ntrain['sent_len'] = train['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.to_csv('Train_data_cleaned.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Spanish\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_es['comment_text'] = train_es['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_es['sent_len'] = train_es['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# French\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_fr['comment_text'] = train_fr['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_fr['sent_len'] = train_fr['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Portuguese\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_pt['comment_text'] = train_pt['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_pt['sent_len'] = train_pt['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Russian\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_ru['comment_text'] = train_ru['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_ru['sent_len'] = train_ru['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Italian\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_it['comment_text'] = train_it['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_it['sent_len'] = train_it['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Turkish\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#train_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntrain_tr['comment_text'] = train_tr['comment_text'].progress_apply(lambda x: remove_symbols(x))\n\ntrain_tr['sent_len'] = train_tr['comment_text'].progress_apply(lambda x: sen_len(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('------------- Training Set --> English -------------')\nprint(train['sent_len'].max())\nprint(train['sent_len'].min())\nprint(train['sent_len'].mean())\nprint('------------- Training Set --> Spanish -------------')\nprint(train_es['sent_len'].max())\nprint(train_es['sent_len'].min())\nprint(train_es['sent_len'].mean())\nprint('------------- Training Set --> French -------------')\nprint(train_fr['sent_len'].max())\nprint(train_fr['sent_len'].min())\nprint(train_fr['sent_len'].mean())\nprint('------------- Training Set --> Portuguese -------------')\nprint(train_pt['sent_len'].max())\nprint(train_pt['sent_len'].min())\nprint(train_pt['sent_len'].mean())\nprint('------------- Training Set --> Russian -------------')\nprint(train_ru['sent_len'].max())\nprint(train_ru['sent_len'].min())\nprint(train_ru['sent_len'].mean())\nprint('------------- Training Set --> Italian -------------')\nprint(train_it['sent_len'].max())\nprint(train_it['sent_len'].min())\nprint(train_it['sent_len'].mean())\nprint('------------- Training Set --> Turkish -------------')\nprint(train_tr['sent_len'].max())\nprint(train_tr['sent_len'].min())\nprint(train_tr['sent_len'].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, axes = plt.subplots(2, 4, figsize=(25, 7), sharex=True)\nsns.despine(left=True)\n\n#plt.figure(1)\nsns.distplot(train['sent_len'], ax=axes[0, 0]).set_title('English')\n#plt.figure(2)\nsns.distplot(train_es['sent_len'], ax=axes[0, 1]).set_title('Spanish')\nsns.distplot(train_fr['sent_len'], ax=axes[0, 2]).set_title('French')\nsns.distplot(train_pt['sent_len'], ax=axes[0, 3]).set_title('Portuguese')\nsns.distplot(train_ru['sent_len'], ax=axes[1, 0]).set_title('Russian')\nsns.distplot(train_it['sent_len'], ax=axes[1, 1]).set_title('Italian')\nsns.distplot(train_tr['sent_len'], ax=axes[1, 2]).set_title('Turkish')\n\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Selecting comments having length less than 1500 words**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train[train['sent_len'] < 1500]\ntrain_es = train_es[train_es['sent_len'] < 1500]\ntrain_fr = train_fr[train_fr['sent_len'] < 1500]\ntrain_it = train_it[train_it['sent_len'] < 1500]\ntrain_pt = train_pt[train_pt['sent_len'] < 1500]\ntrain_ru = train_ru[train_ru['sent_len'] < 1500]\ntrain_tr = train_tr[train_tr['sent_len'] < 1500]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('------------- Training Set --> English -------------')\nprint(train.toxic.value_counts())\nprint('------------- Training Set --> Spanish -------------')\nprint(train_es.toxic.value_counts())\nprint('------------- Training Set --> French -------------')\nprint(train_fr.toxic.value_counts())\nprint('------------- Training Set --> Portuguese -------------')\nprint(train_pt.toxic.value_counts())\nprint('------------- Training Set --> Russian -------------')\nprint(train_ru.toxic.value_counts())\nprint('------------- Training Set --> Italian -------------')\nprint(train_it.toxic.value_counts())\nprint('------------- Training Set --> Turkish -------------')\nprint(train_tr.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* Since, Non toxic comments are approximately 10 times in 7 languages except English. Beacause I already added toxic comments from **'inintended-bias-train'**. So, Balance the datasets for other 6 languages and maintain the ratio as of English language","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_es = pd.concat([\n    train_es[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_es[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\ntrain_fr = pd.concat([\n    train_fr[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_fr[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\ntrain_pt = pd.concat([\n    train_pt[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_pt[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\ntrain_ru = pd.concat([\n    train_ru[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_ru[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\ntrain_it = pd.concat([\n    train_it[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_it[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\ntrain_tr = pd.concat([\n    train_tr[['comment_text', 'lang', 'toxic']].query('toxic==1'),\n    train_tr[['comment_text', 'lang', 'toxic']].query('toxic==0').sample(n=60000, random_state=0)\n    ])\n\n\n\nprint(train_es.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.concat([\n    \n    train[['comment_text', 'lang', 'toxic']],\n    train_es[['comment_text', 'lang', 'toxic']],\n    train_tr[['comment_text', 'lang', 'toxic']],\n    train_fr[['comment_text', 'lang', 'toxic']],\n    train_pt[['comment_text', 'lang', 'toxic']],\n    train_ru[['comment_text', 'lang', 'toxic']],\n    train_it[['comment_text', 'lang', 'toxic']]\n    \n])\n\n#del train1, train_es, train_fr, train_pt, train_ru, train_it, train_tr, train2\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.sample(550000)\nprint(train.toxic.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.utils import shuffle\n\ntrain = shuffle(train, random_state=20)\nprint(train.toxic.value_counts())\ntrain.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"valid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#valid['comment_text'] = valid['comment_text'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\nvalid['comment_text'] = valid['comment_text'].progress_apply(lambda x: remove_symbols(x))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"test['content'] = test['content'].progress_apply(lambda x: re.sub(r\"(https?\\S+)|(\\s+)\",' ',str(x)))\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"(www?\\S+)|(\\s+)\",' ',str(x)))\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\\.\\d{1,3}\",' ',str(x)))\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"\\[\\[User.*\",' ',str(x)))\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"\\\\n\", ' ',str(x)))\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"^\\s+\", '',str(x))) #remove spaces from begining\ntest['content'] = test['content'].progress_apply(lambda x: re.sub(r\"\\s+$\", ' ',str(x))) #remove spaces from ending\n#test['content'] = test['content'].progress_apply(lambda x: re.sub(r\"\\s+[a-zA-Z]\\s+\", ' ',str(x))) #remove a single character\ntest['content'] = test['content'].progress_apply(lambda x: remove_symbols(x))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Modelling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n        return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]    # 0 refers to output for the [CLS] token OR [all sentences,token(0 for CLS),hiddne units output]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)\n\n# tokenizer = XLMRobertaTokenizer.from_pretrained(MODEL)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train = regular_encode(train.comment_text.values, tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(valid.comment_text.values, tokenizer, maxlen=MAX_LEN)\nx_test = regular_encode(test.content.values, tokenizer, maxlen=MAX_LEN)\n\ny_train = train.toxic.values\ny_valid = valid.toxic.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL)\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline\n\n# Plot training & validation accuracy values\nplt.plot(train_history.history['accuracy'])\nplt.plot(train_history.history['val_accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plot training & validation accuracy values\nplt.plot(train_history_2.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}