{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport re\nimport matplotlib.pyplot as plt\nimport glob","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:40:46.807182Z","iopub.execute_input":"2022-01-28T05:40:46.807901Z","iopub.status.idle":"2022-01-28T05:40:46.812182Z","shell.execute_reply.started":"2022-01-28T05:40:46.807860Z","shell.execute_reply":"2022-01-28T05:40:46.811539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, GridSearchCV, StratifiedKFold\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.pipeline import Pipeline\nfrom gensim import utils\nimport gensim.parsing.preprocessing as gsp\n\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import accuracy_score, f1_score, roc_auc_score","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:21.972663Z","iopub.execute_input":"2022-01-28T04:33:21.972984Z","iopub.status.idle":"2022-01-28T04:33:21.980018Z","shell.execute_reply.started":"2022-01-28T04:33:21.972944Z","shell.execute_reply":"2022-01-28T04:33:21.979244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Logistic Regression","metadata":{}},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:22.889962Z","iopub.execute_input":"2022-01-28T04:33:22.890381Z","iopub.status.idle":"2022-01-28T04:33:38.727551Z","shell.execute_reply.started":"2022-01-28T04:33:22.890333Z","shell.execute_reply":"2022-01-28T04:33:38.726794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2.toxic = train2.toxic.round().astype(int)\ntrain = pd.concat([train1[['comment_text', 'toxic']],\n    train2[['comment_text', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=100000)\n    ])\n#rate=10\n#train = train[::rate]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:38.729214Z","iopub.execute_input":"2022-01-28T04:33:38.729512Z","iopub.status.idle":"2022-01-28T04:33:39.370615Z","shell.execute_reply.started":"2022-01-28T04:33:38.729473Z","shell.execute_reply":"2022-01-28T04:33:39.369955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:39.371766Z","iopub.execute_input":"2022-01-28T04:33:39.372017Z","iopub.status.idle":"2022-01-28T04:33:39.383871Z","shell.execute_reply.started":"2022-01-28T04:33:39.371981Z","shell.execute_reply":"2022-01-28T04:33:39.383142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:39.386348Z","iopub.execute_input":"2022-01-28T04:33:39.386783Z","iopub.status.idle":"2022-01-28T04:33:39.397406Z","shell.execute_reply.started":"2022-01-28T04:33:39.386748Z","shell.execute_reply":"2022-01-28T04:33:39.396559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Validation data set size:\",valid.shape)\nprint(\"Test data set size:\",test.shape)\nprint(\"Training data set size:\",train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:39.398639Z","iopub.execute_input":"2022-01-28T04:33:39.398893Z","iopub.status.idle":"2022-01-28T04:33:39.405212Z","shell.execute_reply.started":"2022-01-28T04:33:39.398860Z","shell.execute_reply":"2022-01-28T04:33:39.404377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_language(text):\n    return Detector(\"\".join(x for x in text if x.isprintable()),quiet=True).languages[0].name","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:39.406733Z","iopub.execute_input":"2022-01-28T04:33:39.407244Z","iopub.status.idle":"2022-01-28T04:33:39.412416Z","shell.execute_reply.started":"2022-01-28T04:33:39.407207Z","shell.execute_reply":"2022-01-28T04:33:39.411550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pyicu","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:39.413543Z","iopub.execute_input":"2022-01-28T04:33:39.414096Z","iopub.status.idle":"2022-01-28T04:33:46.823101Z","shell.execute_reply.started":"2022-01-28T04:33:39.414060Z","shell.execute_reply":"2022-01-28T04:33:46.822204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pycld2","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:46.824658Z","iopub.execute_input":"2022-01-28T04:33:46.824932Z","iopub.status.idle":"2022-01-28T04:33:53.908860Z","shell.execute_reply.started":"2022-01-28T04:33:46.824902Z","shell.execute_reply":"2022-01-28T04:33:53.908003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from polyglot.detect import Detector\nfrom polyglot.utils import pretty_list","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:53.910817Z","iopub.execute_input":"2022-01-28T04:33:53.911168Z","iopub.status.idle":"2022-01-28T04:33:53.916308Z","shell.execute_reply.started":"2022-01-28T04:33:53.911059Z","shell.execute_reply":"2022-01-28T04:33:53.915041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['language'] = train[\"comment_text\"].apply(get_language)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:33:53.920963Z","iopub.execute_input":"2022-01-28T04:33:53.921249Z","iopub.status.idle":"2022-01-28T04:34:34.904974Z","shell.execute_reply.started":"2022-01-28T04:33:53.921191Z","shell.execute_reply":"2022-01-28T04:34:34.904214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:34:34.906506Z","iopub.execute_input":"2022-01-28T04:34:34.906765Z","iopub.status.idle":"2022-01-28T04:34:34.916973Z","shell.execute_reply.started":"2022-01-28T04:34:34.906730Z","shell.execute_reply":"2022-01-28T04:34:34.916277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['toxic'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:34:34.918540Z","iopub.execute_input":"2022-01-28T04:34:34.919040Z","iopub.status.idle":"2022-01-28T04:34:34.931726Z","shell.execute_reply.started":"2022-01-28T04:34:34.919003Z","shell.execute_reply":"2022-01-28T04:34:34.930848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"re_tok = re.compile(f'([{string.punctuation}“”¨«»®´·º½¾¿¡§£₤‘’])')\ndef tokenize(s): \n    return re_tok.sub(r' \\1 ', s).split()\n \nfilters = [\n           gsp.strip_tags, #remove tags \n           gsp.strip_punctuation, #remove punctuation\n           gsp.strip_multiple_whitespaces, #standarized the spaces \n           gsp.strip_numeric,\n           gsp.remove_stopwords, #stop words  \n           gsp.strip_short, \n           gsp.stem_text #stemming \n          ]\n\ndef clean_text(s):\n    s = str(s).lower() \n    s = utils.to_unicode(s)\n    for f in filters:\n        s = f(s)\n    return s","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:34:34.933252Z","iopub.execute_input":"2022-01-28T04:34:34.933669Z","iopub.status.idle":"2022-01-28T04:34:34.940723Z","shell.execute_reply.started":"2022-01-28T04:34:34.933633Z","shell.execute_reply":"2022-01-28T04:34:34.939957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train.language=='xx']","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:34:34.942488Z","iopub.execute_input":"2022-01-28T04:34:34.942999Z","iopub.status.idle":"2022-01-28T04:34:35.053817Z","shell.execute_reply.started":"2022-01-28T04:34:34.942959Z","shell.execute_reply":"2022-01-28T04:34:35.053160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#clean the text first \ntrain['comment_text'].fillna(\"unknown\", inplace=True)\ntrain[\"comment_text\"] = train[\"comment_text\"].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:34:35.055089Z","iopub.execute_input":"2022-01-28T04:34:35.055351Z","iopub.status.idle":"2022-01-28T04:37:18.079785Z","shell.execute_reply.started":"2022-01-28T04:34:35.055301Z","shell.execute_reply":"2022-01-28T04:37:18.078995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#vectorization of the model \nvec = TfidfVectorizer(ngram_range=(1,2), tokenizer=tokenize, strip_accents='unicode', use_idf=1,smooth_idf=1, sublinear_tf=1)\n\npipeline = Pipeline([\n    ('tfidf', vec),\n    ('logreg', LogisticRegression(penalty='elasticnet')),\n])\n\nparameters = {\n    'tfidf__max_features': [None, 1000, 5000, 50000],\n    'tfidf__ngram_range': [(1, 1), (1, 2)],  # unigrams or unigrams + bigrams\n    'logreg__penalty' : ['l1', 'l2'],\n    'logreg__C' : np.logspace(-4, 4, 20),\n    'logreg__solver' : ['liblinear'],\n}","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:37:18.085000Z","iopub.execute_input":"2022-01-28T04:37:18.087429Z","iopub.status.idle":"2022-01-28T04:37:18.099672Z","shell.execute_reply.started":"2022-01-28T04:37:18.087388Z","shell.execute_reply":"2022-01-28T04:37:18.098908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ROC AUC","metadata":{}},{"cell_type":"code","source":"cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nrandom_search = RandomizedSearchCV(pipeline, parameters, n_jobs=-1, verbose=1,scoring=['roc_auc'],cv=cv, n_iter=10, refit='roc_auc')","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:37:18.105120Z","iopub.execute_input":"2022-01-28T04:37:18.108068Z","iopub.status.idle":"2022-01-28T04:37:18.116002Z","shell.execute_reply.started":"2022-01-28T04:37:18.108029Z","shell.execute_reply":"2022-01-28T04:37:18.114974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nrandom_search.fit(train[\"comment_text\"], train['toxic']);","metadata":{"execution":{"iopub.status.busy":"2022-01-28T04:37:18.121336Z","iopub.execute_input":"2022-01-28T04:37:18.123731Z","iopub.status.idle":"2022-01-28T05:02:27.487153Z","shell.execute_reply.started":"2022-01-28T04:37:18.123692Z","shell.execute_reply":"2022-01-28T05:02:27.486410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(random_search.cv_results_).sort_values('mean_test_roc_auc', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:02:27.488722Z","iopub.execute_input":"2022-01-28T05:02:27.489434Z","iopub.status.idle":"2022-01-28T05:02:27.528035Z","shell.execute_reply.started":"2022-01-28T05:02:27.489386Z","shell.execute_reply":"2022-01-28T05:02:27.527344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Validation dataset fitting\n- With AUC model","metadata":{}},{"cell_type":"code","source":"#valid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\nvalid['comment_text'].fillna(\"unknown\", inplace=True)\nvalid[\"comment_text\"] = valid[\"comment_text\"].apply(clean_text)\n\n#score the validation dataset\ny_valid = valid['toxic']\ny_pred_valid = random_search.best_estimator_.predict_proba(valid[\"comment_text\"])\n# print('Testing accuracy %s' % accuracy_score(y_valid, y_pred_valid))\n# print('Testing F1 score: {}'.format(f1_score(y_valid, y_pred_valid, average='weighted')))\nprint('Validation AUC score %s' % roc_auc_score(y_valid, y_pred_valid[:, 1]))","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:19:24.295816Z","iopub.execute_input":"2022-01-28T05:19:24.296082Z","iopub.status.idle":"2022-01-28T05:19:28.002888Z","shell.execute_reply.started":"2022-01-28T05:19:24.296054Z","shell.execute_reply":"2022-01-28T05:19:28.002010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"comment_text\"] = test[\"content\"].apply(clean_text)\n\n#score the submission file \ny_pred = random_search.best_estimator_.predict_proba(test[\"comment_text\"])\n\n#load the sample submission file \nsample_sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\n\nsubmid = pd.DataFrame({'id': sample_sub[\"id\"]})\nsubmission = pd.concat([submid,test.comment_text, pd.DataFrame(y_pred[:, 1],columns=['toxic'])], axis=1)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:25:02.096223Z","iopub.execute_input":"2022-01-28T05:25:02.096783Z","iopub.status.idle":"2022-01-28T05:25:38.292925Z","shell.execute_reply.started":"2022-01-28T05:25:02.096742Z","shell.execute_reply":"2022-01-28T05:25:38.292182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.toxic = submission.toxic.round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:26:55.241355Z","iopub.execute_input":"2022-01-28T05:26:55.241807Z","iopub.status.idle":"2022-01-28T05:26:55.249127Z","shell.execute_reply.started":"2022-01-28T05:26:55.241771Z","shell.execute_reply":"2022-01-28T05:26:55.248262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:27:17.932067Z","iopub.execute_input":"2022-01-28T05:27:17.932358Z","iopub.status.idle":"2022-01-28T05:27:17.939884Z","shell.execute_reply.started":"2022-01-28T05:27:17.932304Z","shell.execute_reply":"2022-01-28T05:27:17.938945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:27:47.974909Z","iopub.execute_input":"2022-01-28T05:27:47.975167Z","iopub.status.idle":"2022-01-28T05:27:47.986983Z","shell.execute_reply.started":"2022-01-28T05:27:47.975137Z","shell.execute_reply":"2022-01-28T05:27:47.986186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DistilBERT","metadata":{}},{"cell_type":"code","source":"!pip install -U transformers","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:40:54.085789Z","iopub.execute_input":"2022-01-28T05:40:54.086765Z","iopub.status.idle":"2022-01-28T05:41:08.978287Z","shell.execute_reply.started":"2022-01-28T05:40:54.086715Z","shell.execute_reply":"2022-01-28T05:41:08.977276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Dropout, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom tqdm.notebook import tqdm\nimport tokenizers\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:08.980663Z","iopub.execute_input":"2022-01-28T05:41:08.980934Z","iopub.status.idle":"2022-01-28T05:41:16.915493Z","shell.execute_reply.started":"2022-01-28T05:41:08.980901Z","shell.execute_reply":"2022-01-28T05:41:16.914553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    text = str(text)\n    text = re.sub(r'[0-9\"]', '', text) # number\n    text = re.sub(r'#[\\S]+\\b', '', text) # hash\n    text = re.sub(r'@[\\S]+\\b', '', text) # mention\n    text = re.sub(r'https?\\S+', '', text) # link\n    text = re.sub(r'\\s+', ' ', text) # multiple white spaces\n#     text = re.sub(r'\\W+', ' ', text) # non-alphanumeric\n    return text.strip()\n\ndef text_process(text):\n    ws = text.split(' ')\n    if(len(ws)>160):\n        text = ' '.join(ws[:160]) + ' ' + ' '.join(ws[-32:])\n    return text\n\ndef fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    print('encoding with', tokenizer)\n    \n    # for transformers 3.5\n    if isinstance(tokenizer, transformers.DistilBertTokenizer) or \\\n        isinstance(tokenizer, transformers.DistilBertTokenizerFast):\n    #     tokenizer.enable_truncation(max_length=maxlen)\n    #     tokenizer.enable_padding(max_length=maxlen)\n        all_ids = []\n\n        for i in tqdm(range(0, len(texts), chunk_size)):\n            text_chunk = texts[i:i+chunk_size].tolist()\n    #         encs = tokenizer.encode_batch(text_chunk)\n            encs = tokenizer(text_chunk, padding='max_length', truncation=True, max_length=maxlen)\n    #         all_ids.extend([enc.ids for enc in encs])\n            all_ids.extend(encs['input_ids']) \n    elif isinstance(fast_tokenizer, tokenizers.implementations.bert_wordpiece.BertWordPieceTokenizer): \n        tokenizer.enable_truncation(max_length=maxlen)\n        tokenizer.enable_padding(max_length=maxlen)\n        all_ids = []\n\n        for i in tqdm(range(0, len(texts), chunk_size)):\n            text_chunk = texts[i:i+chunk_size].tolist()\n            encs = tokenizer.encode_batch(text_chunk)\n            all_ids.extend([enc.ids for enc in encs])\n\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:16.917968Z","iopub.execute_input":"2022-01-28T05:41:16.918201Z","iopub.status.idle":"2022-01-28T05:41:16.932631Z","shell.execute_reply.started":"2022-01-28T05:41:16.918175Z","shell.execute_reply":"2022-01-28T05:41:16.931749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.DistilBertTokenizerFast.from_pretrained('distilbert-base-multilingual-cased')\n\nsave_path = '/kaggle/working/distilbert_base_cased/'\nif not os.path.exists(save_path):\n    os.makedirs(save_path)\ntokenizer.save_pretrained(save_path)\nfast_tokenizer = tokenizer\n\n# \"faster as the tokenizers from transformers because they are implemented in Rust.\"\n# fast_tokenizer = BertWordPieceTokenizer('distilbert_base_cased/vocab.txt', lowercase=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:16.934849Z","iopub.execute_input":"2022-01-28T05:41:16.935076Z","iopub.status.idle":"2022-01-28T05:41:19.958480Z","shell.execute_reply.started":"2022-01-28T05:41:16.935049Z","shell.execute_reply":"2022-01-28T05:41:19.957475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TPU Config","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:19.961151Z","iopub.execute_input":"2022-01-28T05:41:19.961508Z","iopub.status.idle":"2022-01-28T05:41:25.701717Z","shell.execute_reply.started":"2022-01-28T05:41:19.961457Z","shell.execute_reply":"2022-01-28T05:41:25.700593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configuration\nAUTO = tf.data.experimental.AUTOTUNE\nSHUFFLE = 2048\nEPOCHS1 = 20\nEPOCHS2 = 4\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\nVERBOSE = 2","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:25.703905Z","iopub.execute_input":"2022-01-28T05:41:25.704371Z","iopub.status.idle":"2022-01-28T05:41:25.713913Z","shell.execute_reply.started":"2022-01-28T05:41:25.704325Z","shell.execute_reply":"2022-01-28T05:41:25.712735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:41:25.715131Z","iopub.execute_input":"2022-01-28T05:41:25.715496Z","iopub.status.idle":"2022-01-28T05:42:00.144576Z","shell.execute_reply.started":"2022-01-28T05:41:25.715462Z","shell.execute_reply":"2022-01-28T05:42:00.143815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2.toxic = train2.toxic.round().astype(int)\ntrain = pd.concat([train1[['comment_text', 'toxic']],\n    train2[['comment_text', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=100000)\n    ])\n#rate=10\n#train = train[::rate]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:42:00.145712Z","iopub.execute_input":"2022-01-28T05:42:00.146373Z","iopub.status.idle":"2022-01-28T05:42:00.935523Z","shell.execute_reply.started":"2022-01-28T05:42:00.146335Z","shell.execute_reply":"2022-01-28T05:42:00.934563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def len_sent(data):\n    return len(data.split())\ntrain[\"num_words_comment_text\"] = train[\"comment_text\"].apply(lambda x : len_sent(x))\n#sns.kdeplot(train[train[\"toxic\"] == 0][\"num_words_comment_text\"].values, shade = True, color = \"red\", label='non_toxity')\n#sns.kdeplot(train[train[\"toxic\"] == 1][\"num_words_comment_text\"].values, shade = True, color = \"blue\", label='toxity')\n\ny_train = train['toxic'].values\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:42:00.936850Z","iopub.execute_input":"2022-01-28T05:42:00.937105Z","iopub.status.idle":"2022-01-28T05:42:03.087088Z","shell.execute_reply.started":"2022-01-28T05:42:00.937075Z","shell.execute_reply":"2022-01-28T05:42:03.086052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:42:03.090568Z","iopub.execute_input":"2022-01-28T05:42:03.090864Z","iopub.status.idle":"2022-01-28T05:42:03.129083Z","shell.execute_reply.started":"2022-01-28T05:42:03.090828Z","shell.execute_reply":"2022-01-28T05:42:03.126988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train['toxic']; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:42:03.132478Z","iopub.execute_input":"2022-01-28T05:42:03.134088Z","iopub.status.idle":"2022-01-28T05:42:03.736803Z","shell.execute_reply.started":"2022-01-28T05:42:03.134005Z","shell.execute_reply":"2022-01-28T05:42:03.733587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['comment_text'] = train['comment_text'].apply(lambda x: clean_text(x))\ntrain['comment_text'] = train['comment_text'].apply(lambda x: text_process(x))\nx_train = fast_encode(train['comment_text'].astype(str), fast_tokenizer, maxlen=MAX_LEN)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:42:03.741048Z","iopub.execute_input":"2022-01-28T05:42:03.742193Z","iopub.status.idle":"2022-01-28T05:43:59.574823Z","shell.execute_reply.started":"2022-01-28T05:42:03.741905Z","shell.execute_reply":"2022-01-28T05:43:59.573648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(SHUFFLE)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\ndel x_train; gc.collect()\n\n\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\nvalid['comment_text'] = valid.apply(lambda x: clean_text(x['comment_text']), axis=1)\nvalid['comment_text'] = valid['comment_text'].apply(lambda x: text_process(x))\nx_valid = fast_encode(valid['comment_text'].astype(str), fast_tokenizer, maxlen=MAX_LEN)\ny_valid = valid['toxic'].values\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ndel x_valid; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:43:59.582628Z","iopub.execute_input":"2022-01-28T05:43:59.583023Z","iopub.status.idle":"2022-01-28T05:44:06.881395Z","shell.execute_reply.started":"2022-01-28T05:43:59.582986Z","shell.execute_reply":"2022-01-28T05:44:06.880409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Callbacks","metadata":{}},{"cell_type":"code","source":"lrs = ReduceLROnPlateau(monitor='val_auc', mode ='max', factor = 0.7, min_lr= 1e-7, verbose = 1, patience = 2)\nes1 = EarlyStopping(monitor='val_auc', mode='max', verbose = 1, patience = 5, restore_best_weights=True)\nes2 = EarlyStopping(monitor='auc', mode='max', verbose = 1, patience = 1, restore_best_weights=True)\ncallbacks_list1 = [lrs,es1]\ncallbacks_list2 = [lrs,es2]","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:44:06.882828Z","iopub.execute_input":"2022-01-28T05:44:06.883158Z","iopub.status.idle":"2022-01-28T05:44:06.890655Z","shell.execute_reply.started":"2022-01-28T05:44:06.883116Z","shell.execute_reply":"2022-01-28T05:44:06.889940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build Model","metadata":{}},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    x = tf.keras.layers.Dropout(0.4)(cls_token)\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=[tf.keras.metrics.AUC(name='auc'), 'accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:44:06.891791Z","iopub.execute_input":"2022-01-28T05:44:06.892062Z","iopub.status.idle":"2022-01-28T05:44:06.909772Z","shell.execute_reply.started":"2022-01-28T05:44:06.892029Z","shell.execute_reply":"2022-01-28T05:44:06.908904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load model in TPU","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:44:06.911064Z","iopub.execute_input":"2022-01-28T05:44:06.911756Z","iopub.status.idle":"2022-01-28T05:44:48.749844Z","shell.execute_reply.started":"2022-01-28T05:44:06.911714Z","shell.execute_reply":"2022-01-28T05:44:48.748902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Run Model","metadata":{}},{"cell_type":"code","source":"# not train on order to save memory\nn_steps = len(y_train) // (BATCH_SIZE*8)\n\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS1,\n    callbacks=callbacks_list1,\n    verbose=VERBOSE\n)\n\ndel train_dataset; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:44:48.752549Z","iopub.execute_input":"2022-01-28T05:44:48.753207Z","iopub.status.idle":"2022-01-28T05:49:43.585580Z","shell.execute_reply.started":"2022-01-28T05:44:48.753155Z","shell.execute_reply":"2022-01-28T05:49:43.584332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history_df = pd.DataFrame.from_dict(train_history.history)\ntrain_history_df","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:49:43.587765Z","iopub.execute_input":"2022-01-28T05:49:43.588041Z","iopub.status.idle":"2022-01-28T05:49:43.655580Z","shell.execute_reply.started":"2022-01-28T05:49:43.588004Z","shell.execute_reply":"2022-01-28T05:49:43.654532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(10, 5))\nplt.plot(train_history_df['val_auc'], label='val_auc')\nplt.plot(train_history_df['auc'], label='auc')\nplt.legend(fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:49:43.657078Z","iopub.execute_input":"2022-01-28T05:49:43.657442Z","iopub.status.idle":"2022-01-28T05:49:43.991688Z","shell.execute_reply.started":"2022-01-28T05:49:43.657410Z","shell.execute_reply":"2022-01-28T05:49:43.990525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(train_history_df['accuracy'], label='accuracy')\nplt.plot(train_history_df['val_accuracy'], label='val_accuracy')\nplt.legend(fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:49:43.993893Z","iopub.execute_input":"2022-01-28T05:49:43.994311Z","iopub.status.idle":"2022-01-28T05:49:44.273041Z","shell.execute_reply.started":"2022-01-28T05:49:43.994205Z","shell.execute_reply":"2022-01-28T05:49:44.272352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = len(y_valid) // (BATCH_SIZE)\n\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS2,\n    callbacks=callbacks_list2,\n    verbose=VERBOSE\n)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:49:44.274127Z","iopub.execute_input":"2022-01-28T05:49:44.274413Z","iopub.status.idle":"2022-01-28T05:50:33.381354Z","shell.execute_reply.started":"2022-01-28T05:49:44.274377Z","shell.execute_reply":"2022-01-28T05:50:33.380376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history2_df = pd.DataFrame.from_dict(train_history_2.history)\ntrain_history2_df","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:50:33.383256Z","iopub.execute_input":"2022-01-28T05:50:33.383665Z","iopub.status.idle":"2022-01-28T05:50:33.398312Z","shell.execute_reply.started":"2022-01-28T05:50:33.383610Z","shell.execute_reply":"2022-01-28T05:50:33.397503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(train_history2_df['loss'], label='loss')\nplt.plot(train_history2_df['auc'], label='auc')\nplt.plot(train_history2_df['accuracy'], label='accuracy')\nplt.legend(fontsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:50:33.399628Z","iopub.execute_input":"2022-01-28T05:50:33.400543Z","iopub.status.idle":"2022-01-28T05:50:33.699360Z","shell.execute_reply.started":"2022-01-28T05:50:33.400503Z","shell.execute_reply":"2022-01-28T05:50:33.698593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = fast_encode(test['content'].astype(str), fast_tokenizer, maxlen=MAX_LEN)\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:50:33.700873Z","iopub.execute_input":"2022-01-28T05:50:33.701428Z","iopub.status.idle":"2022-01-28T05:50:48.760497Z","shell.execute_reply.started":"2022-01-28T05:50:33.701380Z","shell.execute_reply":"2022-01-28T05:50:48.759655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\nsub['toxic'] = model.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:50:48.761684Z","iopub.execute_input":"2022-01-28T05:50:48.763604Z","iopub.status.idle":"2022-01-28T05:51:08.582995Z","shell.execute_reply.started":"2022-01-28T05:50:48.763558Z","shell.execute_reply":"2022-01-28T05:51:08.581994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"comment_text\"] = test[\"content\"].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:57:55.337892Z","iopub.execute_input":"2022-01-28T05:57:55.338323Z","iopub.status.idle":"2022-01-28T05:57:59.253779Z","shell.execute_reply.started":"2022-01-28T05:57:55.338282Z","shell.execute_reply":"2022-01-28T05:57:59.252738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"comment_text\"] = test.comment_text\nsub.toxic = sub.toxic.round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:59:42.520855Z","iopub.execute_input":"2022-01-28T05:59:42.521703Z","iopub.status.idle":"2022-01-28T05:59:42.533146Z","shell.execute_reply.started":"2022-01-28T05:59:42.521660Z","shell.execute_reply":"2022-01-28T05:59:42.532293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-01-28T05:59:42.999144Z","iopub.execute_input":"2022-01-28T05:59:42.999791Z","iopub.status.idle":"2022-01-28T05:59:43.011815Z","shell.execute_reply.started":"2022-01-28T05:59:42.999745Z","shell.execute_reply":"2022-01-28T05:59:43.010893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-28T06:01:32.688385Z","iopub.execute_input":"2022-01-28T06:01:32.688758Z","iopub.status.idle":"2022-01-28T06:01:32.701015Z","shell.execute_reply.started":"2022-01-28T06:01:32.688720Z","shell.execute_reply":"2022-01-28T06:01:32.700231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}