{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About this notebook\n\n*[Jigsaw Multilingual Toxic Comment Classification](https://www.kaggle.com/c/jigsaw-multilingual-toxic-comment-classification)* is the 3rd annual competition organized by the Jigsaw team. It follows *[Toxic Comment Classification Challenge](https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge)*, the original 2018 competition, and *[Jigsaw Unintended Bias in Toxicity Classification](https://www.kaggle.com/c/jigsaw-unintended-bias-in-toxicity-classification)*, which required the competitors to consider biased ML predictions in their new models. This year, the goal is to use english only training data to run toxicity predictions on many different languages, which can be done using multilingual models, and speed up using TPUs.\n\nMany awesome notebooks has already been made so far. Many of them used really cool technologies like [Pytorch XLA](https://www.kaggle.com/theoviel/bert-pytorch-huggingface-starter). This notebook instead aims at constructing a **fast, concise, reusable, and beginner-friendly model scaffold**. \n\n**THIS DOES NOT USE ANY TRANSLATED DATA, BUT IT DOES TRAIN ON THE VALIDATION SET.**\n\n\n### References\n* Original Author: [@xhlulu](https://www.kaggle.com/xhlulu/)\n* Original notebook: [Link](https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras)","metadata":{}},{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-10T02:24:01.848240Z","iopub.execute_input":"2021-07-10T02:24:01.848955Z","iopub.status.idle":"2021-07-10T02:24:10.392831Z","shell.execute_reply.started":"2021-07-10T02:24:01.848823Z","shell.execute_reply":"2021-07-10T02:24:10.391949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper Functions","metadata":{}},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-07-10T02:24:13.142420Z","iopub.execute_input":"2021-07-10T02:24:13.142827Z","iopub.status.idle":"2021-07-10T02:24:13.150109Z","shell.execute_reply.started":"2021-07-10T02:24:13.142790Z","shell.execute_reply":"2021-07-10T02:24:13.148757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef regular_encode(texts, tokenizer, maxlen=512):\n    input_ids = []\n    attention_masks = []\n    enc_di = tokenizer.batch_encode_plus(\n        texts,\n        max_length=maxlen ,\n        add_special_tokens=True, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        return_attention_mask=False\n        \n        )\n    \n    \n    return np.array(enc_di.get('input_ids'))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:24:14.979289Z","iopub.execute_input":"2021-07-10T02:24:14.979851Z","iopub.status.idle":"2021-07-10T02:24:14.985674Z","shell.execute_reply.started":"2021-07-10T02:24:14.979802Z","shell.execute_reply":"2021-07-10T02:24:14.984709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:24:15.298044Z","iopub.execute_input":"2021-07-10T02:24:15.298388Z","iopub.status.idle":"2021-07-10T02:24:15.306338Z","shell.execute_reply.started":"2021-07-10T02:24:15.298353Z","shell.execute_reply":"2021-07-10T02:24:15.305274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TPU Configs","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:24:27.064548Z","iopub.execute_input":"2021-07-10T02:24:27.064932Z","iopub.status.idle":"2021-07-10T02:24:32.473794Z","shell.execute_reply.started":"2021-07-10T02:24:27.064893Z","shell.execute_reply":"2021-07-10T02:24:32.472904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\n#GCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 50\nMODEL = 'jplu/tf-xlm-roberta-large'","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:24:36.502404Z","iopub.execute_input":"2021-07-10T02:24:36.502774Z","iopub.status.idle":"2021-07-10T02:24:36.507823Z","shell.execute_reply.started":"2021-07-10T02:24:36.502745Z","shell.execute_reply":"2021-07-10T02:24:36.506874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create fast tokenizer","metadata":{}},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:24:38.740976Z","iopub.execute_input":"2021-07-10T02:24:38.741347Z","iopub.status.idle":"2021-07-10T02:24:42.506376Z","shell.execute_reply.started":"2021-07-10T02:24:38.741316Z","shell.execute_reply":"2021-07-10T02:24:42.505387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:25:04.132396Z","iopub.execute_input":"2021-07-10T02:25:04.132744Z","iopub.status.idle":"2021-07-10T02:25:04.164920Z","shell.execute_reply.started":"2021-07-10T02:25:04.132715Z","shell.execute_reply":"2021-07-10T02:25:04.163869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load text data into memory","metadata":{}},{"cell_type":"code","source":"#train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\n#train2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\n#train2.toxic = train2.toxic.round().astype(int)\n\n#valid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\n#test = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\n#sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\n\n#train1= pd.read_csv('/kaggle/input/hatespeechmulti/cleaned_data_hatespeech.csv')\ntrain1= pd.read_csv('/kaggle/input/datamulti/data_combined.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:25:23.243679Z","iopub.execute_input":"2021-07-10T02:25:23.244059Z","iopub.status.idle":"2021-07-10T02:25:23.502781Z","shell.execute_reply.started":"2021-07-10T02:25:23.244026Z","shell.execute_reply":"2021-07-10T02:25:23.501523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = train1.sample(frac=0.2,random_state=200 )\nvalid.shape\ntrain1=train1.drop(valid.index)\ntrain1.shape,valid.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:25:25.999031Z","iopub.execute_input":"2021-07-10T02:25:25.999384Z","iopub.status.idle":"2021-07-10T02:25:26.024385Z","shell.execute_reply.started":"2021-07-10T02:25:25.999355Z","shell.execute_reply":"2021-07-10T02:25:26.023277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine train1 with a subset of train2\n#train = pd.concat([\n#    train1[['comment_text', 'toxic']],\n#    train2[['comment_text', 'toxic']].query('toxic==1'),\n#    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=100000, random_state=0)\n#])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\n#x_train = regular_encode(train1.comment_text.tolist(), tokenizer, maxlen=MAX_LEN)\n#x_valid = regular_encode(valid.comment_text.tolist(), tokenizer, maxlen=MAX_LEN)\n\nx_train = regular_encode(train1.tweet.tolist(), tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(valid.tweet.tolist(), tokenizer, maxlen=MAX_LEN)\n#x_test = regular_encode(test.content.tolist(), tokenizer, maxlen=MAX_LEN)\n\ny_train = train1.label.values\ny_valid = valid.label.values","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:26:06.830268Z","iopub.execute_input":"2021-07-10T02:26:06.830640Z","iopub.status.idle":"2021-07-10T02:26:10.908286Z","shell.execute_reply.started":"2021-07-10T02:26:06.830606Z","shell.execute_reply":"2021-07-10T02:26:10.906828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build datasets objects","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\n#test_dataset = (\n#    tf.data.Dataset\n#    .from_tensor_slices(x_test)\n#    .batch(BATCH_SIZE)\n#)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:26:12.955353Z","iopub.execute_input":"2021-07-10T02:26:12.955863Z","iopub.status.idle":"2021-07-10T02:26:13.012412Z","shell.execute_reply.started":"2021-07-10T02:26:12.955830Z","shell.execute_reply":"2021-07-10T02:26:13.011406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load model into the TPU","metadata":{}},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL)\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:26:14.489554Z","iopub.execute_input":"2021-07-10T02:26:14.489914Z","iopub.status.idle":"2021-07-10T02:28:35.387168Z","shell.execute_reply.started":"2021-07-10T02:26:14.489883Z","shell.execute_reply":"2021-07-10T02:28:35.386182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"markdown","source":"First, we train on the subset of the training set, which is completely in English.","metadata":{}},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=5\n)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:28:40.963415Z","iopub.execute_input":"2021-07-10T02:28:40.963976Z","iopub.status.idle":"2021-07-10T02:36:07.971339Z","shell.execute_reply.started":"2021-07-10T02:28:40.963923Z","shell.execute_reply":"2021-07-10T02:36:07.970276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have pretty much saturated the learning potential of the model on english only data, we train it for one more epoch on the `validation` set, which is significantly smaller but contains a mixture of different languages.","metadata":{}},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=5\n)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-10T02:46:54.955783Z","iopub.execute_input":"2021-07-10T02:46:54.956218Z","iopub.status.idle":"2021-07-10T02:49:16.695274Z","shell.execute_reply.started":"2021-07-10T02:46:54.956181Z","shell.execute_reply":"2021-07-10T02:49:16.694110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('/kaggle/working/mymodel1.h5', overwrite=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T03:02:26.854129Z","iopub.execute_input":"2021-07-10T03:02:26.854692Z","iopub.status.idle":"2021-07-10T03:02:36.917845Z","shell.execute_reply.started":"2021-07-10T03:02:26.854639Z","shell.execute_reply":"2021-07-10T03:02:36.916687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/working'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        ","metadata":{"execution":{"iopub.status.busy":"2021-07-10T03:02:58.105645Z","iopub.execute_input":"2021-07-10T03:02:58.105991Z","iopub.status.idle":"2021-07-10T03:02:58.112148Z","shell.execute_reply.started":"2021-07-10T03:02:58.105963Z","shell.execute_reply":"2021-07-10T03:02:58.111090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_roc(probs, y_true):\n    \"\"\"\n    - Print AUC and accuracy on the test set\n    - Plot ROC\n    @params    probs (np.array): an array of predicted probabilities with shape (len(y_true), 2)\n    @params    y_true (np.array): an array of the true values with shape (len(y_true),)\n    \"\"\"\n    preds = probs[:, 1]\n    fpr, tpr, threshold = roc_curve(y_true, preds)\n    roc_auc = auc(fpr, tpr)\n    print(f'AUC: {roc_auc:.4f}')\n       \n    # Get accuracy over the test set\n    y_pred = np.where(preds >= 0.5, 1, 0)\n    accuracy = accuracy_score(y_true, y_pred)\n    print(f'Accuracy: {accuracy*100:.2f}%')\n    \n    # Plot ROC AUC\n    plt.title('Receiver Operating Characteristic')\n    plt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\n    plt.legend(loc = 'lower right')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlim([0, 1])\n    plt.ylim([0, 1])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-14T04:02:04.769136Z","iopub.execute_input":"2021-06-14T04:02:04.769584Z","iopub.status.idle":"2021-06-14T04:02:04.780774Z","shell.execute_reply.started":"2021-06-14T04:02:04.769549Z","shell.execute_reply":"2021-06-14T04:02:04.779151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-14T04:04:04.697084Z","iopub.execute_input":"2021-06-14T04:04:04.697666Z","iopub.status.idle":"2021-06-14T04:04:14.23004Z","shell.execute_reply.started":"2021-06-14T04:04:04.697621Z","shell.execute_reply":"2021-06-14T04:04:14.22926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Evaluate on test data\")\nresults = model.evaluate(x_valid, y_valid, batch_size=128)\nprint(\"test loss, test acc:\", results)","metadata":{"execution":{"iopub.status.busy":"2021-06-14T04:06:04.230802Z","iopub.execute_input":"2021-06-14T04:06:04.231437Z","iopub.status.idle":"2021-06-14T04:06:10.338063Z","shell.execute_reply.started":"2021-06-14T04:06:04.231386Z","shell.execute_reply":"2021-06-14T04:06:10.337062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred= model.predict(x_valid)\n# calculate accuracy\nfrom sklearn import metrics\ny_pred = (y_pred > 0.5)\nprint(metrics.accuracy_score(y_valid, y_pred))","metadata":{"execution":{"iopub.status.busy":"2021-06-14T04:16:13.773663Z","iopub.execute_input":"2021-06-14T04:16:13.774015Z","iopub.status.idle":"2021-06-14T04:16:23.558743Z","shell.execute_reply.started":"2021-06-14T04:16:13.773984Z","shell.execute_reply":"2021-06-14T04:16:23.557673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n%matplotlib inline\n\nfpr, tpr, thresholds = metrics.roc_curve(y_valid, y_pred)\nroc_auc = metrics.auc(fpr, tpr)\n# Plot ROC AUC\nplt.title('Receiver Operating Characteristic')\nplt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\nplt.legend(loc = 'lower right')\nplt.plot([0, 1], [0, 1],'r--')\nplt.xlim([0, 1])\nplt.ylim([0, 1])\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-14T04:20:38.882326Z","iopub.execute_input":"2021-06-14T04:20:38.882689Z","iopub.status.idle":"2021-06-14T04:20:39.092231Z","shell.execute_reply.started":"2021-06-14T04:20:38.88266Z","shell.execute_reply":"2021-06-14T04:20:39.091199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/working'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"sub['toxic'] = model.predict(test_dataset, verbose=1)\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}