{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About this notebook\n\n*[Jigsaw Multilingual Toxic Comment Classification](https://www.kaggle.com/c/jigsaw-multilingual-toxic-comment-classification)* is the 3rd annual competition organized by the Jigsaw team. It follows *[Toxic Comment Classification Challenge](https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge)*, the original 2018 competition, and *[Jigsaw Unintended Bias in Toxicity Classification](https://www.kaggle.com/c/jigsaw-unintended-bias-in-toxicity-classification)*, which required the competitors to consider biased ML predictions in their new models. This year, the goal is to use english only training data to run toxicity predictions on many different languages, which can be done using multilingual models, and speed up using TPUs.\n\nMany awesome notebooks has already been made so far. Many of them used really cool technologies like [Pytorch XLA](https://www.kaggle.com/theoviel/bert-pytorch-huggingface-starter). This notebook instead aims at constructing a **fast, concise, reusable, and beginner-friendly model scaffold**. \n\n**THIS DOES NOT USE ANY TRANSLATED DATA, BUT IT DOES TRAIN ON THE VALIDATION SET.**\n\n\n### References\n* Original Author: [@xhlulu](https://www.kaggle.com/xhlulu/)\n* Original notebook: [Link](https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras)","metadata":{}},{"cell_type":"code","source":"import os\n\n\nimport numpy as np\nimport pandas as pd","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-04-27T20:10:56.416703Z","iopub.execute_input":"2023-04-27T20:10:56.417372Z","iopub.status.idle":"2023-04-27T20:10:56.442896Z","shell.execute_reply.started":"2023-04-27T20:10:56.417275Z","shell.execute_reply":"2023-04-27T20:10:56.442051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-04-27T20:11:08.040431Z","iopub.execute_input":"2023-04-27T20:11:08.040813Z","iopub.status.idle":"2023-04-27T20:11:13.588244Z","shell.execute_reply.started":"2023-04-27T20:11:08.040774Z","shell.execute_reply":"2023-04-27T20:11:13.587246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper Functions","metadata":{}},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-04-27T20:11:14.697279Z","iopub.execute_input":"2023-04-27T20:11:14.698166Z","iopub.status.idle":"2023-04-27T20:11:14.706806Z","shell.execute_reply.started":"2023-04-27T20:11:14.698127Z","shell.execute_reply":"2023-04-27T20:11:14.704547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n#         return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:17.002921Z","iopub.execute_input":"2023-04-27T20:11:17.003352Z","iopub.status.idle":"2023-04-27T20:11:17.0128Z","shell.execute_reply.started":"2023-04-27T20:11:17.003313Z","shell.execute_reply":"2023-04-27T20:11:17.011793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import backend as K\n\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\n\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision\n\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:18.5254Z","iopub.execute_input":"2023-04-27T20:11:18.525889Z","iopub.status.idle":"2023-04-27T20:11:18.53664Z","shell.execute_reply.started":"2023-04-27T20:11:18.525843Z","shell.execute_reply":"2023-04-27T20:11:18.535646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    x1 = Dense(256, activation = 'sigmoid')(cls_token)\n    out = Dense(2, activation='softmax')(x1)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy', recall_m, precision_m, f1_m])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:18.692647Z","iopub.execute_input":"2023-04-27T20:11:18.695534Z","iopub.status.idle":"2023-04-27T20:11:18.7054Z","shell.execute_reply.started":"2023-04-27T20:11:18.695486Z","shell.execute_reply":"2023-04-27T20:11:18.704455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TPU Configs","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:22.391356Z","iopub.execute_input":"2023-04-27T20:11:22.391717Z","iopub.status.idle":"2023-04-27T20:11:22.404952Z","shell.execute_reply.started":"2023-04-27T20:11:22.391685Z","shell.execute_reply":"2023-04-27T20:11:22.403751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\n#GCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 1\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 275\nMODEL = 'jplu/tf-xlm-roberta-base'","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:24.46123Z","iopub.execute_input":"2023-04-27T20:11:24.461586Z","iopub.status.idle":"2023-04-27T20:11:24.466692Z","shell.execute_reply.started":"2023-04-27T20:11:24.461555Z","shell.execute_reply":"2023-04-27T20:11:24.465574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create fast tokenizer","metadata":{}},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:25.006229Z","iopub.execute_input":"2023-04-27T20:11:25.00693Z","iopub.status.idle":"2023-04-27T20:11:32.009456Z","shell.execute_reply.started":"2023-04-27T20:11:25.006893Z","shell.execute_reply":"2023-04-27T20:11:32.00832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load text data into memory","metadata":{}},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\ntrain2.toxic = train2.toxic.round().astype(int)\n\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\n# test = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\n# sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\n# submission = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:32.011517Z","iopub.execute_input":"2023-04-27T20:11:32.012183Z","iopub.status.idle":"2023-04-27T20:11:56.451534Z","shell.execute_reply.started":"2023-04-27T20:11:32.012143Z","shell.execute_reply":"2023-04-27T20:11:56.450562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine train1 with a subset of train2\ntrain = pd.concat([\n    train1[['comment_text', 'toxic']],\n    train2[['comment_text', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=100000, random_state=0)\n])","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:56.45371Z","iopub.execute_input":"2023-04-27T20:11:56.45447Z","iopub.status.idle":"2023-04-27T20:11:56.942041Z","shell.execute_reply.started":"2023-04-27T20:11:56.454431Z","shell.execute_reply":"2023-04-27T20:11:56.941003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.groupby('toxic').apply(\n    lambda x: x.sample(frac=0.5)\n)\ntrain = train.droplevel(0)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:56.944245Z","iopub.execute_input":"2023-04-27T20:11:56.944824Z","iopub.status.idle":"2023-04-27T20:11:57.115163Z","shell.execute_reply.started":"2023-04-27T20:11:56.944785Z","shell.execute_reply":"2023-04-27T20:11:57.114145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train.index)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:11:57.116446Z","iopub.execute_input":"2023-04-27T20:11:57.116992Z","iopub.status.idle":"2023-04-27T20:11:57.125961Z","shell.execute_reply.started":"2023-04-27T20:11:57.116953Z","shell.execute_reply":"2023-04-27T20:11:57.124684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nontoxic = train[train['toxic']==0]","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:16:27.564052Z","iopub.execute_input":"2023-04-27T20:16:27.56442Z","iopub.status.idle":"2023-04-27T20:16:27.582877Z","shell.execute_reply.started":"2023-04-27T20:16:27.56439Z","shell.execute_reply":"2023-04-27T20:16:27.58197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nontoxic","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:16:35.27034Z","iopub.execute_input":"2023-04-27T20:16:35.270688Z","iopub.status.idle":"2023-04-27T20:16:35.284406Z","shell.execute_reply.started":"2023-04-27T20:16:35.270657Z","shell.execute_reply":"2023-04-27T20:16:35.283393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic = train[train[\"toxic\"]==1]","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:18:10.726371Z","iopub.execute_input":"2023-04-27T20:18:10.726763Z","iopub.status.idle":"2023-04-27T20:18:10.739596Z","shell.execute_reply.started":"2023-04-27T20:18:10.726709Z","shell.execute_reply":"2023-04-27T20:18:10.738719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:18:11.554086Z","iopub.execute_input":"2023-04-27T20:18:11.554436Z","iopub.status.idle":"2023-04-27T20:18:11.568446Z","shell.execute_reply.started":"2023-04-27T20:18:11.554404Z","shell.execute_reply":"2023-04-27T20:18:11.567217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_nontoxic = nontoxic.sample(toxic.shape[0])\nnew_nontoxic.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:19:20.723057Z","iopub.execute_input":"2023-04-27T20:19:20.723522Z","iopub.status.idle":"2023-04-27T20:19:20.755463Z","shell.execute_reply.started":"2023-04-27T20:19:20.723481Z","shell.execute_reply":"2023-04-27T20:19:20.754445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import shuffle\ntrain = shuffle(pd.concat([toxic, new_nontoxic]))","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:21:14.926018Z","iopub.execute_input":"2023-04-27T20:21:14.926472Z","iopub.status.idle":"2023-04-27T20:21:15.173764Z","shell.execute_reply.started":"2023-04-27T20:21:14.92643Z","shell.execute_reply":"2023-04-27T20:21:15.172786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(valid.index)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:21:45.774864Z","iopub.execute_input":"2023-04-27T20:21:45.775233Z","iopub.status.idle":"2023-04-27T20:21:45.78159Z","shell.execute_reply.started":"2023-04-27T20:21:45.775205Z","shell.execute_reply":"2023-04-27T20:21:45.780669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train1\ndel train2","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:21:46.19715Z","iopub.execute_input":"2023-04-27T20:21:46.197457Z","iopub.status.idle":"2023-04-27T20:21:46.202441Z","shell.execute_reply.started":"2023-04-27T20:21:46.197429Z","shell.execute_reply":"2023-04-27T20:21:46.201414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n\n# x_train = regular_encode(train.comment_text.values.tolist()[0:1000], tokenizer, maxlen=MAX_LEN)\nx_train = regular_encode(train.comment_text.values.tolist(), tokenizer, maxlen=MAX_LEN)\n# x_valid = regular_encode(valid.comment_text.values.tolist()[0:100], tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(valid.comment_text.values.tolist(), tokenizer, maxlen=MAX_LEN)\n\n# x_test = regular_encode(test.content.values.tolist(), tokenizer, maxlen=MAX_LEN)\n\ny_train = train.toxic.values.tolist()\ny_valid = valid.toxic.values.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:21:52.454524Z","iopub.execute_input":"2023-04-27T20:21:52.454901Z","iopub.status.idle":"2023-04-27T20:22:28.808583Z","shell.execute_reply.started":"2023-04-27T20:21:52.454868Z","shell.execute_reply":"2023-04-27T20:22:28.807599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = tf.keras.utils.to_categorical(y_train, num_classes = 2)\ny_valid = tf.keras.utils.to_categorical(y_valid, num_classes = 2)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:28.811364Z","iopub.execute_input":"2023-04-27T20:22:28.811759Z","iopub.status.idle":"2023-04-27T20:22:28.825249Z","shell.execute_reply.started":"2023-04-27T20:22:28.811707Z","shell.execute_reply":"2023-04-27T20:22:28.824269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ndel valid","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:28.826917Z","iopub.execute_input":"2023-04-27T20:22:28.827325Z","iopub.status.idle":"2023-04-27T20:22:28.834364Z","shell.execute_reply.started":"2023-04-27T20:22:28.82729Z","shell.execute_reply":"2023-04-27T20:22:28.833449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"c = 0\nfor i in y_train:\n    if (i==0):\n        c+=1\nprint(c)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T16:54:25.20351Z","iopub.execute_input":"2023-04-27T16:54:25.203943Z","iopub.status.idle":"2023-04-27T16:54:25.511738Z","shell.execute_reply.started":"2023-04-27T16:54:25.203911Z","shell.execute_reply":"2023-04-27T16:54:25.509205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build datasets objects","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\n# test_dataset = (\n#     tf.data.Dataset\n#     .from_tensor_slices(x_test)\n#     .batch(BATCH_SIZE)\n# )","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:28.839555Z","iopub.execute_input":"2023-04-27T20:22:28.839924Z","iopub.status.idle":"2023-04-27T20:22:31.45192Z","shell.execute_reply.started":"2023-04-27T20:22:28.839881Z","shell.execute_reply":"2023-04-27T20:22:31.450961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\nn_steps_valid = x_valid.shape[0] // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:31.453828Z","iopub.execute_input":"2023-04-27T20:22:31.454184Z","iopub.status.idle":"2023-04-27T20:22:31.459649Z","shell.execute_reply.started":"2023-04-27T20:22:31.454149Z","shell.execute_reply":"2023-04-27T20:22:31.458635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load model into the TPU","metadata":{}},{"cell_type":"code","source":"del x_train\ndel y_train","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:31.830375Z","iopub.execute_input":"2023-04-27T20:22:31.83134Z","iopub.status.idle":"2023-04-27T20:22:31.837213Z","shell.execute_reply.started":"2023-04-27T20:22:31.831296Z","shell.execute_reply":"2023-04-27T20:22:31.835806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id2label = {0: \"NONTOXIC\", 1: \"TOXIC\"}\nlabel2id = {\"NONTOXIC\": 0, \"TOXIC\": 1}","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:22:46.350613Z","iopub.execute_input":"2023-04-27T20:22:46.351221Z","iopub.status.idle":"2023-04-27T20:22:46.356234Z","shell.execute_reply.started":"2023-04-27T20:22:46.351185Z","shell.execute_reply":"2023-04-27T20:22:46.355278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL, num_labels=2, id2label=id2label, label2id=label2id)\n    \n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:23:03.531293Z","iopub.execute_input":"2023-04-27T20:23:03.531645Z","iopub.status.idle":"2023-04-27T20:24:19.647234Z","shell.execute_reply.started":"2023-04-27T20:23:03.531614Z","shell.execute_reply":"2023-04-27T20:24:19.646227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"markdown","source":"First, we train on the subset of the training set, which is completely in English.","metadata":{}},{"cell_type":"code","source":"EPOCHS = 3\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T20:24:19.64921Z","iopub.execute_input":"2023-04-27T20:24:19.650035Z","iopub.status.idle":"2023-04-28T00:34:27.240527Z","shell.execute_reply.started":"2023-04-27T20:24:19.649996Z","shell.execute_reply":"2023-04-28T00:34:27.239583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have pretty much saturated the learning potential of the model on english only data, we train it for one more epoch on the `validation` set, which is significantly smaller but contains a mixture of different languages.","metadata":{}},{"cell_type":"code","source":"train_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps_valid,\n    epochs=3\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-28T00:34:27.243754Z","iopub.execute_input":"2023-04-28T00:34:27.244096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import classification_report, confusion_matrix\ndef plot_confusion_matrix(cm, classes, normalize=True, title='Confusion matrix', cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting normalize=True.\n    \"\"\"\n    plt.figure(figsize=(10,10))\n\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n        cm = np.around(cm, decimals=2)\n        cm[np.isnan(cm)] = 0.0\n        print(\"Normalized confusion matrix\")\n    else:\n        print('Confusion matrix, without normalization')\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###Overall Model\ntarget_names = ['Toxic', 'Non-Toxic']\n\nY_pred = model.predict(x_valid)\ny_preds = np.argmax(Y_pred, axis=1)\nprint('Confusion Matrix')\nrounded_labels=np.argmax(y_valid, axis=1)\ncm = confusion_matrix(y_true = rounded_labels, y_pred = y_preds)\nprint(cm)\nplot_confusion_matrix(cm, target_names, title='Confusion Matrix')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\ndef plot_confusion_matrix2(y_true, y_pred, classes):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, cmap=plt.cm.Blues, xticklabels=classes, yticklabels=classes, fmt='g')\n    plt.xlabel('Predicted labels')\n    plt.ylabel('True labels')\n    plt.title('Confusion Matrix')\n    plt.show()\nplot_confusion_matrix2(y_true = rounded_labels, y_pred = y_preds, classes=target_names)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\nprecision = dict()\nrecall = dict()\nfor i in range(2):\n    precision[i], recall[i], _ = precision_recall_curve(y_valid[:,i],\n                                                        Y_pred[:, i])\n    if (i==0):\n      plt.plot(recall[i], precision[i], lw=2, label='Toxic')\n    elif (i==1):\n      plt.plot(recall[i], precision[i], lw=2, label='NonToxic')\n    \nplt.xlabel(\"recall\")\nplt.ylabel(\"precision\")\nplt.legend(loc=\"best\")\nplt.title(\"precision vs. recall curve\")\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"axes = plt.gca()\nacc = train_history_2.history['accuracy']\nloss = train_history_2.history['loss']\nepochs = range(1, len(acc) + 1)\n#Train and validation accuracy\nplt.plot(epochs, acc, 'b', label='Training accuracy')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title('Training accuracy and Validation accuracy')\n\nplt.legend()\n\nplt.figure()\n#Train and validation loss\nplt.plot(epochs, loss, 'b', label='Training loss')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title('Training loss and Validation loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"axes = plt.gca()\nacc = train_history.history['accuracy']\nval_acc = train_history.history['val_accuracy']\nloss = train_history.history['loss']\nval_loss = train_history.history['val_loss']\nepochs = range(1, len(acc) + 1)\n#Train and validation accuracy\nplt.plot(epochs, acc, 'b', label='Training accuracy')\nplt.plot(epochs, val_acc, 'r', label='Validation accuracy')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title('Training accuracy and Validation accuracy')\n\nplt.legend()\n\nplt.figure()\n#Train and validation loss\nplt.plot(epochs, loss, 'b', label='Training loss')\nplt.plot(epochs, val_loss, 'r', label='Validation loss')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title('Training loss and Validation loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(valid_dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nimport time\nimport numpy as np\nstart_time = time.time()\ntest_predictions = model.predict(x_valid)\n# Comparing the predictions to actual forest cover types for the test rows\n# test is the data right after splitting into train, test and val (shuffle was false in dataset so the order will match)\nrounded_labels=np.argmax(y_valid, axis=1)\ntest_predictions = np.argmax(test_predictions, axis=1)\nprint(classification_report(rounded_labels,test_predictions))\nprint(\"Time taken to predict the model \" + str(time.time() - start_time))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dataset\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:06:53.307026Z","iopub.execute_input":"2023-04-27T11:06:53.307405Z","iopub.status.idle":"2023-04-27T11:06:53.314081Z","shell.execute_reply.started":"2023-04-27T11:06:53.307368Z","shell.execute_reply":"2023-04-27T11:06:53.312145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del valid_dataset","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:06:53.316839Z","iopub.execute_input":"2023-04-27T11:06:53.317554Z","iopub.status.idle":"2023-04-27T11:06:53.330582Z","shell.execute_reply.started":"2023-04-27T11:06:53.317517Z","shell.execute_reply":"2023-04-27T11:06:53.32937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\nsubmission = pd.read_csv","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:06:53.334538Z","iopub.execute_input":"2023-04-27T11:06:53.335301Z","iopub.status.idle":"2023-04-27T11:06:54.357151Z","shell.execute_reply.started":"2023-04-27T11:06:53.335264Z","shell.execute_reply":"2023-04-27T11:06:54.356117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test.index)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:06:54.358435Z","iopub.execute_input":"2023-04-27T11:06:54.358775Z","iopub.status.idle":"2023-04-27T11:06:54.36698Z","shell.execute_reply.started":"2023-04-27T11:06:54.358739Z","shell.execute_reply":"2023-04-27T11:06:54.366004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x_test = regular_encode(test.content.values.tolist()[0:63812], tokenizer, maxlen=MAX_LEN)\nx_test = regular_encode(test.content.values.tolist(), tokenizer, maxlen=MAX_LEN)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:06:54.368293Z","iopub.execute_input":"2023-04-27T11:06:54.368748Z","iopub.status.idle":"2023-04-27T11:07:17.948273Z","shell.execute_reply.started":"2023-04-27T11:06:54.368711Z","shell.execute_reply":"2023-04-27T11:07:17.947245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub_df = sub.iloc[0:63812]\nsub_df = sub","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:07:17.949572Z","iopub.execute_input":"2023-04-27T11:07:17.949933Z","iopub.status.idle":"2023-04-27T11:07:17.955811Z","shell.execute_reply.started":"2023-04-27T11:07:17.94988Z","shell.execute_reply":"2023-04-27T11:07:17.954935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"sub_df['toxic'] = model.predict(test_dataset, verbose=1)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:07:17.95724Z","iopub.execute_input":"2023-04-27T11:07:17.958348Z","iopub.status.idle":"2023-04-27T11:20:02.186767Z","shell.execute_reply.started":"2023-04-27T11:07:17.958312Z","shell.execute_reply":"2023-04-27T11:20:02.185824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=sub_df[['id','toxic']]","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:20:02.189842Z","iopub.execute_input":"2023-04-27T11:20:02.190141Z","iopub.status.idle":"2023-04-27T11:20:02.201377Z","shell.execute_reply.started":"2023-04-27T11:20:02.190115Z","shell.execute_reply":"2023-04-27T11:20:02.200392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission-Dense2Konjam.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:20:02.205189Z","iopub.execute_input":"2023-04-27T11:20:02.206207Z","iopub.status.idle":"2023-04-27T11:20:02.311273Z","shell.execute_reply.started":"2023-04-27T11:20:02.20617Z","shell.execute_reply":"2023-04-27T11:20:02.310327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[submission.toxic <= 0.5]","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:23:34.69721Z","iopub.execute_input":"2023-04-27T11:23:34.697595Z","iopub.status.idle":"2023-04-27T11:23:34.716945Z","shell.execute_reply.started":"2023-04-27T11:23:34.697563Z","shell.execute_reply":"2023-04-27T11:23:34.715754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-04-27T11:24:17.150284Z","iopub.execute_input":"2023-04-27T11:24:17.151118Z","iopub.status.idle":"2023-04-27T11:24:17.200966Z","shell.execute_reply.started":"2023-04-27T11:24:17.15106Z","shell.execute_reply":"2023-04-27T11:24:17.196553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}