{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"* Introduction and Import Packages\n\n* Load and Explore the Data\n\n* Data Preparation — Tokenize and Pad Text Data\n\n* Prepare Embedding Matrix with Pre-trained GloVe Embeddings\n\n* Create the Embedding Layer\n\n* Build the Model\n\n* Train the Model\n\n* Model Evaluation - Classify Toxic Comments","metadata":{}},{"cell_type":"markdown","source":"### Imports","metadata":{}},{"cell_type":"code","source":"try:\n  # %tensorflow_version only exists in Colab.\n  %tensorflow_version 2.x\nexcept Exception:\n    pass\n  \nimport tensorflow as tf\nimport tensorflow_datasets as tfds\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-09T08:34:50.388184Z","iopub.execute_input":"2023-03-09T08:34:50.388694Z","iopub.status.idle":"2023-03-09T08:34:50.397650Z","shell.execute_reply.started":"2023-03-09T08:34:50.388649Z","shell.execute_reply":"2023-03-09T08:34:50.395872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setting TPU","metadata":{}},{"cell_type":"code","source":"import os\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:34:52.004207Z","iopub.execute_input":"2023-03-09T08:34:52.004672Z","iopub.status.idle":"2023-03-09T08:35:00.307515Z","shell.execute_reply.started":"2023-03-09T08:34:52.004611Z","shell.execute_reply":"2023-03-09T08:35:00.305842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:00.309930Z","iopub.execute_input":"2023-03-09T08:35:00.310317Z","iopub.status.idle":"2023-03-09T08:35:00.317136Z","shell.execute_reply.started":"2023-03-09T08:35:00.310278Z","shell.execute_reply":"2023-03-09T08:35:00.315530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot Utility\ndef plot_graphs(history, string):\n    plt.plot(history.history[string])\n    plt.plot(history.history['val_'+string])\n    plt.xlabel(\"Epochs\")\n    plt.ylabel(string)\n    plt.legend([string, 'val_'+string])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:00.318921Z","iopub.execute_input":"2023-03-09T08:35:00.319479Z","iopub.status.idle":"2023-03-09T08:35:00.332145Z","shell.execute_reply.started":"2023-03-09T08:35:00.319418Z","shell.execute_reply":"2023-03-09T08:35:00.330532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Retreiving the english data\n\nWe are just using english comments","metadata":{}},{"cell_type":"code","source":"#Train data and labels\ndata = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:00.335624Z","iopub.execute_input":"2023-03-09T08:35:00.336167Z","iopub.status.idle":"2023-03-09T08:35:03.290127Z","shell.execute_reply.started":"2023-03-09T08:35:00.336106Z","shell.execute_reply":"2023-03-09T08:35:03.288464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.292110Z","iopub.execute_input":"2023-03-09T08:35:03.292571Z","iopub.status.idle":"2023-03-09T08:35:03.308376Z","shell.execute_reply.started":"2023-03-09T08:35:03.292528Z","shell.execute_reply":"2023-03-09T08:35:03.307043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data[[\"comment_text\"]]\ny = data[[\"toxic\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.310367Z","iopub.execute_input":"2023-03-09T08:35:03.311348Z","iopub.status.idle":"2023-03-09T08:35:03.360210Z","shell.execute_reply.started":"2023-03-09T08:35:03.311295Z","shell.execute_reply":"2023-03-09T08:35:03.358828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.under_sampling import RandomUnderSampler\nrus = RandomUnderSampler(random_state=42)\nX_res, y_res = rus.fit_resample(X,y)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.361969Z","iopub.execute_input":"2023-03-09T08:35:03.362407Z","iopub.status.idle":"2023-03-09T08:35:03.447492Z","shell.execute_reply.started":"2023-03-09T08:35:03.362367Z","shell.execute_reply":"2023-03-09T08:35:03.445248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.451045Z","iopub.execute_input":"2023-03-09T08:35:03.451539Z","iopub.status.idle":"2023-03-09T08:35:03.463876Z","shell.execute_reply.started":"2023-03-09T08:35:03.451491Z","shell.execute_reply":"2023-03-09T08:35:03.461922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.concat([X_res, y_res], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.465942Z","iopub.execute_input":"2023-03-09T08:35:03.467003Z","iopub.status.idle":"2023-03-09T08:35:03.476261Z","shell.execute_reply.started":"2023-03-09T08:35:03.466944Z","shell.execute_reply":"2023-03-09T08:35:03.474657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.sample(len(data)).reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.477828Z","iopub.execute_input":"2023-03-09T08:35:03.478207Z","iopub.status.idle":"2023-03-09T08:35:03.499084Z","shell.execute_reply.started":"2023-03-09T08:35:03.478170Z","shell.execute_reply":"2023-03-09T08:35:03.497978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.500811Z","iopub.execute_input":"2023-03-09T08:35:03.501453Z","iopub.status.idle":"2023-03-09T08:35:03.509481Z","shell.execute_reply.started":"2023-03-09T08:35:03.501412Z","shell.execute_reply":"2023-03-09T08:35:03.507613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = 0\nprint(f\"Toxicity:--> {data.toxic[example]} \\n\\nComment:\\n{data.comment_text[example]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.511441Z","iopub.execute_input":"2023-03-09T08:35:03.512057Z","iopub.status.idle":"2023-03-09T08:35:03.525202Z","shell.execute_reply.started":"2023-03-09T08:35:03.511984Z","shell.execute_reply":"2023-03-09T08:35:03.523839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting the dataset","metadata":{}},{"cell_type":"code","source":"training_size = 30000\ntesting_size = 6000\n# Split the sentences\ntraining_sentences = data.comment_text[0:training_size].to_numpy()\nvalidation_sentences = data.comment_text[training_size:training_size + testing_size].to_numpy()\ntesting_sentences = data.comment_text[training_size + testing_size:].to_numpy()\n\n# Split the labels\ntraining_labels = data.toxic[0:training_size].to_numpy()\nvalidation_labels = data.toxic[training_size:training_size + testing_size].to_numpy()\ntesting_labels = data.toxic[training_size + testing_size:].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:45:00.920863Z","iopub.execute_input":"2023-03-09T08:45:00.921290Z","iopub.status.idle":"2023-03-09T08:45:00.930679Z","shell.execute_reply.started":"2023-03-09T08:45:00.921254Z","shell.execute_reply":"2023-03-09T08:45:00.929551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Training shape:-->{training_sentences.shape}\")\nprint(f\"Validation shape:-->{validation_sentences.shape}\")\nprint(f\"Testing shape:-->{testing_sentences.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:45:01.226572Z","iopub.execute_input":"2023-03-09T08:45:01.228060Z","iopub.status.idle":"2023-03-09T08:45:01.236295Z","shell.execute_reply.started":"2023-03-09T08:45:01.227993Z","shell.execute_reply":"2023-03-09T08:45:01.235029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:44:48.207042Z","iopub.execute_input":"2023-03-09T08:44:48.207530Z","iopub.status.idle":"2023-03-09T08:44:48.218001Z","shell.execute_reply.started":"2023-03-09T08:44:48.207486Z","shell.execute_reply":"2023-03-09T08:44:48.216427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Repreprocessing the data","metadata":{}},{"cell_type":"code","source":"validation_sentences[0]","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.567818Z","iopub.execute_input":"2023-03-09T08:35:03.568689Z","iopub.status.idle":"2023-03-09T08:35:03.580657Z","shell.execute_reply.started":"2023-03-09T08:35:03.568630Z","shell.execute_reply":"2023-03-09T08:35:03.579291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = {\"text\":[], \"label\":[]}\nval_data = {\"text\":[], \"label\":[]}\ntest_data = {\"text\":[], \"label\":[]}\n\nfor i in range(len(training_sentences)):\n    train_data['text'].append(training_sentences[i])\n    train_data['label'].append(training_labels[i])\n\nfor j in range(len(validation_sentences)):\n    val_data['text'].append(validation_sentences[j])\n    val_data['label'].append(validation_labels[j])\n    \nfor k in range(len(testing_sentences)):\n    test_data['text'].append(testing_sentences[k])\n    test_data['label'].append(testing_labels[k])","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.582705Z","iopub.execute_input":"2023-03-09T08:35:03.583581Z","iopub.status.idle":"2023-03-09T08:35:03.633629Z","shell.execute_reply.started":"2023-03-09T08:35:03.583509Z","shell.execute_reply":"2023-03-09T08:35:03.632526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenization","metadata":{}},{"cell_type":"code","source":"!pip install transformers","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:03.635311Z","iopub.execute_input":"2023-03-09T08:35:03.636349Z","iopub.status.idle":"2023-03-09T08:35:16.420391Z","shell.execute_reply.started":"2023-03-09T08:35:03.636286Z","shell.execute_reply":"2023-03-09T08:35:16.418578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer\nmodel_ckpt = \"distilbert-base-uncased\"\ntokenizer = AutoTokenizer.from_pretrained(model_ckpt)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.427012Z","iopub.execute_input":"2023-03-09T08:35:16.427443Z","iopub.status.idle":"2023-03-09T08:35:16.693225Z","shell.execute_reply.started":"2023-03-09T08:35:16.427396Z","shell.execute_reply":"2023-03-09T08:35:16.692040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Example of tokenization","metadata":{}},{"cell_type":"code","source":"text = \"I am trying to learn how to use transformers\"\nencoded_text = tokenizer(text)\nprint(encoded_text)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.696828Z","iopub.execute_input":"2023-03-09T08:35:16.697507Z","iopub.status.idle":"2023-03-09T08:35:16.704255Z","shell.execute_reply.started":"2023-03-09T08:35:16.697463Z","shell.execute_reply":"2023-03-09T08:35:16.702580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Detokenize","metadata":{}},{"cell_type":"code","source":"tokens = tokenizer.convert_ids_to_tokens(encoded_text.input_ids)\nprint(tokens)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.705954Z","iopub.execute_input":"2023-03-09T08:35:16.706436Z","iopub.status.idle":"2023-03-09T08:35:16.719396Z","shell.execute_reply.started":"2023-03-09T08:35:16.706349Z","shell.execute_reply":"2023-03-09T08:35:16.717634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dealing with our dataset","metadata":{}},{"cell_type":"code","source":"#https://medium.com/@ashwinnaidu1991/text-classification-with-transformers-70acaf65c4a4\n#Tokenizing batch data\n#I've modified the function according to my data\ndef tokenize(batch):\n    dico = dict()\n    dico = tokenizer(batch[\"text\"], padding=True, truncation=True)\n    dico[\"text\"] = batch[\"text\"][:]\n    dico[\"label\"] = batch[\"label\"][:]\n    return dico","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.721381Z","iopub.execute_input":"2023-03-09T08:35:16.722744Z","iopub.status.idle":"2023-03-09T08:35:16.734851Z","shell.execute_reply.started":"2023-03-09T08:35:16.722656Z","shell.execute_reply":"2023-03-09T08:35:16.733422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(tokenize(train_data)) ","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.736578Z","iopub.execute_input":"2023-03-09T08:35:16.737115Z","iopub.status.idle":"2023-03-09T08:35:16.747689Z","shell.execute_reply.started":"2023-03-09T08:35:16.737058Z","shell.execute_reply":"2023-03-09T08:35:16.746509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds_encoded = tokenize(train_data)\nval_ds_encoded = tokenize(val_data)\ntest_ds_encoded = tokenize(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:16.749108Z","iopub.execute_input":"2023-03-09T08:35:16.749659Z","iopub.status.idle":"2023-03-09T08:35:27.612756Z","shell.execute_reply.started":"2023-03-09T08:35:16.749603Z","shell.execute_reply":"2023-03-09T08:35:27.610954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"train_ds = tf.data.Dataset.from_tensor_slices(train_ds_encoded)\nval_ds = tf.data.Dataset.from_tensor_slices(val_ds_encoded)\ntest_ds = tf.data.Dataset.from_tensor_slices(test_ds_encoded)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:27.616226Z","iopub.execute_input":"2023-03-09T08:35:27.616841Z","iopub.status.idle":"2023-03-09T08:35:27.628766Z","shell.execute_reply.started":"2023-03-09T08:35:27.616784Z","shell.execute_reply":"2023-03-09T08:35:27.627003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Retreiving our data for model training, validation and testing","metadata":{}},{"cell_type":"code","source":"x_train = train_ds_encoded[\"input_ids\"]\ny_train = train_ds_encoded[\"label\"]\nx_val = val_ds_encoded[\"input_ids\"]\ny_val = val_ds_encoded[\"label\"]\nx_test = test_ds_encoded[\"input_ids\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:27.630502Z","iopub.execute_input":"2023-03-09T08:35:27.631611Z","iopub.status.idle":"2023-03-09T08:35:27.840085Z","shell.execute_reply.started":"2023-03-09T08:35:27.631482Z","shell.execute_reply":"2023-03-09T08:35:27.838753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating batch dataset for model training...","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 512\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:27.845784Z","iopub.execute_input":"2023-03-09T08:35:27.846261Z","iopub.status.idle":"2023-03-09T08:35:27.853777Z","shell.execute_reply.started":"2023-03-09T08:35:27.846217Z","shell.execute_reply":"2023-03-09T08:35:27.852437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(1024)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\ntrain_dataset = strategy.experimental_distribute_dataset(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:35:29.541125Z","iopub.execute_input":"2023-03-09T08:35:29.541883Z","iopub.status.idle":"2023-03-09T08:36:14.659671Z","shell.execute_reply.started":"2023-03-09T08:35:29.541842Z","shell.execute_reply":"2023-03-09T08:36:14.658325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_val, y_val))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\nvalid_dataset = strategy.experimental_distribute_dataset(valid_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:36:14.661752Z","iopub.execute_input":"2023-03-09T08:36:14.662116Z","iopub.status.idle":"2023-03-09T08:36:23.770276Z","shell.execute_reply.started":"2023-03-09T08:36:14.662081Z","shell.execute_reply":"2023-03-09T08:36:23.768743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:36:23.771801Z","iopub.execute_input":"2023-03-09T08:36:23.772213Z","iopub.status.idle":"2023-03-09T08:36:33.986874Z","shell.execute_reply.started":"2023-03-09T08:36:23.772158Z","shell.execute_reply":"2023-03-09T08:36:33.985528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building the Transformer Model","metadata":{}},{"cell_type":"markdown","source":"Inspired from the model at this link\nhttps://www.kaggle.com/code/fahaddalwai/toxicity-detection-using-rnn-bert","metadata":{}},{"cell_type":"code","source":"import transformers\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint","metadata":{"execution":{"iopub.status.busy":"2023-03-09T08:36:33.989982Z","iopub.execute_input":"2023-03-09T08:36:33.990492Z","iopub.status.idle":"2023-03-09T08:36:33.997637Z","shell.execute_reply.started":"2023-03-09T08:36:33.990453Z","shell.execute_reply":"2023-03-09T08:36:33.996010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    layer1 = Dense(4, activation='relu')(cls_token)#Adding a layer\n    drp = Dropout(0.3)(layer1)\n    output = Dense(1, activation='sigmoid')(drp)\n    \n    model = Model(inputs=input_word_ids, outputs=output)\n    model.compile(Adam(learning_rate=1e-6), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:06.279111Z","iopub.execute_input":"2023-03-09T09:11:06.280451Z","iopub.status.idle":"2023-03-09T09:11:06.290544Z","shell.execute_reply.started":"2023-03-09T09:11:06.280394Z","shell.execute_reply":"2023-03-09T09:11:06.288739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN = 512","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:06.914307Z","iopub.execute_input":"2023-03-09T09:11:06.914783Z","iopub.status.idle":"2023-03-09T09:11:06.921053Z","shell.execute_reply.started":"2023-03-09T09:11:06.914736Z","shell.execute_reply":"2023-03-09T09:11:06.919165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:10.583439Z","iopub.execute_input":"2023-03-09T09:11:10.584162Z","iopub.status.idle":"2023-03-09T09:11:22.333424Z","shell.execute_reply.started":"2023-03-09T09:11:10.584118Z","shell.execute_reply":"2023-03-09T09:11:22.331996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"batch_size = 1024\ntrain_ds_batched = train_ds.batch(batch_size).prefetch(1)\nval_ds_batched = val_ds.batch(batch_size).prefetch(1)\ntest_ds_batched = test_ds.batch(batch_size).prefetch(1)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:22.335456Z","iopub.execute_input":"2023-03-09T09:11:22.335951Z","iopub.status.idle":"2023-03-09T09:11:22.346330Z","shell.execute_reply.started":"2023-03-09T09:11:22.335882Z","shell.execute_reply":"2023-03-09T09:11:22.344305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's load the distilbert-base-uncased pretrained model checkpoint","metadata":{}},{"cell_type":"code","source":"\"\"\"from transformers import TFAutoModel \ntf_model = TFAutoModel.from_pretrained(model_ckpt)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:22.348689Z","iopub.execute_input":"2023-03-09T09:11:22.349949Z","iopub.status.idle":"2023-03-09T09:11:22.359953Z","shell.execute_reply.started":"2023-03-09T09:11:22.349862Z","shell.execute_reply":"2023-03-09T09:11:22.358695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = len(x_train) // BATCH_SIZE\nn_val_step = len(x_val)//BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    validation_steps = n_val_step,\n    epochs=100\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:11:24.027080Z","iopub.execute_input":"2023-03-09T09:11:24.027528Z","iopub.status.idle":"2023-03-09T09:59:28.776960Z","shell.execute_reply.started":"2023-03-09T09:11:24.027476Z","shell.execute_reply":"2023-03-09T09:59:28.775486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(test_dataset, verbose=1)\n#prediction = {}\n#prediction['toxic'] = y_pred","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:59:39.455223Z","iopub.execute_input":"2023-03-09T09:59:39.455686Z","iopub.status.idle":"2023-03-09T09:59:52.309862Z","shell.execute_reply.started":"2023-03-09T09:59:39.455636Z","shell.execute_reply":"2023-03-09T09:59:52.308551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the accuracy and loss history\nplot_graphs(train_history, 'accuracy')\nplot_graphs(train_history, 'loss')","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:59:52.313429Z","iopub.execute_input":"2023-03-09T09:59:52.313994Z","iopub.status.idle":"2023-03-09T09:59:52.676125Z","shell.execute_reply.started":"2023-03-09T09:59:52.313940Z","shell.execute_reply":"2023-03-09T09:59:52.674734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_series(x, y, format=\"-\", start=0, end=None, \n                title=None, xlabel=None, ylabel=None, legend=None ):\n    \"\"\"\n    Visualizes time series data\n\n    Args:\n      x (array of int) - contains values for the x-axis\n      y (array of int or tuple of arrays) - contains the values for the y-axis\n      format (string) - line style when plotting the graph\n      label (string) - tag for the line\n      start (int) - first time step to plot\n      end (int) - last time step to plot\n      title (string) - title of the plot\n      xlabel (string) - label for the x-axis\n      ylabel (string) - label for the y-axis\n      legend (list of strings) - legend for the plot\n    \"\"\"\n\n    # Setup dimensions of the graph figure\n    plt.figure(figsize=(18, 10))\n    \n    # Check if there are more than two series to plot\n    if type(y) is tuple:\n\n      # Loop over the y elements\n      for y_curr in y:\n\n        # Plot the x and current y values\n        plt.plot(x[start:end], y_curr[start:end], format)\n\n    else:\n      # Plot the x and y values\n      plt.plot(x[start:end], y[start:end], format)\n      \n\n    # Label the x-axis\n    plt.xlabel(xlabel)\n\n    # Label the y-axis\n    plt.ylabel(ylabel)\n\n    # Set the legend\n    if legend:\n        plt.legend(legend)\n\n    # Set the title\n    plt.title(title)\n\n    # Overlay a grid on the graph\n    plt.grid(True)\n\n    # Draw the graph on screen\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T09:59:59.308857Z","iopub.execute_input":"2023-03-09T09:59:59.309857Z","iopub.status.idle":"2023-03-09T09:59:59.320531Z","shell.execute_reply.started":"2023-03-09T09:59:59.309809Z","shell.execute_reply":"2023-03-09T09:59:59.318942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get mae and loss from history log\nacc=train_history.history['accuracy']\nloss=train_history.history['loss']\n\nval_acc=train_history.history['val_accuracy']\nval_loss=train_history.history['val_loss']\n\n# Get number of epochs\nepochs=range(len(loss)) \n\n# Plot mae and loss\nplot_series(\n    x=epochs, \n    y=(acc, loss, val_acc, val_loss), \n    title='Accuracy and Loss', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )\n\n# Only plot the last 80% of the epochs\nzoom_split = int(epochs[-1] * 0.2)\nepochs_zoom = epochs[zoom_split:]\nacc_zoom = acc[zoom_split:]\nloss_zoom = loss[zoom_split:]\nval_acc_zoom = val_acc[zoom_split:]\nval_loss_zoom = val_loss[zoom_split:]\n\n# Plot zoomed mae and loss\nplot_series(\n    x=epochs_zoom, \n    y=(acc_zoom, loss_zoom, val_acc_zoom, val_loss_zoom), \n    title='Accuracy and Loss Zoom', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:01.352342Z","iopub.execute_input":"2023-03-09T10:00:01.353698Z","iopub.status.idle":"2023-03-09T10:00:02.092171Z","shell.execute_reply.started":"2023-03-09T10:00:01.353639Z","shell.execute_reply":"2023-03-09T10:00:02.090568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_pred = model.predict(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:09.136305Z","iopub.execute_input":"2023-03-09T10:00:09.136790Z","iopub.status.idle":"2023-03-09T10:00:09.141789Z","shell.execute_reply.started":"2023-03-09T10:00:09.136731Z","shell.execute_reply":"2023-03-09T10:00:09.140558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:13.002583Z","iopub.execute_input":"2023-03-09T10:00:13.003015Z","iopub.status.idle":"2023-03-09T10:00:25.500372Z","shell.execute_reply.started":"2023-03-09T10:00:13.002977Z","shell.execute_reply":"2023-03-09T10:00:25.498370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:25.503214Z","iopub.execute_input":"2023-03-09T10:00:25.503774Z","iopub.status.idle":"2023-03-09T10:00:38.000182Z","shell.execute_reply.started":"2023-03-09T10:00:25.503682Z","shell.execute_reply":"2023-03-09T10:00:37.998834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = [0 if i<0.7 else 1 for i in y_pred]","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:43.558032Z","iopub.execute_input":"2023-03-09T10:00:43.558563Z","iopub.status.idle":"2023-03-09T10:00:43.577561Z","shell.execute_reply.started":"2023-03-09T10:00:43.558501Z","shell.execute_reply":"2023-03-09T10:00:43.575825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ncm = confusion_matrix(testing_labels, y_hat)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:46.442945Z","iopub.execute_input":"2023-03-09T10:00:46.443425Z","iopub.status.idle":"2023-03-09T10:00:46.455533Z","shell.execute_reply.started":"2023-03-09T10:00:46.443374Z","shell.execute_reply":"2023-03-09T10:00:46.454374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.heatmap(cm, annot=True, linewidth=.5)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:00:48.689706Z","iopub.execute_input":"2023-03-09T10:00:48.690176Z","iopub.status.idle":"2023-03-09T10:00:48.992749Z","shell.execute_reply.started":"2023-03-09T10:00:48.690136Z","shell.execute_reply":"2023-03-09T10:00:48.990999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(testing_labels, y_hat))","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:01:10.325659Z","iopub.execute_input":"2023-03-09T10:01:10.326168Z","iopub.status.idle":"2023-03-09T10:01:10.350336Z","shell.execute_reply.started":"2023-03-09T10:01:10.326118Z","shell.execute_reply":"2023-03-09T10:01:10.348924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save the model","metadata":{}},{"cell_type":"code","source":"# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nmodel.save('saved_model/ToxicityTransformerModel.h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-09T10:01:29.743857Z","iopub.execute_input":"2023-03-09T10:01:29.744375Z","iopub.status.idle":"2023-03-09T10:01:40.826679Z","shell.execute_reply.started":"2023-03-09T10:01:29.744325Z","shell.execute_reply":"2023-03-09T10:01:40.824937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}