{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\n\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\n\n\nfrom tensorflow import keras\n\nfrom tensorflow.keras.preprocessing import sequence,text\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, GRU,SimpleRNN\nfrom tensorflow.keras.layers import Dense, Activation, Dropout\nfrom tensorflow.keras.layers import Embedding\nfrom tensorflow.keras.layers import BatchNormalization\n\nfrom tensorflow.keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom tensorflow.keras.preprocessing import sequence, text\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n \nimport os\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-06T12:38:13.979583Z","iopub.execute_input":"2023-03-06T12:38:13.980270Z","iopub.status.idle":"2023-03-06T12:38:13.988543Z","shell.execute_reply.started":"2023-03-06T12:38:13.980227Z","shell.execute_reply":"2023-03-06T12:38:13.987103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting Up TPU  \nIn order to speed up the process of our training, we'll run the notebook on a TPU. The below code is used to set up the TPU, and is a boilerplate which can be used with any notebook!","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:17.692541Z","iopub.execute_input":"2023-03-06T12:38:17.693231Z","iopub.status.idle":"2023-03-06T12:38:23.601387Z","shell.execute_reply.started":"2023-03-06T12:38:17.693167Z","shell.execute_reply":"2023-03-06T12:38:23.600015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our TPU seems to be working with 8 REPLICAS. This is like having 8 different GPU cards running at the same time.","metadata":{}},{"cell_type":"markdown","source":"# Exploring Our Data  \nWe'll load up and have a look through the format of our data.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:26.007488Z","iopub.execute_input":"2023-03-06T12:38:26.007870Z","iopub.status.idle":"2023-03-06T12:38:29.738011Z","shell.execute_reply.started":"2023-03-06T12:38:26.007831Z","shell.execute_reply":"2023-03-06T12:38:29.736549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:32.954381Z","iopub.execute_input":"2023-03-06T12:38:32.954745Z","iopub.status.idle":"2023-03-06T12:38:32.986735Z","shell.execute_reply.started":"2023-03-06T12:38:32.954711Z","shell.execute_reply":"2023-03-06T12:38:32.985360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:35.335190Z","iopub.execute_input":"2023-03-06T12:38:35.335587Z","iopub.status.idle":"2023-03-06T12:38:35.348314Z","shell.execute_reply.started":"2023-03-06T12:38:35.335555Z","shell.execute_reply":"2023-03-06T12:38:35.346621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll turn this into a simple binary classification problem and remove all the other types of classification. Also, to speed up training, we'll use only 12000 rows rather than the entire dataset for training","metadata":{}},{"cell_type":"code","source":"train.drop(['severe_toxic','obscene','threat','insult','identity_hate'],axis=1,inplace=True)\ntrain = train.loc[:12000,:]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:37.512526Z","iopub.execute_input":"2023-03-06T12:38:37.512930Z","iopub.status.idle":"2023-03-06T12:38:37.540053Z","shell.execute_reply.started":"2023-03-06T12:38:37.512889Z","shell.execute_reply":"2023-03-06T12:38:37.538034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For our models embedding, we are required to set a max length. We'll pad all the texts upto a maximum length so that we can have a uniform input format for our model. This is required.","metadata":{}},{"cell_type":"code","source":"train['comment_text'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:39.629461Z","iopub.execute_input":"2023-03-06T12:38:39.629848Z","iopub.status.idle":"2023-03-06T12:38:39.699925Z","shell.execute_reply.started":"2023-03-06T12:38:39.629812Z","shell.execute_reply":"2023-03-06T12:38:39.697869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below is a helper function in order to help us define the AUC percentage of our model.","metadata":{}},{"cell_type":"code","source":"import sklearn.metrics as metrics\ndef roc_auc(predictions,target):\n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:44:53.440262Z","iopub.execute_input":"2023-03-06T12:44:53.440682Z","iopub.status.idle":"2023-03-06T12:44:53.447254Z","shell.execute_reply.started":"2023-03-06T12:44:53.440644Z","shell.execute_reply":"2023-03-06T12:44:53.445670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Train Test Split  \nWe make the train test split using scikits built in function and also use the stratify parameter, in order to get equal proportions of toxic and non toxic comments in each split.","metadata":{}},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train.comment_text.values, train.toxic.values, \n                                                  stratify=train.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:43.691135Z","iopub.execute_input":"2023-03-06T12:38:43.691578Z","iopub.status.idle":"2023-03-06T12:38:43.706025Z","shell.execute_reply.started":"2023-03-06T12:38:43.691542Z","shell.execute_reply":"2023-03-06T12:38:43.703714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(xtrain[0])\nprint(ytrain[0])","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:46.293196Z","iopub.execute_input":"2023-03-06T12:38:46.294620Z","iopub.status.idle":"2023-03-06T12:38:46.300272Z","shell.execute_reply.started":"2023-03-06T12:38:46.294579Z","shell.execute_reply":"2023-03-06T12:38:46.298737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our x value contains the sentences and our y values contain the value of either toxic or not! Lets get started with our models.","metadata":{}},{"cell_type":"markdown","source":"# Using Simple RNN Model  \nRNN (Recurrent Neural Network) is a type of artificial neural network that allows for processing of sequential data, such as time-series data or natural language processing tasks. Unlike feedforward neural networks that process input data in a single pass, RNNs can take into account the previous inputs as they process the current input. This allows for modeling of dependencies between sequential inputs and outputs. RNNs have a feedback loop that allows them to store information about previous inputs, which is then used to influence the output at the current time step. \n","metadata":{}},{"cell_type":"markdown","source":"The below code does the following:\n1. Create our tokenizer for our RNN. Our model cannot input details in a sentence format. Instead, we are required to use a tokenizer which will perform tokenizatino and split the words in each sentence. We use Keras Tokenizer for this purpose.\n2. We set the max_len as 1500, since we previously saw that the max characters in a tweet was 1403\n3. Convert the texts to sequences, which transforms each text in texts to a sequence of integers for our model. THis is a mandatory step as well.\n4. Perform text padding so all inputs are of same length","metadata":{}},{"cell_type":"code","source":"\n#defining our tokenizer\ntoken=text.Tokenizer(num_words=None)\n\nmax_len=1500\n\n#required to fit on text before using texts to sequences\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\n\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n\n#zero pad the sequences\nxtrain_pad = sequence.pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = sequence.pad_sequences(xvalid_seq, maxlen=max_len)\n\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:49.540141Z","iopub.execute_input":"2023-03-06T12:38:49.540579Z","iopub.status.idle":"2023-03-06T12:38:51.184066Z","shell.execute_reply.started":"2023-03-06T12:38:49.540542Z","shell.execute_reply":"2023-03-06T12:38:51.182404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The word index contains the list of all the words in our data/corpus sorted in their descending order in the dictionary for our model to use.","metadata":{}},{"cell_type":"code","source":"len(word_index)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:54.538759Z","iopub.execute_input":"2023-03-06T12:38:54.539705Z","iopub.status.idle":"2023-03-06T12:38:54.545981Z","shell.execute_reply.started":"2023-03-06T12:38:54.539659Z","shell.execute_reply":"2023-03-06T12:38:54.544749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now lets define our model! We use a simple RNN with 100 cells for this purpose. Notice the Embedding layer, which is responsible for performing embedding. An embedding layer takes as input a sequence of integer-encoded tokens (i.e., words or characters) and converts them into a dense vector representation. This dense vector representation is learned during training and can capture the semantic relationships between the tokens.\n\n","metadata":{}},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    Embedding(len(word_index) + 1,100,input_length=max_len),\n    SimpleRNN(100),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:56.641642Z","iopub.execute_input":"2023-03-06T12:38:56.642340Z","iopub.status.idle":"2023-03-06T12:38:56.830604Z","shell.execute_reply.started":"2023-03-06T12:38:56.642275Z","shell.execute_reply":"2023-03-06T12:38:56.828929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad,ytrain,epochs=2)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:38:59.954033Z","iopub.execute_input":"2023-03-06T12:38:59.955589Z","iopub.status.idle":"2023-03-06T12:43:51.581883Z","shell.execute_reply.started":"2023-03-06T12:38:59.955516Z","shell.execute_reply":"2023-03-06T12:43:51.580768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:44:57.922353Z","iopub.execute_input":"2023-03-06T12:44:57.923305Z","iopub.status.idle":"2023-03-06T12:45:02.024707Z","shell.execute_reply.started":"2023-03-06T12:44:57.923174Z","shell.execute_reply":"2023-03-06T12:45:02.023055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We get an AUC score of roughly 70 percent, which is pretty respectable for just a single layer! Training on more layers will yield a better result as it will be able to idetify more features from the data.","metadata":{}},{"cell_type":"markdown","source":"# Using LSTM Model  \nUnlike traditional RNNs, LSTMs are able to capture long-term dependencies in sequential data by using a memory cell and gates that control the flow of information. The memory cell is able to store information over long periods of time, while the gates are able to selectively add or remove information from the cell. ","metadata":{}},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    Embedding(len(word_index) + 1,100,input_length=max_len),\n    LSTM(100),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:45:05.165972Z","iopub.execute_input":"2023-03-06T12:45:05.168899Z","iopub.status.idle":"2023-03-06T12:45:05.496328Z","shell.execute_reply.started":"2023-03-06T12:45:05.168839Z","shell.execute_reply":"2023-03-06T12:45:05.495124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(xtrain_pad,ytrain,epochs=2)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:45:08.676622Z","iopub.execute_input":"2023-03-06T12:45:08.676982Z","iopub.status.idle":"2023-03-06T12:51:14.835395Z","shell.execute_reply.started":"2023-03-06T12:45:08.676948Z","shell.execute_reply":"2023-03-06T12:51:14.834296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.predict(xvalid_pad)\nprint(\"Auc: %.2f%%\" % (roc_auc(scores,yvalid)))","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:51:40.684856Z","iopub.execute_input":"2023-03-06T12:51:40.685294Z","iopub.status.idle":"2023-03-06T12:51:53.639459Z","shell.execute_reply.started":"2023-03-06T12:51:40.685246Z","shell.execute_reply":"2023-03-06T12:51:53.637926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Already, we have a much larger increase in the AUC value and it is upto 93%. ","metadata":{}},{"cell_type":"markdown","source":"# Using BERT  \nBERT (Bidirectional Encoder Representations from Transformers) is a state-of-the-art pre-trained language model developed by Google. It is based on the Transformer architecture and is trained on massive amounts of text data using an unsupervised learning approach. BERT is designed to generate high-quality embeddings of words and sentences that can be fine-tuned for a wide range of natural language processing tasks, such as sentiment analysis, named entity recognition, and question answering. One of the key innovations of BERT is the use of a \"masked language modeling\" objective, which trains the model to predict missing words in a sentence based on the context.","metadata":{}},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nimport numpy as np\nimport pandas as pd\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:51:57.946526Z","iopub.execute_input":"2023-03-06T12:51:57.947605Z","iopub.status.idle":"2023-03-06T12:51:58.343481Z","shell.execute_reply.started":"2023-03-06T12:51:57.947557Z","shell.execute_reply":"2023-03-06T12:51:58.341797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll reload our dataset to make it easier for ourselves","metadata":{}},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:52:00.902712Z","iopub.execute_input":"2023-03-06T12:52:00.903084Z","iopub.status.idle":"2023-03-06T12:52:02.618642Z","shell.execute_reply.started":"2023-03-06T12:52:00.903049Z","shell.execute_reply":"2023-03-06T12:52:02.617817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below code is used for speeding up the training process and for the configuration of our model for fine tuning.","metadata":{}},{"cell_type":"code","source":"#IMP DATA FOR CONFIG\nAUTO = tf.data.experimental.AUTOTUNE\n\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 512","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:52:05.070632Z","iopub.execute_input":"2023-03-06T12:52:05.071325Z","iopub.status.idle":"2023-03-06T12:52:05.077349Z","shell.execute_reply.started":"2023-03-06T12:52:05.071285Z","shell.execute_reply":"2023-03-06T12:52:05.075494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are a few steps before we can start training and fine tuning our model. First of all, we begine by loading the BERT model from the HuggingFace library. We'll use DistilBert, which is a condensed version of the bigger model. We also load our tokenizer up. Bert has its own tokenizerwhich is based on the WordPiece tokenization.","metadata":{}},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:52:07.491401Z","iopub.execute_input":"2023-03-06T12:52:07.491778Z","iopub.status.idle":"2023-03-06T12:52:09.423247Z","shell.execute_reply.started":"2023-03-06T12:52:07.491744Z","shell.execute_reply":"2023-03-06T12:52:09.421092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next step is to feed our data into the tokenizer. In order to do that, we are required to feed a list of the sentences into our tokenizer. We'll convert the train, test and validation texts into a list for this purpose","metadata":{}},{"cell_type":"code","source":"list_of_train_words=train1['comment_text'].to_list()\nlist_of_val_words=valid['comment_text'].to_list()\nlist_of_test_words=test['content'].to_list()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:52:17.072058Z","iopub.execute_input":"2023-03-06T12:52:17.072470Z","iopub.status.idle":"2023-03-06T12:52:17.091954Z","shell.execute_reply.started":"2023-03-06T12:52:17.072436Z","shell.execute_reply":"2023-03-06T12:52:17.090584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The below code performs the tokenization. We use only 20,000 rows for faster training and make sure that the return type of our data is in the form of np arrays.","metadata":{}},{"cell_type":"code","source":"x_train = tokenizer(list_of_train_words[:20000], padding=True, truncation=True, return_tensors=\"np\")\nx_test = tokenizer(list_of_test_words, padding=True, truncation=True, return_tensors=\"np\")\nx_valid = tokenizer(list_of_val_words, padding=True, truncation=True, return_tensors=\"np\")","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:52:19.566090Z","iopub.execute_input":"2023-03-06T12:52:19.566509Z","iopub.status.idle":"2023-03-06T12:55:14.910150Z","shell.execute_reply.started":"2023-03-06T12:52:19.566472Z","shell.execute_reply":"2023-03-06T12:55:14.908153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We perform the same conversion of our y values into an np array","metadata":{}},{"cell_type":"code","source":"y_train=train1['toxic'][:20000].to_numpy()\ny_valid=valid['toxic'].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:56:41.271878Z","iopub.execute_input":"2023-03-06T12:56:41.272412Z","iopub.status.idle":"2023-03-06T12:56:41.281477Z","shell.execute_reply.started":"2023-03-06T12:56:41.272367Z","shell.execute_reply":"2023-03-06T12:56:41.279940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To train our model, we only require the input_id's and not the masking, etc.We take this out from our returned np array from our tokenizer","metadata":{}},{"cell_type":"code","source":"x_train=x_train['input_ids']\nx_test=x_test['input_ids']\nx_valid=x_valid['input_ids']","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:56:43.696652Z","iopub.execute_input":"2023-03-06T12:56:43.697083Z","iopub.status.idle":"2023-03-06T12:56:43.706088Z","shell.execute_reply.started":"2023-03-06T12:56:43.697045Z","shell.execute_reply":"2023-03-06T12:56:43.704397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next step is converting our data into Dataset format to feed into our model for performing predictions. The Dataset format helsp us perform batching and easily feeding the data into our TPU's","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:56:46.735819Z","iopub.execute_input":"2023-03-06T12:56:46.737250Z","iopub.status.idle":"2023-03-06T12:56:48.582143Z","shell.execute_reply.started":"2023-03-06T12:56:46.737168Z","shell.execute_reply":"2023-03-06T12:56:48.580473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:56:52.142525Z","iopub.execute_input":"2023-03-06T12:56:52.142974Z","iopub.status.idle":"2023-03-06T12:56:52.151336Z","shell.execute_reply.started":"2023-03-06T12:56:52.142933Z","shell.execute_reply":"2023-03-06T12:56:52.150247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-06T12:56:55.256687Z","iopub.execute_input":"2023-03-06T12:56:55.257075Z","iopub.status.idle":"2023-03-06T12:57:47.142372Z","shell.execute_reply.started":"2023-03-06T12:56:55.257042Z","shell.execute_reply":"2023-03-06T12:57:47.139328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=5\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T13:00:02.773557Z","iopub.execute_input":"2023-03-06T13:00:02.773993Z","iopub.status.idle":"2023-03-06T13:02:19.694165Z","shell.execute_reply.started":"2023-03-06T13:00:02.773954Z","shell.execute_reply":"2023-03-06T13:02:19.692529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Submissions  \nWe'll use this model to perform predictions on our test data and submit them for the competition.","metadata":{}},{"cell_type":"code","source":"sub['toxic'] = model.predict(test_dataset, verbose=1)\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-06T13:03:17.050691Z","iopub.execute_input":"2023-03-06T13:03:17.051073Z","iopub.status.idle":"2023-03-06T13:03:48.790974Z","shell.execute_reply.started":"2023-03-06T13:03:17.051037Z","shell.execute_reply":"2023-03-06T13:03:48.789650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}