{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"* Introduction and Import Packages\n\n* Load and Explore the Data\n\n* Data Preparation — Tokenize and Pad Text Data\n\n* Prepare Embedding Matrix with Pre-trained GloVe Embeddings\n\n* Create the Embedding Layer\n\n* Build the Model\n\n* Train the Model\n\n* Model Evaluation - Classify Toxic Comments","metadata":{}},{"cell_type":"markdown","source":"### Imports","metadata":{}},{"cell_type":"code","source":"try:\n  # %tensorflow_version only exists in Colab.\n  %tensorflow_version 2.x\nexcept Exception:\n    pass\n  \nimport tensorflow as tf\nimport tensorflow_datasets as tfds\n\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-08T10:29:54.855674Z","iopub.execute_input":"2023-03-08T10:29:54.856084Z","iopub.status.idle":"2023-03-08T10:29:58.203756Z","shell.execute_reply.started":"2023-03-08T10:29:54.856049Z","shell.execute_reply":"2023-03-08T10:29:58.202631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setting TPU","metadata":{}},{"cell_type":"code","source":"import os\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:29:58.205786Z","iopub.execute_input":"2023-03-08T10:29:58.206496Z","iopub.status.idle":"2023-03-08T10:29:58.216937Z","shell.execute_reply.started":"2023-03-08T10:29:58.206457Z","shell.execute_reply":"2023-03-08T10:29:58.215713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:29:58.218565Z","iopub.execute_input":"2023-03-08T10:29:58.219053Z","iopub.status.idle":"2023-03-08T10:29:58.226591Z","shell.execute_reply.started":"2023-03-08T10:29:58.219014Z","shell.execute_reply":"2023-03-08T10:29:58.225515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot Utility\ndef plot_graphs(history, string):\n    plt.plot(history.history[string])\n    plt.plot(history.history['val_'+string])\n    plt.xlabel(\"Epochs\")\n    plt.ylabel(string)\n    plt.legend([string, 'val_'+string])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:29:58.229947Z","iopub.execute_input":"2023-03-08T10:29:58.230369Z","iopub.status.idle":"2023-03-08T10:29:58.236720Z","shell.execute_reply.started":"2023-03-08T10:29:58.230320Z","shell.execute_reply":"2023-03-08T10:29:58.235583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Retreiving the english data\n\nWe are just using english comments","metadata":{}},{"cell_type":"code","source":"#Train data and labels\ndata = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:29:58.238411Z","iopub.execute_input":"2023-03-08T10:29:58.239149Z","iopub.status.idle":"2023-03-08T10:30:00.710234Z","shell.execute_reply.started":"2023-03-08T10:29:58.239114Z","shell.execute_reply":"2023-03-08T10:30:00.709145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.columns","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.711603Z","iopub.execute_input":"2023-03-08T10:30:00.712216Z","iopub.status.idle":"2023-03-08T10:30:00.722142Z","shell.execute_reply.started":"2023-03-08T10:30:00.712178Z","shell.execute_reply":"2023-03-08T10:30:00.720941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data[['comment_text', 'toxic']]","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.723479Z","iopub.execute_input":"2023-03-08T10:30:00.724474Z","iopub.status.idle":"2023-03-08T10:30:00.748813Z","shell.execute_reply.started":"2023-03-08T10:30:00.724433Z","shell.execute_reply":"2023-03-08T10:30:00.747922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data.sample(len(data)).reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.752926Z","iopub.execute_input":"2023-03-08T10:30:00.755151Z","iopub.status.idle":"2023-03-08T10:30:00.803188Z","shell.execute_reply.started":"2023-03-08T10:30:00.755115Z","shell.execute_reply":"2023-03-08T10:30:00.802221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.807576Z","iopub.execute_input":"2023-03-08T10:30:00.809854Z","iopub.status.idle":"2023-03-08T10:30:00.825419Z","shell.execute_reply.started":"2023-03-08T10:30:00.809808Z","shell.execute_reply":"2023-03-08T10:30:00.824589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = 0\nprint(f\"Toxicity:--> {data.toxic[example]} \\n\\nComment:\\n{data.comment_text[example]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.832337Z","iopub.execute_input":"2023-03-08T10:30:00.834480Z","iopub.status.idle":"2023-03-08T10:30:00.844116Z","shell.execute_reply.started":"2023-03-08T10:30:00.834444Z","shell.execute_reply":"2023-03-08T10:30:00.842963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.848349Z","iopub.execute_input":"2023-03-08T10:30:00.850542Z","iopub.status.idle":"2023-03-08T10:30:00.894953Z","shell.execute_reply.started":"2023-03-08T10:30:00.850492Z","shell.execute_reply":"2023-03-08T10:30:00.894105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Splitting the dataset","metadata":{}},{"cell_type":"code","source":"training_size = 160000\nvalidation_size = 50000\n\n# Split the sentences\ntraining_sentences = pd.DataFrame(data.comment_text[0:training_size])\nvalidation_sentences = pd.DataFrame(data.comment_text[training_size:training_size + validation_size])\ntesting_sentences = pd.DataFrame(data.comment_text[training_size + validation_size:])\n\n# Split the labels\ntraining_labels = pd.DataFrame(data.toxic[0:training_size])\nvalidation_labels = pd.DataFrame(data.toxic[training_size:training_size + validation_size])\ntesting_labels = pd.DataFrame(data.toxic[training_size + validation_size:])","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.898844Z","iopub.execute_input":"2023-03-08T10:30:00.901043Z","iopub.status.idle":"2023-03-08T10:30:00.923399Z","shell.execute_reply.started":"2023-03-08T10:30:00.901006Z","shell.execute_reply":"2023-03-08T10:30:00.922205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Training shape:-->{training_sentences.shape}\")\nprint(f\"Validation shape:-->{validation_sentences.shape}\")\nprint(f\"Testing shape:-->{testing_sentences.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.926226Z","iopub.execute_input":"2023-03-08T10:30:00.927760Z","iopub.status.idle":"2023-03-08T10:30:00.938345Z","shell.execute_reply.started":"2023-03-08T10:30:00.927722Z","shell.execute_reply":"2023-03-08T10:30:00.933700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.940617Z","iopub.execute_input":"2023-03-08T10:30:00.941579Z","iopub.status.idle":"2023-03-08T10:30:00.955012Z","shell.execute_reply.started":"2023-03-08T10:30:00.941537Z","shell.execute_reply":"2023-03-08T10:30:00.953746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.956156Z","iopub.execute_input":"2023-03-08T10:30:00.956423Z","iopub.status.idle":"2023-03-08T10:30:00.967339Z","shell.execute_reply.started":"2023-03-08T10:30:00.956398Z","shell.execute_reply":"2023-03-08T10:30:00.966245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testing_labels.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.968767Z","iopub.execute_input":"2023-03-08T10:30:00.969128Z","iopub.status.idle":"2023-03-08T10:30:00.979149Z","shell.execute_reply.started":"2023-03-08T10:30:00.969092Z","shell.execute_reply":"2023-03-08T10:30:00.978127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Under sampling","metadata":{}},{"cell_type":"code","source":"#from imblearn.under_sampling import RandomUnderSampler\n#rus = RandomUnderSampler(random_state=42)\n#training_sentences, training_labels = rus.fit_resample(training_sentences, training_labels)\n#validation_sentences, validation_labels = rus.fit_resample(validation_sentences, validation_labels)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.981758Z","iopub.execute_input":"2023-03-08T10:30:00.982164Z","iopub.status.idle":"2023-03-08T10:30:00.988505Z","shell.execute_reply.started":"2023-03-08T10:30:00.982130Z","shell.execute_reply":"2023-03-08T10:30:00.987075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert to numpy array\ntraining_sentences = np.squeeze(training_sentences.to_numpy())\nvalidation_sentences = np.squeeze(validation_sentences.to_numpy())\ntesting_sentences = np.squeeze(testing_sentences.to_numpy())\n\n# Split the labels\ntraining_labels = np.squeeze(training_labels.to_numpy())\nvalidation_labels = np.squeeze(validation_labels.to_numpy())\ntesting_labels = np.squeeze(testing_labels.to_numpy())","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.990159Z","iopub.execute_input":"2023-03-08T10:30:00.990626Z","iopub.status.idle":"2023-03-08T10:30:00.998128Z","shell.execute_reply.started":"2023-03-08T10:30:00.990589Z","shell.execute_reply":"2023-03-08T10:30:00.997206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Training shape:-->{training_sentences.shape}\")\nprint(f\"Validation shape:-->{validation_sentences.shape}\")\nprint(f\"Testing shape:-->{testing_sentences.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:00.999742Z","iopub.execute_input":"2023-03-08T10:30:01.000398Z","iopub.status.idle":"2023-03-08T10:30:01.007357Z","shell.execute_reply.started":"2023-03-08T10:30:01.000363Z","shell.execute_reply":"2023-03-08T10:30:01.006073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check the size of the most long sentence in our dataset\nIt will be required when doing padding","metadata":{}},{"cell_type":"code","source":"#The size of my vocabulary or my dictionary\nvocab_size = 100000\n#Where to truncate if a sentence have more word than the max_len\ntrunc_type='post'\n#Where to pad if we have less than the max length for a sentence\npadding_type='post'\n#how to annotate unknown token\noov_tok = \"<OOV>\"\n#The maximum length of a sentence in our dataset\nmax_length = np.max([len(str(x).split()) for x in data.comment_text])","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:30:01.009103Z","iopub.execute_input":"2023-03-08T10:30:01.009976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the Tokenizer class\ntokenizer = Tokenizer(num_words=vocab_size, oov_token=oov_tok)\n\n# Generate the word index dictionary\ntokenizer.fit_on_texts(training_sentences)\nword_index = tokenizer.word_index\n\n# Generate and pad the training sequences\ntraining_sequences = tokenizer.texts_to_sequences(training_sentences)\ntraining_padded = pad_sequences(training_sequences, maxlen=max_length, padding=padding_type, truncating=trunc_type)\n\n# Generate and pad the Validation sequences\nvalidation_sequences = tokenizer.texts_to_sequences(validation_sentences)\nvalidation_padded = pad_sequences(validation_sequences, maxlen=max_length, padding=padding_type, truncating=trunc_type)\n\n\n# Generate and pad the testing sequences\ntesting_sequences = tokenizer.texts_to_sequences(testing_sentences)\ntesting_padded = pad_sequences(testing_sequences, maxlen=max_length, padding=padding_type, truncating=trunc_type)\n\n# Convert the labels lists into numpy arrays\ntraining_labels = np.array(training_labels)\nvalidation_labels = np.array(validation_labels)\ntesting_labels = np.array(testing_labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#IMP DATA FOR CONFIG\nAUTO = tf.data.experimental.AUTOTUNE\n\n\n# Configuration\nEPOCHS = 10\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTM","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((training_padded, training_labels))\n    .repeat()\n    .shuffle(1024)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((validation_padded, validation_labels))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(testing_padded)\n    .batch(BATCH_SIZE)\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\nembedding_dim = 16\nlstm_dim = 16\ndense_dim = 4\n\n# Model Definition with LSTM\nmodel_lstm = tf.keras.Sequential([\n    tf.keras.layers.Embedding(vocab_size, embedding_dim, input_length=max_length),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(lstm_dim)),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(dense_dim, activation='relu'),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(1, activation='sigmoid')\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # Set the training parameters\n    model_lstm.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\n# Print the model summary\nmodel_lstm.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Using callbacks to stop training when a metric is reached","metadata":{}},{"cell_type":"code","source":"class myCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs={}):\n        '''\n        Halts the training after reaching 99 percent accuracy\n\n        Args:\n          epoch (integer) - index of epoch (required but unused in the function definition below)\n          logs (dict) - metric results from the training epoch\n        '''\n            # Check accuracy\n        if(logs.get('accuracy') > 0.98): #challenge stop the training when the accuracy exceed 60%\n\n            # Stop if threshold is met\n            print(\"\\nAccuracy is upper than 0.98 so cancelling training!\")\n            self.model.stop_training = True\nclass DetectOverfittingCallback(tf.keras.callbacks.Callback):\n    def __init__(self, threshold=10):#just for test\n        super(DetectOverfittingCallback, self).__init__()\n        self.threshold = threshold\n\n    def on_epoch_end(self, epoch, logs=None):\n        ratio = logs[\"val_loss\"] / logs[\"loss\"]\n        print(\"Epoch: {}, Val/Train loss ratio: {:.2f}\".format(epoch, ratio))\n\n        if ratio > self.threshold:\n            print(\"Stopping training...\")\n            self.model.stop_training = True\n# Instantiate class\ndetectOverfittingCallback = DetectOverfittingCallback()\ncallbacks = myCallback()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from tensorflow.keras.callbacks import EarlyStopping ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = training_padded.shape[0] // BATCH_SIZE\nEPOCHS = 10\nNUM_EPOCHS = EPOCHS\n\n# Train the model\nhistory_lstm = model_lstm.fit(train_dataset, batch_size = BATCH_SIZE, steps_per_epoch=n_steps, epochs=NUM_EPOCHS, validation_data=valid_dataset, callbacks=[callbacks,detectOverfittingCallback])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the accuracy and loss history\nplot_graphs(history_lstm, 'accuracy')\nplot_graphs(history_lstm, 'loss')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_series(x, y, format=\"-\", start=0, end=None, \n                title=None, xlabel=None, ylabel=None, legend=None ):\n    \"\"\"\n    Visualizes time series data\n\n    Args:\n      x (array of int) - contains values for the x-axis\n      y (array of int or tuple of arrays) - contains the values for the y-axis\n      format (string) - line style when plotting the graph\n      label (string) - tag for the line\n      start (int) - first time step to plot\n      end (int) - last time step to plot\n      title (string) - title of the plot\n      xlabel (string) - label for the x-axis\n      ylabel (string) - label for the y-axis\n      legend (list of strings) - legend for the plot\n    \"\"\"\n\n    # Setup dimensions of the graph figure\n    plt.figure(figsize=(18, 10))\n    \n    # Check if there are more than two series to plot\n    if type(y) is tuple:\n\n      # Loop over the y elements\n      for y_curr in y:\n\n        # Plot the x and current y values\n        plt.plot(x[start:end], y_curr[start:end], format)\n\n    else:\n      # Plot the x and y values\n      plt.plot(x[start:end], y[start:end], format)\n      \n\n    # Label the x-axis\n    plt.xlabel(xlabel)\n\n    # Label the y-axis\n    plt.ylabel(ylabel)\n\n    # Set the legend\n    if legend:\n        plt.legend(legend)\n\n    # Set the title\n    plt.title(title)\n\n    # Overlay a grid on the graph\n    plt.grid(True)\n\n    # Draw the graph on screen\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get mae and loss from history log\nacc=history_lstm.history['accuracy']\nloss=history_lstm.history['loss']\n\nval_acc=history_lstm.history['val_accuracy']\nval_loss=history_lstm.history['val_loss']\n\n# Get number of epochs\nepochs=range(len(loss)) \n\n# Plot mae and loss\nplot_series(\n    x=epochs, \n    y=(acc, loss, val_acc, val_loss), \n    title='Accuracy and Loss', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )\n\n# Only plot the last 80% of the epochs\nzoom_split = int(epochs[-1] * 0.2)\nepochs_zoom = epochs[zoom_split:]\nacc_zoom = acc[zoom_split:]\nloss_zoom = loss[zoom_split:]\nval_acc_zoom = val_acc[zoom_split:]\nval_loss_zoom = val_loss[zoom_split:]\n\n# Plot zoomed mae and loss\nplot_series(\n    x=epochs_zoom, \n    y=(acc_zoom, loss_zoom, val_acc_zoom, val_loss_zoom), \n    title='Accuracy and Loss Zoom', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predictions","metadata":{}},{"cell_type":"code","source":"y_pred = model_lstm.predict(test_dataset)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = [0 if i<0.7 else 1 for i in y_pred]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\ncm = confusion_matrix(testing_labels, y_hat)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install seaborn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.heatmap(cm, annot=True, linewidth=.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(testing_labels, y_hat))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save the model","metadata":{}},{"cell_type":"code","source":"# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nmodel_lstm.save('saved_model/ToxicityLSTMModel.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GRU","metadata":{}},{"cell_type":"code","source":"# Parameters\nembedding_dim = 16\ngru_dim = 16\ndense_dim = 4\n# Model Definition with GRU\nmodel_gru = tf.keras.Sequential([\n    tf.keras.layers.Embedding(vocab_size, embedding_dim, input_length=max_length),\n    tf.keras.layers.Bidirectional(tf.keras.layers.GRU(gru_dim)),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(dense_dim, activation='relu'),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(1, activation='sigmoid')\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # Set the training parameters\n    model_gru.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\n# Print the model summary\nmodel_gru.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = training_padded.shape[0] // BATCH_SIZE\nEPOCHS = 10\nNUM_EPOCHS = EPOCHS\n\n# Train the model\nhistory_gru = model_gru.fit(train_dataset, batch_size = 512, steps_per_epoch=n_steps, epochs=NUM_EPOCHS, validation_data=valid_dataset, callbacks=[callbacks, detectOverfittingCallback])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the accuracy and loss history\nplot_graphs(history_gru, 'accuracy')\nplot_graphs(history_gru, 'loss')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get mae and loss from history log\nacc=history_gru.history['accuracy']\nloss=history_gru.history['loss']\n\nval_acc=history_gru.history['val_accuracy']\nval_loss=history_gru.history['val_loss']\n\n# Get number of epochs\nepochs=range(len(loss)) \n\n# Plot mae and loss\nplot_series(\n    x=epochs, \n    y=(acc, loss, val_acc, val_loss), \n    title='Accuracy and Loss', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )\n\n# Only plot the last 80% of the epochs\nzoom_split = int(epochs[-1] * 0.2)\nepochs_zoom = epochs[zoom_split:]\nacc_zoom = acc[zoom_split:]\nloss_zoom = loss[zoom_split:]\nval_acc_zoom = val_acc[zoom_split:]\nval_loss_zoom = val_loss[zoom_split:]\n\n# Plot zoomed mae and loss\nplot_series(\n    x=epochs_zoom, \n    y=(acc_zoom, loss_zoom, val_acc_zoom, val_loss_zoom), \n    title='Accuracy and Loss Zoom', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model_gru.predict(test_dataset)\ny_hat = [0 if i<0.7 else 1 for i in y_pred]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(testing_labels, y_hat)\nsns.heatmap(cm, annot=True, linewidth=.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(testing_labels, y_hat))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nmodel_gru.save('saved_model/ToxicityGRUModel.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convolutions","metadata":{}},{"cell_type":"code","source":"# Parameters\nembedding_dim = 16\nfilters = 128\nkernel_size = 5\ndense_dim = 6\n\n# Model Definition with Conv1D\nmodel_conv = tf.keras.Sequential([\n    tf.keras.layers.Embedding(vocab_size, embedding_dim, input_length=max_length),\n    tf.keras.layers.Conv1D(filters, kernel_size, activation='relu'),\n    tf.keras.layers.GlobalAveragePooling1D(),\n    tf.keras.layers.Dense(dense_dim, activation='relu'),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(1, activation='sigmoid')\n])\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    # Set the training parameters\n    model_conv.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])\n# Print the model summary\nmodel_conv.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = training_padded.shape[0] // BATCH_SIZE\nEPOCHS = 3\nNUM_EPOCHS = EPOCHS\n\n# Train the model\nhistory_conv = model_conv.fit(train_dataset, batch_size = 512, steps_per_epoch=n_steps, epochs=NUM_EPOCHS, validation_data=valid_dataset, callbacks=[callbacks, detectOverfittingCallback])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the accuracy and loss history\nplot_graphs(history_conv, 'accuracy')\nplot_graphs(history_conv, 'loss')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get mae and loss from history log\nacc=history_conv.history['accuracy']\nloss=history_conv.history['loss']\n\nval_acc=history_conv.history['val_accuracy']\nval_loss=history_conv.history['val_loss']\n\n# Get number of epochs\nepochs=range(len(loss)) \n\n# Plot mae and loss\nplot_series(\n    x=epochs, \n    y=(acc, loss, val_acc, val_loss), \n    title='Accuracy and Loss', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )\n\n# Only plot the last 80% of the epochs\nzoom_split = int(epochs[-1] * 0.2)\nepochs_zoom = epochs[zoom_split:]\nacc_zoom = acc[zoom_split:]\nloss_zoom = loss[zoom_split:]\nval_acc_zoom = val_acc[zoom_split:]\nval_loss_zoom = val_loss[zoom_split:]\n\n# Plot zoomed mae and loss\nplot_series(\n    x=epochs_zoom, \n    y=(acc_zoom, loss_zoom, val_acc_zoom, val_loss_zoom), \n    title='Accuracy and Loss Zoom', \n    xlabel='Epochs',\n    legend=['Accuracy', 'Loss', 'Val Accuracy', 'Val Loss']\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model_conv.predict(test_dataset)\ny_hat = [0 if i<0.7 else 1 for i in y_pred]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(testing_labels, y_hat)\nsns.heatmap(cm, annot=True, linewidth=.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(testing_labels, y_hat))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the entire model as a SavedModel.\n!mkdir -p saved_model\nmodel_conv.save('saved_model/ToxicityConvModel.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Looking at embeddings dimensions","metadata":{}},{"cell_type":"markdown","source":"* `vecs.tsv` - contains the vector weights of each word in the vocabulary\n* `meta.tsv` - contains the words in the vocabulary","metadata":{}},{"cell_type":"code","source":"\"\"\"# Get the index-word dictionary\nreverse_word_index = tokenizer.index_word\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"import io\n\n# Open writeable files\nout_v = io.open('vecs.tsv', 'w', encoding='utf-8')\nout_m = io.open('meta.tsv', 'w', encoding='utf-8')\n\n# Initialize the loop. Start counting at `1` because `0` is just for the padding\nfor word_num in range(1, vocab_size):\n\n  # Get the word associated at the current index\n  word_name = reverse_word_index[word_num]\n\n  # Get the embedding weights associated with the current index\n  word_embedding = embedding_weights[word_num]\n\n  # Write the word name\n  out_m.write(word_name + \"\\n\")\n\n  # Write the word embedding\n  out_v.write('\\t'.join([str(x) for x in word_embedding]) + \"\\n\")\n\n# Close the files\nout_v.close()\nout_m.close()\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now you can go to the [Tensorflow Embedding Projector](https://projector.tensorflow.org/) and load the two files: vecs and meta to visualize the vectors","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}