{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip uninstall transformers -y\n!pip install transformers\n!pip install loguru","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nglobal user-defined variables\n\"\"\"\n\nrandom_state: int = 0\n\nIN_KAGGLE = True\n\ndata_folder: str = 'data'\nraw_data_folder: str = f'{data_folder}/raw_data'\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nutils functions (utils.py)\n\"\"\"\n\nfrom os.path import abspath, join\nfrom pathlib import Path\nfrom sklearn.metrics import roc_curve, auc\nfrom typing import Any\nimport tensorflow as tf\n#from variables import raw_data_folder, IN_KAGGLE\n\ndefault_base_folder: str = raw_data_folder if not IN_KAGGLE else 'jigsaw-multilingual-toxic-comment-classification'\n\n\ndef file_path_relative(rel_path: str, base_folder: str = default_base_folder) -> str:\n    \"\"\"\n    get file path relative to base folder\n    \"\"\"\n    if IN_KAGGLE:\n        current_path = '/kaggle/input/'\n    else:\n        current_path = join(Path(__file__).absolute(), '../../')\n    return join(abspath(current_path), base_folder, rel_path)\n\n\ndef roc_auc(predictions, target):\n    \"\"\"\n    This methods returns the AUC Score when given the Predictions\n    and Labels\n    \"\"\"\n\n    fpr, tpr, _thresholds = roc_curve(target, predictions[0: len(target)])\n    return auc(fpr, tpr)\n\n\ndef build_model(transformer: Any, max_len: int) -> tf.keras.Model:\n    \"\"\"\n    function for building a model given a transformer and max length\n    \"\"\"\n    input_word_ids = tf.keras.Input(\n        shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = tf.keras.layers.Dense(1, activation='sigmoid')(cls_token)\n\n    model = tf.keras.Model(inputs=input_word_ids, outputs=out)\n    model.compile(tf.keras.optimizers.Adam(lr=1e-5), loss='binary_crossentropy',\n                  metrics=['accuracy'])\n\n    return model\n\ndef plot_train_val_loss(history: tf.keras.callbacks.History, model_name: str) -> None:\n    \"\"\"\n    plots the training and validation loss given training history\n    \"\"\"\n\n    plt.figure()\n\n    print(history.history)\n    loss_train = history.history['loss']\n    loss_val = history.history['loss']\n\n    num_epochs = len(loss_train)\n    nums = range(1, num_epochs + 1)\n\n    plt.plot(nums, loss_train, label=\"train\")\n    plt.plot(nums, loss_val, label=\"validation\")\n    plt.title(f\"Training and Validation loss over {num_epochs} epochs\")\n    plt.xlabel(\"Epoch #\")\n    plt.ylabel(\"Loss\")\n    plt.legend()\n    file_path = file_path_relative(\n        f'{output_folder}/{model_name}.png')\n    plt.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ls","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\ndata file\n\nread in data\n\"\"\"\n\nimport pandas as pd\nimport tensorflow as tf\nfrom loguru import logger\nfrom typing import List, Tuple, Dict\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport numpy as np\n# from variables import raw_data_folder\n# from utils import file_path_relative\n\n\ndef fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n\n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n\n    return np.array(all_ids)\n\n\nNUM_ROWS_TRAIN: int = 15000\nTEST_RATIO: float = 0.2\n\n\ndef read_data() -> Tuple[np.array, np.array, np.array, np.array, int, Dict[str, int]]:\n    \"\"\"\n    read data from raw data, convert to dataframes\n    \"\"\"\n    logger.info('reading data')\n\n    train = pd.read_csv(file_path_relative('jigsaw-toxic-comment-train.csv'))\n\n    # drop unused columns\n    train.drop(['severe_toxic', 'obscene', 'threat', 'insult',\n                'identity_hate'], axis=1, inplace=True)\n\n    # only use first n rows\n    train = train.loc[:NUM_ROWS_TRAIN, :]\n    logger.info(f'shape of training data: {train.shape}')\n\n    max_len = train['comment_text'].apply(\n        lambda x: len(str(x).split())).max()\n    logger.info(f'max len: {max_len}')\n\n    x_train, x_valid, y_train, y_valid = train_test_split(train['comment_text'].values, train['toxic'].values,\n                                                          stratify=train['toxic'].values,\n                                                          test_size=TEST_RATIO, shuffle=True)\n\n    tokens = tf.keras.preprocessing.text.Tokenizer(num_words=None)\n\n    all_data: List[str] = list(x_train)\n    all_data.extend(list(x_valid))\n    tokens.fit_on_texts(all_data)\n    x_train_sequences = tokens.texts_to_sequences(x_train)\n    x_valid_sequences = tokens.texts_to_sequences(x_valid)\n\n    # pad the data with zeros\n    x_train_padded = tf.keras.preprocessing.sequence.pad_sequences(\n        x_train_sequences, maxlen=max_len)\n    x_valid_padded = tf.keras.preprocessing.sequence.pad_sequences(\n        x_valid_sequences, maxlen=max_len)\n\n    word_indexes = tokens.word_index\n\n    return x_train_padded, x_valid_padded, y_train, y_valid, max_len, word_indexes\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run data on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\ndata attention file\n\nread in data for attention\n\"\"\"\n\nfrom typing import Tuple, List, Any\nimport pandas as pd\nimport tensorflow as tf\nfrom loguru import logger\nfrom tqdm import tqdm\nimport numpy as np\nfrom transformers import DistilBertTokenizer\n# from utils import file_path_relative\n# from variables import raw_data_folder\n\nNUM_ROWS_TRAIN: int = 15000\nTEST_RATIO: float = 0.2\n\n\ndef _run_encode(texts: np.array, tokenizer: Any, maxlen: int = 512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for transformer Input\n    \"\"\"\n    logger.info('encode')\n    encodings = tokenizer(\n        texts.tolist(),\n        return_token_type_ids=False,\n        padding='max_length',\n        truncation=True,\n        max_length=maxlen\n    )\n\n    return np.array(encodings['input_ids'])\n\n\ndef read_data_attention(strategy: Any,\n                        max_len: int,\n                        ) -> Tuple[np.array, np.array, np.array, np.array, tf.data.Dataset, tf.data.Dataset, tf.data.Dataset, int]:\n    \"\"\"\n    read data from attention models\n    \"\"\"\n    logger.info('reading data for attention models')\n\n    batch_size = 16 * strategy.num_replicas_in_sync\n    auto = tf.data.experimental.AUTOTUNE\n\n    # First load the tokenizer\n    tokenizer = DistilBertTokenizer.from_pretrained(\n        'distilbert-base-multilingual-cased')\n\n    train = pd.read_csv(file_path_relative('jigsaw-toxic-comment-train.csv'))\n    valid = pd.read_csv(file_path_relative('validation.csv'))\n    test = pd.read_csv(file_path_relative('test.csv'))\n\n    x_train = _run_encode(train['comment_text'].astype(str),\n                          tokenizer, maxlen=max_len)\n    x_valid = _run_encode(valid['comment_text'].astype(str),\n                          tokenizer, maxlen=max_len)\n    x_test = _run_encode(test['content'].astype(\n        str), tokenizer, maxlen=max_len)\n\n    y_train = train['toxic'].values\n    y_valid = valid['toxic'].values\n\n    train_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((x_train, y_train))\n        .repeat()\n        .shuffle(2048)\n        .batch(batch_size)\n        .prefetch(auto)\n    )\n\n    valid_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((x_valid, y_valid))\n        .batch(batch_size)\n        .cache()\n        .prefetch(auto)\n    )\n\n    test_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices(x_test)\n        .batch(batch_size)\n    )\n\n    return x_train, x_valid, y_train, y_valid, train_dataset, valid_dataset, \\\n        test_dataset, batch_size\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run data attention on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\ndistilibert file\n\nrun distilibert on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom transformers import TFDistilBertModel\nfrom typing import Any\n# from utils import roc_auc, build_model\n\nMODEL: str = 'distilbert-base-multilingual-cased'\n\n\ndef run_distilibert(strategy: Any, x_train: np.array,\n                    x_valid: np.array, _y_train: np.array, y_valid: np.array,\n                    train_dataset: tf.data.Dataset, valid_dataset: tf.data.Dataset,\n                    test_dataset: tf.data.Dataset, max_len: int, epochs: int,\n                    batch_size: int) -> tf.keras.models.Model:\n    \"\"\"\n    create and run distilibert on training and testing data\n    \"\"\"\n    logger.info('build distilibert')\n\n    with strategy.scope():\n        transformer_layer = TFDistilBertModel.from_pretrained(MODEL)\n        model = build_model(transformer_layer, max_len=max_len)\n    model.summary()\n\n    n_steps = x_train.shape[0] // batch_size\n    _train_history = model.fit(\n        train_dataset,\n        steps_per_epoch=n_steps,\n        validation_data=valid_dataset,\n        epochs=epochs\n    )\n\n    n_steps = x_valid.shape[0] // batch_size\n    _train_history_2 = model.fit(\n        valid_dataset.repeat(),\n        steps_per_epoch=n_steps,\n        epochs=epochs*2\n    )\n\n    scores = model.predict(test_dataset, verbose=1)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run distilibert on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nbuild embeddings\n\nbuild embeddings for all data\n\"\"\"\n\nimport numpy as np\nfrom loguru import logger\nfrom typing import Dict\nfrom tqdm import tqdm\n# from utils import file_path_relative, default_base_folder\n# from variables import IN_KAGGLE\n\n\ndef build_embeddings(embedding_size_y: int, word_indexes: Dict[str, int]) -> np.array:\n    \"\"\"\n    build embeddings to be used with models\n    \"\"\"\n    logger.info('build glove embeddings')\n\n    embeddings_indexes: Dict[str, np.array] = {}\n    with open(file_path_relative(f'glove.840B.{embedding_size_y}d.txt',\n                                 base_folder=default_base_folder if not IN_KAGGLE else 'glove840b300dtxt'),\n              encoding='utf-8') as glove_file:\n        for line in tqdm(glove_file):\n            words = line.split(' ')\n            word = words[0]\n            coefficients = np.asarray([float(val) for val in words[1:]])\n            embeddings_indexes[word] = coefficients\n\n    logger.info(f'Found {len(embeddings_indexes)} word vectors.')\n\n    embedding_size_x: int = len(word_indexes) + 1\n\n    embeddings_output = np.zeros((embedding_size_x, embedding_size_y))\n    for word, i in tqdm(word_indexes.items()):\n        word_embedding = embeddings_indexes.get(word)\n        if word_embedding is not None:\n            embeddings_output[i] = word_embedding\n\n    return embeddings_output\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run embeddings on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\ngru file\n\nrun gru on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom typing import Any\n# from utils import roc_auc\n\n\ndef run_gru(strategy: Any, x_train_padded: np.array,\n            x_valid_padded: np.array, y_train: np.array, y_valid: np.array,\n            max_len: int, embedding_size_x: int, embedding_size_y: int,\n            embedding_matrix: np.array) -> tf.keras.models.Sequential:\n    \"\"\"\n    create and run gru on training and testing data\n    \"\"\"\n    logger.info('build gru')\n\n    with strategy.scope():\n        # GRU with glove embeddings and two dense layers\n        model = tf.keras.models.Sequential()\n        model.add(tf.keras.layers.Embedding(embedding_size_x,\n                                            embedding_size_y,\n                                            weights=[embedding_matrix],\n                                            input_length=max_len,\n                                            trainable=False))\n        model.add(tf.keras.layers.SpatialDropout1D(0.3))\n        model.add(tf.keras.layers.GRU(embedding_size_y))\n        model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n\n        model.compile(loss='binary_crossentropy',\n                      optimizer='adam', metrics=['accuracy'])\n\n    model.summary()\n\n    model.fit(x_train_padded, y_train, batch_size=64*strategy.num_replicas_in_sync)\n\n    scores = model.predict(x_valid_padded)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run gru on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nlstm file\n\nrun lstm on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom typing import Any\n# from utils import roc_auc, plot_train_val_loss\n\n\ndef run_lstm(strategy: Any, x_train_padded: np.array,\n             x_valid_padded: np.array, y_train: np.array, y_valid: np.array,\n             max_len: int, embedding_size_x: int, embedding_size_y: int,\n             embedding_matrix: np.array) -> tf.keras.models.Sequential:\n    \"\"\"\n    create and run lstm on training and testing data\n    \"\"\"\n    logger.info('build lstm')\n\n    with strategy.scope():\n        # A simple LSTM with glove embeddings and one dense layer\n        model = tf.keras.models.Sequential()\n        model.add(tf.keras.layers.Embedding(embedding_size_x,\n                                            embedding_size_y,\n                                            weights=[embedding_matrix],\n                                            input_length=max_len,\n                                            trainable=False))\n\n        model.add(tf.keras.layers.LSTM(\n            100, dropout=0.3, recurrent_dropout=0.3))\n        model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n        model.compile(loss='binary_crossentropy',\n                      optimizer='adam', metrics=['accuracy'])\n\n    model.summary()\n\n    history = model.fit(x_train_padded, y_train, batch_size=64*strategy.num_replicas_in_sync)\n    plot_train_val_loss(history, 'lstm')\n\n    scores = model.predict(x_valid_padded)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run lstm on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nrnn file\n\nrun rnn on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom typing import Any\n# from utils import roc_auc\n\n\ndef run_rnn(strategy: Any, x_train_padded: np.array,\n            x_valid_padded: np.array, y_train: np.array, y_valid: np.array,\n            max_len: int, embedding_size_x: int, embedding_size_y: int,\n            embedding_matrix: np.array) -> tf.keras.models.Sequential:\n    \"\"\"\n    create and run bidirectional rnn on training and testing data\n    \"\"\"\n    logger.info('build rnn')\n\n    with strategy.scope():\n        # A simple bidirectional LSTM with glove embeddings and one dense layer\n        model = tf.keras.models.Sequential()\n        model.add(tf.keras.layers.Embedding(embedding_size_x,\n                                            embedding_size_y,\n                                            weights=[embedding_matrix],\n                                            input_length=max_len,\n                                            trainable=False))\n        model.add(tf.keras.layers.Bidirectional(\n            tf.keras.layers.LSTM(embedding_size_y, dropout=0.3, recurrent_dropout=0.3)))\n\n        model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n        model.compile(loss='binary_crossentropy',\n                      optimizer='adam', metrics=['accuracy'])\n\n    model.summary()\n\n    model.fit(x_train_padded, y_train, batch_size=64*strategy.num_replicas_in_sync)\n\n    scores = model.predict(x_valid_padded)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run rnn on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nroberta file\n\nrun roberta on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom transformers import TFAutoModel\nfrom typing import Any\n# from utils import roc_auc, build_model\n\nMODEL: str = 'jplu/tf-xlm-roberta-large'\n\n\ndef run_roberta(strategy: Any, x_train: np.array,\n                x_valid: np.array, _y_train: np.array, y_valid: np.array,\n                train_dataset: tf.data.Dataset, valid_dataset: tf.data.Dataset,\n                test_dataset: tf.data.Dataset, max_len: int, epochs: int,\n                batch_size: int) -> tf.keras.models.Model:\n    \"\"\"\n    create and run distilibert on training and testing data\n    \"\"\"\n    logger.info('build roberta')\n\n    with strategy.scope():\n        transformer_layer = TFAutoModel.from_pretrained(MODEL)\n        model = build_model(transformer_layer, max_len=max_len)\n    model.summary()\n\n    n_steps = x_train.shape[0] // batch_size\n    _train_history = model.fit(\n        train_dataset,\n        steps_per_epoch=n_steps,\n        validation_data=valid_dataset,\n        epochs=epochs\n    )\n\n    n_steps = x_valid.shape[0] // batch_size\n    _train_history_2 = model.fit(\n        valid_dataset.repeat(),\n        steps_per_epoch=n_steps,\n        epochs=epochs\n    )\n\n    scores = model.predict(test_dataset, verbose=1)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run roberta on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nsimple rnn file\n\nrun simple rnn on dataset\n\"\"\"\n\nimport tensorflow as tf\nimport numpy as np\nfrom loguru import logger\nfrom typing import Any\n# from utils import roc_auc\n\n\ndef simple_rnn(strategy: Any, x_train_padded: np.array,\n               x_valid_padded: np.array, y_train: np.array, y_valid: np.array,\n               max_len: int, embedding_size_x: int, embedding_size_y: int) -> tf.keras.models.Sequential:\n    \"\"\"\n    create and run simple rnn on training and testing data\n    \"\"\"\n    logger.info('build simple RNN')\n\n    with strategy.scope():\n        # A simpleRNN without any pretrained embeddings and one dense layer\n        model = tf.keras.models.Sequential()\n        model.add(tf.keras.layers.Embedding(embedding_size_x, embedding_size_y,\n                                            input_length=max_len))\n        model.add(tf.keras.layers.SimpleRNN(100))\n        model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n        model.compile(loss='binary_crossentropy',\n                      optimizer='adam', metrics=['accuracy'])\n\n    model.summary()\n\n    model.fit(x_train_padded, y_train, batch_size=64 *\n              strategy.num_replicas_in_sync)\n\n    scores = model.predict(x_valid_padded)\n    logger.info(f\"AUC: {roc_auc(scores, y_valid):.4f}\")\n\n    return model\n\n\nif __name__ == '__main__':\n    # raise RuntimeError('cannot run simple rnn on its own')\n    pass\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nmain file\n\nentry point for running final project\n\"\"\"\n\nimport numpy as np\nimport tensorflow as tf\nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom loguru import logger\nfrom typing import Any\n# from variables import random_state\n# from data import read_data\n# from data_attention import read_data_attention\n# from embeddings import build_embeddings\n# from simple_rnn import simple_rnn\n# from lstm import run_lstm\n# from gru import run_gru\n# from rnn import run_rnn\n# from distilibert import run_distilibert\n# from roberta import run_roberta\n\nEMBEDDING_SIZE_Y: int = 300\n# epochs for attention models\nEPOCHS: int = 1\n\n\ndef initialize() -> Any:\n    \"\"\"\n    initialize before running anything\n    \"\"\"\n    tf.random.set_seed(random_state)\n    random.seed(random_state)\n    np.random.seed(random_state)\n    sns.set_style('whitegrid')\n    plt.style.use('fivethirtyeight')\n\n    try:\n        # TPU detection\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        logger.info(f'Running on TPU {tpu.master()}')\n    except ValueError:\n        tpu = None\n\n    if tpu:\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    else:\n        strategy = tf.distribute.get_strategy()\n\n    logger.info(f'Num Replicas: {strategy.num_replicas_in_sync}')\n\n    # show if there is a GPU. this will allow for faster training\n    logger.info(\n        f\"Num GPUs Available: {len(tf.config.experimental.list_physical_devices('GPU'))}\")\n\n    return strategy\n\n\nstrategy = initialize()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# read in the data\nx_train_padded, x_valid_padded, y_train, y_valid, max_len, word_indexes = read_data()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"simple_rnn(strategy, x_train_padded, x_valid_padded, y_train,\n            y_valid, max_len, len(word_indexes) + 1, EMBEDDING_SIZE_Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings_output = build_embeddings(EMBEDDING_SIZE_Y, word_indexes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run_lstm(strategy, x_train_padded, x_valid_padded, y_train, y_valid,\n            max_len, len(word_indexes) + 1, EMBEDDING_SIZE_Y, embeddings_output)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run_gru(strategy, x_train_padded, x_valid_padded, y_train, y_valid,\n        max_len, len(word_indexes) + 1, EMBEDDING_SIZE_Y, embeddings_output)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run_rnn(strategy, x_train_padded, x_valid_padded, y_train, y_valid,\n        max_len, len(word_indexes) + 1, EMBEDDING_SIZE_Y, embeddings_output)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"attention_max_len = 192","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_valid, y_train, y_valid, train_dataset, \\\n    valid_dataset, test_dataset, batch_size = read_data_attention(\n        strategy, attention_max_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run_distilibert(strategy, x_train, x_valid, y_train, y_valid,\n                train_dataset, valid_dataset, test_dataset, attention_max_len, EPOCHS, batch_size)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_valid, y_train, y_valid, train_dataset, \\\n    valid_dataset, test_dataset, batch_size = read_data_attention(\n        strategy, attention_max_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run_roberta(strategy, x_train, x_valid, y_train, y_valid,\n            train_dataset, valid_dataset, test_dataset, attention_max_len, EPOCHS, batch_size)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}