{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T14:52:20.018073Z","iopub.execute_input":"2022-07-29T14:52:20.018879Z","iopub.status.idle":"2022-07-29T14:52:20.125465Z","shell.execute_reply.started":"2022-07-29T14:52:20.018785Z","shell.execute_reply":"2022-07-29T14:52:20.124667Z"},"trusted":true},"execution_count":1,"outputs":[{"name":"stdout","text":"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test_labels.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation-processed-seqlen128.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test-processed-seqlen128.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train-processed-seqlen128.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\n/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train-processed-seqlen128.csv\n","output_type":"stream"}]},{"cell_type":"code","source":"\n!pip install keras","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:53:34.564038Z","iopub.execute_input":"2022-07-29T14:53:34.5646Z","iopub.status.idle":"2022-07-29T14:53:43.895073Z","shell.execute_reply.started":"2022-07-29T14:53:34.564566Z","shell.execute_reply":"2022-07-29T14:53:43.894196Z"},"trusted":true},"execution_count":2,"outputs":[{"name":"stdout","text":"Collecting keras\n  Downloading keras-2.9.0-py2.py3-none-any.whl (1.6 MB)\n\u001b[K     |████████████████████████████████| 1.6 MB 4.4 MB/s eta 0:00:01\n\u001b[?25hInstalling collected packages: keras\nSuccessfully installed keras-2.9.0\n\u001b[33mWARNING: Running pip as the 'root' user can result in broken permissions and conflicting behaviour with the system package manager. It is recommended to use a virtual environment instead: https://pip.pypa.io/warnings/venv\u001b[0m\n","output_type":"stream"}]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\n#import contractions\nimport re\n#import matplotlib.pyplot as plt\n#from tqdm.notebook import tqdm_notebook\n#from nltk.corpus import stopwords\n#from nltk.corpus import words\n#from nltk.stem import WordNetLemmatizer\n#import keras\n#from keras.preprocessing.text import Tokenizer\n#from keras.preprocessing.sequence import pad_sequences\n#from sklearn.feature_extraction.text import TfidfVectorizer\nimport tensorflow as tf\n#from tensorflow.keras.layers import Dense, Dropout, Input, Concatenate, Average, GlobalAveragePooling1D, GlobalMaxPooling1D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model, load_model\n#from tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping, ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom tensorflow.keras.layers import Input, Dense","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:54:17.884051Z","iopub.execute_input":"2022-07-29T14:54:17.884503Z","iopub.status.idle":"2022-07-29T14:54:17.891392Z","shell.execute_reply.started":"2022-07-29T14:54:17.884466Z","shell.execute_reply":"2022-07-29T14:54:17.890618Z"},"trusted":true},"execution_count":5,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n\"Batched inputs are often different lengths, so they can’t be converted to fixed-size tensors.\n\"Padding and truncation are strategies for dealing with this problem, to create rectangular tensors from batches of varying lengths\"\n    tokenizer.enable_truncation(max_length=maxlen)()\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:55:32.243479Z","iopub.execute_input":"2022-07-29T14:55:32.243737Z","iopub.status.idle":"2022-07-29T14:55:32.250913Z","shell.execute_reply.started":"2022-07-29T14:55:32.24371Z","shell.execute_reply":"2022-07-29T14:55:32.250041Z"},"trusted":true},"execution_count":15,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen = 512 ):\n    enc_di = tokenizer.batch_encode_plus(texts,return_attention_mask = False, return_token_type_ids  = False,\n                                        pad_to_max_length = True, max_length = maxlen, truncation = True)\n    return np.array(enc_di[\"input_ids\"])\n           \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:55:35.053569Z","iopub.execute_input":"2022-07-29T14:55:35.053885Z","iopub.status.idle":"2022-07-29T14:55:35.058853Z","shell.execute_reply.started":"2022-07-29T14:55:35.053855Z","shell.execute_reply":"2022-07-29T14:55:35.05822Z"},"trusted":true},"execution_count":16,"outputs":[]},{"cell_type":"code","source":"import transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:55:44.173664Z","iopub.execute_input":"2022-07-29T14:55:44.173939Z","iopub.status.idle":"2022-07-29T14:55:44.178383Z","shell.execute_reply.started":"2022-07-29T14:55:44.173896Z","shell.execute_reply":"2022-07-29T14:55:44.176699Z"},"trusted":true},"execution_count":17,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer,maxlen = 512):\n    input_word_ids = Input(shape=(maxlen,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:55:48.983778Z","iopub.execute_input":"2022-07-29T14:55:48.984503Z","iopub.status.idle":"2022-07-29T14:55:48.99102Z","shell.execute_reply.started":"2022-07-29T14:55:48.984464Z","shell.execute_reply":"2022-07-29T14:55:48.989901Z"},"trusted":true},"execution_count":19,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:55:56.599041Z","iopub.execute_input":"2022-07-29T14:55:56.599821Z","iopub.status.idle":"2022-07-29T14:55:56.613668Z","shell.execute_reply.started":"2022-07-29T14:55:56.599771Z","shell.execute_reply":"2022-07-29T14:55:56.612143Z"},"trusted":true},"execution_count":20,"outputs":[{"name":"stdout","text":"REPLICAS:  1\n","output_type":"stream"}]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nAUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 152\nMODEL = 'jplu/tf-xlm-roberta-large'","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:56:14.803353Z","iopub.execute_input":"2022-07-29T14:56:14.803978Z","iopub.status.idle":"2022-07-29T14:56:15.190473Z","shell.execute_reply.started":"2022-07-29T14:56:14.803918Z","shell.execute_reply":"2022-07-29T14:56:15.189693Z"},"trusted":true},"execution_count":23,"outputs":[]},{"cell_type":"code","source":"#from transformers import TFAutoModel, AutoTokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:56:18.953862Z","iopub.execute_input":"2022-07-29T14:56:18.954467Z","iopub.status.idle":"2022-07-29T14:56:21.723153Z","shell.execute_reply.started":"2022-07-29T14:56:18.954429Z","shell.execute_reply":"2022-07-29T14:56:21.72221Z"},"trusted":true},"execution_count":24,"outputs":[{"output_type":"display_data","data":{"text/plain":"Downloading:   0%|          | 0.00/513 [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"522613c3ae104f31ab06d13c6bb59c85"}},"metadata":{}},{"output_type":"display_data","data":{"text/plain":"Downloading:   0%|          | 0.00/5.07M [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"9df1fd3f66994d9eb219ae69fc344da9"}},"metadata":{}}]},{"cell_type":"code","source":"'''# Configuration\nMODEL = 'jplu/tf-xlm-roberta-large'\nAUTO = tf.data.experimental.AUTOTUNE\nSEED = 2020\nEPOCHS_1 = 20\nEPOCHS_2 = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\nSHUFFLE = 2048\nVERBOSE = 1'''","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:56:57.129277Z","iopub.execute_input":"2022-07-29T14:56:57.129542Z","iopub.status.idle":"2022-07-29T14:56:57.134753Z","shell.execute_reply.started":"2022-07-29T14:56:57.129516Z","shell.execute_reply":"2022-07-29T14:56:57.133948Z"},"trusted":true},"execution_count":25,"outputs":[{"execution_count":25,"output_type":"execute_result","data":{"text/plain":"\"# Configuration\\nMODEL = 'jplu/tf-xlm-roberta-large'\\nAUTO = tf.data.experimental.AUTOTUNE\\nSEED = 2020\\nEPOCHS_1 = 20\\nEPOCHS_2 = 2\\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\\nMAX_LEN = 192\\nSHUFFLE = 2048\\nVERBOSE = 1\""},"metadata":{}}]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:57:05.924119Z","iopub.execute_input":"2022-07-29T14:57:05.924649Z","iopub.status.idle":"2022-07-29T14:57:05.92814Z","shell.execute_reply.started":"2022-07-29T14:57:05.924615Z","shell.execute_reply":"2022-07-29T14:57:05.927299Z"},"trusted":true},"execution_count":26,"outputs":[]},{"cell_type":"code","source":"train_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nval_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")\nunintended_bias_train = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:57:09.443817Z","iopub.execute_input":"2022-07-29T14:57:09.44442Z","iopub.status.idle":"2022-07-29T14:57:37.564784Z","shell.execute_reply.started":"2022-07-29T14:57:09.444381Z","shell.execute_reply":"2022-07-29T14:57:37.564018Z"},"trusted":true},"execution_count":27,"outputs":[]},{"cell_type":"code","source":"unintended_bias_train.toxic = unintended_bias_train.toxic.round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:57:59.753023Z","iopub.execute_input":"2022-07-29T14:57:59.753568Z","iopub.status.idle":"2022-07-29T14:57:59.919788Z","shell.execute_reply.started":"2022-07-29T14:57:59.753534Z","shell.execute_reply":"2022-07-29T14:57:59.919041Z"},"trusted":true},"execution_count":30,"outputs":[]},{"cell_type":"code","source":"unintended_bias_train.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:05.533444Z","iopub.execute_input":"2022-07-29T14:58:05.533718Z","iopub.status.idle":"2022-07-29T14:58:05.557596Z","shell.execute_reply.started":"2022-07-29T14:58:05.533692Z","shell.execute_reply":"2022-07-29T14:58:05.556836Z"},"trusted":true},"execution_count":31,"outputs":[{"execution_count":31,"output_type":"execute_result","data":{"text/plain":"0    1789968\n1     112226\nName: toxic, dtype: int64"},"metadata":{}}]},{"cell_type":"code","source":"# Combine train1 with a subset of train2\ntrain = pd.concat([\n    train_data[['comment_text', 'toxic']],\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==1'),\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==0').sample(n=50000, random_state=0)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:11.523891Z","iopub.execute_input":"2022-07-29T14:58:11.524571Z","iopub.status.idle":"2022-07-29T14:58:12.041005Z","shell.execute_reply.started":"2022-07-29T14:58:11.524541Z","shell.execute_reply":"2022-07-29T14:58:12.040135Z"},"trusted":true},"execution_count":32,"outputs":[]},{"cell_type":"code","source":"import numpy as np \n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:19.988425Z","iopub.execute_input":"2022-07-29T14:58:19.988703Z","iopub.status.idle":"2022-07-29T14:58:19.992395Z","shell.execute_reply.started":"2022-07-29T14:58:19.988676Z","shell.execute_reply":"2022-07-29T14:58:19.991305Z"},"trusted":true},"execution_count":33,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n        #return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:25.877311Z","iopub.execute_input":"2022-07-29T14:58:25.878012Z","iopub.status.idle":"2022-07-29T14:58:25.885088Z","shell.execute_reply.started":"2022-07-29T14:58:25.877978Z","shell.execute_reply":"2022-07-29T14:58:25.884061Z"},"trusted":true},"execution_count":34,"outputs":[]},{"cell_type":"code","source":"type(test_data.content.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:37.928871Z","iopub.execute_input":"2022-07-29T14:58:37.929435Z","iopub.status.idle":"2022-07-29T14:58:37.936444Z","shell.execute_reply.started":"2022-07-29T14:58:37.929399Z","shell.execute_reply":"2022-07-29T14:58:37.935501Z"},"trusted":true},"execution_count":35,"outputs":[{"execution_count":35,"output_type":"execute_result","data":{"text/plain":"numpy.ndarray"},"metadata":{}}]},{"cell_type":"code","source":"%%time \n\nx_train = regular_encode(list(train.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(list(val_data.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_test = regular_encode(list(test_data.content.values), tokenizer, maxlen=MAX_LEN)\n\n#y_train = train_data.toxic.values\ny_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2022-07-29T14:58:41.738417Z","iopub.execute_input":"2022-07-29T14:58:41.738974Z","iopub.status.idle":"2022-07-29T15:00:58.800794Z","shell.execute_reply.started":"2022-07-29T14:58:41.738905Z","shell.execute_reply":"2022-07-29T15:00:58.799776Z"},"trusted":true},"execution_count":36,"outputs":[{"name":"stderr","text":"Truncation was not explicitly activated but `max_length` is provided a specific value, please use `truncation=True` to explicitly truncate examples to max length. Defaulting to 'longest_first' truncation strategy. If you encode pairs of sequences (GLUE-style) with the tokenizer you can select this strategy more precisely by providing a specific strategy to `truncation`.\n","output_type":"stream"},{"name":"stdout","text":"CPU times: user 3min 42s, sys: 6.14 s, total: 3min 48s\nWall time: 2min 17s\n","output_type":"stream"}]},{"cell_type":"code","source":"y_train = train.toxic.values\n#y_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2022-07-29T15:04:03.228895Z","iopub.execute_input":"2022-07-29T15:04:03.229196Z","iopub.status.idle":"2022-07-29T15:04:03.233365Z","shell.execute_reply.started":"2022-07-29T15:04:03.229168Z","shell.execute_reply":"2022-07-29T15:04:03.232396Z"},"trusted":true},"execution_count":37,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T15:04:09.543807Z","iopub.execute_input":"2022-07-29T15:04:09.544094Z","iopub.status.idle":"2022-07-29T15:04:11.476235Z","shell.execute_reply.started":"2022-07-29T15:04:09.544065Z","shell.execute_reply":"2022-07-29T15:04:11.475481Z"},"trusted":true},"execution_count":38,"outputs":[{"name":"stderr","text":"2022-07-29 15:04:09.554288: I tensorflow/compiler/jit/xla_cpu_device.cc:41] Not creating XLA devices, tf_xla_enable_xla_devices not set\n2022-07-29 15:04:09.562110: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcuda.so.1\n2022-07-29 15:04:09.627286: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.628055: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1720] Found device 0 with properties: \npciBusID: 0000:00:04.0 name: Tesla P100-PCIE-16GB computeCapability: 6.0\ncoreClock: 1.3285GHz coreCount: 56 deviceMemorySize: 15.90GiB deviceMemoryBandwidth: 681.88GiB/s\n2022-07-29 15:04:09.628123: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcudart.so.11.0\n2022-07-29 15:04:09.660412: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublas.so.11\n2022-07-29 15:04:09.660510: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublasLt.so.11\n2022-07-29 15:04:09.684787: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcufft.so.10\n2022-07-29 15:04:09.712322: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcurand.so.10\n2022-07-29 15:04:09.744434: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcusolver.so.10\n2022-07-29 15:04:09.752745: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcusparse.so.11\n2022-07-29 15:04:09.756888: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcudnn.so.8\n2022-07-29 15:04:09.757111: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.757910: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.759543: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1862] Adding visible gpu devices: 0\n2022-07-29 15:04:09.762793: I tensorflow/core/platform/cpu_feature_guard.cc:142] This TensorFlow binary is optimized with oneAPI Deep Neural Network Library (oneDNN) to use the following CPU instructions in performance-critical operations:  AVX2 AVX512F FMA\nTo enable them in other operations, rebuild TensorFlow with the appropriate compiler flags.\n2022-07-29 15:04:09.763596: I tensorflow/compiler/jit/xla_gpu_device.cc:99] Not creating XLA devices, tf_xla_enable_xla_devices not set\n2022-07-29 15:04:09.763790: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.764543: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1720] Found device 0 with properties: \npciBusID: 0000:00:04.0 name: Tesla P100-PCIE-16GB computeCapability: 6.0\ncoreClock: 1.3285GHz coreCount: 56 deviceMemorySize: 15.90GiB deviceMemoryBandwidth: 681.88GiB/s\n2022-07-29 15:04:09.764595: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcudart.so.11.0\n2022-07-29 15:04:09.764635: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublas.so.11\n2022-07-29 15:04:09.764654: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublasLt.so.11\n2022-07-29 15:04:09.764672: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcufft.so.10\n2022-07-29 15:04:09.764690: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcurand.so.10\n2022-07-29 15:04:09.764707: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcusolver.so.10\n2022-07-29 15:04:09.764726: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcusparse.so.11\n2022-07-29 15:04:09.764744: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcudnn.so.8\n2022-07-29 15:04:09.764841: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.765524: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:09.766115: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1862] Adding visible gpu devices: 0\n2022-07-29 15:04:09.767628: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcudart.so.11.0\n2022-07-29 15:04:11.301015: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1261] Device interconnect StreamExecutor with strength 1 edge matrix:\n2022-07-29 15:04:11.301066: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1267]      0 \n2022-07-29 15:04:11.301077: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1280] 0:   N \n2022-07-29 15:04:11.303088: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:11.303820: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:11.304557: I tensorflow/stream_executor/cuda/cuda_gpu_executor.cc:941] successful NUMA node read from SysFS had negative value (-1), but there must be at least one NUMA node, so returning NUMA node zero\n2022-07-29 15:04:11.305177: I tensorflow/core/common_runtime/gpu/gpu_device.cc:1406] Created TensorFlow device (/job:localhost/replica:0/task:0/device:GPU:0 with 14957 MB memory) -> physical GPU (device: 0, name: Tesla P100-PCIE-16GB, pci bus id: 0000:00:04.0, compute capability: 6.0)\n2022-07-29 15:04:11.315188: W tensorflow/core/framework/cpu_allocator_impl.cc:80] Allocation of 469102400 exceeds 10% of free system memory.\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL)\n    model = build_model(transformer_layer, maxlen=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T15:04:29.604177Z","iopub.execute_input":"2022-07-29T15:04:29.604642Z","iopub.status.idle":"2022-07-29T15:06:17.345784Z","shell.execute_reply.started":"2022-07-29T15:04:29.604608Z","shell.execute_reply":"2022-07-29T15:06:17.345079Z"},"trusted":true},"execution_count":39,"outputs":[{"output_type":"display_data","data":{"text/plain":"Downloading:   0%|          | 0.00/3.27G [00:00<?, ?B/s]","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"09fd9b5e5c8741eba3d427577040176c"}},"metadata":{}},{"name":"stderr","text":"2022-07-29 15:05:52.850986: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublas.so.11\n2022-07-29 15:05:53.755228: I tensorflow/stream_executor/platform/default/dso_loader.cc:49] Successfully opened dynamic library libcublasLt.so.11\n2022-07-29 15:06:08.128097: W tensorflow/core/framework/cpu_allocator_impl.cc:80] Allocation of 1024008192 exceeds 10% of free system memory.\nSome layers from the model checkpoint at jplu/tf-xlm-roberta-large were not used when initializing TFXLMRobertaModel: ['lm_head']\n- This IS expected if you are initializing TFXLMRobertaModel from the checkpoint of a model trained on another task or with another architecture (e.g. initializing a BertForSequenceClassification model from a BertForPreTraining model).\n- This IS NOT expected if you are initializing TFXLMRobertaModel from the checkpoint of a model that you expect to be exactly identical (initializing a BertForSequenceClassification model from a BertForSequenceClassification model).\nAll the layers of TFXLMRobertaModel were initialized from the model checkpoint at jplu/tf-xlm-roberta-large.\nIf your task is similar to the task the model of the checkpoint was trained on, you can already use TFXLMRobertaModel for predictions without further training.\n","output_type":"stream"},{"name":"stdout","text":"Model: \"model\"\n_________________________________________________________________\nLayer (type)                 Output Shape              Param #   \n=================================================================\ninput_word_ids (InputLayer)  [(None, 152)]             0         \n_________________________________________________________________\ntfxlm_roberta_model (TFXLMRo TFBaseModelOutputWithPool 559890432 \n_________________________________________________________________\ntf.__operators__.getitem (Sl (None, 1024)              0         \n_________________________________________________________________\ndense (Dense)                (None, 1)                 1025      \n=================================================================\nTotal params: 559,891,457\nTrainable params: 559,891,457\nNon-trainable params: 0\n_________________________________________________________________\nCPU times: user 1min 15s, sys: 10.9 s, total: 1min 26s\nWall time: 1min 47s\n","output_type":"stream"}]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" y_t = model.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:47:49.70496Z","iopub.execute_input":"2021-11-28T04:47:49.705317Z","iopub.status.idle":"2021-11-28T04:49:00.336763Z","shell.execute_reply.started":"2021-11-28T04:47:49.705269Z","shell.execute_reply":"2021-11-28T04:49:00.335358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:09.086741Z","iopub.execute_input":"2021-11-28T04:49:09.087459Z","iopub.status.idle":"2021-11-28T04:49:09.151551Z","shell.execute_reply.started":"2021-11-28T04:49:09.087405Z","shell.execute_reply":"2021-11-28T04:49:09.150769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({\"id\": s.id , \"toxic\":y_t.squeeze()}, index = None)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:12.734429Z","iopub.execute_input":"2021-11-28T04:49:12.735326Z","iopub.status.idle":"2021-11-28T04:49:12.742664Z","shell.execute_reply.started":"2021-11-28T04:49:12.735255Z","shell.execute_reply":"2021-11-28T04:49:12.74182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:50.288873Z","iopub.execute_input":"2021-11-27T22:06:50.289183Z","iopub.status.idle":"2021-11-27T22:06:50.502459Z","shell.execute_reply.started":"2021-11-27T22:06:50.289154Z","shell.execute_reply":"2021-11-27T22:06:50.501491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:42.579983Z","iopub.execute_input":"2021-11-27T22:06:42.58029Z","iopub.status.idle":"2021-11-27T22:06:42.599863Z","shell.execute_reply.started":"2021-11-27T22:06:42.580259Z","shell.execute_reply":"2021-11-27T22:06:42.598756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}