{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-28T04:08:15.508046Z","iopub.execute_input":"2021-11-28T04:08:15.508460Z","iopub.status.idle":"2021-11-28T04:08:15.515826Z","shell.execute_reply.started":"2021-11-28T04:08:15.508427Z","shell.execute_reply":"2021-11-28T04:08:15.515156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install keras","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:08:29.865689Z","iopub.execute_input":"2021-11-28T04:08:29.866250Z","iopub.status.idle":"2021-11-28T04:08:41.077085Z","shell.execute_reply.started":"2021-11-28T04:08:29.866213Z","shell.execute_reply":"2021-11-28T04:08:41.076247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\n#import contractions\nimport re\n#import matplotlib.pyplot as plt\n#from tqdm.notebook import tqdm_notebook\n#from nltk.corpus import stopwords\n#from nltk.corpus import words\n#from nltk.stem import WordNetLemmatizer\n#import keras\n#from keras.preprocessing.text import Tokenizer\n#from keras.preprocessing.sequence import pad_sequences\n#from sklearn.feature_extraction.text import TfidfVectorizer\nimport tensorflow as tf\n#from tensorflow.keras.layers import Dense, Dropout, Input, Concatenate, Average, GlobalAveragePooling1D, GlobalMaxPooling1D\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model, load_model\n#from tensorflow.keras.callbacks import ReduceLROnPlateau, LearningRateScheduler, EarlyStopping, ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom tensorflow.keras.layers import Input, Dense","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:08:41.079585Z","iopub.execute_input":"2021-11-28T04:08:41.079953Z","iopub.status.idle":"2021-11-28T04:08:49.606227Z","shell.execute_reply.started":"2021-11-28T04:08:41.079906Z","shell.execute_reply":"2021-11-28T04:08:49.605222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''train_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nval_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")\nunintended_bias_train = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\n'''","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:08:49.607809Z","iopub.execute_input":"2021-11-28T04:08:49.608053Z","iopub.status.idle":"2021-11-28T04:08:49.617534Z","shell.execute_reply.started":"2021-11-28T04:08:49.608024Z","shell.execute_reply":"2021-11-28T04:08:49.616426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#display(train_data[\"comment_text\"][10])","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:08:49.619966Z","iopub.execute_input":"2021-11-28T04:08:49.620299Z","iopub.status.idle":"2021-11-28T04:08:49.628649Z","shell.execute_reply.started":"2021-11-28T04:08:49.620261Z","shell.execute_reply":"2021-11-28T04:08:49.627936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#[re.search(\"https?\\S+\", x) for x in train_data[\"comment_text\"]]\n#[re.findall(\"\\w+.jpg\\W*\", x) for x in train_data[\"comment_text\"]]\n#[re.findall(\"@\\S+\\b\", x) for x in train_data[\"comment_text\"]]\n","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.743885Z","iopub.execute_input":"2021-11-27T21:24:04.744225Z","iopub.status.idle":"2021-11-27T21:24:04.752753Z","shell.execute_reply.started":"2021-11-27T21:24:04.744194Z","shell.execute_reply":"2021-11-27T21:24:04.751808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_data[\"comment_text\"][0]","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.755455Z","iopub.execute_input":"2021-11-27T21:24:04.755818Z","iopub.status.idle":"2021-11-27T21:24:04.767330Z","shell.execute_reply.started":"2021-11-27T21:24:04.755770Z","shell.execute_reply":"2021-11-27T21:24:04.766427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" '''def preprocess(train_data): \n        train_data = train_data.apply(lambda x : contractions.fix(x, slang = False)) \n        train_data = train_data.apply(lambda x : x.lower()  )\n        train_data = train_data.apply(lambda x : re.sub(r'[^\\w\\s]','',x)  )\n        train_data = train_data.apply(lambda x : re.sub(r'https?\\S+','', x)  )\n        train_data = train_data.apply(lambda x : re.sub(r'\\w+.jpg\\W*','', x)  )\n        train_data = train_data.apply(lambda x : re.sub(r'@\\S+\\b','',  x)  )\n        train_data = train_data.apply(lambda x : re.sub(r'#\\S+\\b','', x)  )\n        stop_words = set(stopwords.words(\"english\"))\n        #stop_words = set(stop_words) \n        train_data = train_data.apply(lambda x: re.sub(\"[@_!#$%^&*()<>?/|}{~:0-9]\",'', x))\n        train_data = train_data.apply(lambda x : re.sub(r'\\s+',' ', x) )\n        train_data = train_data.apply(lambda x: x.strip())\n        train_data = train_data.apply(lambda x : [word for word in x.split() if word not in stop_words])\n         \n        #train_data= train_data.apply(lambda x: ' '.join(x))\n        return train_data'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.768629Z","iopub.execute_input":"2021-11-27T21:24:04.768874Z","iopub.status.idle":"2021-11-27T21:24:04.781464Z","shell.execute_reply.started":"2021-11-27T21:24:04.768847Z","shell.execute_reply":"2021-11-27T21:24:04.780828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''x = \"explanation\\nWhy sat\"\ncontractions.fix(x, slang = False)\nx = x.lower()\nx = re.sub(r'[^\\w\\s]','',x)\nx = re.sub(r'https?\\S+','', x)\nx = re.sub(r'\\w+.jpg\\W*','', x)\n#re.sub(r'\\w+.jpg\\W*','',\"explanation\")\nx = re.sub(r'@\\S+\\b','',  x)\nx = re.sub(r'#\\S+\\b','', x) \nx = re.sub(\"[@_!#$%^&*()<>?/|}{~:0-9]\",'', x)\nx = re.sub(r'#\\S+\\b','', x)\nx = re.sub(r'\\s+',' ', x)\nx = [word for word in x.split() if word not in stop_words]\nx'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.782337Z","iopub.execute_input":"2021-11-27T21:24:04.782609Z","iopub.status.idle":"2021-11-27T21:24:04.798398Z","shell.execute_reply.started":"2021-11-27T21:24:04.782579Z","shell.execute_reply":"2021-11-27T21:24:04.797648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''stop_words = set(stopwords.words(\"english\"))\n\"why\" in stop_words'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.802068Z","iopub.execute_input":"2021-11-27T21:24:04.802359Z","iopub.status.idle":"2021-11-27T21:24:04.811492Z","shell.execute_reply.started":"2021-11-27T21:24:04.802326Z","shell.execute_reply":"2021-11-27T21:24:04.810486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''#remove all those special texts whivh won't be helpful \ntrain_data[\"comment_text\"] = preprocess(train_data[\"comment_text\"])'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.814358Z","iopub.execute_input":"2021-11-27T21:24:04.814663Z","iopub.status.idle":"2021-11-27T21:24:04.822429Z","shell.execute_reply.started":"2021-11-27T21:24:04.814621Z","shell.execute_reply":"2021-11-27T21:24:04.821524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''train_data[\"comment_text\"][0]'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.823838Z","iopub.execute_input":"2021-11-27T21:24:04.824169Z","iopub.status.idle":"2021-11-27T21:24:04.835060Z","shell.execute_reply.started":"2021-11-27T21:24:04.824128Z","shell.execute_reply":"2021-11-27T21:24:04.834028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' def text_trunc(text):\n    if len(text)>512:\n        text =  text[:512]\n    return text'''","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.836004Z","iopub.execute_input":"2021-11-27T21:24:04.836227Z","iopub.status.idle":"2021-11-27T21:24:04.846424Z","shell.execute_reply.started":"2021-11-27T21:24:04.836202Z","shell.execute_reply":"2021-11-27T21:24:04.845692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:24:04.847569Z","iopub.execute_input":"2021-11-27T21:24:04.847822Z","iopub.status.idle":"2021-11-27T21:24:04.858132Z","shell.execute_reply.started":"2021-11-27T21:24:04.847794Z","shell.execute_reply":"2021-11-27T21:24:04.857513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen = 512 ):\n    enc_di = tokenizer.batch_encode_plus(texts,return_attention_mask = False, return_token_type_ids  = False,\n                                        pad_to_max_length = True, max_length = maxlen, truncation = True)\n    return np.array(enc_di[\"input_ids\"])\n           \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:29.361326Z","iopub.execute_input":"2021-11-28T04:09:29.361640Z","iopub.status.idle":"2021-11-28T04:09:29.367293Z","shell.execute_reply.started":"2021-11-28T04:09:29.361611Z","shell.execute_reply":"2021-11-28T04:09:29.366389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:31.613713Z","iopub.execute_input":"2021-11-28T04:09:31.614007Z","iopub.status.idle":"2021-11-28T04:09:31.618541Z","shell.execute_reply.started":"2021-11-28T04:09:31.613976Z","shell.execute_reply":"2021-11-28T04:09:31.617528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer,maxlen = 512):\n    input_word_ids = Input(shape=(maxlen,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:34.023985Z","iopub.execute_input":"2021-11-28T04:09:34.024516Z","iopub.status.idle":"2021-11-28T04:09:34.032607Z","shell.execute_reply.started":"2021-11-28T04:09:34.024464Z","shell.execute_reply":"2021-11-28T04:09:34.031602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:41.345404Z","iopub.execute_input":"2021-11-28T04:09:41.345730Z","iopub.status.idle":"2021-11-28T04:09:47.668733Z","shell.execute_reply.started":"2021-11-28T04:09:41.345695Z","shell.execute_reply":"2021-11-28T04:09:47.667778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nAUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 152\nMODEL = 'jplu/tf-xlm-roberta-large'","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:47.670339Z","iopub.execute_input":"2021-11-28T04:09:47.670566Z","iopub.status.idle":"2021-11-28T04:09:48.096583Z","shell.execute_reply.started":"2021-11-28T04:09:47.670539Z","shell.execute_reply":"2021-11-28T04:09:48.095808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from transformers import TFAutoModel, AutoTokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:48.097801Z","iopub.execute_input":"2021-11-28T04:09:48.098147Z","iopub.status.idle":"2021-11-28T04:09:52.174236Z","shell.execute_reply.started":"2021-11-28T04:09:48.098106Z","shell.execute_reply":"2021-11-28T04:09:52.173411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''# Configuration\nMODEL = 'jplu/tf-xlm-roberta-large'\nAUTO = tf.data.experimental.AUTOTUNE\nSEED = 2020\nEPOCHS_1 = 20\nEPOCHS_2 = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\nSHUFFLE = 2048\nVERBOSE = 1'''","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:52.176193Z","iopub.execute_input":"2021-11-28T04:09:52.176431Z","iopub.status.idle":"2021-11-28T04:09:52.182708Z","shell.execute_reply.started":"2021-11-28T04:09:52.176405Z","shell.execute_reply":"2021-11-28T04:09:52.181682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:52.184359Z","iopub.execute_input":"2021-11-28T04:09:52.184601Z","iopub.status.idle":"2021-11-28T04:09:52.194924Z","shell.execute_reply.started":"2021-11-28T04:09:52.184564Z","shell.execute_reply":"2021-11-28T04:09:52.193715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nval_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")\nunintended_bias_train = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:09:52.196135Z","iopub.execute_input":"2021-11-28T04:09:52.196398Z","iopub.status.idle":"2021-11-28T04:10:22.188613Z","shell.execute_reply.started":"2021-11-28T04:09:52.196369Z","shell.execute_reply":"2021-11-28T04:10:22.187657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unintended_bias_train.toxic = unintended_bias_train.toxic.round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.189854Z","iopub.execute_input":"2021-11-28T04:10:22.190076Z","iopub.status.idle":"2021-11-28T04:10:22.194936Z","shell.execute_reply.started":"2021-11-28T04:10:22.190051Z","shell.execute_reply":"2021-11-28T04:10:22.193800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unintended_bias_train.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.196316Z","iopub.execute_input":"2021-11-28T04:10:22.196581Z","iopub.status.idle":"2021-11-28T04:10:22.208840Z","shell.execute_reply.started":"2021-11-28T04:10:22.196552Z","shell.execute_reply":"2021-11-28T04:10:22.207798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine train1 with a subset of train2\ntrain = pd.concat([\n    train_data[['comment_text', 'toxic']],\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==1'),\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==0').sample(n=50000, random_state=0)\n])","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.210482Z","iopub.execute_input":"2021-11-28T04:10:22.210729Z","iopub.status.idle":"2021-11-28T04:10:22.578354Z","shell.execute_reply.started":"2021-11-28T04:10:22.210700Z","shell.execute_reply":"2021-11-28T04:10:22.577467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \n","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.580818Z","iopub.execute_input":"2021-11-28T04:10:22.581084Z","iopub.status.idle":"2021-11-28T04:10:22.586040Z","shell.execute_reply.started":"2021-11-28T04:10:22.581054Z","shell.execute_reply":"2021-11-28T04:10:22.584930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n        #return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.587813Z","iopub.execute_input":"2021-11-28T04:10:22.588127Z","iopub.status.idle":"2021-11-28T04:10:22.599929Z","shell.execute_reply.started":"2021-11-28T04:10:22.588088Z","shell.execute_reply":"2021-11-28T04:10:22.599327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(test_data.content.values)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.602920Z","iopub.execute_input":"2021-11-28T04:10:22.603658Z","iopub.status.idle":"2021-11-28T04:10:22.615546Z","shell.execute_reply.started":"2021-11-28T04:10:22.603624Z","shell.execute_reply":"2021-11-28T04:10:22.614537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nx_train = regular_encode(list(train.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(list(val_data.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_test = regular_encode(list(test_data.content.values), tokenizer, maxlen=MAX_LEN)\n\n#y_train = train_data.toxic.values\ny_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:10:22.616580Z","iopub.execute_input":"2021-11-28T04:10:22.617318Z","iopub.status.idle":"2021-11-28T04:11:55.487151Z","shell.execute_reply.started":"2021-11-28T04:10:22.617267Z","shell.execute_reply":"2021-11-28T04:11:55.486433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train.toxic.values\n#y_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:11:55.488603Z","iopub.execute_input":"2021-11-28T04:11:55.489153Z","iopub.status.idle":"2021-11-28T04:11:55.495370Z","shell.execute_reply.started":"2021-11-28T04:11:55.489121Z","shell.execute_reply":"2021-11-28T04:11:55.494235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:11:55.496755Z","iopub.execute_input":"2021-11-28T04:11:55.496987Z","iopub.status.idle":"2021-11-28T04:11:57.100103Z","shell.execute_reply.started":"2021-11-28T04:11:55.496962Z","shell.execute_reply":"2021-11-28T04:11:57.098792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL)\n    model = build_model(transformer_layer, maxlen=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:11:57.101626Z","iopub.execute_input":"2021-11-28T04:11:57.101900Z","iopub.status.idle":"2021-11-28T04:15:53.743706Z","shell.execute_reply.started":"2021-11-28T04:11:57.101869Z","shell.execute_reply":"2021-11-28T04:15:53.742415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:16:35.405910Z","iopub.execute_input":"2021-11-28T04:16:35.406263Z","iopub.status.idle":"2021-11-28T04:43:10.331048Z","shell.execute_reply.started":"2021-11-28T04:16:35.406228Z","shell.execute_reply":"2021-11-28T04:43:10.330071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:43:18.605327Z","iopub.execute_input":"2021-11-28T04:43:18.606172Z","iopub.status.idle":"2021-11-28T04:47:20.147038Z","shell.execute_reply.started":"2021-11-28T04:43:18.606060Z","shell.execute_reply":"2021-11-28T04:47:20.146069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" y_t = model.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:47:49.704960Z","iopub.execute_input":"2021-11-28T04:47:49.705317Z","iopub.status.idle":"2021-11-28T04:49:00.336763Z","shell.execute_reply.started":"2021-11-28T04:47:49.705269Z","shell.execute_reply":"2021-11-28T04:49:00.335358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:09.086741Z","iopub.execute_input":"2021-11-28T04:49:09.087459Z","iopub.status.idle":"2021-11-28T04:49:09.151551Z","shell.execute_reply.started":"2021-11-28T04:49:09.087405Z","shell.execute_reply":"2021-11-28T04:49:09.150769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({\"id\": s.id , \"toxic\":y_t.squeeze()}, index = None)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:12.734429Z","iopub.execute_input":"2021-11-28T04:49:12.735326Z","iopub.status.idle":"2021-11-28T04:49:12.742664Z","shell.execute_reply.started":"2021-11-28T04:49:12.735255Z","shell.execute_reply":"2021-11-28T04:49:12.741820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:50.288873Z","iopub.execute_input":"2021-11-27T22:06:50.289183Z","iopub.status.idle":"2021-11-27T22:06:50.502459Z","shell.execute_reply.started":"2021-11-27T22:06:50.289154Z","shell.execute_reply":"2021-11-27T22:06:50.501491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:10.469314Z","iopub.execute_input":"2021-11-27T22:06:10.469654Z","iopub.status.idle":"2021-11-27T22:06:10.528652Z","shell.execute_reply.started":"2021-11-27T22:06:10.469620Z","shell.execute_reply":"2021-11-27T22:06:10.527640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:42.579983Z","iopub.execute_input":"2021-11-27T22:06:42.580290Z","iopub.status.idle":"2021-11-27T22:06:42.599863Z","shell.execute_reply.started":"2021-11-27T22:06:42.580259Z","shell.execute_reply":"2021-11-27T22:06:42.598756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}