{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport re\nimport tensorflow as tf\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model, load_model\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom tensorflow.keras.layers import Input, Dense\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T09:43:09.972677Z","iopub.execute_input":"2022-07-30T09:43:09.972983Z","iopub.status.idle":"2022-07-30T09:43:10.082028Z","shell.execute_reply.started":"2022-07-30T09:43:09.972895Z","shell.execute_reply":"2022-07-30T09:43:10.081332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install keras","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:43:17.047198Z","iopub.execute_input":"2022-07-30T09:43:17.047968Z","iopub.status.idle":"2022-07-30T09:43:26.578983Z","shell.execute_reply.started":"2022-07-30T09:43:17.047919Z","shell.execute_reply":"2022-07-30T09:43:26.578138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"Batched inputs are often different lengths, so they can’t be converted to fixed-size tensors.\"\n    \"Padding and truncation are strategies for dealing with this problem, to create rectangular tensors from batches of varying lengths\"\n    tokenizer.enable_truncation(max_length=maxlen)()\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:07.607135Z","iopub.execute_input":"2022-07-30T09:44:07.607447Z","iopub.status.idle":"2022-07-30T09:44:07.615011Z","shell.execute_reply.started":"2022-07-30T09:44:07.607415Z","shell.execute_reply":"2022-07-30T09:44:07.613901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen = 512 ):\n    enc_di = tokenizer.batch_encode_plus(texts,return_attention_mask = False, return_token_type_ids  = False,\n                                        pad_to_max_length = True, max_length = maxlen, truncation = True)\n    return np.array(enc_di[\"input_ids\"])\n           \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:13.04139Z","iopub.execute_input":"2022-07-30T09:44:13.041666Z","iopub.status.idle":"2022-07-30T09:44:13.046877Z","shell.execute_reply.started":"2022-07-30T09:44:13.041636Z","shell.execute_reply":"2022-07-30T09:44:13.046071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer,maxlen = 512):\n    input_word_ids = Input(shape=(maxlen,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:26.486228Z","iopub.execute_input":"2022-07-30T09:44:26.486763Z","iopub.status.idle":"2022-07-30T09:44:26.492806Z","shell.execute_reply.started":"2022-07-30T09:44:26.486726Z","shell.execute_reply":"2022-07-30T09:44:26.492139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:32.907365Z","iopub.execute_input":"2022-07-30T09:44:32.907889Z","iopub.status.idle":"2022-07-30T09:44:32.927054Z","shell.execute_reply.started":"2022-07-30T09:44:32.907801Z","shell.execute_reply":"2022-07-30T09:44:32.9261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nAUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 152\nMODEL = 'jplu/tf-xlm-roberta-large'","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:38.276993Z","iopub.execute_input":"2022-07-30T09:44:38.277755Z","iopub.status.idle":"2022-07-30T09:44:38.748482Z","shell.execute_reply.started":"2022-07-30T09:44:38.277719Z","shell.execute_reply":"2022-07-30T09:44:38.747594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from transformers import TFAutoModel, AutoTokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:43.957285Z","iopub.execute_input":"2022-07-30T09:44:43.957545Z","iopub.status.idle":"2022-07-30T09:44:49.036242Z","shell.execute_reply.started":"2022-07-30T09:44:43.957518Z","shell.execute_reply":"2022-07-30T09:44:49.035429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''# Configuration\nMODEL = 'jplu/tf-xlm-roberta-large'\nAUTO = tf.data.experimental.AUTOTUNE\nSEED = 2020\nEPOCHS_1 = 20\nEPOCHS_2 = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\nSHUFFLE = 2048\nVERBOSE = 1'''","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:44:51.414946Z","iopub.execute_input":"2022-07-30T09:44:51.415486Z","iopub.status.idle":"2022-07-30T09:44:51.421707Z","shell.execute_reply.started":"2022-07-30T09:44:51.415451Z","shell.execute_reply":"2022-07-30T09:44:51.421069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nval_data =  pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")\nunintended_bias_train = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:45:01.464765Z","iopub.execute_input":"2022-07-30T09:45:01.465488Z","iopub.status.idle":"2022-07-30T09:45:26.583958Z","shell.execute_reply.started":"2022-07-30T09:45:01.465452Z","shell.execute_reply":"2022-07-30T09:45:26.583199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended_bias_train.toxic = unintended_bias_train.toxic.round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:45:35.028339Z","iopub.execute_input":"2022-07-30T09:45:35.028651Z","iopub.status.idle":"2022-07-30T09:45:35.192448Z","shell.execute_reply.started":"2022-07-30T09:45:35.02862Z","shell.execute_reply":"2022-07-30T09:45:35.191671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended_bias_train.toxic.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:45:41.986367Z","iopub.execute_input":"2022-07-30T09:45:41.986647Z","iopub.status.idle":"2022-07-30T09:45:42.011266Z","shell.execute_reply.started":"2022-07-30T09:45:41.986618Z","shell.execute_reply":"2022-07-30T09:45:42.010601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine train1 with a subset of train2\ntrain = pd.concat([\n    train_data[['comment_text', 'toxic']],\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==1'),\n    unintended_bias_train[['comment_text', 'toxic']].query('toxic==0').sample(n=50000, random_state=0)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:45:46.460131Z","iopub.execute_input":"2022-07-30T09:45:46.460401Z","iopub.status.idle":"2022-07-30T09:45:46.963235Z","shell.execute_reply.started":"2022-07-30T09:45:46.460373Z","shell.execute_reply":"2022-07-30T09:45:46.962462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def regular_encode(texts, tokenizer, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n        #return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    return np.array(enc_di['input_ids'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:45:57.515809Z","iopub.execute_input":"2022-07-30T09:45:57.516389Z","iopub.status.idle":"2022-07-30T09:45:57.520929Z","shell.execute_reply.started":"2022-07-30T09:45:57.516353Z","shell.execute_reply":"2022-07-30T09:45:57.52004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(test_data.content.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:46:02.713021Z","iopub.execute_input":"2022-07-30T09:46:02.713291Z","iopub.status.idle":"2022-07-30T09:46:02.719255Z","shell.execute_reply.started":"2022-07-30T09:46:02.713265Z","shell.execute_reply":"2022-07-30T09:46:02.71848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nx_train = regular_encode(list(train.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_valid = regular_encode(list(val_data.comment_text.values), tokenizer, maxlen=MAX_LEN)\nx_test = regular_encode(list(test_data.content.values), tokenizer, maxlen=MAX_LEN)\n\n#y_train = train_data.toxic.values\ny_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:46:08.651511Z","iopub.execute_input":"2022-07-30T09:46:08.651777Z","iopub.status.idle":"2022-07-30T09:48:24.363478Z","shell.execute_reply.started":"2022-07-30T09:46:08.65175Z","shell.execute_reply":"2022-07-30T09:48:24.362736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train.toxic.values\n#y_valid = val_data.toxic.values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:51:30.530232Z","iopub.execute_input":"2022-07-30T09:51:30.530508Z","iopub.status.idle":"2022-07-30T09:51:30.534619Z","shell.execute_reply.started":"2022-07-30T09:51:30.53048Z","shell.execute_reply":"2022-07-30T09:51:30.533687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:51:38.632386Z","iopub.execute_input":"2022-07-30T09:51:38.632871Z","iopub.status.idle":"2022-07-30T09:51:40.558353Z","shell.execute_reply.started":"2022-07-30T09:51:38.632816Z","shell.execute_reply":"2022-07-30T09:51:40.556913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = TFAutoModel.from_pretrained(MODEL)\n    model = build_model(transformer_layer, maxlen=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:51:47.567104Z","iopub.execute_input":"2022-07-30T09:51:47.567371Z","iopub.status.idle":"2022-07-30T09:53:32.206639Z","shell.execute_reply.started":"2022-07-30T09:51:47.567343Z","shell.execute_reply":"2022-07-30T09:53:32.205926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T09:53:53.640243Z","iopub.execute_input":"2022-07-30T09:53:53.640507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T07:14:12.456535Z","iopub.execute_input":"2022-07-30T07:14:12.457225Z","iopub.status.idle":"2022-07-30T07:30:38.089998Z","shell.execute_reply.started":"2022-07-30T07:14:12.457191Z","shell.execute_reply":"2022-07-30T07:30:38.089286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" y_t = model.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:47:49.70496Z","iopub.execute_input":"2021-11-28T04:47:49.705317Z","iopub.status.idle":"2021-11-28T04:49:00.336763Z","shell.execute_reply.started":"2021-11-28T04:47:49.705269Z","shell.execute_reply":"2021-11-28T04:49:00.335358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:09.086741Z","iopub.execute_input":"2021-11-28T04:49:09.087459Z","iopub.status.idle":"2021-11-28T04:49:09.151551Z","shell.execute_reply.started":"2021-11-28T04:49:09.087405Z","shell.execute_reply":"2021-11-28T04:49:09.150769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({\"id\": s.id , \"toxic\":y_t.squeeze()}, index = None)","metadata":{"execution":{"iopub.status.busy":"2021-11-28T04:49:12.734429Z","iopub.execute_input":"2021-11-28T04:49:12.735326Z","iopub.status.idle":"2021-11-28T04:49:12.742664Z","shell.execute_reply.started":"2021-11-28T04:49:12.735255Z","shell.execute_reply":"2021-11-28T04:49:12.74182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:50.288873Z","iopub.execute_input":"2021-11-27T22:06:50.289183Z","iopub.status.idle":"2021-11-27T22:06:50.502459Z","shell.execute_reply.started":"2021-11-27T22:06:50.289154Z","shell.execute_reply":"2021-11-27T22:06:50.501491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s","metadata":{"execution":{"iopub.status.busy":"2021-11-27T22:06:42.579983Z","iopub.execute_input":"2021-11-27T22:06:42.58029Z","iopub.status.idle":"2021-11-27T22:06:42.599863Z","shell.execute_reply.started":"2021-11-27T22:06:42.580259Z","shell.execute_reply":"2021-11-27T22:06:42.598756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}