{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Loading Dependencies\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-26T13:44:35.697102Z","iopub.execute_input":"2023-08-26T13:44:35.697462Z","iopub.status.idle":"2023-08-26T13:44:44.347547Z","shell.execute_reply.started":"2023-08-26T13:44:35.697433Z","shell.execute_reply":"2023-08-26T13:44:44.346178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:44:44.349463Z","iopub.execute_input":"2023-08-26T13:44:44.350216Z","iopub.status.idle":"2023-08-26T13:44:48.275173Z","shell.execute_reply.started":"2023-08-26T13:44:44.350176Z","shell.execute_reply":"2023-08-26T13:44:48.274189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n#     tokenizer.enable_padding(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:44:48.276446Z","iopub.execute_input":"2023-08-26T13:44:48.278818Z","iopub.status.idle":"2023-08-26T13:44:48.286485Z","shell.execute_reply.started":"2023-08-26T13:44:48.278781Z","shell.execute_reply":"2023-08-26T13:44:48.285294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#IMP DATA FOR CONFIG\n\nAUTO = tf.data.experimental.AUTOTUNE\nstrategy = tf.distribute.get_strategy()\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:44:48.289389Z","iopub.execute_input":"2023-08-26T13:44:48.289864Z","iopub.status.idle":"2023-08-26T13:44:48.311695Z","shell.execute_reply.started":"2023-08-26T13:44:48.289831Z","shell.execute_reply":"2023-08-26T13:44:48.311118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\ntokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:44:48.312996Z","iopub.execute_input":"2023-08-26T13:44:48.313358Z","iopub.status.idle":"2023-08-26T13:44:49.651903Z","shell.execute_reply.started":"2023-08-26T13:44:48.313315Z","shell.execute_reply":"2023-08-26T13:44:49.650890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = fast_encode(train1.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_valid = fast_encode(valid.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=MAX_LEN)\n\ny_train = train1.toxic.values\ny_valid = valid.toxic.values\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:44:49.653303Z","iopub.execute_input":"2023-08-26T13:44:49.653680Z","iopub.status.idle":"2023-08-26T13:46:11.087134Z","shell.execute_reply.started":"2023-08-26T13:44:49.653646Z","shell.execute_reply":"2023-08-26T13:46:11.086134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:46:11.088931Z","iopub.execute_input":"2023-08-26T13:46:11.089405Z","iopub.status.idle":"2023-08-26T13:46:14.777808Z","shell.execute_reply.started":"2023-08-26T13:46:11.089371Z","shell.execute_reply":"2023-08-26T13:46:14.776803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bert Model","metadata":{}},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    function for training the BERT model\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:46:14.779233Z","iopub.execute_input":"2023-08-26T13:46:14.779582Z","iopub.status.idle":"2023-08-26T13:46:14.786497Z","shell.execute_reply.started":"2023-08-26T13:46:14.779547Z","shell.execute_reply":"2023-08-26T13:46:14.785381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:46:14.787924Z","iopub.execute_input":"2023-08-26T13:46:14.788565Z","iopub.status.idle":"2023-08-26T13:46:39.426248Z","shell.execute_reply.started":"2023-08-26T13:46:14.788533Z","shell.execute_reply":"2023-08-26T13:46:39.425492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reberta Model","metadata":{}},{"cell_type":"code","source":"# from transformers import TFRobertaModel, RobertaTokenizer\n\n# # Build the model function\n# def build_model(transformer, max_len=512):\n#     input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n#     sequence_output = transformer(input_word_ids)[0]\n#     cls_token = sequence_output[:, 0, :]\n#     out = Dense(1, activation='sigmoid')(cls_token)\n    \n#     model = Model(inputs=input_word_ids, outputs=out)\n#     model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n#     return model","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:33:29.127692Z","iopub.execute_input":"2023-08-26T13:33:29.128028Z","iopub.status.idle":"2023-08-26T13:33:29.171208Z","shell.execute_reply.started":"2023-08-26T13:33:29.128002Z","shell.execute_reply":"2023-08-26T13:33:29.170086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# with strategy.scope():\n#     transformer_layer = (\n#         TFRobertaModel\n#         .from_pretrained('roberta-base')  # Use the RoBERTa base model\n#     )\n#     model = build_model(transformer_layer, max_len=MAX_LEN)\n    \n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:34:42.903916Z","iopub.execute_input":"2023-08-26T13:34:42.904331Z","iopub.status.idle":"2023-08-26T13:34:42.910574Z","shell.execute_reply.started":"2023-08-26T13:34:42.904299Z","shell.execute_reply":"2023-08-26T13:34:42.908922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\n# print(n_steps)\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:34:42.999184Z","iopub.execute_input":"2023-08-26T13:34:43.000666Z","iopub.status.idle":"2023-08-26T13:38:54.098034Z","shell.execute_reply.started":"2023-08-26T13:34:43.000605Z","shell.execute_reply":"2023-08-26T13:38:54.096415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:46:52.563879Z","iopub.execute_input":"2023-08-26T13:46:52.564638Z","iopub.status.idle":"2023-08-26T13:46:52.571037Z","shell.execute_reply.started":"2023-08-26T13:46:52.564602Z","shell.execute_reply":"2023-08-26T13:46:52.570113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = model.predict(x_valid)\n# custom_metrics = ['accuracy', 'precision', 'recall']\n# evaluation_results = model.evaluate(test_dataset)\n# evaluation_results\nmapped_values = [1 if value > 0.5 else 0 for value in sub]\nmapped_values = np.array(mapped_values)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:58:23.318543Z","iopub.execute_input":"2023-08-26T13:58:23.318898Z","iopub.status.idle":"2023-08-26T13:58:23.347227Z","shell.execute_reply.started":"2023-08-26T13:58:23.318865Z","shell.execute_reply":"2023-08-26T13:58:23.346392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = model.evaluate(mapped_values, y_valid)\n# type(mapped_values) # mapped_values","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:58:23.481302Z","iopub.status.idle":"2023-08-26T13:58:23.481649Z","shell.execute_reply.started":"2023-08-26T13:58:23.481485Z","shell.execute_reply":"2023-08-26T13:58:23.481501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a","metadata":{"execution":{"iopub.status.busy":"2023-08-26T13:50:20.970323Z","iopub.status.idle":"2023-08-26T13:50:20.971058Z","shell.execute_reply.started":"2023-08-26T13:50:20.970782Z","shell.execute_reply":"2023-08-26T13:50:20.970805Z"},"trusted":true},"execution_count":null,"outputs":[]}]}