{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install transformers\n!pip install tqdm\n# !pip install tensorflow\n\n# Loading Dependencies\nimport os\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nimport pandas as pd\nfrom tokenizers import BertWordPieceTokenizer\nfrom tqdm import tqdm\nimport numpy as np\nprint('okay')","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:27:40.647008Z","iopub.execute_input":"2023-08-07T21:27:40.647757Z","iopub.status.idle":"2023-08-07T21:28:43.806435Z","shell.execute_reply.started":"2023-08-07T21:27:40.647719Z","shell.execute_reply":"2023-08-07T21:28:43.805435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up tpus \n# detect and init the TPU\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n# tf.config.experimental_connect_to_cluster(tpu)\n# tf.tpu.experimental.initialize_tpu_system(tpu)\nstrategy = tf.distribute.TPUStrategy(tpu)\ntpu_cores=strategy.num_replicas_in_sync\n\nprint(\"REPLICAS: \", tpu_cores)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:35:54.640240Z","iopub.execute_input":"2023-08-07T21:35:54.640859Z","iopub.status.idle":"2023-08-07T21:36:02.672375Z","shell.execute_reply.started":"2023-08-07T21:35:54.640828Z","shell.execute_reply":"2023-08-07T21:36:02.671099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LOADING THE DATA\ntrain1 = pd.read_csv('/kaggle/input/c/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalid = pd.read_csv('/kaggle/input/c/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/c/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/c/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\n\n# Load the tokenizer\ntokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:36:12.734379Z","iopub.execute_input":"2023-08-07T21:36:12.734800Z","iopub.status.idle":"2023-08-07T21:36:17.709663Z","shell.execute_reply.started":"2023-08-07T21:36:12.734769Z","shell.execute_reply":"2023-08-07T21:36:17.708307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up some global variables\nAUTO = tf.data.experimental.AUTOTUNE\nEPOCHS = 3\nBATCH_SIZE = 32 * tpu_cores\nMAX_LEN = 192","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:36:30.278583Z","iopub.execute_input":"2023-08-07T21:36:30.279054Z","iopub.status.idle":"2023-08-07T21:36:30.284536Z","shell.execute_reply.started":"2023-08-07T21:36:30.279017Z","shell.execute_reply":"2023-08-07T21:36:30.283476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)\n\n\n\n# Encode full data\nx_train = fast_encode(train1.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_valid = fast_encode(valid.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=MAX_LEN)\n\ny_train = train1.toxic.values\ny_valid = valid.toxic.values","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:36:32.275961Z","iopub.execute_input":"2023-08-07T21:36:32.276351Z","iopub.status.idle":"2023-08-07T21:37:04.160950Z","shell.execute_reply.started":"2023-08-07T21:36:32.276319Z","shell.execute_reply":"2023-08-07T21:37:04.159652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:37:10.308853Z","iopub.execute_input":"2023-08-07T21:37:10.309813Z","iopub.status.idle":"2023-08-07T21:37:10.695254Z","shell.execute_reply.started":"2023-08-07T21:37:10.309773Z","shell.execute_reply":"2023-08-07T21:37:10.694044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Construct the tensor model\ndef build_model(transformer, max_len=512):\n    \"\"\"\n    function for training the BERT model\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(learning_rate=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model\n\n\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:38:44.090026Z","iopub.execute_input":"2023-08-07T21:38:44.090481Z","iopub.status.idle":"2023-08-07T21:38:52.991696Z","shell.execute_reply.started":"2023-08-07T21:38:44.090446Z","shell.execute_reply":"2023-08-07T21:38:52.990543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the training data\nn_steps = x_train.shape[0] // BATCH_SIZE\nmodel.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:39:28.134730Z","iopub.execute_input":"2023-08-07T21:39:28.135277Z","iopub.status.idle":"2023-08-07T21:45:26.096083Z","shell.execute_reply.started":"2023-08-07T21:39:28.135230Z","shell.execute_reply":"2023-08-07T21:45:26.094666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the validation data\nn_steps = x_valid.shape[0] // BATCH_SIZE\nmodel.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS*4\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T22:24:58.390645Z","iopub.execute_input":"2023-08-06T22:24:58.391040Z","iopub.status.idle":"2023-08-06T22:25:39.864629Z","shell.execute_reply.started":"2023-08-06T22:24:58.391011Z","shell.execute_reply":"2023-08-06T22:25:39.863305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict and submit\nsub['toxic'] = model.predict(test_dataset, verbose=1)\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T21:45:48.914244Z","iopub.execute_input":"2023-08-07T21:45:48.914680Z","iopub.status.idle":"2023-08-07T21:46:07.944624Z","shell.execute_reply.started":"2023-08-07T21:45:48.914645Z","shell.execute_reply":"2023-08-07T21:46:07.943044Z"},"trusted":true},"execution_count":null,"outputs":[]}]}