{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nfrom transformers import RobertaTokenizerFast, TFRobertaModel\nfrom tqdm import tqdm\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:14.921408Z","iopub.execute_input":"2025-04-19T18:06:14.921794Z","iopub.status.idle":"2025-04-19T18:06:28.303025Z","shell.execute_reply.started":"2025-04-19T18:06:14.921717Z","shell.execute_reply":"2025-04-19T18:06:28.302092Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper Functions","metadata":{}},{"cell_type":"code","source":"def encode(texts, tokenizer, max_len=512):\n    enc = tokenizer(\n        texts.tolist(),\n        padding='max_length',\n        truncation=True,\n        max_length=max_len,\n        return_tensors='np'\n    )\n    return enc['input_ids']\n","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:33.005683Z","iopub.execute_input":"2025-04-19T18:06:33.006283Z","iopub.status.idle":"2025-04-19T18:06:33.01139Z","shell.execute_reply.started":"2025-04-19T18:06:33.00625Z","shell.execute_reply":"2025-04-19T18:06:33.010434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:35.218528Z","iopub.execute_input":"2025-04-19T18:06:35.219114Z","iopub.status.idle":"2025-04-19T18:06:35.225308Z","shell.execute_reply.started":"2025-04-19T18:06:35.219075Z","shell.execute_reply":"2025-04-19T18:06:35.224376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 32\nMAX_LEN = 192","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:36.907833Z","iopub.execute_input":"2025-04-19T18:06:36.908283Z","iopub.status.idle":"2025-04-19T18:06:37.350943Z","shell.execute_reply.started":"2025-04-19T18:06:36.908251Z","shell.execute_reply":"2025-04-19T18:06:37.350269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create fast tokenizer","metadata":{}},{"cell_type":"code","source":"tokenizer = RobertaTokenizerFast.from_pretrained('roberta-base')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:39.787593Z","iopub.execute_input":"2025-04-19T18:06:39.787935Z","iopub.status.idle":"2025-04-19T18:06:42.343222Z","shell.execute_reply.started":"2025-04-19T18:06:39.787905Z","shell.execute_reply":"2025-04-19T18:06:42.342502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load text data into memory","metadata":{}},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\ntrain2.toxic = train2.toxic.round().astype(int)\n\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:06:43.980349Z","iopub.execute_input":"2025-04-19T18:06:43.980951Z","iopub.status.idle":"2025-04-19T18:07:13.500727Z","shell.execute_reply.started":"2025-04-19T18:06:43.980919Z","shell.execute_reply":"2025-04-19T18:07:13.50005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine train1 with a subset of train2\ntrain = pd.concat([\n    train1[['comment_text', 'toxic']],\n    train2[['comment_text', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=150000, random_state=0)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:07:15.95597Z","iopub.execute_input":"2025-04-19T18:07:15.956786Z","iopub.status.idle":"2025-04-19T18:07:16.502996Z","shell.execute_reply.started":"2025-04-19T18:07:15.956756Z","shell.execute_reply":"2025-04-19T18:07:16.502293Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### tokenizing...","metadata":{}},{"cell_type":"code","source":"x_train = encode(train.comment_text.astype(str), tokenizer, max_len=MAX_LEN)\nx_valid = encode(valid.comment_text.astype(str), tokenizer, max_len=MAX_LEN)\nx_test = encode(test.content.astype(str), tokenizer, max_len=MAX_LEN)\n\ny_train = train.toxic.values\ny_valid = valid.toxic.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:07:18.905282Z","iopub.execute_input":"2025-04-19T18:07:18.905854Z","iopub.status.idle":"2025-04-19T18:09:29.880215Z","shell.execute_reply.started":"2025-04-19T18:07:18.90582Z","shell.execute_reply":"2025-04-19T18:09:29.879391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build datasets objects","metadata":{}},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:09:34.264946Z","iopub.execute_input":"2025-04-19T18:09:34.26532Z","iopub.status.idle":"2025-04-19T18:09:35.913725Z","shell.execute_reply.started":"2025-04-19T18:09:34.26528Z","shell.execute_reply":"2025-04-19T18:09:35.912755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load model ","metadata":{}},{"cell_type":"code","source":"%%time\ntransformer_layer = TFRobertaModel.from_pretrained('roberta-base')\nmodel = build_model(transformer_layer, max_len=MAX_LEN)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:09:41.543463Z","iopub.execute_input":"2025-04-19T18:09:41.544183Z","iopub.status.idle":"2025-04-19T18:09:59.56259Z","shell.execute_reply.started":"2025-04-19T18:09:41.544149Z","shell.execute_reply":"2025-04-19T18:09:59.561614Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"markdown","source":"First, we train on the subset of the training set, which is completely in English.","metadata":{}},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T18:09:59.56388Z","iopub.execute_input":"2025-04-19T18:09:59.564265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now that we have pretty much saturated the learning potential of the model on english only data, we train it for one more epoch on the `validation` set, which is significantly smaller but contains a mixture of different languages.","metadata":{}},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS*2\n)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-19T18:05:22.623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"sub['toxic'] = model.predict(test_dataset, verbose=1)\nsub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save model and tokenizer\nmodel.save_pretrained('./roberta-toxic-model')\ntokenizer.save_pretrained('./roberta-toxic-model')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}