{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8493351,"sourceType":"datasetVersion","datasetId":5067525}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install bert-for-tf2","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:26:01.999972Z","iopub.execute_input":"2024-05-23T10:26:02.000381Z","iopub.status.idle":"2024-05-23T10:26:22.569832Z","shell.execute_reply.started":"2024-05-23T10:26:02.00035Z","shell.execute_reply":"2024-05-23T10:26:22.568896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow-text","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:26:22.57166Z","iopub.execute_input":"2024-05-23T10:26:22.571947Z","iopub.status.idle":"2024-05-23T10:26:37.366001Z","shell.execute_reply.started":"2024-05-23T10:26:22.571921Z","shell.execute_reply":"2024-05-23T10:26:37.364926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow-gpu\n!pip install --upgrade tensorflow-hub\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:26:37.367434Z","iopub.execute_input":"2024-05-23T10:26:37.367752Z","iopub.status.idle":"2024-05-23T10:26:52.367779Z","shell.execute_reply.started":"2024-05-23T10:26:37.367723Z","shell.execute_reply":"2024-05-23T10:26:52.366844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport bert\nfrom bert import tokenization\n\n# Ensure mixed precision for faster training\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:26:52.370217Z","iopub.execute_input":"2024-05-23T10:26:52.370545Z","iopub.status.idle":"2024-05-23T10:27:05.797373Z","shell.execute_reply.started":"2024-05-23T10:26:52.370517Z","shell.execute_reply":"2024-05-23T10:27:05.79593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.mixed_precision.set_global_policy('mixed_float16')\n\nprint(\"TensorFlow version:\", tf.__version__)\nprint(\"Keras API available:\", hasattr(tf, 'keras'))","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:27:05.798961Z","iopub.execute_input":"2024-05-23T10:27:05.800349Z","iopub.status.idle":"2024-05-23T10:27:05.996366Z","shell.execute_reply.started":"2024-05-23T10:27:05.800318Z","shell.execute_reply":"2024-05-23T10:27:05.995393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if GPU is available and set batch size accordingly\nif tf.config.list_physical_devices('GPU'):\n    device = '/GPU:0'\n    batch_size = 64\nelse:\n    device = '/CPU:0'\n    batch_size = 32\n\nprint(f'Using device: {device}, Batch size: {batch_size}')","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:27:05.99754Z","iopub.execute_input":"2024-05-23T10:27:05.997894Z","iopub.status.idle":"2024-05-23T10:27:06.032288Z","shell.execute_reply.started":"2024-05-23T10:27:05.997827Z","shell.execute_reply":"2024-05-23T10:27:06.031098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/model-dataset/train_preprocess.csv')\ntest_data = pd.read_csv('/kaggle/input/model-dataset/test_preprocess.csv')\nvalidation_data = pd.read_csv('/kaggle/input/model-dataset/validation_preprocess.csv')\n\n# Display the first few rows of the training data\ntrain_data.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:27:16.868695Z","iopub.execute_input":"2024-05-23T10:27:16.869625Z","iopub.status.idle":"2024-05-23T10:27:17.187731Z","shell.execute_reply.started":"2024-05-23T10:27:16.869594Z","shell.execute_reply":"2024-05-23T10:27:17.186775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\nimport tensorflow_text as text  # Import TensorFlow Text\n\n# Initialize the BERT layer and tokenizer\nBERT_MODEL_URL = \"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/2\"\nbert_layer = hub.KerasLayer(BERT_MODEL_URL, trainable=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:27:23.281098Z","iopub.execute_input":"2024-05-23T10:27:23.281499Z","iopub.status.idle":"2024-05-23T10:27:40.7167Z","shell.execute_reply.started":"2024-05-23T10:27:23.281468Z","shell.execute_reply":"2024-05-23T10:27:40.715857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use TensorFlow Text's tokenizers\ntokenizer = text.BertTokenizer(bert_layer.resolved_object.vocab_file.asset_path, \n                                lower_case=bert_layer.resolved_object.do_lower_case.numpy())","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:27:40.718307Z","iopub.execute_input":"2024-05-23T10:27:40.718608Z","iopub.status.idle":"2024-05-23T10:27:40.736569Z","shell.execute_reply.started":"2024-05-23T10:27:40.718582Z","shell.execute_reply":"2024-05-23T10:27:40.735855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer\n\n# def encode_sentences(sentences, tokenizer, max_seq_length):\n#     cls_token_id = tokenizer.cls_token_id\n#     sep_token_id = tokenizer.sep_token_id\n\n#     encoded_dict = tokenizer(sentences, add_special_tokens=False, max_length=max_seq_length, truncation=True, padding=False, return_tensors='tf')\n#     tokens = encoded_dict['input_ids']\n\n#     start_tokens = tf.fill([tf.shape(tokens)[0], 1], cls_token_id)\n#     end_tokens = tf.fill([tf.shape(tokens)[0], 1], sep_token_id)\n\n#     # Ensure all tensors have the same dtype\n#     start_tokens = tf.cast(start_tokens, tokens.dtype)\n#     end_tokens = tf.cast(end_tokens, tokens.dtype)\n\n#     # Concatenate '[CLS]', tokens, and '[SEP]'\n#     tokens = tf.concat([start_tokens, tokens, end_tokens], axis=1)\n\n#     # Pad to max_seq_length\n#     pad_len = max_seq_length - tf.shape(tokens)[1]\n#     padding = tf.zeros([tf.shape(tokens)[0], pad_len], dtype=tokens.dtype)\n#     tokens = tf.concat([tokens, padding], axis=1)\n\n#     # Create attention masks (1 for real tokens and 0 for padding)\n#     attention_masks = tf.cast(tokens != 0, dtype=tokens.dtype)\n\n#     # No segment ids needed for single sentence inputs (BERT)\n#     segment_ids = tf.zeros_like(tokens, dtype=tokens.dtype)\n\n#     return tokens, attention_masks, segment_ids\n\n\n\ndef encode_sentences(sentences, tokenizer, max_seq_length):\n    encoded_dict = tokenizer(\n        sentences, \n        add_special_tokens=True, \n        max_length=max_seq_length, \n        truncation=True, \n        padding='max_length', \n        return_tensors='tf'\n    )\n    \n    input_ids = encoded_dict['input_ids']\n    attention_masks = encoded_dict['attention_mask']\n    segment_ids = tf.zeros_like(input_ids)\n\n    return input_ids, attention_masks, segment_ids\n\nmax_seq_length = 128\n\n# Encode training data\ntrain_sentences = train_data['comment'].tolist()\ntrain_input_ids, train_input_masks, train_segment_ids = encode_sentences(train_sentences, tokenizer, max_seq_length)\n\n# Encode validation data\nval_sentences = validation_data['comment'].tolist()\nval_input_ids, val_input_masks, val_segment_ids = encode_sentences(val_sentences, tokenizer, max_seq_length)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:28:58.085452Z","iopub.execute_input":"2024-05-23T10:28:58.085802Z","iopub.status.idle":"2024-05-23T10:29:58.750524Z","shell.execute_reply.started":"2024-05-23T10:28:58.085775Z","shell.execute_reply":"2024-05-23T10:29:58.749496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Example usage\ntokenizer = BertTokenizer.from_pretrained('bert-base-uncased')\nsentences = ['example sentence', 'another example']\nmax_seq_length = 128\ntrain_input_ids, train_input_masks, train_segment_ids = encode_sentences(sentences, tokenizer, max_seq_length)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:30:12.104979Z","iopub.execute_input":"2024-05-23T10:30:12.105764Z","iopub.status.idle":"2024-05-23T10:30:12.286161Z","shell.execute_reply.started":"2024-05-23T10:30:12.105732Z","shell.execute_reply":"2024-05-23T10:30:12.285109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the outputs\nprint(\"Input IDs:\\n\", train_input_ids.numpy())\nprint(\"Attention Masks:\\n\", train_input_masks.numpy())\nprint(\"Segment IDs:\\n\", train_segment_ids.numpy())","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:30:16.263142Z","iopub.execute_input":"2024-05-23T10:30:16.263568Z","iopub.status.idle":"2024-05-23T10:30:16.273509Z","shell.execute_reply.started":"2024-05-23T10:30:16.263538Z","shell.execute_reply":"2024-05-23T10:30:16.272585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.layers import Dense","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:30:20.45357Z","iopub.execute_input":"2024-05-23T10:30:20.453945Z","iopub.status.idle":"2024-05-23T10:30:20.459864Z","shell.execute_reply.started":"2024-05-23T10:30:20.453917Z","shell.execute_reply":"2024-05-23T10:30:20.458866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_bert_model(max_seq_length):\n    input_word_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"input_word_ids\")\n    input_mask = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"input_mask\")\n    segment_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"segment_ids\")\n    \n    pooled_output, _ = bert_layer([input_word_ids, input_mask, segment_ids])\n    \n    x = tf.keras.layers.Dense(32, activation='relu')(pooled_output)\n    output = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    \n    model = tf.keras.Model(inputs=[input_word_ids, input_mask, segment_ids], outputs=output)\n    \n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=2e-5),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\nmodel = build_bert_model(max_seq_length)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:30:21.885132Z","iopub.execute_input":"2024-05-23T10:30:21.885524Z","iopub.status.idle":"2024-05-23T10:30:22.69203Z","shell.execute_reply.started":"2024-05-23T10:30:21.885494Z","shell.execute_reply":"2024-05-23T10:30:22.691257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:30:25.894515Z","iopub.execute_input":"2024-05-23T10:30:25.894893Z","iopub.status.idle":"2024-05-23T10:30:25.931571Z","shell.execute_reply.started":"2024-05-23T10:30:25.894863Z","shell.execute_reply":"2024-05-23T10:30:25.929581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def encode_sentences(sentences, tokenizer, max_seq_length):\n    encoded_dict = tokenizer(\n        sentences, \n        add_special_tokens=True, \n        max_length=max_seq_length, \n        truncation=True, \n        padding='max_length', \n        return_tensors='tf'\n    )\n    \n    input_ids = encoded_dict['input_ids']\n    attention_masks = encoded_dict['attention_mask']\n    segment_ids = tf.zeros_like(input_ids)\n\n    return input_ids, attention_masks, segment_ids\n\nmax_seq_length = 128\n\n# Encode training data\ntrain_sentences = train_data['comment'].tolist()\ntrain_input_ids, train_input_masks, train_segment_ids = encode_sentences(train_sentences, tokenizer, max_seq_length)\n\n# Encode validation data\nval_sentences = validation_data['comment'].tolist()\nval_input_ids, val_input_masks, val_segment_ids = encode_sentences(val_sentences, tokenizer, max_seq_length)\n\n# Define and build the BERT model\ndef build_bert_model(max_seq_length):\n    input_word_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"input_word_ids\")\n    input_mask = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"input_mask\")\n    segment_ids = tf.keras.layers.Input(shape=(max_seq_length,), dtype=tf.int32, name=\"segment_ids\")\n    \n    pooled_output, _ = bert_layer([input_word_ids, input_mask, segment_ids])\n    \n    x = tf.keras.layers.Dense(32, activation='relu')(pooled_output)\n    output = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n    \n    model = tf.keras.Model(inputs=[input_word_ids, input_mask, segment_ids], outputs=output)\n    \n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=2e-5),\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\nmodel = build_bert_model(max_seq_length)\n\n# Train the model\nhistory = model.fit(\n    [train_input_ids, train_input_masks, train_segment_ids],\n    train_data['binary_score'],\n    validation_data=([val_input_ids, val_input_masks, val_segment_ids], validation_data['binary_score']),\n    epochs=3,\n    batch_size=batch_size\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T10:31:22.936585Z","iopub.execute_input":"2024-05-23T10:31:22.936967Z","iopub.status.idle":"2024-05-23T12:07:45.004981Z","shell.execute_reply.started":"2024-05-23T10:31:22.936939Z","shell.execute_reply":"2024-05-23T12:07:45.003945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dont run this\n\nhistory = model.fit(\n    [train_input_ids, train_input_masks, train_segment_ids],\n    train_data['binary_score'],\n    validation_data=([val_input_ids, val_input_masks, val_segment_ids], validation_data['binary_score']),\n    epochs=3,\n    batch_size=batch_size\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dont run this \n\ntest_loss, test_accuracy = model.evaluate([test_input_ids, test_input_masks, test_segment_ids], test_data['binary_score'])\nprint(f'Test Accuracy: {test_accuracy:.4f}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_comment(comment, model, tokenizer, max_seq_length=128):\n    input_ids, input_masks, segment_ids = encode_sentences([comment], tokenizer, max_seq_length)\n    \n    prediction = model.predict([input_ids, input_masks, segment_ids])[0][0]\n    return 'Gender Biased' if prediction > 0.6 else 'Not Gender Biased'\n\n# Example usage\nuser_comment = input(\"Enter a comment: \")\nprediction = predict_comment(user_comment, model, tokenizer)\nprint(f'The comment is: {prediction}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}