{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"},{"sourceId":1246668,"sourceType":"datasetVersion","datasetId":715814}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Content of this Notebook\n* Simple RNN\n* Word Embeddings\n* LSTM\n* GRU\n* Bi-directional\n* Encoder-Decoder Models\n* Attention Models\n* Transformers\n* BERT","metadata":{}},{"cell_type":"markdown","source":"#### **Note** I will explain each topic in detailed so Don't worry, But Be focus With me, so let's start😁😁","metadata":{}},{"cell_type":"markdown","source":"### Load Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import GRU, SimpleRNN,LSTM\nfrom keras.layers import Dense, Activation, Dropout\nfrom keras.layers import Embedding\nfrom keras.layers import BatchNormalization\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom tensorflow.keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-05T22:11:30.550787Z","iopub.execute_input":"2025-12-05T22:11:30.551034Z","iopub.status.idle":"2025-12-05T22:11:47.848863Z","shell.execute_reply.started":"2025-12-05T22:11:30.551007Z","shell.execute_reply":"2025-12-05T22:11:47.848344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nvalidation_df = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest_df = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:40.454325Z","iopub.execute_input":"2025-07-23T08:35:40.454843Z","iopub.status.idle":"2025-07-23T08:35:43.537428Z","shell.execute_reply.started":"2025-07-23T08:35:40.454823Z","shell.execute_reply":"2025-07-23T08:35:43.536811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explore Data","metadata":{}},{"cell_type":"code","source":"# Load First 5 rows\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.538474Z","iopub.execute_input":"2025-07-23T08:35:43.538678Z","iopub.status.idle":"2025-07-23T08:35:43.559761Z","shell.execute_reply.started":"2025-07-23T08:35:43.538660Z","shell.execute_reply":"2025-07-23T08:35:43.559231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sum the null values for each columns\n\ntrain_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.561376Z","iopub.execute_input":"2025-07-23T08:35:43.561586Z","iopub.status.idle":"2025-07-23T08:35:43.609801Z","shell.execute_reply.started":"2025-07-23T08:35:43.561569Z","shell.execute_reply":"2025-07-23T08:35:43.609227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### there aren't null values so won't drop any rows","metadata":{}},{"cell_type":"code","source":"# Show the shapeof the data\n\ntrain_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.610617Z","iopub.execute_input":"2025-07-23T08:35:43.610952Z","iopub.status.idle":"2025-07-23T08:35:43.619149Z","shell.execute_reply.started":"2025-07-23T08:35:43.610918Z","shell.execute_reply":"2025-07-23T08:35:43.618633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### we have 223549 rows and this is large number that will take long time for processing and training so we will take a samplei think 15k rows good number.","metadata":{}},{"cell_type":"code","source":"train_df = train_df.iloc[:15000,:]\ntrain_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.619689Z","iopub.execute_input":"2025-07-23T08:35:43.619874Z","iopub.status.idle":"2025-07-23T08:35:43.631326Z","shell.execute_reply.started":"2025-07-23T08:35:43.619859Z","shell.execute_reply":"2025-07-23T08:35:43.630679Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### This is MultiClassification problem but we can covert it to Binary classification let's do that.","metadata":{}},{"cell_type":"code","source":"# drop classes except one and drop id because non useful columns\n\ntrain_df.drop(columns=[\"id\",\"identity_hate\", \"insult\", \"threat\", \"obscene\",\"severe_toxic\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.632051Z","iopub.execute_input":"2025-07-23T08:35:43.632343Z","iopub.status.idle":"2025-07-23T08:35:43.652386Z","shell.execute_reply.started":"2025-07-23T08:35:43.632326Z","shell.execute_reply":"2025-07-23T08:35:43.651597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the length of longest sentence\n\nmax_len = train_df['comment_text'].apply(lambda x: len(str(x).split())).max()\nmax_len","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.653263Z","iopub.execute_input":"2025-07-23T08:35:43.653677Z","iopub.status.idle":"2025-07-23T08:35:43.720503Z","shell.execute_reply.started":"2025-07-23T08:35:43.653650Z","shell.execute_reply":"2025-07-23T08:35:43.719856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make function for getting auc score\n\ndef roc_auc(y_pred, y_true):\n\n    fpr,tpr, threshold = metrics.roc_curve(y_true, y_pred)\n    roc_auc = metrics.auc(fpr,tpr)\n    return roc_auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.721283Z","iopub.execute_input":"2025-07-23T08:35:43.721523Z","iopub.status.idle":"2025-07-23T08:35:43.724810Z","shell.execute_reply.started":"2025-07-23T08:35:43.721508Z","shell.execute_reply":"2025-07-23T08:35:43.724327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lets discove hoe many classes at toxic columns\n\ntrain_df['toxic'].value_counts(normalize=True).plot(\n    kind=\"pie\",\n    autopct='%1.1f%%',\n    title=\"Toxic Comment Distribution\",\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.726305Z","iopub.execute_input":"2025-07-23T08:35:43.726502Z","iopub.status.idle":"2025-07-23T08:35:43.970361Z","shell.execute_reply.started":"2025-07-23T08:35:43.726487Z","shell.execute_reply":"2025-07-23T08:35:43.969622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(data=train_df, x=\"toxic\")\nplt.title(\"Distribution of Toxic Comments\")\nplt.xlabel(\"Toxic Label\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:43.970975Z","iopub.execute_input":"2025-07-23T08:35:43.971160Z","iopub.status.idle":"2025-07-23T08:35:44.087566Z","shell.execute_reply.started":"2025-07-23T08:35:43.971146Z","shell.execute_reply":"2025-07-23T08:35:44.087016Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Based on vsualization We have imbalanced Data and we must correct that. but first let's make numerical value because SMOTE technique works only with Numerical values","metadata":{}},{"cell_type":"markdown","source":"### Tokenizer","metadata":{}},{"cell_type":"code","source":"xtrain, xvalid, ytrain, yvalid = train_test_split(train_df.comment_text.values, train_df.toxic.values, \n                                                  stratify=train_df.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:45.682894Z","iopub.execute_input":"2025-07-23T08:35:45.683189Z","iopub.status.idle":"2025-07-23T08:35:45.694716Z","shell.execute_reply.started":"2025-07-23T08:35:45.683143Z","shell.execute_reply":"2025-07-23T08:35:45.694093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokens = text.Tokenizer(num_words=None)\nmax_length=1500\n\ntokens.fit_on_texts(list(xtrain) + list(xvalid))\ntrain_seq = tokens.texts_to_sequences(xtrain)\nvalid_seq = tokens.texts_to_sequences(xvalid)\n\n# zero pad the sentenecs\n\ntrain_pad = sequence.pad_sequences(train_seq, maxlen = max_len)\nvalid_pad = sequence.pad_sequences(valid_seq, maxlen = max_len)\n\nword_index = tokens.word_index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:46.805741Z","iopub.execute_input":"2025-07-23T08:35:46.805983Z","iopub.status.idle":"2025-07-23T08:35:48.070776Z","shell.execute_reply.started":"2025-07-23T08:35:46.805964Z","shell.execute_reply":"2025-07-23T08:35:48.070216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pad","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:48.866669Z","iopub.execute_input":"2025-07-23T08:35:48.866913Z","iopub.status.idle":"2025-07-23T08:35:48.871994Z","shell.execute_reply.started":"2025-07-23T08:35:48.866887Z","shell.execute_reply":"2025-07-23T08:35:48.871487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\nsmote = SOMTE(random_state=42)\nx_sampled, y_sampled = smote.fit_transform(train_pad, train_df['toxic'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-22T10:37:17.489345Z","iopub.execute_input":"2025-07-22T10:37:17.489502Z","iopub.status.idle":"2025-07-22T10:37:17.524601Z","shell.execute_reply.started":"2025-07-22T10:37:17.489489Z","shell.execute_reply":"2025-07-22T10:37:17.523691Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What is RNN?\n\n#### Recurrent Neural Network(RNN) are a type of Neural Network where the output from previous step are fed as input to the current step. In traditional neural networks, all the inputs and outputs are independent of each other, but in cases like when it is required to predict the next word of a sentence, the previous words are required and hence there is a need to remember the previous words. Thus RNN came into existence, which solved this issue with the help of a Hidden Layer.","metadata":{}},{"cell_type":"code","source":"# Strategy is used disribute to the model corss the GPU\nimport tensorflow as tf\nstrategy = tf.distribute.MirroredStrategy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:35:57.624940Z","iopub.execute_input":"2025-07-23T08:35:57.625240Z","iopub.status.idle":"2025-07-23T08:35:58.755300Z","shell.execute_reply.started":"2025-07-23T08:35:57.625219Z","shell.execute_reply":"2025-07-23T08:35:58.754486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model = Sequential()\n    model.add(Embedding(\n        input_dim=len(word_index) + 1,\n        output_dim=300,\n        input_length=max_len\n    ))\n    model.add(SimpleRNN(100))\n    model.add(Dense(1, activation=\"sigmoid\"))\n    model.build(input_shape=(None, max_len)) \n    model.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\n    model.summary()  # ✅ INSIDE the scope\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:48:06.201118Z","iopub.execute_input":"2025-07-23T08:48:06.201435Z","iopub.status.idle":"2025-07-23T08:48:06.269741Z","shell.execute_reply.started":"2025-07-23T08:48:06.201412Z","shell.execute_reply":"2025-07-23T08:48:06.269212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_pad,ytrain, epochs=10, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T08:48:10.548026Z","iopub.execute_input":"2025-07-23T08:48:10.548630Z","iopub.status.idle":"2025-07-23T09:15:12.214423Z","shell.execute_reply.started":"2025-07-23T08:48:10.548606Z","shell.execute_reply":"2025-07-23T09:15:12.213798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 64 * strategy.num_replicas_in_sync\nprint(f\"Using batch size: {batch_size}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:12.215476Z","iopub.execute_input":"2025-07-23T09:15:12.215675Z","iopub.status.idle":"2025-07-23T09:15:12.219851Z","shell.execute_reply.started":"2025-07-23T09:15:12.215660Z","shell.execute_reply":"2025-07-23T09:15:12.219152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### One-hot vectors of dimension \"vocabulary size + 1\" refer to binary vectors used to represent words, where each vector has a length equal to the number of unique words in the vocabulary plus one additional slot to account for special tokens like padding (<PAD>) or unknown words (<UNK>). In this representation, only one element in the vector is set to 1 (indicating the word's index), while all others are 0, ensuring that each word or token has a distinct, sparse binary encoding.","metadata":{}},{"cell_type":"markdown","source":"# Word Embedding","metadata":{}},{"cell_type":"markdown","source":"### ### Embedding means that we capture the semantic meaning of a word in numerical form because machines can only process numbers. In the past, scientists used a table-based method to represent semantic meaning by asking a wide range of questions and assigning fractional values to the answers, creating a manually curated semantic representation. However, this approach was inefficient, time-consuming, and limited in scalability. Modern word embeddings, like Word2Vec or GloVe, overcome these issues by learning vector representations automatically from large text corpora. These learned vectors capture relationships and contextual meanings between words—such as similarity, analogy, and syntactic roles—allowing machines to understand language in a more meaningful and efficient way.","metadata":{}},{"cell_type":"code","source":"embeddings_index = {}\n\nwith open(\"/kaggle/input/glove6b100dtxt/glove.6B.100d.txt\", \"r\", encoding=\"utf-8\") as f:\n    for line in f:\n        value = line.strip().split()\n        word = value[0]\n        bs = np.asarray([float(val) for val in value[1:]], dtype='float32')\n        embeddings_index[word] = bs\n\nprint(f\"Found {len(embeddings_index)} word vectors.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:21:18.329714Z","iopub.execute_input":"2025-07-23T09:21:18.329981Z","iopub.status.idle":"2025-07-23T09:21:27.765816Z","shell.execute_reply.started":"2025-07-23T09:21:18.329960Z","shell.execute_reply":"2025-07-23T09:21:27.765205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LSTM ","metadata":{}},{"cell_type":"markdown","source":"### ### What is LSTM?\n\n**LSTM (Long Short-Term Memory)** is a special type of Recurrent Neural Network (RNN) designed to learn and remember long-term dependencies in sequential data, such as text, time series, or speech. Traditional RNNs struggle with remembering information over long sequences due to the *vanishing gradient problem*, where gradients become too small for effective learning.\n\nLSTM solves this by introducing a **memory cell** and **three gates**:\n\n1. **Forget Gate** – decides what information to discard from the cell.\n2. **Input Gate** – decides which new information to store.\n3. **Output Gate** – decides what information to pass to the next step.\n\nThese gates allow LSTM to retain useful information for many time steps and ignore irrelevant parts, making it ideal for tasks like language modeling, machine translation, speech recognition, and sentiment analysis.\n","metadata":{}},{"cell_type":"code","source":"# create an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 100))\nfor word, i in word_index.items():\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:21:29.495446Z","iopub.execute_input":"2025-07-23T09:21:29.495694Z","iopub.status.idle":"2025-07-23T09:21:29.590415Z","shell.execute_reply.started":"2025-07-23T09:21:29.495674Z","shell.execute_reply":"2025-07-23T09:21:29.589787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Strategy is used disribute to the model corss the GPU\nstrategy = tf.distribute.MirroredStrategy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:30:31.102084Z","iopub.execute_input":"2025-07-23T09:30:31.102632Z","iopub.status.idle":"2025-07-23T09:30:31.107982Z","shell.execute_reply.started":"2025-07-23T09:30:31.102608Z","shell.execute_reply":"2025-07-23T09:30:31.107215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model = Sequential()\n    model.add(Embedding(\n        len(word_index)+1,\n        100,\n        weights=[embedding_matrix],\n        input_length = max_len,\n        trainable=False\n    ))\n    model.add(LSTM(100, dropout=0.3, recurrent_dropout=0.2))\n    model.add(Dense(1,activation = \"sigmoid\"))\n    model.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:30:32.387622Z","iopub.execute_input":"2025-07-23T09:30:32.387833Z","iopub.status.idle":"2025-07-23T09:30:32.464025Z","shell.execute_reply.started":"2025-07-23T09:30:32.387816Z","shell.execute_reply":"2025-07-23T09:30:32.463438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explain code","metadata":{}},{"cell_type":"markdown","source":"### This code builds an LSTM-based neural network using pre-trained word embeddings. First, an `embeddings_matrix` is created by mapping each word in the dataset (`word_index`) to its corresponding 300-dimensional vector from the `embeddings_index`, filling in zeros for words not found. Inside a TensorFlow `strategy.scope()` for distributed training, a Sequential model is defined. It starts with an `Embedding` layer initialized with the pre-trained embeddings, set to be non-trainable (so their values remain fixed), and matches the vocabulary size and input sequence length. Then, an `LSTM` layer with 100 units is added, using dropout and recurrent dropout to reduce overfitting. Finally, a `Dense` output layer with sigmoid activation performs binary classification. The model is compiled with binary crossentropy loss, the Adam optimizer, and accuracy as the evaluation metric.\n","metadata":{}},{"cell_type":"code","source":"model.fit(train_pad, ytrain, epochs=5, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:30:50.867918Z","iopub.execute_input":"2025-07-23T09:30:50.868198Z","iopub.status.idle":"2025-07-23T09:30:58.867335Z","shell.execute_reply.started":"2025-07-23T09:30:50.868157Z","shell.execute_reply":"2025-07-23T09:30:58.866235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.predict(valid_pad)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.736249Z","iopub.status.idle":"2025-07-23T09:15:21.736548Z","shell.execute_reply.started":"2025-07-23T09:15:21.736409Z","shell.execute_reply":"2025-07-23T09:15:21.736423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.predict(valid_pad)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.737628Z","iopub.status.idle":"2025-07-23T09:15:21.737964Z","shell.execute_reply.started":"2025-07-23T09:15:21.737778Z","shell.execute_reply":"2025-07-23T09:15:21.737793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n\n\nroc_auc = roc_auc(score, yvalid)\n\n# Plotting\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (AUC = {:.2f})'.format(roc_auc))\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')  # diagonal line\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC)')\nplt.legend(loc='lower right')\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.738886Z","iopub.status.idle":"2025-07-23T09:15:21.739125Z","shell.execute_reply.started":"2025-07-23T09:15:21.739020Z","shell.execute_reply":"2025-07-23T09:15:21.739030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GRU","metadata":{}},{"cell_type":"markdown","source":"### GRU (Gated Recurrent Unit) is a type of Recurrent Neural Network (RNN) architecture, designed to solve the vanishing gradient problem that standard RNNs face. It was introduced by Cho et al. in 2014 as a simpler and faster alternative to LSTM (Long Short-Term Memory), while still being effective at capturing dependencies in sequential data like text, time series, or speech.\n","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    model = Sequential()\n    model.add(Embedding(len(word_index)+1)\n             , 100,\n             weights=[embeddings_matrix],\n             input_length=max_len,\n             trainable=False)\n    model.add(SaptialDrpout(0.3))\n    model.add(GRU(300))\n    model.Dense(1, activation=\"sigmoid\")\n    model.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.740112Z","iopub.status.idle":"2025-07-23T09:15:21.740467Z","shell.execute_reply.started":"2025-07-23T09:15:21.740283Z","shell.execute_reply":"2025-07-23T09:15:21.740299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_pad,ytrain, epochs=10, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.741436Z","iopub.status.idle":"2025-07-23T09:15:21.741740Z","shell.execute_reply.started":"2025-07-23T09:15:21.741583Z","shell.execute_reply":"2025-07-23T09:15:21.741597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score_gru = model.predict(valid_pad)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.742684Z","iopub.status.idle":"2025-07-23T09:15:21.742983Z","shell.execute_reply.started":"2025-07-23T09:15:21.742817Z","shell.execute_reply":"2025-07-23T09:15:21.742831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"accuracy: {roc_auc(score, yvalid)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.744551Z","iopub.status.idle":"2025-07-23T09:15:21.744798Z","shell.execute_reply.started":"2025-07-23T09:15:21.744682Z","shell.execute_reply":"2025-07-23T09:15:21.744695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n\n\nroc_auc = roc_auc(score_gru, yvalid)\n\n# Plotting\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label='ROC curve (AUC = {:.2f})'.format(roc_auc))\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')  # diagonal line\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC)')\nplt.legend(loc='lower right')\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.745880Z","iopub.status.idle":"2025-07-23T09:15:21.746230Z","shell.execute_reply.started":"2025-07-23T09:15:21.746043Z","shell.execute_reply":"2025-07-23T09:15:21.746056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Bi-directional RNN","metadata":{}},{"cell_type":"markdown","source":"### A Bidirectional RNN is a type of Recurrent Neural Network where two RNNs are run in parallel:\n\n### One processes the sequence forward (from past to future)\n\n### The other processes it backward (from future to past)","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n\n    model = Sequential()\n    model.add(Embedding(len(word_index) +1),\n             100,\n             input_length=max_len,\n             weights=[embeddings_matrix],\n             trainable=False)\n    model.add(Bidirectional(LSTM(100, dropout=0.3, recurrent_dropout=0.3)))\n    model.add(Dense(1,activation=\"sigmoid\"))\n    model.compile(loss=\"binary_crossentropy\", opimizer=\"adam\", metrics=[\"accuracy\"])\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.747426Z","iopub.status.idle":"2025-07-23T09:15:21.747727Z","shell.execute_reply.started":"2025-07-23T09:15:21.747571Z","shell.execute_reply":"2025-07-23T09:15:21.747584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(train_pad,ytrain, epochs=10, batch_size=64*strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.749019Z","iopub.status.idle":"2025-07-23T09:15:21.749297Z","shell.execute_reply.started":"2025-07-23T09:15:21.749138Z","shell.execute_reply":"2025-07-23T09:15:21.749150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(valid_pad)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.750982Z","iopub.status.idle":"2025-07-23T09:15:21.751280Z","shell.execute_reply.started":"2025-07-23T09:15:21.751132Z","shell.execute_reply":"2025-07-23T09:15:21.751142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"accuracy: {roc_auc(y_pred, yvalid)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.751830Z","iopub.status.idle":"2025-07-23T09:15:21.752078Z","shell.execute_reply.started":"2025-07-23T09:15:21.751944Z","shell.execute_reply":"2025-07-23T09:15:21.751962Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transformers : Attention is all you need¶\n","metadata":{}},{"cell_type":"markdown","source":"### We’ve now arrived at the turning point in modern NLP—the breakthrough that redefined everything: **Transformers**. Introduced by Google in the groundbreaking paper *“Attention Is All You Need”*, transformers revolutionized how machines understand language. If you’ve grasped the concept of attention mechanisms, understanding transformers will be a smooth ride. Let’s dive into how this powerful architecture works and why it forms the backbone of today’s state-of-the-art NLP models.\n","metadata":{}},{"cell_type":"code","source":"# loading libararies\n\nimport os\nimport tensorflow as tf\nimport tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.model import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.752717Z","iopub.status.idle":"2025-07-23T09:15:21.753009Z","shell.execute_reply.started":"2025-07-23T09:15:21.752874Z","shell.execute_reply":"2025-07-23T09:15:21.752887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LOADING THE DATA\n\ntrain_data = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nvalid_data = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest_data = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub_data = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:15:21.754489Z","iopub.status.idle":"2025-07-23T09:15:21.754711Z","shell.execute_reply.started":"2025-07-23T09:15:21.754609Z","shell.execute_reply":"2025-07-23T09:15:21.754618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Encoding","metadata":{}},{"cell_type":"code","source":"def encoding(text, tokenizer, chunk_size=256, max_len=512):\n\n    \"\"\"\n        it's making encoding to convert the row text to sequence numerical token ids for bert model understanding the text.\n    \"\"\"\n\n    tokenizer.enable_padding(max_length=max_len)\n    tokenizer.enable_truncation(max_length=max_len)\n    ids = []\n    \n    for i in range(0, len(text), chunk_size):\n        text_chunk = text[i: i+chunk_size].tolist()\n        encode = tokenizer.encode_batch(text_chunk)\n        ids.extend([encode.ids for e in encode])\n\n    return ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:16:20.188387Z","iopub.execute_input":"2025-07-23T09:16:20.188635Z","iopub.status.idle":"2025-07-23T09:16:20.193352Z","shell.execute_reply.started":"2025-07-23T09:16:20.188619Z","shell.execute_reply":"2025-07-23T09:16:20.192699Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explain Code","metadata":{}},{"cell_type":"markdown","source":"### This `encoding` function is designed to preprocess raw text data for input into a BERT-based model. It takes in a list of text samples, a tokenizer, and optional parameters for chunk size and maximum sequence length. First, it enables padding and truncation so that all tokenized sequences are of equal length (defined by `max_len`). It then splits the input text into smaller chunks (based on `chunk_size`), converts each chunk into a list, and tokenizes them in batches using the tokenizer’s `enable_batch` method. The resulting token IDs (numerical representations of text) are extracted and accumulated into a single list called `ids`, which is returned for use as model input.\n","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\nepochs = 3\nbatch_size = 16 * strategy.num_replicas_in_sync\nmax_len = 192","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-23T09:21:01.804404Z","iopub.execute_input":"2025-07-23T09:21:01.804943Z","iopub.status.idle":"2025-07-23T09:21:01.808310Z","shell.execute_reply.started":"2025-07-23T09:21:01.804920Z","shell.execute_reply":"2025-07-23T09:21:01.807672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}