{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"GPU","colab":{"gpuType":"T4","provenance":[]},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"},{"sourceId":11650,"sourceType":"datasetVersion","datasetId":8327}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip show tensorflow\n!pip show nvidia-cuda-toolkit\n!pip show nvidia-cudnn\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T06:30:15.346834Z","iopub.execute_input":"2025-04-25T06:30:15.347584Z","iopub.status.idle":"2025-04-25T06:30:19.810556Z","shell.execute_reply.started":"2025-04-25T06:30:15.347555Z","shell.execute_reply":"2025-04-25T06:30:19.809334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nimport joblib\nimport re\nimport gc\nfrom tqdm import tqdm\nfrom sklearn import metrics\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import text\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Input, Flatten, GlobalMaxPooling1D, Bidirectional, SpatialDropout1D, LSTM, GRU, SimpleRNN, Dense, Activation, Dropout, Embedding, BatchNormalization, TextVectorization\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing import sequence\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport spacy # spacy after tensorflow to avoid conflicting GPUs\nimport warnings\nwarnings.filterwarnings('ignore')\n\nnp.random.seed(571)\n\nprint(tf.__version__)","metadata":{"id":"QwtPl7tIZBvB","outputId":"77ea511c-b675-469d-809b-be85c6528f66","execution":{"iopub.status.busy":"2025-04-25T06:34:33.283352Z","iopub.execute_input":"2025-04-25T06:34:33.284231Z","iopub.status.idle":"2025-04-25T06:34:42.187897Z","shell.execute_reply.started":"2025-04-25T06:34:33.2842Z","shell.execute_reply":"2025-04-25T06:34:42.18693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T06:31:03.367667Z","iopub.execute_input":"2025-04-25T06:31:03.368018Z","iopub.status.idle":"2025-04-25T06:31:03.373305Z","shell.execute_reply.started":"2025-04-25T06:31:03.367994Z","shell.execute_reply":"2025-04-25T06:31:03.372402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv', engine=\"python\")\ndf_all.head()\n","metadata":{"id":"33bvV823ZBvE","outputId":"c5c00c86-d4de-48e4-ddbe-4a03da9eccca","execution":{"iopub.status.busy":"2025-04-25T06:39:08.934637Z","iopub.execute_input":"2025-04-25T06:39:08.935451Z","iopub.status.idle":"2025-04-25T06:39:12.794934Z","shell.execute_reply.started":"2025-04-25T06:39:08.935426Z","shell.execute_reply":"2025-04-25T06:39:12.794092Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/jigsaw-multilingual-toxic-comment-classification')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T06:38:35.398905Z","iopub.execute_input":"2025-04-25T06:38:35.39922Z","iopub.status.idle":"2025-04-25T06:38:35.406557Z","shell.execute_reply.started":"2025-04-25T06:38:35.3992Z","shell.execute_reply":"2025-04-25T06:38:35.405698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all.info()","metadata":{"id":"gHCkmlmmZBvF","outputId":"a3fac4c3-3beb-48fe-d592-cff5ee564c29","trusted":true,"execution":{"iopub.status.busy":"2025-04-25T06:39:30.071979Z","iopub.execute_input":"2025-04-25T06:39:30.072581Z","iopub.status.idle":"2025-04-25T06:39:30.139215Z","shell.execute_reply.started":"2025-04-25T06:39:30.072558Z","shell.execute_reply":"2025-04-25T06:39:30.138249Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**See the num.of presented Samples in DataFrame**","metadata":{"id":"DHU4AsWyASL7"}},{"cell_type":"code","source":"(df_all.drop(['id','comment_text'], axis=1)).apply(lambda a: a.value_counts())","metadata":{"id":"lWOiru3SZBvF","outputId":"6cfef4db-50d0-49aa-985a-05c1c68dc050","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Merge Columns in One Column 'is_offensive' expressing to any form of offensive sentence (toxic, severe_toxic, obscene, threat, insult, identity_hate) -IF NEEDED-**\n\nI will just use the toxic column to the competition Submittion","metadata":{"id":"0QB8z1stASL8"}},{"cell_type":"code","source":"df_all['is_offensive'] = df_all['toxic'] # df_all['toxic'] | df_all['severe_toxic'] | df_all['obscene'] | df_all['threat'] | df_all['insult'] | df_all['identity_hate'] | 0\n(df_all.drop(['id','comment_text'], axis=1)).apply(lambda a: a.value_counts())","metadata":{"id":"RNYAJ38eESCW","outputId":"dedf1292-af7a-4b4f-86e3-b1e70b9690a8","execution":{"iopub.status.busy":"2025-04-25T06:40:19.553837Z","iopub.execute_input":"2025-04-25T06:40:19.554148Z","iopub.status.idle":"2025-04-25T06:40:19.583925Z","shell.execute_reply.started":"2025-04-25T06:40:19.554126Z","shell.execute_reply":"2025-04-25T06:40:19.583146Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Apply Pre-Process on text to improve the models Perforamance**\n\n**[Remove Punctuation, Special Characters & Stop Words - Lowercasing - Tokenization - Lemmatization - Handle Contraction]**","metadata":{"id":"DVcNcMRNnISJ"}},{"cell_type":"code","source":"# Disable GPU usage for SpaCy (GPUs Conflict Error in TensorFlow)\nspacy.require_cpu()\n\n# Load the SpaCy small model to tokenize the texts\nnlp = spacy.load(\"en_core_web_sm\", enable=[\"tokenizer\", \"lemmatizer\"])\n\n# Initialize tqdm for pandas\ntqdm.pandas()\n\ndef preprocess_text(text):\n    \"\"\"\n    Optimized text preprocessing using spaCy.\n    \"\"\"\n    # Pre-Process the text\n    contractions_dict = {\"'m\": \"am\", \"'s\": \"is\", \"'re\": \"are\", \"n't\": \"not\",\"'ll\": \"will\", \n                         \"'d\": \"would\", \"'ve\": \"have\", \"ca\": \"can\",\"sha\": \"shall\", \"wo\":\"will\"}\n    text = re.sub(r'[^\\w\\s\\']', '', text)\n    doc = nlp(text.lower())\n    tokens = [contractions_dict.get(token.text, token.lemma_) for token in doc if not token.is_stop]\n    preprocessed_text = ' '.join(tokens)\n    return preprocessed_text\n\n# Apply the function to the DataFrame\ndf_all['comment_text'] = df_all['comment_text'].progress_apply(preprocess_text)\n\n# Release spaCy model from memory\ndel nlp\ngc.collect()\n\n# Display The tail of the data\ndf_all.tail()","metadata":{"id":"Y-KS3YWpnISK","outputId":"7af1e2ba-6667-4607-fe24-71173a3c7220","execution":{"iopub.status.busy":"2025-04-25T06:40:40.284577Z","iopub.execute_input":"2025-04-25T06:40:40.284928Z","iopub.status.idle":"2025-04-25T06:42:55.748028Z","shell.execute_reply.started":"2025-04-25T06:40:40.284903Z","shell.execute_reply":"2025-04-25T06:42:55.747293Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Split DataFrame to train & test sets to compute the accuracy of the models**","metadata":{"id":"5aAxzru9ASL8"}},{"cell_type":"code","source":"num = 200_000\ndf = df_all.iloc[:num, :]\ntest = df_all.iloc[num:, :]\n\nprint(\"The Shape of Train Data: \", df.shape)\nprint(\"The Shape of Test Data: \", test.shape)","metadata":{"id":"s5FwW212ESCX","execution":{"iopub.status.busy":"2025-04-25T06:44:04.204476Z","iopub.execute_input":"2025-04-25T06:44:04.204856Z","iopub.status.idle":"2025-04-25T06:44:04.212334Z","shell.execute_reply.started":"2025-04-25T06:44:04.204832Z","shell.execute_reply":"2025-04-25T06:44:04.211211Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**UnderSampling for the class 0 to the length of class 1 to solve imbalanced data and decreased the size of the data**","metadata":{"id":"_YnKF55gZBvG"}},{"cell_type":"markdown","source":"**Trying to get smaller subset of data and overcome the imbalanced data problem [IF Needed]**","metadata":{"id":"7RNJbQ2wASL9"}},{"cell_type":"code","source":"# the count of rows in each class of the is_offensive 1/0\nlen_class = np.min(df.is_offensive.value_counts())\n\n# separate according to `label`\ndf_class_0 = df[df['is_offensive'] == 0]\ndf_class_1 = df[df['is_offensive'] == 1]\n\n# sample only from class 0 quantity of rows of class 1\ndf_class_0 = df_class_0.sample(len_class, replace=True)\ndf_class_1 = df_class_1.sample(len_class, replace=True)\n\ndf = pd.concat([df_class_0, df_class_1], ignore_index = True, axis=0).sample(frac = 1).reset_index()\n(df['is_offensive'].value_counts())","metadata":{"id":"ID02otJHZBvH","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Bag of Words - Traditional ML","metadata":{"id":"YVnmeoEJZBvH"}},{"cell_type":"markdown","source":"**This is the matrix will used in the competition**","metadata":{"id":"pOQXeoXDESCZ"}},{"cell_type":"code","source":"def AUC_accuracy(target, predictions):\n    '''\n    Compute the Area Under Curve (AUC) Score to get the accuracy of the model\n\n    Paramas:\n        target -> list of real numbers\n        predictions -> list of predictions numbers\n\n    return auc accuracy (float)\n    '''\n\n    fpr, tpr, thresholds = metrics.roc_curve(target, predictions)\n    roc_auc = metrics.auc(fpr, tpr)\n    return roc_auc","metadata":{"id":"ZlvCyKwdESCZ","execution":{"iopub.status.busy":"2025-04-25T06:45:04.250548Z","iopub.execute_input":"2025-04-25T06:45:04.250854Z","iopub.status.idle":"2025-04-25T06:45:04.255718Z","shell.execute_reply.started":"2025-04-25T06:45:04.250835Z","shell.execute_reply":"2025-04-25T06:45:04.25485Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get the comment text in X and is_offensive in Y\nX = df['comment_text']\ny = df['is_offensive']\n\n# spliting the train dataset to sub-train and validation data\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, stratify=y.values, random_state=42)\nX_test, y_test = test['comment_text'], test['is_offensive']\n\nprint(\"The Shape of Train Data: \", X_train.shape)\nprint(\"The Shape of Validation Data: \", X_val.shape)\nprint(\"The Shape of Test Data: \", X_test.shape)","metadata":{"id":"LKpTqT71ZBvI","execution":{"iopub.status.busy":"2025-04-25T06:45:09.883969Z","iopub.execute_input":"2025-04-25T06:45:09.885045Z","iopub.status.idle":"2025-04-25T06:45:09.913669Z","shell.execute_reply.started":"2025-04-25T06:45:09.88501Z","shell.execute_reply":"2025-04-25T06:45:09.912705Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vectorize text\nvectorizer = TfidfVectorizer(stop_words='english')\nX_train_vec = vectorizer.fit_transform(X_train)\nX_val_vec = vectorizer.transform(X_val)\n\nprint(\"The Shape of Train Data After Victorizer: \", X_train_vec.shape)\nprint(\"The Shape of Validation Data After Victorizer: \", X_val_vec.shape)","metadata":{"id":"5XPYTc4_ZBvJ","outputId":"79cd362f-4a3b-4410-f4ee-85eda40790f1","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train simple ML model for text classification task\nLRmodel  =  LogisticRegression(max_iter=1000, n_jobs=-1)\nLRmodel.fit(X_train_vec, y_train)\n\n# Predict on validation data\ny_predLR = LRmodel.predict(X_val_vec)\n\n# Evaluate the model\nprint(\"Logistic Regression on Validation Data\")\nprint(classification_report(y_val, y_predLR))","metadata":{"id":"SzIPyc3pESCb","outputId":"379538e2-d906-4cd3-f89c-9bf7799016bd","execution":{"iopub.status.busy":"2025-04-25T06:46:05.414231Z","iopub.execute_input":"2025-04-25T06:46:05.414929Z","iopub.status.idle":"2025-04-25T06:46:08.032901Z","shell.execute_reply.started":"2025-04-25T06:46:05.414901Z","shell.execute_reply":"2025-04-25T06:46:08.031999Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Let's try to get the accuracy on test dataset**","metadata":{"id":"znVDXq4BESCb"}},{"cell_type":"code","source":"test_vec = vectorizer.transform(X_test)\ny_preds = LRmodel.predict(test_vec)\n\n# Evaluate the model\nprint(\"Logistic Regression on Test Data\")\nprint(classification_report(y_test, y_preds))","metadata":{"id":"wp1kqylFESCc","outputId":"fc8e9bc5-2ad1-4534-9d4b-f5fc76164d2a","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Note we need the decision_function to predict confidence scores for samples\n# (the distance between point and hyperplane)\n# for getting area under curve (auc) as an accuracy matrix\ny_preds_prob = LRmodel.decision_function(test_vec)\nlr_score = AUC_accuracy(y_test, y_preds_prob)\nlr_score","metadata":{"id":"oA9otQZLESCd","outputId":"e653285f-5214-4aa4-8f7a-716258dbca5a","execution":{"iopub.status.busy":"2025-04-25T06:46:50.248753Z","iopub.execute_input":"2025-04-25T06:46:50.249115Z","iopub.status.idle":"2025-04-25T06:46:50.263504Z","shell.execute_reply.started":"2025-04-25T06:46:50.249093Z","shell.execute_reply":"2025-04-25T06:46:50.2626Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Deep Learning","metadata":{"id":"JrcNEwEiZBvK"}},{"cell_type":"markdown","source":"**Get The longest sequance on the Train data**","metadata":{"id":"zYRzE-tAASMA"}},{"cell_type":"code","source":"df['comment_text'].apply(lambda comment: len(comment.split())).max()","metadata":{"id":"rWHmKBmdhhm_","outputId":"6f604d95-15fc-42aa-c62c-dcdd7db810cb","execution":{"iopub.status.busy":"2025-04-25T06:48:05.455812Z","iopub.execute_input":"2025-04-25T06:48:05.45615Z","iopub.status.idle":"2025-04-25T06:48:05.577974Z","shell.execute_reply.started":"2025-04-25T06:48:05.456126Z","shell.execute_reply":"2025-04-25T06:48:05.576864Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Tokenize & Padding Sequance of the training data from tensorflow pre-processing**","metadata":{"id":"gOKgGt6JASMA"}},{"cell_type":"code","source":"# Define embedding dimensions and maximum sequence length\n# embedding_dim: the number of features that represent the word\nembedding_dim = 300\nmax_len = 1300\n\n# using keras tokenizer here\ntoken = text.Tokenizer(oov_token=\"<OOV>\")\n\n# tokenize the texts to sort desc\n# the highest counts for the word\n# in the corpus and get the indeces\ntoken.fit_on_texts(X_train.to_list() + X_val.to_list())\nX_train_seq = token.texts_to_sequences(X_train)\nX_val_seq = token.texts_to_sequences(X_val)\n\n# padding the sequences to 2000 indices [constant shape]\n# for fitting input layer of the Deep Learning model\nX_train_pad = sequence.pad_sequences(X_train_seq, maxlen=max_len, padding='post')\nX_val_pad = sequence.pad_sequences(X_val_seq, maxlen=max_len, padding='post')\n\nword2idx = token.word_index\nvocab_size = len(token.word_index) + 1\n\nprint(\"The distinct number of words in corpus\", vocab_size)","metadata":{"id":"BAbVgqOGESCf","outputId":"a400eade-04f2-4e47-d094-89b22c981e62","execution":{"iopub.status.busy":"2025-04-25T06:48:25.19101Z","iopub.execute_input":"2025-04-25T06:48:25.191332Z","iopub.status.idle":"2025-04-25T06:48:27.341517Z","shell.execute_reply.started":"2025-04-25T06:48:25.191311Z","shell.execute_reply":"2025-04-25T06:48:27.340644Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Tokenize & Padding Sequance of the TEST data**","metadata":{"id":"xt5fqVXfASME"}},{"cell_type":"code","source":"X_test_seq = token.texts_to_sequences(X_test)\nX_test_pad = sequence.pad_sequences(X_test_seq, maxlen=max_len, padding='post')\nprint(\"tokenize the test dataset ..\")\nprint(\"The Test Shape: \", X_test_pad.shape)","metadata":{"id":"6xheO5M7ESCg","outputId":"3fe173d8-45ec-40c0-c0dd-74ecdf34f484","execution":{"iopub.status.busy":"2025-04-25T06:48:53.812177Z","iopub.execute_input":"2025-04-25T06:48:53.812469Z","iopub.status.idle":"2025-04-25T06:48:54.477745Z","shell.execute_reply.started":"2025-04-25T06:48:53.81245Z","shell.execute_reply":"2025-04-25T06:48:54.47684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Let's Make Simple Neural Nets for BagOfWords [Count matrix] model to get the best accuracy for Traditional ML**","metadata":{"id":"3Y8ljQVhASL_"}},{"cell_type":"code","source":"with strategy.scope():\n    # Create the model\n    model = Sequential()\n\n    # Add layers\n    model.add(Embedding(input_dim=vocab_size,\n                        output_dim=embedding_dim,\n                        input_length=max_len,\n                        mask_zero=True))\n\n    model.add(GlobalMaxPooling1D())\n\n    model.add(Dense(1024))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(512))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(256))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    # For binary classification\n    model.add(Dense(1, activation='sigmoid'))\n\n    # Compile the model\n    model.compile(optimizer='adam',\n                  loss='binary_crossentropy',\n                  metrics=['accuracy'])\n\n# Train the model\nhistory = model.fit(X_train_pad, y_train,\n                    validation_data=(X_val_pad, y_val),\n                    epochs=10,\n                    batch_size=64*strategy.num_replicas_in_sync,\n                    callbacks=[EarlyStopping(monitor='val_loss', patience=3, verbose=1)])\n\n\n# Summary of the model\nmodel.summary()\n\n","metadata":{"id":"2u26ug40fDNX","outputId":"bda656e3-5812-42f1-f98b-c8cfd9e748b1","execution":{"iopub.status.busy":"2025-04-25T07:11:38.05786Z","iopub.execute_input":"2025-04-25T07:11:38.058164Z","iopub.status.idle":"2025-04-25T07:35:15.711539Z","shell.execute_reply.started":"2025-04-25T07:11:38.058143Z","shell.execute_reply":"2025-04-25T07:35:15.710861Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds_prob = model.predict(X_test_pad)\nprint(\"Simple Neural Nets BagOfWords Result\\n\")\nprint(classification_report(y_test, (preds_prob>0.5).astype(int))) # convert probabilities to 1 or 0 according the threshold 0.5","metadata":{"id":"o4e3doDiv0N2","outputId":"559b6ba0-8498-4a78-9464-dff8b16157e6","execution":{"iopub.status.busy":"2025-04-25T07:35:48.650668Z","iopub.execute_input":"2025-04-25T07:35:48.650984Z","iopub.status.idle":"2025-04-25T07:36:10.068804Z","shell.execute_reply.started":"2025-04-25T07:35:48.650964Z","shell.execute_reply":"2025-04-25T07:36:10.06784Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BagOfWords_score = AUC_accuracy(y_test, preds_prob)\nBagOfWords_score","metadata":{"id":"F3_ve7lxESCh","outputId":"d519594b-0aed-49e5-e4b7-6b8cbad384bb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**It is Limited, The Traditional Machine Learning and Simple Neural Nets is not the solution**","metadata":{"id":"lCfUm_r2sxqc"}},{"cell_type":"markdown","source":"## GloVe (Word Embeddings)","metadata":{"id":"h4YC4fB3ESCh"}},{"cell_type":"code","source":"def word2GloVeEmbed(corpus, filepath= r'/kaggle/input/glove840b300dtxt/glove.840B.300d.txt'):\n    \"\"\"\n    Load GloVe embeddings for specific words from a file\n\n    Args:\n        corpus (set): A set of words to extract embeddings for\n        filepath (str): Path to the GloVe embeddings file\n\n    Returns:\n        dict: A dictionary mapping words to their GloVe embeddings\n    \"\"\"\n    word2vec = {}\n    with open(filepath, 'r', encoding='utf-8') as file:\n        for line in tqdm(file):\n            values = line.rstrip().split(' ')\n            word = ' '.join(values[:-300])\n            if word in corpus: # if the word in the corpus\n                vector = np.asarray(values[-300:], dtype='float32') # 300 is the number of the length of the embeddings\n                word2vec[word] = vector\n            if len(word2vec) == len(corpus):  # Stop early if all words are found\n                break\n    return word2vec","metadata":{"id":"zZ-yroIhESCh","execution":{"iopub.status.busy":"2025-04-25T07:36:56.856705Z","iopub.execute_input":"2025-04-25T07:36:56.857071Z","iopub.status.idle":"2025-04-25T07:36:56.864354Z","shell.execute_reply.started":"2025-04-25T07:36:56.857048Z","shell.execute_reply":"2025-04-25T07:36:56.863466Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**We have already tokenized and paded our text to extract the words from train dateset**","metadata":{"id":"hvs2ZEDdESCi"}},{"cell_type":"code","source":"# convert all the words of the train-data into GloVe Embedding Vectors\nword2vec = word2GloVeEmbed(word2idx.keys())\nprint(f'The shape of the word2vec: {len(word2vec)}')","metadata":{"id":"29njm17KZBvK","outputId":"a0f92b1f-24cd-4ac2-be20-8682038e9a42","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Save the subset to a text file\n# with open(\"dataset_glove.txt\", 'w', encoding='utf-8') as output_file:\n#     for word, vector in word2vec.items():\n#         vector_str = ' '.join(map(str, vector))\n#         output_file.write(f\"{word} {vector_str}\\n\")","metadata":{"id":"iUjEuAMWrneb","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"The vector of the word 'The'\", word2vec['the'])","metadata":{"id":"vC9roIwYESCi","outputId":"276f6d4e-aadb-4fdb-8009-ce0bcdf4f57b","execution":{"iopub.status.busy":"2025-04-25T07:39:49.413934Z","iopub.execute_input":"2025-04-25T07:39:49.414239Z","iopub.status.idle":"2025-04-25T07:39:50.289425Z","shell.execute_reply.started":"2025-04-25T07:39:49.41422Z","shell.execute_reply":"2025-04-25T07:39:50.288507Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Extract Vectors of the unique words in dataset**","metadata":{"id":"7ECTPAzunISS"}},{"cell_type":"code","source":"# create an embeddings matrix for the words we have in the dataset\nembed_matrix = np.zeros((vocab_size, embedding_dim))\nfor word, i in tqdm(word2idx.items()):\n    vector = word2vec.get(word, None)\n    if vector is not None:\n        embed_matrix[i] = vector\n\nprint(f'\\nThe shape of embeddings matrix: {embed_matrix.shape}')","metadata":{"id":"06Lcg_0HESCj","outputId":"5fafe84d-f9c1-4bcd-d669-8338cc8aa355","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RNN model","metadata":{"id":"PF6ittpUESCj"}},{"cell_type":"code","source":"# A SimpleRNN with GloVe embeddings\nwith strategy.scope():\n    # Create the model\n    model = Sequential()\n\n    # Add layers\n    model.add(Embedding(input_dim=vocab_size,\n                        output_dim=embedding_dim,\n                        input_length=max_len,\n                        mask_zero=True,\n                        weights=[embed_matrix],\n                        trainable=False\n                       ))\n\n    model.add(SimpleRNN(128))\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1024))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(512))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(256))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    # Output layer\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nmodel.fit(X_train_pad, y_train,\n          validation_data=(X_val_pad, y_val),\n          epochs=10,\n          batch_size=64*strategy.num_replicas_in_sync,\n          callbacks=[EarlyStopping(monitor='val_loss', patience=3, verbose=1)])\n\nmodel.summary()","metadata":{"id":"vTHtFkG7ZBvK","outputId":"42769ac8-6d63-4ee8-c35f-c0368e368992","execution":{"iopub.status.busy":"2025-04-25T07:40:31.364415Z","iopub.execute_input":"2025-04-25T07:40:31.365174Z","iopub.status.idle":"2025-04-25T08:18:47.929708Z","shell.execute_reply.started":"2025-04-25T07:40:31.365148Z","shell.execute_reply":"2025-04-25T08:18:47.928607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_test_pad, y_test)[1]","metadata":{"id":"FF3E0igPxfTu","outputId":"a32f2811-4b2c-4352-e8f1-48bff0d720e4","execution":{"iopub.status.busy":"2025-04-25T08:18:47.931567Z","iopub.execute_input":"2025-04-25T08:18:47.931891Z","iopub.status.idle":"2025-04-25T08:20:55.736866Z","shell.execute_reply.started":"2025-04-25T08:18:47.931871Z","shell.execute_reply":"2025-04-25T08:20:55.735965Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get predictions\nSimpleRNN_preds = np.round(model.predict(X_test_pad)).astype(int)\nprint(\"Simple RNN predictions \\n\", classification_report(y_test, SimpleRNN_preds))","metadata":{"id":"D46cNs4_ZBvL","outputId":"15348287-469b-4724-c42c-3514bc893549","execution":{"iopub.status.busy":"2025-04-25T08:20:55.737832Z","iopub.execute_input":"2025-04-25T08:20:55.738105Z","iopub.status.idle":"2025-04-25T08:23:25.991884Z","shell.execute_reply.started":"2025-04-25T08:20:55.738084Z","shell.execute_reply":"2025-04-25T08:23:25.990821Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rnn_score = AUC_accuracy(y_test, SimpleRNN_preds)\nrnn_score","metadata":{"id":"tLcj4xH2ESCr","outputId":"f2c67c9f-c126-4a85-a845-ccbab417c32d","execution":{"iopub.status.busy":"2025-04-25T08:23:25.994562Z","iopub.execute_input":"2025-04-25T08:23:25.994935Z","iopub.status.idle":"2025-04-25T08:23:26.00492Z","shell.execute_reply.started":"2025-04-25T08:23:25.994912Z","shell.execute_reply":"2025-04-25T08:23:26.004053Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LSTM","metadata":{"id":"PGZE6OiDESCr"}},{"cell_type":"code","source":"# A LSTM with GloVe embeddings\nwith strategy.scope():\n    # Create the model\n    model = Sequential()\n\n    # Add layers\n    model.add(Embedding(input_dim=vocab_size,\n                        output_dim=embedding_dim,\n                        input_length=max_len,\n                        mask_zero=True,\n                        weights=[embed_matrix],\n                        trainable=False\n                       ))\n\n    model.add(LSTM(128, use_cudnn=False))\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1024))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(512))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    model.add(Dense(256))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n\n    # for binary classification\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nmodel.fit(X_train_pad, y_train,\n          validation_data=(X_val_pad, y_val),\n          epochs=10,\n          batch_size=64*strategy.num_replicas_in_sync,\n          callbacks=[EarlyStopping(monitor='val_loss', patience=3, verbose=1)])\n\nmodel.summary()","metadata":{"id":"wH8w4iBWESCr","outputId":"64951a03-8d7c-4bb7-e301-2efb17f5cfaa","execution":{"iopub.status.busy":"2025-04-25T08:23:26.005856Z","iopub.execute_input":"2025-04-25T08:23:26.006282Z","iopub.status.idle":"2025-04-25T11:03:26.032365Z","shell.execute_reply.started":"2025-04-25T08:23:26.00626Z","shell.execute_reply":"2025-04-25T11:03:26.030668Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_test_pad, y_test)[1]","metadata":{"id":"3-PviYpwESCs","outputId":"5a47f94c-9edc-442d-ad3a-73576b7c9636","execution":{"iopub.status.busy":"2025-04-25T11:03:26.034648Z","iopub.execute_input":"2025-04-25T11:03:26.035111Z","iopub.status.idle":"2025-04-25T11:10:55.634396Z","shell.execute_reply.started":"2025-04-25T11:03:26.035086Z","shell.execute_reply":"2025-04-25T11:10:55.633569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get predictions\nLSTM_preds = np.round(model.predict(X_test_pad)).astype(int)\nprint(\"Simple LSTM predictions \\n\", classification_report(y_test, LSTM_preds))","metadata":{"id":"ltg3uMMcESCt","outputId":"3b95c2ed-9727-4019-957e-7f54684eb8e5","execution":{"iopub.status.busy":"2025-04-25T11:10:55.635477Z","iopub.execute_input":"2025-04-25T11:10:55.63601Z","iopub.status.idle":"2025-04-25T11:18:21.520024Z","shell.execute_reply.started":"2025-04-25T11:10:55.635987Z","shell.execute_reply":"2025-04-25T11:18:21.518868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LSTM_score = AUC_accuracy(y_test, LSTM_preds)\nLSTM_score","metadata":{"id":"qUXOrjAhESCt","outputId":"a7ac9b60-0eb1-4451-d294-bf09305dc6f7","execution":{"iopub.status.busy":"2025-04-25T11:18:21.521281Z","iopub.execute_input":"2025-04-25T11:18:21.521618Z","iopub.status.idle":"2025-04-25T11:18:21.53196Z","shell.execute_reply.started":"2025-04-25T11:18:21.521591Z","shell.execute_reply":"2025-04-25T11:18:21.531156Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TEST ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:33:21.533635Z","iopub.execute_input":"2025-04-25T11:33:21.534617Z","iopub.status.idle":"2025-04-25T11:33:21.539762Z","shell.execute_reply.started":"2025-04-25T11:33:21.534584Z","shell.execute_reply":"2025-04-25T11:33:21.538846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\n# Assuming you've already defined vocab_size and max_len\ntokenizer = Tokenizer(num_words=vocab_size, oov_token=\"<OOV>\")\ntokenizer.fit_on_texts(X_train)  # Make sure X_train is available and contains the training texts\n\nmax_len = 1300  # Use the same max_len used during training\n\n# Prediction function\ndef predict_toxicity(text):\n    seq = tokenizer.texts_to_sequences([text])\n    padded = pad_sequences(seq, maxlen=max_len)\n    pred = model.predict(padded)[0][0]\n    return \"Toxic\" if pred >= 0.5 else \"Not Toxic\"\n\n# Test it\ncomment = input(\"Enter a comment: \")\nprint(\"Prediction:\", predict_toxicity(comment))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-25T11:33:38.553589Z","iopub.execute_input":"2025-04-25T11:33:38.553963Z","iopub.status.idle":"2025-04-25T11:33:55.777447Z","shell.execute_reply.started":"2025-04-25T11:33:38.55394Z","shell.execute_reply":"2025-04-25T11:33:55.776557Z"}},"outputs":[],"execution_count":null}]}