{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":8076,"databundleVersionId":44219},{"sourceType":"datasetVersion","sourceId":1246668,"datasetId":715814,"databundleVersionId":1278465},{"sourceType":"datasetVersion","sourceId":2624724,"datasetId":1595713,"databundleVersionId":2668532}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# NLP Midterm — Binary Toxic Comment Classification\n**Baglan_Daulet_NLP_Midterm**\n\nThis notebook implements 7 models for binary toxic comment classification:\n- **Part A:** Random Forest + TF-IDF (Model 1)\n- **Part B:** 6 Deep Learning models with various embeddings (Models 2–7)","metadata":{}},{"cell_type":"markdown","source":"## 0. Install & Import Dependencies","metadata":{}},{"cell_type":"code","source":"!pip install gensim --quiet\n!pip install kaggle --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:45:55.966459Z","iopub.execute_input":"2026-04-10T12:45:55.966786Z","iopub.status.idle":"2026-04-10T12:46:02.821067Z","shell.execute_reply.started":"2026-04-10T12:45:55.966761Z","shell.execute_reply":"2026-04-10T12:46:02.820195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nnltk.download('stopwords', quiet=True)\nnltk.download('punkt', quiet=True)\nnltk.download('punkt_tab', quiet=True)\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import (\n    Embedding, LSTM, Bidirectional, Dense, Dropout,\n    GlobalMaxPooling1D, Conv1D, MaxPooling1D, BatchNormalization\n)\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nfrom gensim.models import Word2Vec, KeyedVectors\n\nprint('All libraries imported successfully!')\nprint(f'TensorFlow version: {tf.__version__}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:02.823079Z","iopub.execute_input":"2026-04-10T12:46:02.823424Z","iopub.status.idle":"2026-04-10T12:46:02.832205Z","shell.execute_reply.started":"2026-04-10T12:46:02.823396Z","shell.execute_reply":"2026-04-10T12:46:02.831403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Load & Prepare Dataset","metadata":{}},{"cell_type":"code","source":"import zipfile\n\nZIP_PATH = '/kaggle/input/competitions/jigsaw-toxic-comment-classification-challenge/train.csv.zip'\n\nwith zipfile.ZipFile(ZIP_PATH, 'r') as z:\n    z.extractall('/kaggle/working/')\n    print('Extracted:', z.namelist())\n\nTRAIN_PATH = '/kaggle/working/train.csv'\ndf = pd.read_csv(TRAIN_PATH)\nprint('Shape:', df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:02.833167Z","iopub.execute_input":"2026-04-10T12:46:02.833560Z","iopub.status.idle":"2026-04-10T12:46:04.280972Z","shell.execute_reply.started":"2026-04-10T12:46:02.833539Z","shell.execute_reply":"2026-04-10T12:46:04.280033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:04.283073Z","iopub.execute_input":"2026-04-10T12:46:04.283294Z","iopub.status.idle":"2026-04-10T12:46:04.292004Z","shell.execute_reply.started":"2026-04-10T12:46:04.283274Z","shell.execute_reply":"2026-04-10T12:46:04.291237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"toxic_cols = ['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']\ndf['label'] = (df[toxic_cols].sum(axis=1) > 0).astype(int)\n\nprint('Label distribution:')\nprint(df['label'].value_counts())\nprint(f\"Non-Toxic: {(df['label']==0).sum()} | Toxic: {(df['label']==1).sum()}\")\n\ndf = df[['comment_text', 'label']].copy()\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:04.292953Z","iopub.execute_input":"2026-04-10T12:46:04.293322Z","iopub.status.idle":"2026-04-10T12:46:04.338982Z","shell.execute_reply.started":"2026-04-10T12:46:04.293290Z","shell.execute_reply":"2026-04-10T12:46:04.338334Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Text Preprocessing","metadata":{}},{"cell_type":"code","source":"STOP_WORDS = set(stopwords.words('english'))\n\ndef clean_text(text):\n    \"\"\"Lowercase → remove punctuation/special chars/numbers → remove stop words.\"\"\"\n    text = text.lower()\n    text = re.sub(r'[^a-z\\s]', '', text)\n    text = re.sub(r'\\s+', ' ', text).strip()\n    tokens = text.split()\n    tokens = [t for t in tokens if t not in STOP_WORDS]\n    return ' '.join(tokens)\n\ndef tokenize_text(text):\n    \"\"\"Return list of tokens (for DL embedding lookup).\"\"\"\n    return text.split()\n\nprint('Cleaning text... (this may take ~1-2 minutes)')\ndf['clean'] = df['comment_text'].astype(str).apply(clean_text)\nprint('Done!')\ndf[['comment_text', 'clean', 'label']].head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:04.340021Z","iopub.execute_input":"2026-04-10T12:46:04.340617Z","iopub.status.idle":"2026-04-10T12:46:11.080025Z","shell.execute_reply.started":"2026-04-10T12:46:04.340592Z","shell.execute_reply":"2026-04-10T12:46:11.079380Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Train / Test Split & Tokenization for DL","metadata":{}},{"cell_type":"code","source":"X = df['clean'].values\ny = df['label'].values\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\nprint(f'Train: {X_train.shape[0]}  |  Test: {X_test.shape[0]}')\n\nMAX_VOCAB  = 50_000\nMAX_LEN    = 200       \nEMBED_DIM  = 100      \n\ntokenizer = Tokenizer(num_words=MAX_VOCAB, oov_token='<OOV>')\ntokenizer.fit_on_texts(X_train)\n\nword_index = tokenizer.word_index\nvocab_size  = min(MAX_VOCAB, len(word_index)) + 1\nprint(f'Vocabulary size: {vocab_size}')\n\ndef to_padded(texts):\n    seqs = tokenizer.texts_to_sequences(texts)\n    return pad_sequences(seqs, maxlen=MAX_LEN, padding='post', truncating='post')\n\nX_train_pad = to_padded(X_train)\nX_test_pad  = to_padded(X_test)\nprint(f'Padded train shape: {X_train_pad.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:11.080900Z","iopub.execute_input":"2026-04-10T12:46:11.081268Z","iopub.status.idle":"2026-04-10T12:46:18.723607Z","shell.execute_reply.started":"2026-04-10T12:46:11.081242Z","shell.execute_reply":"2026-04-10T12:46:18.722703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Helper Functions","metadata":{}},{"cell_type":"code","source":"def plot_history(history, title='Model'):\n    fig, axes = plt.subplots(1, 2, figsize=(14, 4))\n\n    axes[0].plot(history.history['accuracy'],     label='Train Acc')\n    axes[0].plot(history.history['val_accuracy'], label='Val Acc')\n    axes[0].set_title(f'{title} — Accuracy')\n    axes[0].set_xlabel('Epoch'); axes[0].set_ylabel('Accuracy')\n    axes[0].legend()\n\n    axes[1].plot(history.history['loss'],     label='Train Loss')\n    axes[1].plot(history.history['val_loss'], label='Val Loss')\n    axes[1].set_title(f'{title} — Loss')\n    axes[1].set_xlabel('Epoch'); axes[1].set_ylabel('Loss')\n    axes[1].legend()\n\n    plt.tight_layout()\n    plt.show()\n\ndef plot_cm(y_true, y_pred, title='Model'):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(5, 4))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n                xticklabels=['Non-Toxic', 'Toxic'],\n                yticklabels=['Non-Toxic', 'Toxic'])\n    plt.title(f'Confusion Matrix — {title}')\n    plt.ylabel('True Label'); plt.xlabel('Predicted Label')\n    plt.tight_layout()\n    plt.show()\n\nresults = {}\n\ndef report(name, model, X_tr, X_te, y_tr, y_te, is_dl=False):\n    if is_dl:\n        train_preds = (model.predict(X_tr, verbose=0) > 0.5).astype(int).flatten()\n        test_preds  = (model.predict(X_te, verbose=0) > 0.5).astype(int).flatten()\n    else:\n        train_preds = model.predict(X_tr)\n        test_preds  = model.predict(X_te)\n\n    tr_acc = accuracy_score(y_tr, train_preds)\n    te_acc = accuracy_score(y_te, test_preds)\n    results[name] = (tr_acc, te_acc)\n\n    print(f'\\n── {name} ──')\n    print(f'  Training Accuracy : {tr_acc:.4f}')\n    print(f'  Testing  Accuracy : {te_acc:.4f}')\n    print(classification_report(y_te, test_preds, target_names=['Non-Toxic', 'Toxic']))\n    plot_cm(y_te, test_preds, title=name)\n    return test_preds\n\ndef get_callbacks():\n    return [\n        EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True),\n        ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, min_lr=1e-6)\n    ]\n\ndef build_dl_model(embedding_layer):\n    \"\"\"LSTM + Conv1D + Pooling + Dropout + BatchNorm + Dense (≥3 layer types).\"\"\"\n    model = Sequential([\n        embedding_layer,\n        Conv1D(128, 5, activation='relu'),\n        MaxPooling1D(2),\n        BatchNormalization(),\n        Dropout(0.3),\n        Bidirectional(LSTM(64, return_sequences=False)),\n        Dropout(0.3),\n        Dense(64, activation='relu'),\n        BatchNormalization(),\n        Dropout(0.3),\n        Dense(1, activation='sigmoid')\n    ])\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    return model\n\nEPOCHS     = 10\nBATCH_SIZE = 256\n\nprint('Helper functions defined!')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:18.725442Z","iopub.execute_input":"2026-04-10T12:46:18.726288Z","iopub.status.idle":"2026-04-10T12:46:18.738403Z","shell.execute_reply.started":"2026-04-10T12:46:18.726259Z","shell.execute_reply":"2026-04-10T12:46:18.737757Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Part A — Model 1: Random Forest + TF-IDF","metadata":{}},{"cell_type":"code","source":"print('Fitting TF-IDF vectorizer...')\ntfidf = TfidfVectorizer(max_features=50_000, ngram_range=(1, 2), sublinear_tf=True)\nX_train_tfidf = tfidf.fit_transform(X_train)\nX_test_tfidf  = tfidf.transform(X_test)\nprint(f'TF-IDF matrix shape: {X_train_tfidf.shape}')\n\nprint('Training Random Forest...')\nrf = RandomForestClassifier(\n    n_estimators=200,\n    max_depth=None,\n    n_jobs=-1,\n    random_state=42,\n    class_weight='balanced'\n)\nrf.fit(X_train_tfidf, y_train)\n\nreport('Model 1: Random Forest + TF-IDF',\n       rf, X_train_tfidf, X_test_tfidf, y_train, y_test, is_dl=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:46:18.739388Z","iopub.execute_input":"2026-04-10T12:46:18.739712Z","iopub.status.idle":"2026-04-10T12:54:00.534304Z","shell.execute_reply.started":"2026-04-10T12:46:18.739680Z","shell.execute_reply":"2026-04-10T12:54:00.533618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Part B — Model 2: DL with Keras Embedding (From Scratch)","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nembed_layer = Embedding(\n    input_dim=vocab_size,\n    output_dim=EMBED_DIM,\n    input_length=MAX_LEN,\n    trainable=True  \n)\n\nmodel2 = build_dl_model(embed_layer)\nmodel2.summary()\n\nhistory2 = model2.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history2, 'Model 2: Keras Embedding (Scratch)')\nreport('Model 2: Keras Embedding (Scratch)',\n       model2, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:54:00.537089Z","iopub.execute_input":"2026-04-10T12:54:00.537499Z","iopub.status.idle":"2026-04-10T12:55:52.744219Z","shell.execute_reply.started":"2026-04-10T12:54:00.537474Z","shell.execute_reply":"2026-04-10T12:55:52.743287Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 3: DL with Self-Trained Word2Vec","metadata":{}},{"cell_type":"code","source":"print('Training Word2Vec on training data...')\ntrain_tokens = [text.split() for text in X_train]\n\nw2v_self = Word2Vec(\n    sentences=train_tokens,\n    vector_size=EMBED_DIM,\n    window=5,\n    min_count=2,\n    workers=4,\n    epochs=5\n)\nprint('Word2Vec trained!')\n\ndef build_embedding_matrix(keyed_vectors, word_index, vocab_size, embed_dim):\n    matrix = np.zeros((vocab_size, embed_dim))\n    hits = 0\n    for word, idx in word_index.items():\n        if idx >= vocab_size:\n            continue\n        if word in keyed_vectors:\n            matrix[idx] = keyed_vectors[word]\n            hits += 1\n    print(f'  Coverage: {hits}/{min(vocab_size, len(word_index))} words ({hits/min(vocab_size,len(word_index))*100:.1f}%)')\n    return matrix\n\nemb_matrix_self = build_embedding_matrix(\n    w2v_self.wv, word_index, vocab_size, EMBED_DIM\n)\n\ntf.keras.backend.clear_session()\n\nembed_layer3 = Embedding(\n    input_dim=vocab_size,\n    output_dim=EMBED_DIM,\n    weights=[emb_matrix_self],\n    input_length=MAX_LEN,\n    trainable=True\n)\n\nmodel3 = build_dl_model(embed_layer3)\n\nhistory3 = model3.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history3, 'Model 3: Self-Trained Word2Vec')\nreport('Model 3: Self-Trained Word2Vec',\n       model3, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:55:52.745551Z","iopub.execute_input":"2026-04-10T12:55:52.745925Z","iopub.status.idle":"2026-04-10T12:58:04.296644Z","shell.execute_reply.started":"2026-04-10T12:55:52.745850Z","shell.execute_reply":"2026-04-10T12:58:04.295949Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 4: DL with Pre-trained Word2Vec (Trainable)","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith('.bin'):\n            print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:58:04.297683Z","iopub.execute_input":"2026-04-10T12:58:04.298393Z","iopub.status.idle":"2026-04-10T12:58:04.361136Z","shell.execute_reply.started":"2026-04-10T12:58:04.298368Z","shell.execute_reply":"2026-04-10T12:58:04.360500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"W2V_PATH = '/kaggle/input/datasets/sugataghosh/google-word2vec/GoogleNews-vectors-negative300.bin'\n\nprint('Loading Google News Word2Vec (this may take ~2 min)...')\nw2v_google = KeyedVectors.load_word2vec_format(W2V_PATH, binary=True)\nprint('Loaded!')\n\nW2V_DIM = 300  \n\nemb_matrix_google = build_embedding_matrix(\n    w2v_google, word_index, vocab_size, W2V_DIM\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:58:04.362027Z","iopub.execute_input":"2026-04-10T12:58:04.362370Z","iopub.status.idle":"2026-04-10T12:58:41.347106Z","shell.execute_reply.started":"2026-04-10T12:58:04.362348Z","shell.execute_reply":"2026-04-10T12:58:41.346190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nembed_layer4 = Embedding(\n    input_dim=vocab_size,\n    output_dim=W2V_DIM,\n    weights=[emb_matrix_google],\n    input_length=MAX_LEN,\n    trainable=True  \n)\n\nmodel4 = build_dl_model(embed_layer4)\n\nhistory4 = model4.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history4, 'Model 4: Pre-trained W2V (Trainable)')\nreport('Model 4: Pre-trained W2V (Trainable)',\n       model4, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T12:58:41.348164Z","iopub.execute_input":"2026-04-10T12:58:41.348634Z","iopub.status.idle":"2026-04-10T13:01:15.361342Z","shell.execute_reply.started":"2026-04-10T12:58:41.348574Z","shell.execute_reply":"2026-04-10T13:01:15.360430Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 5: DL with Pre-trained Word2Vec (Non-trainable / Frozen)","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nembed_layer5 = Embedding(\n    input_dim=vocab_size,\n    output_dim=W2V_DIM,\n    weights=[emb_matrix_google],\n    input_length=MAX_LEN,\n    trainable=False  \n)\n\nmodel5 = build_dl_model(embed_layer5)\n\nhistory5 = model5.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history5, 'Model 5: Pre-trained W2V (Non-trainable)')\nreport('Model 5: Pre-trained W2V (Non-trainable)',\n       model5, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:01:15.362422Z","iopub.execute_input":"2026-04-10T13:01:15.362731Z","iopub.status.idle":"2026-04-10T13:03:08.740420Z","shell.execute_reply.started":"2026-04-10T13:01:15.362709Z","shell.execute_reply":"2026-04-10T13:03:08.739723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 6: DL with GloVe (Trainable)","metadata":{}},{"cell_type":"code","source":"import os\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for f in files:\n        if 'glove' in f.lower():\n            print(os.path.join(root, f))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:04:46.355648Z","iopub.execute_input":"2026-04-10T13:04:46.356433Z","iopub.status.idle":"2026-04-10T13:04:46.388979Z","shell.execute_reply.started":"2026-04-10T13:04:46.356400Z","shell.execute_reply":"2026-04-10T13:04:46.388153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"GLOVE_PATH = '/kaggle/input/datasets/danielwillgeorge/glove6b100dtxt/glove.6B.100d.txt'\n\ndef load_glove(path, embed_dim):\n    glove = {}\n    with open(path, encoding='utf-8') as f:\n        for line in f:\n            parts = line.split()\n            word  = parts[0]\n            vec   = np.array(parts[1:], dtype='float32')\n            glove[word] = vec\n    print(f'GloVe loaded: {len(glove)} vectors of dim {embed_dim}')\n    return glove\n\nprint('Loading GloVe...')\nglove_index = load_glove(GLOVE_PATH, EMBED_DIM)\n\ndef build_glove_matrix(glove_index, word_index, vocab_size, embed_dim):\n    matrix = np.zeros((vocab_size, embed_dim))\n    hits = 0\n    for word, idx in word_index.items():\n        if idx >= vocab_size:\n            continue\n        vec = glove_index.get(word)\n        if vec is not None:\n            matrix[idx] = vec\n            hits += 1\n    print(f'  Coverage: {hits}/{min(vocab_size, len(word_index))} words ({hits/min(vocab_size,len(word_index))*100:.1f}%)')\n    return matrix\n\nemb_matrix_glove = build_glove_matrix(glove_index, word_index, vocab_size, EMBED_DIM)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:04:59.183800Z","iopub.execute_input":"2026-04-10T13:04:59.184840Z","iopub.status.idle":"2026-04-10T13:05:08.827805Z","shell.execute_reply.started":"2026-04-10T13:04:59.184799Z","shell.execute_reply":"2026-04-10T13:05:08.826799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nembed_layer6 = Embedding(\n    input_dim=vocab_size,\n    output_dim=EMBED_DIM,\n    weights=[emb_matrix_glove],\n    input_length=MAX_LEN,\n    trainable=True  \n)\n\nmodel6 = build_dl_model(embed_layer6)\n\nhistory6 = model6.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history6, 'Model 6: GloVe (Trainable)')\nreport('Model 6: GloVe (Trainable)',\n       model6, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:05:41.233510Z","iopub.execute_input":"2026-04-10T13:05:41.234443Z","iopub.status.idle":"2026-04-10T13:07:33.014516Z","shell.execute_reply.started":"2026-04-10T13:05:41.234411Z","shell.execute_reply":"2026-04-10T13:07:33.013745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model 7: DL with GloVe (Non-trainable / Frozen)","metadata":{}},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nembed_layer7 = Embedding(\n    input_dim=vocab_size,\n    output_dim=EMBED_DIM,\n    weights=[emb_matrix_glove],\n    input_length=MAX_LEN,\n    trainable=False  \n)\n\nmodel7 = build_dl_model(embed_layer7)\n\nhistory7 = model7.fit(\n    X_train_pad, y_train,\n    validation_split=0.1,\n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    callbacks=get_callbacks(),\n    verbose=1\n)\n\nplot_history(history7, 'Model 7: GloVe (Non-trainable)')\nreport('Model 7: GloVe (Non-trainable)',\n       model7, X_train_pad, X_test_pad, y_train, y_test, is_dl=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:07:33.016160Z","iopub.execute_input":"2026-04-10T13:07:33.016493Z","iopub.status.idle":"2026-04-10T13:09:09.886726Z","shell.execute_reply.started":"2026-04-10T13:07:33.016468Z","shell.execute_reply":"2026-04-10T13:09:09.886031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Final Summary Table","metadata":{}},{"cell_type":"code","source":"summary_df = pd.DataFrame.from_dict(\n    results, orient='index', columns=['Training Accuracy', 'Testing Accuracy']\n).reset_index().rename(columns={'index': 'Model'})\n\nsummary_df['Training Accuracy'] = summary_df['Training Accuracy'].map('{:.4f}'.format)\nsummary_df['Testing Accuracy']  = summary_df['Testing Accuracy'].map('{:.4f}'.format)\n\nprint('\\n' + '='*65)\nprint('                  FINAL RESULTS SUMMARY')\nprint('='*65)\nprint(summary_df.to_string(index=False))\nprint('='*65)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:09:09.887693Z","iopub.execute_input":"2026-04-10T13:09:09.887989Z","iopub.status.idle":"2026-04-10T13:09:09.897985Z","shell.execute_reply.started":"2026-04-10T13:09:09.887962Z","shell.execute_reply":"2026-04-10T13:09:09.896950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(14, 5))\n\nnames      = list(results.keys())\ntrain_accs = [v[0] for v in results.values()]\ntest_accs  = [v[1] for v in results.values()]\n\nx = np.arange(len(names))\nw = 0.35\n\nbars1 = ax.bar(x - w/2, train_accs, w, label='Train Accuracy', color='steelblue', alpha=0.85)\nbars2 = ax.bar(x + w/2, test_accs,  w, label='Test Accuracy',  color='tomato',    alpha=0.85)\n\nax.set_xticks(x)\nax.set_xticklabels(\n    [f'M{i+1}' for i in range(len(names))],\n    fontsize=10\n)\nax.set_ylim(0.85, 1.01)\nax.set_ylabel('Accuracy')\nax.set_title('Train vs Test Accuracy — All 7 Models')\nax.legend()\n\nfor bar in bars1:\n    ax.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.002,\n            f'{bar.get_height():.3f}', ha='center', va='bottom', fontsize=8)\nfor bar in bars2:\n    ax.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.002,\n            f'{bar.get_height():.3f}', ha='center', va='bottom', fontsize=8)\n\nlegend_text = '\\n'.join([f'M{i+1}: {n}' for i, n in enumerate(names)])\nplt.figtext(0.01, -0.05, legend_text, fontsize=8, va='top')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-10T13:09:09.899461Z","iopub.execute_input":"2026-04-10T13:09:09.899695Z","iopub.status.idle":"2026-04-10T13:09:10.174353Z","shell.execute_reply.started":"2026-04-10T13:09:09.899675Z","shell.execute_reply":"2026-04-10T13:09:10.173703Z"}},"outputs":[],"execution_count":null}]}