{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10737,"databundleVersionId":290346}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\n\n# Check for GPU availability\n#print(\"GPU is\", \"available\" if tf.config.list_physical_devices('GPU') else \"NOT AVAILABLE\")\n\n# Load the training data\n# Kaggle dataset path for Quora Insincere Questions\ntrain_path = '/kaggle/input/competitions/quora-insincere-questions-classification/train.csv'\ntrain_df = pd.read_csv(train_path)\n\n# Quick peek at the structure\nprint(f\"Dataset Shape: {train_df.shape}\")\ntrain_df.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:34.160859Z","iopub.execute_input":"2026-03-13T23:23:34.161178Z","iopub.status.idle":"2026-03-13T23:23:39.215444Z","shell.execute_reply.started":"2026-03-13T23:23:34.161153Z","shell.execute_reply":"2026-03-13T23:23:39.214673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#label 0: Sincere questions (genuine questions for information about something)\n#label 1: Insincere questions (sarcasm, hatespeech, racism, spam etc)\n#total samples: 1.3M+\n\npd.set_option('display.max_colwidth',None)\ntrain_df.sample(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:39.216729Z","iopub.execute_input":"2026-03-13T23:23:39.216937Z","iopub.status.idle":"2026-03-13T23:23:39.252325Z","shell.execute_reply.started":"2026-03-13T23:23:39.216918Z","shell.execute_reply":"2026-03-13T23:23:39.251750Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import string\n\n# 1. Class Imbalance Check\ntarget_counts = train_df['target'].value_counts()\nprint(f\"Sincere (0): {target_counts[0]} ({target_counts[0]/len(train_df)*100:.2f}%)\")\nprint(f\"Insincere (1): {target_counts[1]} ({target_counts[1]/len(train_df)*100:.2f}%)\")\n\n# List of common stop words for our ratio check\n# We could predefined list of stopwords through NLTK or scikit-learn \n#but they could wipe out words which could contain semantic value.\nstop_words = {\"the\", \"a\", \"an\", \"is\", \"are\", \"was\", \"were\", \"and\", \"or\", \"but\", \"if\", \"then\", \"else\", \"what\", \"which\", \"who\", \"whom\"}\n\ndef get_stylometrics(text):\n    words = text.split()\n    word_count = len(words)\n    char_count = len(text)\n    \n    # Avoid division by zero\n    if word_count == 0 or char_count == 0:\n        return 0, 0, 0, 0\n    \n    caps_count = sum(1 for c in text if c.isupper())\n    punc_count = sum(1 for c in text if c in string.punctuation)\n    sw_count = sum(1 for w in words if w.lower() in stop_words)\n\n    # intuition is that insincere questions would contain more caps, more punctuations (!!!) \n    #and sincere questions would contain more stop words\n    return word_count, caps_count/char_count, punc_count/char_count, sw_count/word_count\n\n# Apply to a sample of 100k for faster EDA visualization\neda_df = train_df.sample(100000, random_state=42).copy()\nmetrics = eda_df['question_text'].apply(get_stylometrics)\neda_df[['word_count', 'caps_ratio', 'punc_ratio', 'stop_word_ratio']] = pd.DataFrame(metrics.tolist(), index=eda_df.index)\n\n# Visualization\nfig, axes = plt.subplots(1, 4, figsize=(20, 5))\nfeatures = ['word_count', 'caps_ratio', 'punc_ratio', 'stop_word_ratio']\n\nfor i, feature in enumerate(features):\n    sns.boxplot(x='target', y=feature, data=eda_df, ax=axes[i])\n    axes[i].set_title(f'{feature} by Class')\n\nplt.tight_layout()\nplt.show()\n\n# Descriptive Statistics for Insincere questions\nprint(\"\\nStats for Insincere Questions (Target 1):\")\nprint(eda_df[eda_df['target'] == 1][features].describe())\n\nprint(\"\\nStats for Sincere Questions (Target 0):\")\nprint(eda_df[eda_df['target'] == 0][features].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:39.253183Z","iopub.execute_input":"2026-03-13T23:23:39.253416Z","iopub.status.idle":"2026-03-13T23:23:41.266581Z","shell.execute_reply.started":"2026-03-13T23:23:39.253396Z","shell.execute_reply":"2026-03-13T23:23:41.265902Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"  **So We found out that sincere questions have relatively less word count, more stop words ratio. Caps and Punctuation ratios are more or less the same** ","metadata":{}},{"cell_type":"code","source":"# word count is max 56 and 57 so we can set MAX_LEN = 57 \n# but we have to verify if these signals that we found are legit.\n\n# 1. The \"Marathon\" Questions - Are the longest questions Insincere questions?\nprint(\"--- TOP 10 LONGEST QUESTIONS ---\")\ndisplay(eda_df.nlargest(10, 'word_count')[['question_text', 'word_count', 'target']])\n\n# 2. The \"Shouters\" - Does 50%+ Capitalization mean Insincere question?\nprint(\"\\n--- HIGH CAPITALIZATION SAMPLES (Caps > 50%) ---\")\nshouters = eda_df.query('caps_ratio > 0.5')\ndisplay(shouters[['question_text', 'caps_ratio', 'target']].head(10))\n\n# 3. The \"Punctuation Noise\" - Sincere questions with high punctuation\nprint(\"\\n--- SINCERE QUESTIONS WITH HIGH PUNCTUATION ---\")\n# Finding sincere questions that might confuse a model due to high punctuation\npunc_noise = eda_df.query('target == 0').nlargest(5, 'punc_ratio')\ndisplay(punc_noise[['question_text', 'punc_ratio', 'target']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:41.268176Z","iopub.execute_input":"2026-03-13T23:23:41.268457Z","iopub.status.idle":"2026-03-13T23:23:41.333217Z","shell.execute_reply.started":"2026-03-13T23:23:41.268432Z","shell.execute_reply":"2026-03-13T23:23:41.332662Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**So most of my assumptions have falllen apart. But i have confirmed that i can use .lower() confidently without sweat.Also I will not exclude question mark (?) from punctuation because that's what recognizes a question in the first place.**","metadata":{}},{"cell_type":"code","source":"import re\n\n# Dictionary of common contractions to expand\n# Although python has a library for it called contractions.\ncontractions_dict = {\n    \"ain't\": \"am not\", \"aren't\": \"are not\", \"can't\": \"cannot\", \"can't've\": \"cannot have\",\n    \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\", \"doesn't\": \"does not\",\n    \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\",\n    \"he'd\": \"he would\", \"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\",\n    \"i'd\": \"i would\", \"i'll\": \"i will\", \"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\",\n    \"it'd\": \"it would\", \"it's\": \"it is\", \"let's\": \"let us\", \"might've\": \"might have\",\n    \"must've\": \"must have\", \"shan't\": \"shall not\", \"she'd\": \"she would\", \"she'll\": \"she will\",\n    \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"that's\": \"that is\",\n    \"there's\": \"there is\", \"they'd\": \"they would\", \"they'll\": \"they will\", \"they're\": \"they are\",\n    \"they've\": \"they have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'll\": \"we will\",\n    \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\",\n    \"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\", \"where's\": \"where is\",\n    \"who'll\": \"who will\", \"who's\": \"who is\", \"won't\": \"will not\", \"wouldn't\": \"would not\",\n    \"you'd\": \"you would\", \"you'll\": \"you will\", \"you're\": \"you are\"\n}\n\ndef clean_text(text):\n    text = text.lower()\n    \n    # 1. Replace math blocks\n    text = re.sub(r'\\[math\\].*?\\[/math\\]', 'math', text)\n    \n    # 2. Expand contractions\n    for word in text.split():\n        if word in contractions_dict:\n            text = text.replace(word, contractions_dict[word])\n            \n    # 3. Remove punctuation (keeping only the question mark)\n    # We remove punctuation marks but keep space for words\n    text = re.sub(r\"[^a-zA-Z0-9 ]\", \" \", text)\n    \n    # 4. Remove extra whitespace\n    text = re.sub(r'\\s+', ' ', text).strip()\n    \n    return text\n\n# Apply cleaning to the full training set\nprint(\"Cleaning text... this may take a minute.\")\ntrain_df['question_text'] = train_df['question_text'].apply(clean_text)\n\n# Save to CSV in the Kaggle Output directory\noutput_path = 'quora_train_cleaned.csv'\ntrain_df.to_csv(output_path, index=False)\nprint(f\"Cleaned data saved to {output_path}\")\n\n# Peek at the results\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:41.334114Z","iopub.execute_input":"2026-03-13T23:23:41.334409Z","iopub.status.idle":"2026-03-13T23:23:56.234888Z","shell.execute_reply.started":"2026-03-13T23:23:41.334378Z","shell.execute_reply":"2026-03-13T23:23:56.234308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\n# 1. Initialize the Tokenizer\n# We cap at 100,000 to filter out rare noise/typos \ntokenizer = Tokenizer(filters='') # filters='' because we already cleaned\n# tockenizer will keep all features by deault.\n\n# 2. Build the Vocabulary\nprint(\"Fitting tokenizer on cleaned text...\")\ntokenizer.fit_on_texts(train_df['question_text'])\n\n# 3. Inspect the Vocabulary (As you requested)\n\n# 3.1. Calculate frequencies\nword_counts = sorted(tokenizer.word_counts.items(), key=lambda x: x[1], reverse=True)\n\n# 3.2. Determine thresholds\nmin_freq = 5\nvocab_at_min_freq = len([word for word, count in word_counts if count >= min_freq])\nfreq_of_100k = word_counts[99999][1] if len(word_counts) > 100000 else \"N/A\"\n\n# this will reduce overfitting and increase training speed.\n\n\nprint(f\"Total words in raw vocab: {len(word_counts)}\")\nprint(f\"Words appearing at least {min_freq} times: {vocab_at_min_freq}\")\nprint(f\"Frequency of the 100,000th word: {freq_of_100k}\")\n\n# Recommendation: Set max_features to vocab_at_min_freq or a round number near it\nmax_features = vocab_at_min_freq\n\n\nword_index = tokenizer.word_index\nvocab_size = len(word_index)\nprint(f\"\\nTotal Unique Tokens Found: {vocab_size}\")\nprint(\"-\" * 30)\nprint(\"TOP 50 MOST FREQUENT WORDS:\")\n# Sort by index (1 is most frequent, 2 is second, etc.)\ntop_50 = sorted(word_index.items(), key=lambda x: x[1])[:50]\nfor word, index in top_50:\n    print(f\"{index}: {word}\", end=\" | \")\n    if index % 5 == 0: print() # New line every 5 words\n\ntokenizer.num_words = 55000 # max_features = 50469, rounded it off to 66000 for some cushion.\n\n# 4. Convert Text to Sequences\nprint(\"\\n\" + \"-\" * 30)\nprint(\"Converting text to sequences and padding...\")\nsequences = tokenizer.texts_to_sequences(train_df['question_text'])\n\n# 5. Apply Padding (using your max_len=57)\nX = pad_sequences(sequences, maxlen=57)\ny = train_df['target'].values\n\nprint(f\"Final Input Shape: {X.shape}\") # Should be (1306122, 57)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:23:56.235905Z","iopub.execute_input":"2026-03-13T23:23:56.236158Z","iopub.status.idle":"2026-03-13T23:24:23.068459Z","shell.execute_reply.started":"2026-03-13T23:23:56.236136Z","shell.execute_reply":"2026-03-13T23:24:23.067810Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So before removing the question mark (?) i had got:\n\n* Total words in raw vocab: 247077\n* Words appearing at least 5 times: 65073\n* Frequency of the 100,000th word: 2\n* Total Unique Tokens Found: 247077\n\nAfter removing it, the numbers have reduced to:\n\n* Total words in raw vocab: 192395\n* Words appearing at least 5 times: 50469\n* Frequency of the 100,000th word: 1\n* Total Unique Tokens Found: 192395","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Extract frequencies and ranks\nfrequencies = [count for word, count in word_counts]\nranks = range(1, len(frequencies) + 1)\n\nplt.figure(figsize=(12, 6))\n\n# Main plot (Log-Log scale is best for Zipf's Law)\nplt.loglog(ranks, frequencies, label='Word Frequencies')\nplt.axvline(x=100000, color='r', linestyle='--', label='100k Rank')\nplt.axvline(x=vocab_at_min_freq, color='g', linestyle='--', label=f'Min Freq {min_freq} Rank')\n\nplt.title(\"Zipf's Law: Word Rank vs. Frequency (Log-Log Scale)\")\nplt.xlabel(\"Rank of Word (Most frequent to least)\")\nplt.ylabel(\"Frequency (Number of occurrences)\")\nplt.legend()\nplt.grid(True, which=\"both\", ls=\"-\", alpha=0.5)\nplt.show()\n\n# Zoomed in linear plot for the 'head' of the distribution\nplt.figure(figsize=(12, 6))\nplt.plot(ranks[:50000], frequencies[:50000])\nplt.title(\"Head of Distribution: Top 50,000 Words\")\nplt.xlabel(\"Rank\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:24:23.069418Z","iopub.execute_input":"2026-03-13T23:24:23.069732Z","iopub.status.idle":"2026-03-13T23:24:24.025672Z","shell.execute_reply.started":"2026-03-13T23:24:23.069709Z","shell.execute_reply":"2026-03-13T23:24:24.025026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport os\n\nzip_path = '../input/competitions/quora-insincere-questions-classification/embeddings.zip'\nextract_path = '/kaggle/working/embeddings/'\n\n# We only want these two specifically to save space\nfiles_to_extract = [\n    'glove.840B.300d/glove.840B.300d.txt',\n    'wiki-news-300d-1M/wiki-news-300d-1M.vec'\n]\n\nif not os.path.exists(extract_path):\n    os.makedirs(extract_path)\n\nwith zipfile.ZipFile(zip_path, 'r') as z:\n    for file in files_to_extract:\n        print(f\"Extracting {file}...\")\n        z.extract(file, extract_path)\n\nprint(\"Extraction complete. You can now access the files in:\", extract_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:24:24.026503Z","iopub.execute_input":"2026-03-13T23:24:24.026860Z","iopub.status.idle":"2026-03-13T23:25:20.017116Z","shell.execute_reply.started":"2026-03-13T23:24:24.026838Z","shell.execute_reply":"2026-03-13T23:25:20.016500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tqdm # For progress bars\n\ndef check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    nb_known_words = 0\n    nb_unknown_words = 0\n    \n    for word in vocab.keys():\n        try:\n            known_words[word] = embeddings_index[word]\n            nb_known_words += vocab[word]\n        except:\n            unknown_words[word] = vocab[word]\n            nb_unknown_words += vocab[word]\n            pass\n\n    print(f'Found embeddings for {len(known_words) / len(vocab):.2%} of vocab')\n    print(f'Found embeddings for {nb_known_words / (nb_known_words + nb_unknown_words):.2%} of all text symbols')\n    \n    # Sort unknown words by frequency to see what we are missing\n    unknown_words = sorted(unknown_words.items(), key=lambda x: x[1], reverse=True)\n\n    return unknown_words\n\ndef load_embeddings(file_path):\n    embeddings_index = {}\n    with open(file_path, encoding='utf8', errors='ignore') as f:\n        for line in tqdm.tqdm(f):\n            values = line.split(' ')\n            word = values[0]\n            # FastText and Glove 840B have 300d\n            coefs = np.asarray(values[1:], dtype='float32')\n            embeddings_index[word] = coefs\n    return embeddings_index\n\n# 1. Load Glove\nprint(\"Loading GloVe...\")\nglove_path = '/kaggle/working/embeddings/glove.840B.300d/glove.840B.300d.txt'\nglove_index = load_embeddings(glove_path)\n\n# 2. Run Coverage for Glove\nprint(\"\\n--- GLOVE COVERAGE ---\")\noov_glove = check_coverage(tokenizer.word_counts, glove_index)\n\n# 3. Load FastText\nprint(\"\\nLoading FastText...\")\nfasttext_path = '/kaggle/working/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\nfasttext_index = load_embeddings(fasttext_path)\n\n# 4. Run Coverage for FastText\nprint(\"\\n--- FASTTEXT COVERAGE ---\")\noov_fasttext = check_coverage(tokenizer.word_counts, fasttext_index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:25:20.018145Z","iopub.execute_input":"2026-03-13T23:25:20.018438Z","iopub.status.idle":"2026-03-13T23:28:25.150255Z","shell.execute_reply.started":"2026-03-13T23:25:20.018406Z","shell.execute_reply":"2026-03-13T23:28:25.149546Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Before removing the question mark (?) \\\n--- GLOVE COVERAGE ---\n* Found embeddings for 46.45% of vocab\n* Found embeddings for 91.60% of all text symbols\n\n--- FASTTEXT COVERAGE ---\n* Found embeddings for 35.79% of vocab\n* Found embeddings for 91.07% of all text symbols","metadata":{}},{"cell_type":"markdown","source":"Comaparing GloVe with FastText to pick the best or concatinate both.","metadata":{}},{"cell_type":"code","source":"# Assuming oov_glove was the output from your check_coverage function\nprint(\"TOP 20 WORDS MISSING FROM GLOVE (By Frequency):\")\nfor word, count in oov_glove[:20]:\n    print(f\"Word: '{word}' | Frequency: {count}\")\n\nprint(\"\\n\" + \"-\"*30)\n\nprint(\"TOP 20 WORDS MISSING FROM FASTTEXT (By Frequency):\")\nfor word, count in oov_fasttext[:20]:\n    print(f\"Word: '{word}' | Frequency: {count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:28:25.152065Z","iopub.execute_input":"2026-03-13T23:28:25.152279Z","iopub.status.idle":"2026-03-13T23:28:25.157262Z","shell.execute_reply.started":"2026-03-13T23:28:25.152259Z","shell.execute_reply":"2026-03-13T23:28:25.156528Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"so i found the reason for 91% coverage. The question mark (?) we left spared earlier from cleaning is responsible. For example, the embeddings are familiar with India but not with India?","metadata":{}},{"cell_type":"code","source":"# 1. Update your hyperparameter\nmax_features = 55000 \nembed_size = 600 # 300 (Glove) + 300 (FastText)\n\n# 2. Initialize the matrix\nembedding_matrix = np.zeros((max_features, embed_size))\n\n# 3. Populate the matrix\nfor word, i in tqdm.tqdm(tokenizer.word_index.items()):\n    if i >= max_features: continue # Stay within our 51k limit\n    \n    # Get Glove vector\n    glove_vector = glove_index.get(word)\n    # Get FastText vector\n    fasttext_vector = fasttext_index.get(word)\n    \n    # Concatenate them if they exist, otherwise they stay as zeros\n    if glove_vector is not None and fasttext_vector is not None:\n        embedding_matrix[i] = np.concatenate([glove_vector, fasttext_vector])\n    elif glove_vector is not None:\n        # If only Glove exists, pad the rest with zeros\n        embedding_matrix[i] = np.concatenate([glove_vector, np.zeros(300)])\n    elif fasttext_vector is not None:\n        # If only FastText exists, pad the first half with zeros\n        embedding_matrix[i] = np.concatenate([np.zeros(300), fasttext_vector])\n\nprint(f\"Final Embedding Matrix Shape: {embedding_matrix.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:28:25.158084Z","iopub.execute_input":"2026-03-13T23:28:25.158307Z","iopub.status.idle":"2026-03-13T23:28:25.631989Z","shell.execute_reply.started":"2026-03-13T23:28:25.158281Z","shell.execute_reply":"2026-03-13T23:28:25.631256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Embedding, Bidirectional, LSTM, Dense, GlobalMaxPool1D, Dropout\n\nmodel = Sequential()\n\n# 1. The Embedding Layer (The \"Brain\")\n# We load your 55k x 600 matrix here\nmodel.add(Embedding(input_dim=max_features, \n                    output_dim=embed_size, \n                    weights=[embedding_matrix], \n                    trainable=False)) # We \"Freeze\" the weights because GloVe/FastText are already smart\n\n# 2. The Sequence Processor\nmodel.add(Bidirectional(LSTM(128, return_sequences=True)))\n\n# 3. Dimensionality Reduction\nmodel.add(GlobalMaxPool1D())\n\n# 4. The Classifier\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.2)) # Prevents overfitting\nmodel.add(Dense(1, activation='sigmoid')) # Final probability (0 to 1)\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:28:25.632922Z","iopub.execute_input":"2026-03-13T23:28:25.633162Z","iopub.status.idle":"2026-03-13T23:28:28.645280Z","shell.execute_reply.started":"2026-03-13T23:28:25.633141Z","shell.execute_reply":"2026-03-13T23:28:28.644751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils import class_weight\nimport numpy as np\n\n# Calculate weights based on your training labels\nweights = class_weight.compute_class_weight(class_weight='balanced', \n                                            classes=np.unique(y), \n                                            y=y)\nclass_weights_dict = dict(enumerate(weights))\n\nprint(f\"Class Weights: {class_weights_dict}\")\n# Expect something like {0: 0.53, 1: 8.2}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:33:51.283650Z","iopub.execute_input":"2026-03-13T23:33:51.284254Z","iopub.status.idle":"2026-03-13T23:33:51.527586Z","shell.execute_reply.started":"2026-03-13T23:33:51.284226Z","shell.execute_reply":"2026-03-13T23:33:51.526834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%who","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:32:30.432632Z","iopub.execute_input":"2026-03-13T23:32:30.433445Z","iopub.status.idle":"2026-03-13T23:32:30.438019Z","shell.execute_reply.started":"2026-03-13T23:32:30.433419Z","shell.execute_reply":"2026-03-13T23:32:30.437219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Update your compile step\nmodel.compile(\n    optimizer='adam',\n    loss=tf.keras.losses.BinaryFocalCrossentropy(\n        gamma=2.0, \n        label_smoothing=0.1\n    ),\n    metrics=[tf.keras.metrics.AUC(name='auc'), 'accuracy'] \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:55:12.675712Z","iopub.execute_input":"2026-03-13T23:55:12.676296Z","iopub.status.idle":"2026-03-13T23:55:12.701442Z","shell.execute_reply.started":"2026-03-13T23:55:12.676265Z","shell.execute_reply":"2026-03-13T23:55:12.700695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split the data: 90% for training, 10% for validation\n# We use X and y because those were in your %who list\ntrain_X, val_X, train_y, val_y = train_test_split(X, y, test_size=0.1, random_state=42)\n\nprint(f\"Training shapes: {train_X.shape}, {train_y.shape}\")\nprint(f\"Validation shapes: {val_X.shape}, {val_y.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:55:17.732931Z","iopub.execute_input":"2026-03-13T23:55:17.733698Z","iopub.status.idle":"2026-03-13T23:55:17.946216Z","shell.execute_reply.started":"2026-03-13T23:55:17.733672Z","shell.execute_reply":"2026-03-13T23:55:17.945499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    X, y,\n    batch_size=512, # Large batch size for GPU efficiency\n    epochs=50,       # Start with 50; we can always do more\n    validation_data=(val_X, val_y),\n    class_weight=class_weights_dict, # Apply our 15:1 weight\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T23:55:21.648813Z","iopub.execute_input":"2026-03-13T23:55:21.649414Z","iopub.status.idle":"2026-03-14T00:27:23.458065Z","shell.execute_reply.started":"2026-03-13T23:55:21.649386Z","shell.execute_reply":"2026-03-14T00:27:23.457094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score, precision_score, recall_score\nimport numpy as np\n\n# 1. Generate predictions on validation set\nprint(\"Generating validation predictions...\")\nval_preds = model.predict(val_X, batch_size=1024)\n\n# 2. Search for the best threshold\nbest_threshold = 0\nbest_f1 = 0\n\nprint(\"Searching for optimal threshold...\")\nfor thresh in np.arange(0.1, 0.801, 0.01):\n    preds_at_thresh = (val_preds > thresh).astype(int)\n    score = f1_score(val_y, preds_at_thresh)\n    \n    if score > best_f1:\n        best_f1 = score\n        best_threshold = thresh\n\n# 3. Final Evaluation at the Best Threshold\nfinal_preds = (val_preds > best_threshold).astype(int)\nprecision = precision_score(val_y, final_preds)\nrecall = recall_score(val_y, final_preds)\n\nprint(\"-\" * 40)\nprint(f\"RESULTS AT OPTIMAL THRESHOLD ({best_threshold:.2f})\")\nprint(\"-\" * 40)\nprint(f\"Max F1-Score:  {best_f1:.4f}\")\nprint(f\"Precision:     {precision:.4f}\")\nprint(f\"Recall:        {recall:.4f}\")\nprint(\"-\" * 40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:29:12.755559Z","iopub.execute_input":"2026-03-14T00:29:12.755887Z","iopub.status.idle":"2026-03-14T00:29:18.689591Z","shell.execute_reply.started":"2026-03-14T00:29:12.755862Z","shell.execute_reply":"2026-03-14T00:29:18.688855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\n# 1. Save the Model (Standard Keras Format)\nmodel.save('insincerity_model_v1.h5')\n\n# 2. Save the Tokenizer (Essential for the backend)\nwith open('tokenizer.pickle', 'wb') as handle:\n    pickle.dump(tokenizer, handle, protocol=pickle.HIGHEST_PROTOCOL)\n\n# 3. Record your \"Magic Number\"\nprint(f\"Deployment Note: Use threshold {best_threshold:.2f} in the backend.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:30:32.715000Z","iopub.execute_input":"2026-03-14T00:30:32.715319Z","iopub.status.idle":"2026-03-14T00:30:33.313810Z","shell.execute_reply.started":"2026-03-14T00:30:32.715288Z","shell.execute_reply":"2026-03-14T00:30:33.313100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nfrom tensorflow.keras import backend as K\n\n# Clear the Keras session and force garbage collection\nK.clear_session()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:40:09.183967Z","iopub.execute_input":"2026-03-14T00:40:09.184589Z","iopub.status.idle":"2026-03-14T00:40:10.402327Z","shell.execute_reply.started":"2026-03-14T00:40:09.184561Z","shell.execute_reply":"2026-03-14T00:40:10.401683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Embedding, Bidirectional, GRU, GlobalMaxPool1D, Dense, Dropout\nimport tensorflow as tf\n\n# Define constants to ensure they match your matrix\nVOCAB_SIZE = 55000 \nEMBED_DIM = 600\n\nmodel_gru = Sequential([\n    # FIX: The first two arguments must be input_dim and output_dim\n    Embedding(input_dim=VOCAB_SIZE, \n              output_dim=EMBED_DIM, \n              weights=[embedding_matrix], \n              input_length=57, \n              trainable=False),\n    \n    Bidirectional(GRU(128, return_sequences=True)),\n    \n    GlobalMaxPool1D(),\n    \n    Dense(64, activation='relu'),\n    Dropout(0.2),\n    Dense(1, activation='sigmoid')\n])\n\nmodel_gru.compile(\n    optimizer='adam',\n    loss=tf.keras.losses.BinaryFocalCrossentropy(gamma=2.0, label_smoothing=0.1),\n    metrics=[tf.keras.metrics.AUC(name='auc'), 'accuracy']\n)\n\nmodel_gru.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:40:28.771878Z","iopub.execute_input":"2026-03-14T00:40:28.772437Z","iopub.status.idle":"2026-03-14T00:40:29.126170Z","shell.execute_reply.started":"2026-03-14T00:40:28.772411Z","shell.execute_reply":"2026-03-14T00:40:29.125446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\n# Define Early Stopping\nearly_stop = EarlyStopping(\n    monitor='val_loss', \n    patience=5,             # Stop if val_loss doesn't improve for 3 epochs\n    restore_best_weights=True,\n    verbose=1\n)\n\n# Train the GRU Model\nhistory_gru = model_gru.fit(\n    train_X, train_y,\n    epochs=20,              # We can set this high now; Early Stopping has our back\n    batch_size=512,\n    validation_data=(val_X, val_y),\n    class_weight=class_weights_dict,\n    callbacks=[early_stop]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:42:38.148441Z","iopub.execute_input":"2026-03-14T00:42:38.149265Z","iopub.status.idle":"2026-03-14T00:56:41.320447Z","shell.execute_reply.started":"2026-03-14T00:42:38.149236Z","shell.execute_reply":"2026-03-14T00:56:41.319654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score, precision_score, recall_score\nimport numpy as np\n\n# Generate predictions using the best GRU weights (from Epoch 2)\nprint(\"Generating GRU validation predictions...\")\nval_preds_gru = model_gru.predict(val_X, batch_size=1024)\n\n# Search for the best threshold\nbest_threshold_gru = 0\nbest_f1_gru = 0\n\nfor thresh in np.arange(0.1, 0.801, 0.01):\n    preds_at_thresh = (val_preds_gru > thresh).astype(int)\n    score = f1_score(val_y, preds_at_thresh)\n    if score > best_f1_gru:\n        best_f1_gru = score\n        best_threshold_gru = thresh\n\nfinal_preds_gru = (val_preds_gru > best_threshold_gru).astype(int)\nprint(\"-\" * 45)\nprint(f\"GRU RESULTS (BEST EPOCH: 2)\")\nprint(f\"Optimal Threshold: {best_threshold_gru:.2f}\")\nprint(f\"Max F1-Score:      {best_f1_gru:.4f}\")\nprint(f\"Precision:         {precision_score(val_y, final_preds_gru):.4f}\")\nprint(f\"Recall:            {recall_score(val_y, final_preds_gru):.4f}\")\nprint(\"-\" * 45)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T00:58:24.780376Z","iopub.execute_input":"2026-03-14T00:58:24.780714Z","iopub.status.idle":"2026-03-14T00:58:30.015840Z","shell.execute_reply.started":"2026-03-14T00:58:24.780691Z","shell.execute_reply":"2026-03-14T00:58:30.015048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_comparison(hist_lstm, hist_gru):\n    plt.style.use('dark_background') # Professional \"Dark Mode\" look\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(18, 6))\n\n    # Loss Comparison\n    ax1.plot(hist_lstm.history['val_loss'], label='LSTM Val Loss', color='#00d4ff', linewidth=2)\n    ax1.plot(hist_gru.history['val_loss'], label='GRU Val Loss', color='#ff007f', linewidth=2)\n    ax1.set_title('Validation Loss: LSTM vs GRU', fontsize=15, color='white')\n    ax1.legend()\n\n    # AUC Comparison\n    ax2.plot(hist_lstm.history['val_auc'], label='LSTM Val AUC', color='#00ff41', linewidth=2)\n    ax2.plot(hist_gru.history['val_auc'], label='GRU Val AUC', color='#f9ff00', linewidth=2)\n    ax2.set_title('Validation AUC: LSTM vs GRU', fontsize=15, color='white')\n    ax2.legend()\n\n    plt.suptitle(\"Architecture Comparison: Quora Insincerity Detection\", fontsize=20)\n    plt.show()\n\n# Call this after GRU training\nplot_comparison(history, history_gru)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T01:00:01.254805Z","iopub.execute_input":"2026-03-14T01:00:01.255125Z","iopub.status.idle":"2026-03-14T01:00:01.263439Z","shell.execute_reply.started":"2026-03-14T01:00:01.255099Z","shell.execute_reply":"2026-03-14T01:00:01.262592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Data from your successful runs\nmodels = ['LSTM (Winner)', 'GRU (Fast)']\nf1_scores = [0.8472, 0.6751]\nrecall_scores = [0.9085, 0.7210]\nprecision_scores = [0.7937, 0.6347]\n\nx = np.arange(len(models))\nwidth = 0.25\n\n# Setting up the Dark Mode style for LinkedIn/GitHub\nplt.style.use('dark_background')\nfig, ax = plt.subplots(figsize=(12, 7))\n\n# Create the bars\nrects1 = ax.bar(x - width, f1_scores, width, label='F1-Score', color='#00d4ff', edgecolor='white')\nrects2 = ax.bar(x, recall_scores, width, label='Recall', color='#ff007f', edgecolor='white')\nrects3 = ax.bar(x + width, precision_scores, width, label='Precision', color='#00ff41', edgecolor='white')\n\n# Styling the labels\nax.set_ylabel('Scores', fontsize=12, fontweight='bold')\nax.set_title('Final Benchmark: LSTM vs. GRU Performance', fontsize=18, fontweight='bold', pad=20)\nax.set_xticks(x)\nax.set_xticklabels(models, fontsize=14, fontweight='bold')\nax.set_ylim(0, 1.1)\nax.legend(fontsize=11, loc='upper right')\n\n# Add values on top of the bars\ndef autolabel(rects):\n    for rect in rects:\n        height = rect.get_height()\n        ax.annotate(f'{height:.2f}',\n                    xy=(rect.get_x() + rect.get_width() / 2, height),\n                    xytext=(0, 3), # 3 points vertical offset\n                    textcoords=\"offset points\",\n                    ha='center', va='bottom', fontsize=10, fontweight='bold')\n\nautolabel(rects1)\nautolabel(rects2)\nautolabel(rects3)\n\nplt.tight_layout()\nplt.savefig('final_model_benchmark.png', dpi=300)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T01:01:14.693154Z","iopub.execute_input":"2026-03-14T01:01:14.693487Z","iopub.status.idle":"2026-03-14T01:01:15.331898Z","shell.execute_reply.started":"2026-03-14T01:01:14.693447Z","shell.execute_reply":"2026-03-14T01:01:15.331167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Data reconstructed from your training logs\nlstm_epochs = list(range(1, 13))\nlstm_val_loss = [0.0400, 0.0436, 0.0404, 0.0352, 0.0376, 0.0447, 0.0301, 0.0365, 0.0311, 0.0343, 0.0377, 0.0291]\nlstm_val_auc = [0.9875, 0.9885, 0.9893, 0.9902, 0.9911, 0.9914, 0.9924, 0.9928, 0.9934, 0.9939, 0.9940, 0.9946]\n\ngru_epochs = list(range(1, 8))\ngru_val_loss = [0.0816, 0.0523, 0.0699, 0.0572, 0.0620, 0.0547, 0.0626]\ngru_val_auc = [0.9655, 0.9679, 0.9686, 0.9678, 0.9669, 0.9671, 0.9663]\n\nplt.style.use('dark_background')\n\n# Plot 1: Validation Loss\nplt.figure(figsize=(10, 5))\nplt.plot(lstm_epochs, lstm_val_loss, label='LSTM', color='#00d4ff', marker='o', linewidth=2)\nplt.plot(gru_epochs, gru_val_loss, label='GRU', color='#ff007f', marker='s', linewidth=2)\nplt.title('Validation Loss: Training Stability', fontsize=14, fontweight='bold')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(alpha=0.2)\nplt.savefig('val_loss_comparison.png', dpi=300)\nplt.show()\n\n# Plot 2: Validation AUC\nplt.figure(figsize=(10, 5))\nplt.plot(lstm_epochs, lstm_val_auc, label='LSTM', color='#00ff41', marker='o', linewidth=2)\nplt.plot(gru_epochs, gru_val_auc, label='GRU', color='#f9ff00', marker='s', linewidth=2)\nplt.title('Validation AUC: Model Resolving Power', fontsize=14, fontweight='bold')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid(alpha=0.2)\nplt.savefig('val_auc_comparison.png', dpi=300)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T01:05:13.762844Z","iopub.execute_input":"2026-03-14T01:05:13.763457Z","iopub.status.idle":"2026-03-14T01:05:14.694243Z","shell.execute_reply.started":"2026-03-14T01:05:13.763427Z","shell.execute_reply":"2026-03-14T01:05:14.693457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef plot_confusion_matrices(y_true, preds_lstm, preds_gru):\n    plt.style.use('default') # Switch to light for better matrix readability\n    fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n    # 1. LSTM Confusion Matrix\n    cm_lstm = confusion_matrix(y_true, preds_lstm)\n    sns.heatmap(cm_lstm, annot=True, fmt='d', ax=ax1, cmap='Blues', cbar=False)\n    ax1.set_title('LSTM Confusion Matrix\\n(Threshold: 0.73)', fontsize=14, fontweight='bold')\n    ax1.set_xlabel('Predicted Label')\n    ax1.set_ylabel('True Label')\n    ax1.set_xticklabels(['Sincere', 'Insincere'])\n    ax1.set_yticklabels(['Sincere', 'Insincere'])\n\n    # 2. GRU Confusion Matrix\n    cm_gru = confusion_matrix(y_true, preds_gru)\n    sns.heatmap(cm_gru, annot=True, fmt='d', ax=ax2, cmap='RdPu', cbar=False)\n    ax2.set_title('GRU Confusion Matrix\\n(Threshold: 0.67)', fontsize=14, fontweight='bold')\n    ax2.set_xlabel('Predicted Label')\n    ax2.set_ylabel('True Label')\n    ax2.set_xticklabels(['Sincere', 'Insincere'])\n    ax2.set_yticklabels(['Sincere', 'Insincere'])\n\n    plt.tight_layout()\n    plt.savefig('confusion_matrices_comparison.png', dpi=300)\n    plt.show()\n\n# Run the plotting function\n# Ensure you have final_preds (LSTM) and final_preds_gru (GRU) ready\nplot_confusion_matrices(val_y, final_preds, final_preds_gru)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T01:06:42.948214Z","iopub.execute_input":"2026-03-14T01:06:42.948584Z","iopub.status.idle":"2026-03-14T01:06:43.627377Z","shell.execute_reply.started":"2026-03-14T01:06:42.948558Z","shell.execute_reply":"2026-03-14T01:06:43.626745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# --- 1. DATA RECONSTRUCTION ---\nlstm_epochs = list(range(1, 13))\nlstm_val_loss = [0.0400, 0.0436, 0.0404, 0.0352, 0.0376, 0.0447, 0.0301, 0.0365, 0.0311, 0.0343, 0.0377, 0.0291]\nlstm_val_auc = [0.9875, 0.9885, 0.9893, 0.9902, 0.9911, 0.9914, 0.9924, 0.9928, 0.9934, 0.9939, 0.9940, 0.9946]\n\ngru_epochs = list(range(1, 8))\ngru_val_loss = [0.0816, 0.0523, 0.0699, 0.0572, 0.0620, 0.0547, 0.0626]\ngru_val_auc = [0.9655, 0.9679, 0.9686, 0.9678, 0.9669, 0.9671, 0.9663]\n\n# Peak Metrics from your Threshold Search\nmetrics = ['F1-Score', 'Precision', 'Recall']\nlstm_final = [0.8472, 0.7937, 0.9085]\ngru_final = [0.6751, 0.6347, 0.7210]\n\nplt.style.use('dark_background')\n\n# --- PLOT 1: VALIDATION LOSS ---\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 3, 1)\nplt.plot(lstm_epochs, lstm_val_loss, label='LSTM', color='#00d4ff', marker='o')\nplt.plot(gru_epochs, gru_val_loss, label='GRU', color='#ff007f', marker='s')\nplt.title('Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(alpha=0.2)\n\n# --- PLOT 2: VALIDATION AUC ---\nplt.subplot(1, 3, 2)\nplt.plot(lstm_epochs, lstm_val_auc, label='LSTM', color='#00ff41', marker='o')\nplt.plot(gru_epochs, gru_val_auc, label='GRU', color='#f9ff00', marker='s')\nplt.title('Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid(alpha=0.2)\n\n# --- PLOT 3: FINAL METRICS COMPARISON (F1 & Precision) ---\nplt.subplot(1, 3, 3)\nx = np.arange(len(metrics))\nwidth = 0.35\nplt.bar(x - width/2, lstm_final, width, label='LSTM', color='#00d4ff', alpha=0.8)\nplt.bar(x + width/2, gru_final, width, label='GRU', color='#ff007f', alpha=0.8)\nplt.xticks(x, metrics)\nplt.title('Final Peak Performance')\nplt.ylabel('Score')\nplt.legend()\n\nplt.tight_layout()\nplt.savefig('full_performance_report.png', dpi=300)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T01:07:41.499077Z","iopub.execute_input":"2026-03-14T01:07:41.499708Z","iopub.status.idle":"2026-03-14T01:07:42.408053Z","shell.execute_reply.started":"2026-03-14T01:07:41.499680Z","shell.execute_reply":"2026-03-14T01:07:42.407348Z"}},"outputs":[],"execution_count":null}]}