{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:18:51.985703Z","iopub.execute_input":"2026-07-17T12:18:51.986131Z","iopub.status.idle":"2026-07-17T12:18:51.993235Z","shell.execute_reply.started":"2026-07-17T12:18:51.986097Z","shell.execute_reply":"2026-07-17T12:18:51.992323Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 1: Setup","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:18:51.995228Z","iopub.execute_input":"2026-07-17T12:18:51.996065Z","iopub.status.idle":"2026-07-17T12:18:52.014528Z","shell.execute_reply.started":"2026-07-17T12:18:51.996028Z","shell.execute_reply":"2026-07-17T12:18:52.013573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/competitions/quora-insincere-questions-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/competitions/quora-insincere-questions-classification/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:18:52.015800Z","iopub.execute_input":"2026-07-17T12:18:52.016305Z","iopub.status.idle":"2026-07-17T12:18:57.542279Z","shell.execute_reply.started":"2026-07-17T12:18:52.016264Z","shell.execute_reply":"2026-07-17T12:18:57.541254Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 2: EDA","metadata":{}},{"cell_type":"code","source":"# Create word count feature\ntrain_df['word_count'] = train_df['question_text'].apply(lambda x: len(str(x).split()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:18:57.543632Z","iopub.execute_input":"2026-07-17T12:18:57.544115Z","iopub.status.idle":"2026-07-17T12:18:59.458671Z","shell.execute_reply.started":"2026-07-17T12:18:57.544071Z","shell.execute_reply":"2026-07-17T12:18:59.457371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# features and target split\nX = train_df['question_text']\ny = train_df['target']\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, stratify=y, random_state=42)\n\nprint(f\"Train set shape: {X_train.shape} | Insincere count: {y_train.sum()} ({y_train.mean():.2%})\")\nprint(f\"Val set shape:   {X_val.shape}  | Insincere count: {y_val.sum()} ({y_val.mean():.2%})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:18:59.460228Z","iopub.execute_input":"2026-07-17T12:18:59.460562Z","iopub.status.idle":"2026-07-17T12:19:00.286229Z","shell.execute_reply.started":"2026-07-17T12:18:59.460523Z","shell.execute_reply":"2026-07-17T12:19:00.284742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\nvectorizer = TfidfVectorizer(\n    ngram_range=(1, 2),        # Captures single words and 2-word phrases\n    min_df=5,                  # Drops ultra-rare tokens and typos\n    max_features=50000,        # Keeps the vocab size manageable\n    stop_words='english'       # Removes starndard noise words\n)\n\n# fit on training data and transform both sets\nX_train_tfidf = vectorizer.fit_transform(X_train)\nX_val_tfidf = vectorizer.transform(X_val)\n\nprint(f\"Vocab size: {len(vectorizer.vocabulary_)}\")\nprint(f\"X_train_tfidf shape: {X_train_tfidf.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:19:00.287848Z","iopub.execute_input":"2026-07-17T12:19:00.288190Z","iopub.status.idle":"2026-07-17T12:19:44.323738Z","shell.execute_reply.started":"2026-07-17T12:19:00.288161Z","shell.execute_reply":"2026-07-17T12:19:44.322720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nmodel = LogisticRegression(\n    class_weight='balanced',\n    max_iter=1000,\n    random_state=42,\n    n_jobs=-1\n)\n\nmodel.fit(X_train_tfidf, y_train)\n\ny_val_proba = model.predict_proba(X_val_tfidf)[:, 1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:19:44.326876Z","iopub.execute_input":"2026-07-17T12:19:44.327185Z","iopub.status.idle":"2026-07-17T12:20:03.183314Z","shell.execute_reply.started":"2026-07-17T12:19:44.327158Z","shell.execute_reply":"2026-07-17T12:20:03.182288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score, classification_report\n\ndef find_best_threshold(y_true, y_proba):\n    \"\"\"\n    Sweeps thresholds from 0.01 to 0.99 to find the cutoff that maximizes the F1-score.\n    Returns the best threshold and its corresponding score.\n    \"\"\"\n    thresholds = np.arange(0.1, 0.9, 0.01)\n    best_thresh = 0.5\n    best_f1 = 0.0\n    \n    for thresh in thresholds:\n        # Convert probabilities to binary predictions based on the current threshold\n        y_pred = (y_proba >= thresh).astype(int)\n        score = f1_score(y_true, y_pred)\n        \n        if score > best_f1:\n            best_f1 = score\n            best_thresh = thresh\n            \n    print(f\"Best Threshold: {best_thresh:.2f}\")\n    print(f\"Maximum F1-Score: {best_f1:.4f}\\n\")\n    \n    return best_thresh, best_f1\n\n# 5. Run the sweep on validation predictions\nbest_threshold, max_f1 = find_best_threshold(y_val, y_val_proba)\n\n# Evaluate the final tuned model setup\nfinal_preds = (y_val_proba >= best_threshold).astype(int)\nprint(\"--- Final Classification Report ---\")\nprint(classification_report(y_val, final_preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:20:03.184485Z","iopub.execute_input":"2026-07-17T12:20:03.184852Z","iopub.status.idle":"2026-07-17T12:20:04.123558Z","shell.execute_reply.started":"2026-07-17T12:20:03.184823Z","shell.execute_reply":"2026-07-17T12:20:04.122589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 3: Submit and check","metadata":{}},{"cell_type":"code","source":"X_test_tfidf = vectorizer.transform(test_df['question_text'])\n\ntest_probabilities = model.predict_proba(X_test_tfidf)[:, 1]\ntest_predictions = (test_probabilities >= best_threshold).astype(int)\n\nsubmission_df = pd.DataFrame({\n    'qid': test_df['qid'],\n    'prediction': test_predictions\n})\n\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:20:04.125446Z","iopub.execute_input":"2026-07-17T12:20:04.125747Z","iopub.status.idle":"2026-07-17T12:20:11.599255Z","shell.execute_reply.started":"2026-07-17T12:20:04.125720Z","shell.execute_reply":"2026-07-17T12:20:11.598296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4: Preprocessing/text cleaning","metadata":{}},{"cell_type":"markdown","source":"####  Build a vocabulary counter from the training text","metadata":{}},{"cell_type":"code","source":"from collections import Counter\n\ndef build_vocab(texts):\n    vocab = Counter()\n    for text in texts:\n        for word in text.split():\n            vocab[word] += 1\n    return vocab\n\nvocab = build_vocab(X_train)\nprint(f\"Number of unique words: {len(vocab)}\")\nprint(vocab.most_common(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:20:11.600329Z","iopub.execute_input":"2026-07-17T12:20:11.600626Z","iopub.status.idle":"2026-07-17T12:20:18.741522Z","shell.execute_reply.started":"2026-07-17T12:20:11.600574Z","shell.execute_reply":"2026-07-17T12:20:18.740468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### loading the embedding file","metadata":{}},{"cell_type":"code","source":"import zipfile\n\nzip_path = '/kaggle/input/competitions/quora-insincere-questions-classification/embeddings.zip' \nextract_path = '/kaggle/working/'\n\nwith zipfile.ZipFile(zip_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:20:18.743195Z","iopub.execute_input":"2026-07-17T12:20:18.743622Z","iopub.status.idle":"2026-07-17T12:22:51.684618Z","shell.execute_reply.started":"2026-07-17T12:20:18.743560Z","shell.execute_reply":"2026-07-17T12:22:51.682365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Target directories to wipe out entirely\nfolders_to_delete = [\n    '/kaggle/working/wiki-news-300d-1M',\n    '/kaggle/working/paragram_300_sl999',\n    '/kaggle/working/GoogleNews-vectors-negative300'\n]\n\nfor folder in folders_to_delete:\n    if os.path.exists(folder):\n        print(f\"Deleting heavy directory: {folder}...\")\n        shutil.rmtree(folder)\n        print(\" -> Deleted successfully.\")\n    else:\n        print(f\"Directory not found (already cleared): {folder}\")\n\nprint(\"\\n--- Current Workspace Status ---\")\nprint(os.listdir('/kaggle/working'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:22:51.687251Z","iopub.execute_input":"2026-07-17T12:22:51.687762Z","iopub.status.idle":"2026-07-17T12:22:52.860716Z","shell.execute_reply.started":"2026-07-17T12:22:51.687721Z","shell.execute_reply":"2026-07-17T12:22:52.859599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_embeddings(filepath):\n    embeddings_index = {}\n    fail_count = 0\n    \n    with open(filepath, encoding='utf-8', errors='ignore') as f:\n        for line in f:\n            try:\n                values = line.rstrip().split(' ')\n                \n                # If a line splits into exactly 301 items, values[0] is the word, \n                # values[1:] are the numbers\n                if len(values) == 301:\n                    word = values[0]\n                    vector = np.asarray(values[1:], dtype='float32')\n                    embeddings_index[word] = vector\n                else:\n                    # Catching lines where the word itself contains spaces (like glove.840B often does)\n                    word = \"\".join(values[:-300])\n                    vector = np.asarray(values[-300:], dtype='float32')\n                    embeddings_index[word] = vector\n                    \n            except Exception:\n                fail_count += 1\n                continue\n    \n    print(f\"Loaded {len(embeddings_index)} word vectors\")\n    print(f\"Failed lines: {fail_count}\")\n    return embeddings_index\n\n# Full path to the text file inside the extracted folder\nglove_path = '/kaggle/working/glove.840B.300d/glove.840B.300d.txt'\nembeddings_index = load_embeddings(glove_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:22:52.862094Z","iopub.execute_input":"2026-07-17T12:22:52.863216Z","iopub.status.idle":"2026-07-17T12:25:36.529314Z","shell.execute_reply.started":"2026-07-17T12:22:52.863172Z","shell.execute_reply":"2026-07-17T12:25:36.528106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### coverage checking","metadata":{}},{"cell_type":"code","source":"def check_coverage(vocab, embeddings_index):\n    known_words = {}\n    unknown_words = {}\n    known_count = 0\n    unknown_count = 0\n    \n    for word in vocab.keys():\n        if word in embeddings_index:\n            known_words[word] = embeddings_index[word]\n            known_count += vocab[word]\n        else:\n            unknown_words[word] = vocab[word]\n            unknown_count += vocab[word]\n    \n    vocab_coverage = len(known_words) / len(vocab)\n    text_coverage = known_count / (known_count + unknown_count)\n    \n    print(f\"Found embeddings for {vocab_coverage:.2%} of vocab\")\n    print(f\"Found embeddings for {text_coverage:.2%} of all text\")\n    \n    # sort unknown words by frequency, most common first\n    unknown_words_sorted = sorted(unknown_words.items(), key=lambda x: x[1], reverse=True)\n    \n    return unknown_words_sorted\n\noov = check_coverage(vocab, embeddings_index)\noov[:20] ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:25:36.531075Z","iopub.execute_input":"2026-07-17T12:25:36.531757Z","iopub.status.idle":"2026-07-17T12:25:37.209302Z","shell.execute_reply.started":"2026-07-17T12:25:36.531678Z","shell.execute_reply":"2026-07-17T12:25:37.208311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Text cleaning","metadata":{}},{"cell_type":"code","source":"import re\n\ndef clean_text(text):\n    \"\"\"\n    Cleans raw text by isolating punctuation marks with spaces,\n    allowing pre-trained embeddings to catch them as separate tokens.\n    \"\"\"\n    text = str(text)\n    \n    # List of punctuation characters to isolate\n    # Includes standard punctuation, mathematical symbols, and diverse quotes\n    puncts = [\n        ',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&',\n        '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n        '·', '_', '{', '}', '©', '^', '®', '`',  '→', '°', '€', '™', '›',  '♥', \n        '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', '“', '★', '”', '–', \n        '●', 'â', '►', '−', '¢', '²', '░', '¹', '◦', '°', '♦', 'ã', '³', '✦', \n        'μ', 'ℹ', 'α', 'σ', '⇒', '❌', '👉', '˙', '⚙', '✈', '➡', '🇪🇸', '🍁', '〰', \n        '📍', '😂', '🔥', '📌', '💖', '☑', '◣', '⛳', '🤠', '🤙', '🔵', '🏁', '✨'\n    ]\n    \n    # 1. Isolate every punctuation mark with spaces on either side\n    for p in puncts:\n        if p in text:\n            text = text.replace(p, f' {p} ')\n            \n    # 2. Standardize consecutive whitespaces into a single clean space\n    text = re.sub(r'\\s+', ' ', text).strip()\n    \n    return text","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:25:37.210711Z","iopub.execute_input":"2026-07-17T12:25:37.211490Z","iopub.status.idle":"2026-07-17T12:25:37.220006Z","shell.execute_reply.started":"2026-07-17T12:25:37.211457Z","shell.execute_reply":"2026-07-17T12:25:37.218950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Apply the clean_text function to the dataset\ncleaned_questions = train_df['question_text'].apply(clean_text)\n\n# 2. Build the new vocabulary from the cleaned text\ncleaned_vocab = build_vocab(cleaned_questions)\n\n# 3. Check the updated coverage metrics\nprint(\"\\n--- Coverage After Cleaning Punctuation ---\")\noov_cleaned = check_coverage(cleaned_vocab, embeddings_index)\n\n# 4. Preview the new top 20 OOV words\nprint(\"\\n Top 20 OOV Words after cleaning:\")\ndisplay(oov_cleaned[:20])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:25:37.221196Z","iopub.execute_input":"2026-07-17T12:25:37.221695Z","iopub.status.idle":"2026-07-17T12:25:58.732972Z","shell.execute_reply.started":"2026-07-17T12:25:37.221651Z","shell.execute_reply":"2026-07-17T12:25:58.732108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport operator\nfrom tqdm import tqdm\n\n# 1. COMPREHENSIVE CONTRACTION MAPPING\nCONTRACTION_MAP = {\n    \"ain't\": \"is not\", \"aren't\": \"are not\", \"can't\": \"cannot\", \"'cause\": \"because\",\n    \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",\n    \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\",\n    \"haven't\": \"have not\", \"he'd\": \"he would\", \"he'll\": \"he will\", \"he's\": \"he is\",\n    \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",\n    \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\n    \"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'll\": \"i will\", \"i'm\": \"i am\",\n    \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'll\": \"it will\",\n    \"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\",\n    \"might've\": \"might have\", \"mightn't\": \"might not\", \"must've\": \"must have\",\n    \"mustn't\": \"must not\", \"needn't\": \"need not\", \"oughtn't\": \"ought not\",\n    \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"she'd\": \"she would\",\n    \"she'll\": \"she will\", \"she's\": \"she is\", \"should've\": \"should have\",\n    \"shouldn't\": \"should not\", \"so've\": \"so have\", \"so's\": \"so is\", \"that'd\": \"that would\",\n    \"that's\": \"that is\", \"there'd\": \"there would\", \"there's\": \"there is\",\n    \"they'd\": \"they would\", \"they'll\": \"they will\", \"they're\": \"they are\",\n    \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\",\n    \"we'll\": \"we will\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\",\n    \"what'll\": \"what will\", \"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\",\n    \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\",\n    \"where've\": \"where have\", \"who'll\": \"who will\", \"who's\": \"who is\", \"who've\": \"who have\",\n    \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\",\n    \"would've\": \"would have\", \"wouldn't\": \"would not\", \"y'all\": \"you all\",\n    \"you'd\": \"you would\", \"you'll\": \"you will\", \"you're\": \"you are\", \"you've\": \"you have\"\n}\n\ndef clean_text_v2(text):\n    text = str(text)\n    \n    # STEP 1: Normalize curly/variant quotes and apostrophes to straight ones\n    # This ensures contractions match our dictionary keys perfectly\n    text = text.replace(\"’\", \"'\").replace(\"‘\", \"'\").replace(\"´\", \"'\").replace(\"`\", \"'\")\n    text = text.replace(\"“\", '\"').replace(\"”\", '\"').replace(\"„\", '\"')\n    \n    # STEP 2: Expand Contractions (Word-by-word comparison)\n    # Using space-splitting temporary lookups so we don't accidentally match parts of regular words\n    words = text.split()\n    expanded_words = [CONTRACTION_MAP.get(word, word) for word in words]\n    text = \" \".join(expanded_words)\n    \n    # STEP 3: Isolate Punctuation\n    puncts = [\n        ',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&',\n        '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n        '·', '_', '{', '}', '©', '^', '®', '→', '°', '€', '™', '›',  '♥', \n        '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', '–', '●', 'â', \n        '►', '−', '¢', '²', '░', '¹', '◦', '♦', 'ã', '³', '✦', 'μ', 'ℹ', \n        'α', 'σ', '⇒', '❌', '👉', '˙', '⚙', '✈', '➡', '😂', '🔥', '📌', '💖'\n    ]\n    for p in puncts:\n        if p in text:\n            text = text.replace(p, f' {p} ')\n            \n    # STEP 4: Collapse consecutive whitespaces\n    text = re.sub(r'\\s+', ' ', text).strip()\n    \n    return text","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:25:58.734280Z","iopub.execute_input":"2026-07-17T12:25:58.734629Z","iopub.status.idle":"2026-07-17T12:25:58.799121Z","shell.execute_reply.started":"2026-07-17T12:25:58.734578Z","shell.execute_reply":"2026-07-17T12:25:58.798046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Transform your dataset questions\nv2_cleaned_questions = train_df['question_text'].apply(clean_text_v2)\n\n# 2. Rebuild the vocabulary mapping\nv2_vocab = build_vocab(v2_cleaned_questions)\n\n# 3. Calculate updated embedding scores\nprint(\"\\n--- Diagnostic Check: Coverage Post-Contraction Expansion ---\")\noov_v2 = check_coverage(v2_vocab, embeddings_index)\n\n# 4. Preview the remaining baseline OOV items\nprint(\"\\nRemaining Top 20 OOV Words:\")\ndisplay(oov_v2[:20])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:25:58.801344Z","iopub.execute_input":"2026-07-17T12:25:58.801937Z","iopub.status.idle":"2026-07-17T12:26:23.644398Z","shell.execute_reply.started":"2026-07-17T12:25:58.801907Z","shell.execute_reply":"2026-07-17T12:26:23.643596Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 5: building the embedding matrix and moving toward the LSTM","metadata":{}},{"cell_type":"markdown","source":"#### tokenization (converting text to integer sequences)","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split\n\n# Prepare Cleaned Splits & Clean Test Data\nX_train_clean, X_val_clean, y_train, y_val = train_test_split(\n    v2_cleaned_questions, \n    train_df['target'],\n    test_size=0.1, \n    stratify=train_df['target'], \n    random_state=42\n)\n\n# Clean the test set questions using v2 cleaning as well\ntest_questions_clean = test_df['question_text'].apply(clean_text_v2)\n\nMAX_VOCAB_SIZE = 50000 \nMAX_LEN = 50 \n\n# 1. Fit tokenizer on cleaned TRAINING text only \ntokenizer = Tokenizer(num_words=MAX_VOCAB_SIZE)\ntokenizer.fit_on_texts(X_train_clean)\n\n# 2. Convert cleaned text to integer sequences for train/val/test\ntrain_sequences = tokenizer.texts_to_sequences(X_train_clean)\nval_sequences = tokenizer.texts_to_sequences(X_val_clean)\ntest_sequences = tokenizer.texts_to_sequences(test_questions_clean)\n\n# 3. Pad sequences to uniform length\nX_train_pad = pad_sequences(train_sequences, maxlen=MAX_LEN, padding='pre', truncating='pre')\nX_val_pad = pad_sequences(val_sequences, maxlen=MAX_LEN, padding='pre', truncating='pre')\nX_test_pad = pad_sequences(test_sequences, maxlen=MAX_LEN, padding='pre', truncating='pre')\n\nprint(f\"X_train_pad shape: {X_train_pad.shape}\")\nprint(f\"X_val_pad shape: {X_val_pad.shape}\")\nprint(f\"X_test_pad shape: {X_test_pad.shape}\")\nprint(f\"Word index size: {len(tokenizer.word_index)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:26:23.646066Z","iopub.execute_input":"2026-07-17T12:26:23.646412Z","iopub.status.idle":"2026-07-17T12:27:44.053851Z","shell.execute_reply.started":"2026-07-17T12:26:23.646380Z","shell.execute_reply":"2026-07-17T12:27:44.052747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Building the embedding matrix","metadata":{}},{"cell_type":"code","source":"EMBED_DIM = 300\n\n# 1. Decide matrix size\n# index 0 is reserved for padding tokens, so we add 1 to our maximum vocabulary cutoff\nvocab_size = MAX_VOCAB_SIZE + 1\n\n# 2. Initialize with zeros\nembedding_matrix = np.zeros((vocab_size, EMBED_DIM))\n\n# 3. Fill in vectors for known words\nfor word, i in tokenizer.word_index.items():\n    # Skip words that fall outside the top 50,000 most frequent boundary\n    if i >= MAX_VOCAB_SIZE:\n        continue\n    \n    # Retrieve the pre-trained vector from GloVe\n    embedding_vector = embeddings_index.get(word)\n    \n    if embedding_vector is not None:\n        # Assign the pre-trained 300-dimension vector to its corresponding index row\n        embedding_matrix[i] = embedding_vector\n\nprint(f\"Embedding matrix shape: {embedding_matrix.shape}\")\n\n# 4. Sanity check: how much of the capped vocab actually got a real vector?\n# Check along axis 1 (columns) to see which rows have a non-zero sum, then count them\nnonzero_rows = np.count_nonzero(np.sum(embedding_matrix, axis=1) != 0)\nprint(f\"Words with embeddings: {nonzero_rows} / {vocab_size} ({nonzero_rows/vocab_size:.2%})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.055077Z","iopub.execute_input":"2026-07-17T12:27:44.055808Z","iopub.status.idle":"2026-07-17T12:27:44.351395Z","shell.execute_reply.started":"2026-07-17T12:27:44.055775Z","shell.execute_reply":"2026-07-17T12:27:44.350269Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6: LSTM","metadata":{}},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras.layers import Input, Embedding, Bidirectional, LSTM, GlobalMaxPooling1D, Dense, Dropout\n\n# # Clear any background Keras sessions\n# tf.keras.backend.clear_session()\n\n# model = Sequential([\n#     # Modern Keras entrance point: Tells the model to expect sentences of length 50\n#     Input(shape=(MAX_LEN,), dtype='int32', name=\"input_layer\"),\n    \n#     # 1. Embedding Layer: Plug in the pre-trained GloVe weights and freezing them\n#     Embedding(\n#         input_dim=vocab_size,\n#         output_dim=EMBED_DIM,\n#         weights=[embedding_matrix],\n#         trainable=False,  # Protects the 15M pre-trained parameters from distortion\n#         name=\"glove_embeddings\"\n#     ),\n    \n#     # 2. Recurrent Layer: Bidirectional LSTM\n#     # return_sequences=True is mandatory so the next pooling layer can see every timestep\n#     Bidirectional(\n#         LSTM(64, return_sequences=True, dropout=0.2, recurrent_dropout=0.2),\n#         name=\"bi_lstm\"\n#     ),\n    \n#     # 3. Pooling Layer\n#     GlobalMaxPooling1D(name=\"global_max_pooling\"),\n    \n#     Dropout(0.3, name=\"dropout_layer\"),\n    \n#     Dense(32, activation='relu', name=\"dense_hidden\"),\n    \n#     Dense(1, activation='sigmoid', name=\"output_layer\")\n# ])\n\n# # 6. Compile with class-imbalance aware metrics\n# model.compile(\n#     optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n#     loss='binary_crossentropy',\n#     metrics=[\n#         'accuracy', \n#         tf.keras.metrics.AUC(name='auc'),\n#         tf.keras.metrics.Precision(name='precision'),\n#         tf.keras.metrics.Recall(name='recall')\n#     ]\n# )\n\n# model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.352778Z","iopub.execute_input":"2026-07-17T12:27:44.353152Z","iopub.status.idle":"2026-07-17T12:27:44.358432Z","shell.execute_reply.started":"2026-07-17T12:27:44.353114Z","shell.execute_reply":"2026-07-17T12:27:44.357457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from tensorflow.keras.models import load_model\n\n# # 1. Update these paths to point to your new LSTM dataset\n# LSTM_INPUT_MODEL_PATH = '/kaggle/input/datasets/mdshafayeturrahman/quora-lstm-model/lstm_model.keras'\n# LSTM_WORKING_MODEL_PATH = '/kaggle/working/lstm_model.keras'\n\n# # 2. Check and load\n# if os.path.exists(LSTM_INPUT_MODEL_PATH):\n#     print(f\"Found pinned LSTM model in custom Kaggle Dataset! Loading: {LSTM_INPUT_MODEL_PATH}\")\n#     model = load_model(LSTM_INPUT_MODEL_PATH)\n#     print(\"Model loaded successfully.\")\n    \n# elif os.path.exists(LSTM_WORKING_MODEL_PATH):\n#     print(f\"Found saved LSTM model in working directory! Loading: {LSTM_WORKING_MODEL_PATH}\")\n#     model = load_model(LSTM_WORKING_MODEL_PATH)\n#     print(\"Model loaded successfully.\")\n    \n# else:\n#     print(\"No saved LSTM model found anywhere. Starting 85-minute training loop...\")\n#     early_stop_lstm = EarlyStopping(\n#         monitor='val_auc',\n#         patience=3,\n#         mode='max',\n#         restore_best_weights=True\n#     )\n    \n#     history = model.fit(\n#         X_train_pad, y_train,\n#         validation_data=(X_val_pad, y_val),\n#         epochs=10,\n#         batch_size=512,\n#         class_weight=class_weight_dict,\n#         callbacks=[early_stop_lstm]\n#     )\n    \n#     # Save to working directory\n#     model.save(LSTM_WORKING_MODEL_PATH)\n#     print(f\"Training complete. Model saved to {LSTM_WORKING_MODEL_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.362857Z","iopub.execute_input":"2026-07-17T12:27:44.363263Z","iopub.status.idle":"2026-07-17T12:27:44.377533Z","shell.execute_reply.started":"2026-07-17T12:27:44.363234Z","shell.execute_reply":"2026-07-17T12:27:44.376476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.metrics import classification_report, f1_score\n\n# # 1. Get validation probabilities\n# # model.predict returns a continuous value between 0.0 and 1.0 for each question\n# print(\"Generating predictions on validation set...\")\n# y_val_proba_lstm = model.predict(X_val_pad, batch_size=512).squeeze()\n\n# # 2. Reuse the find_best_threshold function\n# # This runs the exact same threshold sweep logic used for the baseline\n# print(\"\\n--- Tuning Classification Threshold for LSTM ---\")\n# best_lstm_threshold, max_lstm_f1 = find_best_threshold(y_val, y_val_proba_lstm)\n\n# # --- 3. Compare directly against baseline ---\n# baseline_f1 = 0.6048\n# f1_improvement = max_lstm_f1 - baseline_f1\n\n# print(\"--- Performance Comparison ---\")\n# print(f\"Baseline TF-IDF Logistic Regression F1-Score : {baseline_f1:.4f}\")\n# print(f\"Bidirectional LSTM F1-Score                  : {max_lstm_f1:.4f}\")\n# print(f\"Absolute F1-Score Change                     : {f1_improvement:+.4f}\")\n# print(\"----------------------------------------------\\n\")\n\n# # --- 4. Final Classification Report ---\n# final_lstm_preds = (y_val_proba_lstm >= best_lstm_threshold).astype(int)\n# print(\"--- Final LSTM Classification Report ---\")\n# print(classification_report(y_val, final_lstm_preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.378681Z","iopub.execute_input":"2026-07-17T12:27:44.378953Z","iopub.status.idle":"2026-07-17T12:27:44.392617Z","shell.execute_reply.started":"2026-07-17T12:27:44.378927Z","shell.execute_reply":"2026-07-17T12:27:44.391648Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submit the LSTM","metadata":{}},{"cell_type":"code","source":"# # Predict on the already-tokenized/padded test set\n# test_probabilities = model.predict(X_test_pad, batch_size=512).squeeze()\n\n# # Apply the LSTM's tuned threshold\n# test_predictions = (test_probabilities >= best_lstm_threshold).astype(int)\n\n# submission_df = pd.DataFrame({\n#     'qid': test_df['qid'],\n#     'prediction': test_predictions\n# })\n# submission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.394081Z","iopub.execute_input":"2026-07-17T12:27:44.394493Z","iopub.status.idle":"2026-07-17T12:27:44.414272Z","shell.execute_reply.started":"2026-07-17T12:27:44.394451Z","shell.execute_reply":"2026-07-17T12:27:44.413336Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 7: GRU + Attention","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Layer\nimport tensorflow.keras.backend as K\n\nclass AttentionPooling(Layer):\n    def __init__(self, **kwargs):\n        super().__init__(**kwargs)\n\n    def build(self, input_shape):\n        # input_shape example: (None, 50, 128) -> (batch, timesteps, hidden_dim)\n        # We need one weight per hidden_dim feature, to turn each timestep's\n        # 128-dim vector into a single score.\n        self.W = self.add_weight(\n            name=\"attn_weight\",\n            shape=(input_shape[-1], 1),   # (hidden_dim, 1) mapping features down to a 1D score\n            initializer=\"glorot_uniform\",\n            trainable=True\n        )\n        super().build(input_shape)\n\n    def call(self, inputs):\n        # inputs shape example: (batch, 50, 128)\n\n        # Step 1: score each timestep -> shape becomes (batch, 50) after squeeze\n        scores = K.dot(inputs, self.W)          # (batch, 50, 128) @ (128, 1) -> (batch, 50, 1)\n        scores = K.squeeze(scores, axis=-1)     # -> (batch, 50)\n\n        # Step 2: softmax across timesteps so weights sum to 1 per question\n        # Given shape (batch, timesteps), axis=-1 is the timestep axis (axis 1)\n        weights = tf.nn.softmax(scores, axis=-1)\n\n        # Step 3: weighted sum of the original inputs using these weights\n        weights_expanded = K.expand_dims(weights, axis=-1)   # (batch, 50, 1) for broadcasting\n        weighted_inputs = inputs * weights_expanded           # (batch, 50, 128)\n        \n        # Summing across the timesteps collapses axis 1 -> leaving (batch, hidden_dim)\n        output = tf.reduce_sum(weighted_inputs, axis=1)\n\n        return output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.415385Z","iopub.execute_input":"2026-07-17T12:27:44.415825Z","iopub.status.idle":"2026-07-17T12:27:44.433305Z","shell.execute_reply.started":"2026-07-17T12:27:44.415786Z","shell.execute_reply":"2026-07-17T12:27:44.432258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Input, Embedding, Bidirectional, GRU, Dropout, Dense\n\ntf.keras.backend.clear_session()\n\nmodel_gru = Sequential([\n    Input(shape=(MAX_LEN,), dtype='int32', name=\"input_layer\"),\n\n    Embedding(\n        input_dim=vocab_size,\n        output_dim=EMBED_DIM,\n        weights=[embedding_matrix],\n        trainable=False,\n        name=\"glove_embeddings\"\n    ),\n\n    Bidirectional(\n        # Swapped LSTM for GRU with matching parameter style\n        GRU(64, return_sequences=True, dropout=0.2, recurrent_dropout=0.2),\n        name=\"bi_gru\"\n    ),\n\n    AttentionPooling(name=\"attention_pooling\"),  # Dynamic weighted average layer\n\n    Dropout(0.3, name=\"dropout_layer\"),\n    Dense(32, activation='relu', name=\"dense_hidden\"),\n    Dense(1, activation='sigmoid', name=\"output_layer\")\n])\n\nmodel_gru.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n    loss='binary_crossentropy',\n    metrics=[\n        'accuracy', \n        tf.keras.metrics.AUC(name='auc'),\n        tf.keras.metrics.Precision(name='precision'),\n        tf.keras.metrics.Recall(name='recall')\n    ]\n)\n\nmodel_gru.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:44.434388Z","iopub.execute_input":"2026-07-17T12:27:44.434692Z","iopub.status.idle":"2026-07-17T12:27:46.062711Z","shell.execute_reply.started":"2026-07-17T12:27:44.434663Z","shell.execute_reply":"2026-07-17T12:27:46.061780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# 1. Re-calculate class weights (now safely independent of LSTM code)\nunique_classes = np.unique(y_train)\ncomputed_weights = compute_class_weight(\n    class_weight='balanced',\n    classes=unique_classes,\n    y=y_train\n)\nclass_weight_dict = dict(zip(unique_classes, computed_weights))\n\n# 1. Define your custom dataset paths\nGRU_INPUT_MODEL_PATH = '/kaggle/input/datasets/mdshafayeturrahman/quora-gru-attention-model/gru_attention_model.keras'\nGRU_WORKING_MODEL_PATH = '/kaggle/working/gru_attention_model.keras'\n# 2. Check and load\nif os.path.exists(GRU_INPUT_MODEL_PATH):\n    print(f\"Found pinned GRU model in your Kaggle Dataset! Loading: {GRU_INPUT_MODEL_PATH}\")\n    model_gru = load_model(GRU_INPUT_MODEL_PATH, custom_objects={'AttentionPooling': AttentionPooling})\n    print(\"Model loaded successfully.\")\n    \nelif os.path.exists(GRU_WORKING_MODEL_PATH):\n    print(f\"Found saved GRU model in working directory! Loading: {GRU_WORKING_MODEL_PATH}\")\n    model_gru = load_model(GRU_WORKING_MODEL_PATH, custom_objects={'AttentionPooling': AttentionPooling})\n    print(\"Model loaded successfully.\")\n    \nelse:\n    print(\"No saved GRU model found. Training from scratch...\")\n    \n    # Redefining early stopping to reset its internal validation history state\n    early_stop_gru = EarlyStopping(\n        monitor='val_auc',\n        patience=3,\n        mode='max',\n        restore_best_weights=True\n    )\n    \n    history_gru = model_gru.fit(\n        X_train_pad, y_train,\n        validation_data=(X_val_pad, y_val),\n        epochs=10,\n        batch_size=512,\n        class_weight=class_weight_dict,\n        callbacks=[early_stop_gru]\n    )\n    \n    model_gru.save(GRU_WORKING_MODEL_PATH)\n    print(f\"Training complete. Saved to {GRU_WORKING_MODEL_PATH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:46.063769Z","iopub.execute_input":"2026-07-17T12:27:46.064035Z","iopub.status.idle":"2026-07-17T12:27:47.232459Z","shell.execute_reply.started":"2026-07-17T12:27:46.064010Z","shell.execute_reply":"2026-07-17T12:27:47.231411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Generating predictions on validation set (GRU + Attention)...\")\ny_val_proba_gru = model_gru.predict(X_val_pad, batch_size=512).squeeze()\n\nprint(\"\\n--- Tuning Classification Threshold for GRU + Attention ---\")\nbest_gru_threshold, max_gru_f1 = find_best_threshold(y_val, y_val_proba_gru)\n\nprint(\"--- Performance Comparison ---\")\nprint(f\"Baseline TF-IDF Logistic Regression F1 : 0.6048\")\nprint(f\"Bidirectional LSTM F1                  : 0.6805\")\nprint(f\"Bidirectional GRU + Attention F1        : {max_gru_f1:.4f}\")\nprint(f\"Change vs LSTM                          : {max_gru_f1 - 0.6805:+.4f}\")\nprint(\"----------------------------------------------\\n\")\n\nfinal_gru_preds = (y_val_proba_gru >= best_gru_threshold).astype(int)\nprint(\"--- Final GRU + Attention Classification Report ---\")\nprint(classification_report(y_val, final_gru_preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:27:47.233579Z","iopub.execute_input":"2026-07-17T12:27:47.233925Z","iopub.status.idle":"2026-07-17T12:28:26.777269Z","shell.execute_reply.started":"2026-07-17T12:27:47.233888Z","shell.execute_reply":"2026-07-17T12:28:26.776398Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Submiting the GRU model","metadata":{}},{"cell_type":"code","source":"test_probabilities_gru = model_gru.predict(X_test_pad, batch_size=512).squeeze()\ntest_predictions_gru = (test_probabilities_gru >= best_gru_threshold).astype(int)\n\nsubmission_df = pd.DataFrame({\n    'qid': test_df['qid'],\n    'prediction': test_predictions_gru\n})\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-17T12:28:26.778334Z","iopub.execute_input":"2026-07-17T12:28:26.778644Z","iopub.status.idle":"2026-07-17T12:30:14.265900Z","shell.execute_reply.started":"2026-07-17T12:28:26.778590Z","shell.execute_reply":"2026-07-17T12:30:14.264743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}