{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":519432,"sourceType":"modelInstanceVersion","modelInstanceId":408997,"modelId":426851}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Winner's Solution: BiLSTM-Conv1D for Quora Insincere Questions\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import StratifiedKFold\nimport os\nimport zipfile\nfrom tqdm import tqdm\nimport re\nimport gc\n\nprint(\"=\"*60)\nprint(\"BiLSTM-Conv1D\")\nprint(\"=\"*60)\n\n# --- 1. Environment Setup ---\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nFINAL_MODEL_OUTPUT_PATH = './final_model'\nEMBEDDINGS_ZIP = '/kaggle/input/quora-insincere-questions-classification/embeddings.zip'\nEMBEDDINGS_EXTRACT_DIR = './embeddings'\n\nprint(f\"Using device: {device}\")\n\n# Extract embeddings\nif os.path.exists(EMBEDDINGS_ZIP):\n    print(f\"\\nExtracting embeddings...\")\n    os.makedirs(EMBEDDINGS_EXTRACT_DIR, exist_ok=True)\n    with zipfile.ZipFile(EMBEDDINGS_ZIP, 'r') as zip_ref:\n        zip_ref.extractall(EMBEDDINGS_EXTRACT_DIR)\n    \n    GLOVE_PATH = None\n    PARAGRAM_PATH = None\n    for root, dirs, files in os.walk(EMBEDDINGS_EXTRACT_DIR):\n        for file in files:\n            full_path = os.path.join(root, file)\n            if 'glove' in file.lower():\n                GLOVE_PATH = full_path\n            elif 'paragram' in file.lower():\n                PARAGRAM_PATH = full_path\n    \n    print(f\"GloVe: {GLOVE_PATH}\")\n    print(f\"Paragram: {PARAGRAM_PATH}\")\nelse:\n    print(\"Embeddings not found!\")\n    GLOVE_PATH = PARAGRAM_PATH = None\n\n# --- 2. Load Data (Train + Test for vocab) ---\nprint(\"\\nLoading data...\")\ntrain_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')\n\nprint(f\"Train: {len(train_df)} | Test: {len(test_df)}\")\n\n# --- 3. Text Preprocessing  ---\ndef preprocess_text(text):\n    \"\"\"Basic preprocessing from winning solution\"\"\"\n    # Remove URLs\n    text = re.sub(r'http\\S+', '', text)\n    # Split attached words (e.g., \"WhatIs\" -> \"What Is\")\n    text = re.sub(r'([a-z])([A-Z])', r'\\1 \\2', text)\n    # Add spaces around punctuation\n    text = re.sub(r'([?.!,¿])', r' \\1 ', text)\n    # Remove extra spaces\n    text = re.sub(r'\\s+', ' ', text).strip()\n    return text\n\nprint(\"\\nPreprocessing texts...\")\ntrain_df['question_text'] = train_df['question_text'].apply(preprocess_text)\ntest_df['question_text'] = test_df['question_text'].apply(preprocess_text)\n\n# --- 4. Statistical Features ---\ndef extract_statistical_features(text):\n    features = []\n    features.append(len(text))  # Length\n    features.append(len(text.split()))  # Word count\n    features.append(sum(1 for c in text if c.isupper()))  # Uppercase\n    features.append(sum(1 for c in text if c in '!?'))  # Punctuation\n    features.append(text.count(' '))  # Spaces\n    features.append(len(set(text.split())))  # Unique words\n    return features\n\nprint(\"Extracting statistical features...\")\ntrain_df['stat_features'] = train_df['question_text'].apply(extract_statistical_features)\ntest_df['stat_features'] = test_df['question_text'].apply(extract_statistical_features)\n\n# --- 5. Build Vocabulary (NO LOWERCASING, NO LIMIT, TRAIN+TEST) ---\ndef build_vocab_no_limit(texts):\n    \"\"\"Build vocabulary without limits, keeping case\"\"\"\n    word_counts = {}\n    for text in tqdm(texts, desc=\"Building vocab\"):\n        for word in text.split():  # NO .lower()!\n            word_counts[word] = word_counts.get(word, 0) + 1\n    \n    vocab = {word: idx + 2 for idx, word in enumerate(word_counts.keys())}\n    vocab['<PAD>'] = 0\n    vocab['<UNK>'] = 1\n    return vocab\n\nprint(\"\\nBuilding vocabulary from train + test...\")\nall_texts = pd.concat([train_df['question_text'], test_df['question_text']])\nvocab = build_vocab_no_limit(all_texts)\nprint(f\"Vocabulary size: {len(vocab)}\")\n\n# --- 6. Load Embeddings with OOV Handling ---\ndef load_embeddings_dict(filepath):\n    \"\"\"Load embeddings into dictionary\"\"\"\n    if not filepath or not os.path.exists(filepath):\n        return {}\n    \n    embeddings = {}\n    with open(filepath, 'r', encoding='utf-8', errors='ignore') as f:\n        for line in tqdm(f, desc=f\"Loading {os.path.basename(filepath)}\"):\n            try:\n                values = line.rstrip().split(' ')\n                word = values[0]\n                vector = np.array(values[1:], dtype='float32')\n                if len(vector) == 300:  # Ensure correct dimension\n                    embeddings[word] = vector\n            except:\n                continue\n    return embeddings\n\ndef handle_oov(word, embed_dict):\n    \"\"\"Try multiple strategies to find embedding for OOV words\"\"\"\n    # 1. Try exact match\n    if word in embed_dict:\n        return embed_dict[word]\n    \n    # 2. Try lowercase\n    if word.lower() in embed_dict:\n        return embed_dict[word.lower()]\n    \n    # 3. Try uppercase\n    if word.upper() in embed_dict:\n        return embed_dict[word.upper()]\n    \n    # 4. Try title case\n    if word.title() in embed_dict:\n        return embed_dict[word.title()]\n    \n    # 5. Remove special characters\n    cleaned = re.sub(r'[^a-zA-Z0-9]', '', word)\n    if cleaned and cleaned in embed_dict:\n        return embed_dict[cleaned]\n    if cleaned and cleaned.lower() in embed_dict:\n        return embed_dict[cleaned.lower()]\n    \n    # 6. Try singular/plural\n    if word.endswith('s') and word[:-1] in embed_dict:\n        return embed_dict[word[:-1]]\n    if word + 's' in embed_dict:\n        return embed_dict[word + 's']\n    \n    return None\n\nprint(\"\\nLoading embeddings...\")\nglove_dict = load_embeddings_dict(GLOVE_PATH)\nparagram_dict = load_embeddings_dict(PARAGRAM_PATH)\n\nprint(f\"GloVe vectors: {len(glove_dict)}\")\nprint(f\"Paragram vectors: {len(paragram_dict)}\")\n\n# Build embedding matrix with OOV handling\nprint(\"\\nBuilding embedding matrix with OOV handling...\")\nembedding_matrix = np.zeros((len(vocab), 300))\noov_count = 0\n\nfor word, idx in tqdm(vocab.items(), desc=\"Processing vocab\"):\n    if idx < 2:  # Skip PAD and UNK\n        continue\n    \n    # Try to find in GloVe\n    glove_vec = handle_oov(word, glove_dict)\n    # Try to find in Paragram\n    para_vec = handle_oov(word, paragram_dict)\n    \n    if glove_vec is not None and para_vec is not None:\n        # Both found: 0.7 * GloVe + 0.3 * Paragram\n        embedding_matrix[idx] = 0.7 * glove_vec + 0.3 * para_vec\n    elif glove_vec is not None:\n        embedding_matrix[idx] = glove_vec\n    elif para_vec is not None:\n        embedding_matrix[idx] = para_vec\n    else:\n        # OOV: random initialization\n        embedding_matrix[idx] = np.random.randn(300) * 0.01\n        oov_count += 1\n\n# Set UNK token to random\nembedding_matrix[1] = np.random.randn(300) * 0.01\n\ncoverage = 100 * (1 - oov_count / (len(vocab) - 2))\nprint(f\"Embedding coverage: {coverage:.2f}% | OOV: {oov_count}/{len(vocab)-2}\")\n\n# Clean up memory\ndel glove_dict, paragram_dict\ngc.collect()\n\n# --- 7. Text Encoding (NO LOWERCASING) ---\ndef encode_texts(texts, vocab, max_length=72):\n    \"\"\"Encode texts - NO LOWERCASING\"\"\"\n    sequences = []\n    for text in texts:\n        words = text.split()  # NO .lower()!\n        seq = [vocab.get(word, vocab['<UNK>']) for word in words]\n        if len(seq) < max_length:\n            seq = seq + [vocab['<PAD>']] * (max_length - len(seq))\n        else:\n            seq = seq[:max_length]\n        sequences.append(seq)\n    return np.array(sequences)\n\nprint(\"\\nEncoding texts...\")\nX_text_train = encode_texts(train_df['question_text'], vocab)\nX_stats_train = np.array(train_df['stat_features'].tolist())\ny_train = train_df['target'].values\n\nprint(f\"Training data encoded: {X_text_train.shape}\")\n\n# --- 8. Dynamic Padding Dataset ---\nclass DynamicPaddingDataset(Dataset):\n    def __init__(self, sequences, stat_features, labels):\n        self.sequences = sequences\n        self.stat_features = torch.FloatTensor(stat_features)\n        self.labels = torch.LongTensor(labels)\n        self.lengths = [np.sum(seq != 0) for seq in sequences]\n    \n    def __len__(self):\n        return len(self.labels)\n    \n    def __getitem__(self, idx):\n        return self.sequences[idx], self.stat_features[idx], self.labels[idx], self.lengths[idx]\n\ndef collate_dynamic_padding(batch):\n    \"\"\"Dynamic padding based on 95th percentile of batch\"\"\"\n    sequences, stats, labels, lengths = zip(*batch)\n    \n    # Get 95th percentile length\n    max_len = int(np.percentile(lengths, 95))\n    max_len = max(max_len, 10)  # Minimum length\n    \n    # Pad/truncate to this length\n    padded_seqs = []\n    for seq in sequences:\n        if len(seq) > max_len:\n            padded_seqs.append(seq[:max_len])\n        else:\n            padded_seqs.append(np.pad(seq, (0, max_len - len(seq)), constant_values=0))\n    \n    return (torch.LongTensor(padded_seqs), \n            torch.stack(stats), \n            torch.LongTensor(labels))\n\n# --- 9. Model Architecture ---\nclass BiLSTMConv1D(nn.Module):\n    def __init__(self, embedding_matrix, stat_feature_dim=6, lstm_units=128, \n                 conv_filters=64, dense_units=64, spatial_dropout=0.1, dropout=0.1):\n        super(BiLSTMConv1D, self).__init__()\n        \n        vocab_size, embed_dim = embedding_matrix.shape\n        \n        self.embedding = nn.Embedding(vocab_size, embed_dim)\n        self.embedding.weight = nn.Parameter(torch.tensor(embedding_matrix, dtype=torch.float32))\n        self.embedding.weight.requires_grad = False\n        \n        self.spatial_dropout = nn.Dropout2d(spatial_dropout)\n        self.lstm = nn.LSTM(embed_dim, lstm_units, batch_first=True, bidirectional=True)\n        self.conv1d = nn.Conv1d(lstm_units * 2, conv_filters, kernel_size=1)\n        self.stat_dense = nn.Linear(stat_feature_dim, dense_units)\n        self.dropout1 = nn.Dropout(dropout)\n        self.dense = nn.Linear(conv_filters + dense_units, 128)\n        self.dropout2 = nn.Dropout(dropout)\n        self.batch_norm = nn.BatchNorm1d(128)\n        self.output = nn.Linear(128, 2)\n    \n    def forward(self, text, stat_features):\n        x = self.embedding(text)\n        x = x.unsqueeze(1)\n        x = self.spatial_dropout(x)\n        x = x.squeeze(1)\n        \n        lstm_out, _ = self.lstm(x)\n        lstm_out = lstm_out.permute(0, 2, 1)\n        conv_out = self.conv1d(lstm_out)\n        pooled = F.max_pool1d(conv_out, conv_out.size(2)).squeeze(2)\n        \n        stat_out = F.relu(self.stat_dense(stat_features))\n        combined = torch.cat([pooled, stat_out], dim=1)\n        \n        x = self.dropout1(combined)\n        x = F.relu(self.dense(x))\n        x = self.dropout2(x)\n        x = self.batch_norm(x)\n        x = self.output(x)\n        \n        return x\n\n# --- 10. Training with One Cycle Policy ---\nclass OneCycleLR:\n    def __init__(self, optimizer, max_lr, total_steps, pct_start=0.3):\n        self.optimizer = optimizer\n        self.max_lr = max_lr\n        self.total_steps = total_steps\n        self.step_count = 0\n        self.pct_start = pct_start\n        \n    def step(self):\n        self.step_count += 1\n        if self.step_count <= self.total_steps * self.pct_start:\n            # Warmup\n            lr = self.max_lr * (self.step_count / (self.total_steps * self.pct_start))\n        else:\n            # Annealing\n            progress = (self.step_count - self.total_steps * self.pct_start) / (self.total_steps * (1 - self.pct_start))\n            lr = self.max_lr * (1 + np.cos(np.pi * progress)) / 2\n        \n        for param_group in self.optimizer.param_groups:\n            param_group['lr'] = lr\n\ndef train_model(model, train_loader, val_loader, epochs=4, max_lr=0.002):\n    optimizer = torch.optim.NAdam(model.parameters(), lr=max_lr)\n    criterion = nn.CrossEntropyLoss()\n    \n    total_steps = len(train_loader) * epochs\n    scheduler = OneCycleLR(optimizer, max_lr, total_steps)\n    \n    best_f1 = 0\n    \n    for epoch in range(epochs):\n        print(f\"\\nEpoch {epoch + 1}/{epochs}\")\n        \n        # Training\n        model.train()\n        train_loss = 0\n        all_preds, all_labels = [], []\n        \n        for text, stats, labels in tqdm(train_loader, desc=\"Training\"):\n            text, stats, labels = text.to(device), stats.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(text, stats)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            scheduler.step()\n            \n            train_loss += loss.item()\n            preds = torch.argmax(outputs, dim=1)\n            all_preds.extend(preds.cpu().numpy())\n            all_labels.extend(labels.cpu().numpy())\n        \n        train_loss /= len(train_loader)\n        train_f1 = f1_score(all_labels, all_preds)\n        \n        # Validation\n        model.eval()\n        val_loss = 0\n        all_preds, all_labels = [], []\n        \n        with torch.no_grad():\n            for text, stats, labels in tqdm(val_loader, desc=\"Validation\"):\n                text, stats, labels = text.to(device), stats.to(device), labels.to(device)\n                outputs = model(text, stats)\n                loss = criterion(outputs, labels)\n                \n                val_loss += loss.item()\n                preds = torch.argmax(outputs, dim=1)\n                all_preds.extend(preds.cpu().numpy())\n                all_labels.extend(labels.cpu().numpy())\n        \n        val_loss /= len(val_loader)\n        val_f1 = f1_score(all_labels, all_preds)\n        \n        print(f\"Train Loss: {train_loss:.4f} | Train F1: {train_f1:.4f}\")\n        print(f\"Val Loss: {val_loss:.4f} | Val F1: {val_f1:.4f}\")\n        \n        if val_f1 > best_f1:\n            best_f1 = val_f1\n            print(f\"New best F1: {best_f1:.4f}. Saving model...\")\n            os.makedirs(FINAL_MODEL_OUTPUT_PATH, exist_ok=True)\n            torch.save({\n                'model_state_dict': model.state_dict(),\n                'vocab': vocab,\n                'best_f1': best_f1\n            }, os.path.join(FINAL_MODEL_OUTPUT_PATH, 'model.pt'))\n    \n    return best_f1\n\n# --- 11. Train Single Model ---\nprint(\"\\n\" + \"=\"*60)\nprint(\"Training model...\")\nprint(\"=\"*60)\n\n# Simple train/val split for quick training\nfrom sklearn.model_selection import train_test_split\nX_tr, X_val, S_tr, S_val, y_tr, y_val = train_test_split(\n    X_text_train, X_stats_train, y_train, test_size=0.1, random_state=42, stratify=y_train\n)\n\ntrain_dataset = DynamicPaddingDataset(X_tr, S_tr, y_tr)\nval_dataset = DynamicPaddingDataset(X_val, S_val, y_val)\n\ntrain_loader = DataLoader(train_dataset, batch_size=512, shuffle=True, collate_fn=collate_dynamic_padding)\nval_loader = DataLoader(val_dataset, batch_size=512, shuffle=False, collate_fn=collate_dynamic_padding)\n\nprint(f\"Train batches: {len(train_loader)} | Val batches: {len(val_loader)}\")\n\nmodel = BiLSTMConv1D(embedding_matrix, stat_feature_dim=6, spatial_dropout=0.1, dropout=0.1)\nmodel.to(device)\n\nbest_f1 = train_model(model, train_loader, val_loader, epochs=8, max_lr=0.002)\n\nprint(\"\\n\" + \"=\"*60)\nprint(f\"Training complete! Best F1: {best_f1:.4f}\")\nprint(\"=\"*60)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-01T08:04:15.697578Z","iopub.execute_input":"2025-10-01T08:04:15.697857Z","iopub.status.idle":"2025-10-01T08:20:38.789504Z","shell.execute_reply.started":"2025-10-01T08:04:15.697838Z","shell.execute_reply":"2025-10-01T08:20:38.788793Z"}},"outputs":[{"name":"stdout","text":"============================================================\nBiLSTM-Conv1D\n============================================================\nUsing device: cuda\n\nExtracting embeddings...\nGloVe: ./embeddings/glove.840B.300d/glove.840B.300d.txt\nParagram: ./embeddings/paragram_300_sl999/paragram_300_sl999.txt\n\nLoading data...\nTrain: 1306122 | Test: 375806\n\nPreprocessing texts...\nExtracting statistical features...\n\nBuilding vocabulary from train + test...\n","output_type":"stream"},{"name":"stderr","text":"Building vocab: 100%|██████████| 1681928/1681928 [00:04<00:00, 336925.78it/s]\n","output_type":"stream"},{"name":"stdout","text":"Vocabulary size: 433619\n\nLoading embeddings...\n","output_type":"stream"},{"name":"stderr","text":"Loading glove.840B.300d.txt: 2196017it [02:07, 17234.36it/s]\nLoading paragram_300_sl999.txt: 1703756it [01:40, 17018.17it/s]\n","output_type":"stream"},{"name":"stdout","text":"GloVe vectors: 2196016\nParagram vectors: 1703755\n\nBuilding embedding matrix with OOV handling...\n","output_type":"stream"},{"name":"stderr","text":"Processing vocab: 100%|██████████| 433619/433619 [00:06<00:00, 66031.43it/s]\n","output_type":"stream"},{"name":"stdout","text":"Embedding coverage: 75.55% | OOV: 106010/433617\n\nEncoding texts...\nTraining data encoded: (1306122, 72)\n\n============================================================\nTraining model...\n============================================================\nTrain batches: 2296 | Val batches: 256\n\nEpoch 1/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:03<00:00, 36.18it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 76.34it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.2558 | Train F1: 0.3832\nVal Loss: 0.1088 | Val F1: 0.6030\nNew best F1: 0.6030. Saving model...\n\nEpoch 2/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:02<00:00, 36.55it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 76.51it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.1176 | Train F1: 0.5781\nVal Loss: 0.1011 | Val F1: 0.6316\nNew best F1: 0.6316. Saving model...\n\nEpoch 3/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:03<00:00, 35.99it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 75.36it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.1109 | Train F1: 0.6124\nVal Loss: 0.0980 | Val F1: 0.6446\nNew best F1: 0.6446. Saving model...\n\nEpoch 4/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:03<00:00, 35.99it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 75.69it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.1040 | Train F1: 0.6477\nVal Loss: 0.0962 | Val F1: 0.6794\nNew best F1: 0.6794. Saving model...\n\nEpoch 5/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:02<00:00, 36.70it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 76.95it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.0953 | Train F1: 0.6850\nVal Loss: 0.0984 | Val F1: 0.6684\n\nEpoch 6/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:03<00:00, 36.42it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 76.94it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.0843 | Train F1: 0.7341\nVal Loss: 0.1016 | Val F1: 0.6768\n\nEpoch 7/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:03<00:00, 36.38it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 77.51it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.0713 | Train F1: 0.7867\nVal Loss: 0.1156 | Val F1: 0.6651\n\nEpoch 8/8\n","output_type":"stream"},{"name":"stderr","text":"Training: 100%|██████████| 2296/2296 [01:02<00:00, 36.80it/s]\nValidation: 100%|██████████| 256/256 [00:03<00:00, 77.41it/s]\n","output_type":"stream"},{"name":"stdout","text":"Train Loss: 0.0638 | Train F1: 0.8186\nVal Loss: 0.1236 | Val F1: 0.6579\n\n============================================================\nTraining complete! Best F1: 0.6794\n============================================================\n","output_type":"stream"}],"execution_count":5},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}