{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":237213,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":202582,"modelId":224322}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Training LAMA and PyTorch (Transformer) on [Quora Insincere Questions](https://www.kaggle.com/competitions/quora-insincere-questions-classification) competition ","metadata":{}},{"cell_type":"markdown","source":"### **Short description**: the task is basically to classify input text on two classes - good and bad Quora questions. The training dataset consists of approximately 1.3kk rows of text and corresponding binary labels (0 for good and 1 for bad question respectively). The requirement of competition is to use one of listed embeddings:\n- GoogleNews-vectors-negative300;\n- glove.840B.300d;\n- paragram_300_sl999;\n- wiki-news-300d-1M.\n\n### We will implement LAMA and PyTorch solutions, both based on one of that embeddings. These are main points of this notebook:\n- EDA and preprocessing;\n- Implementing LAMA solution (2 configs);\n- Implementing PyTorch solution (Transformer encoder).","metadata":{}},{"cell_type":"code","source":"!pip install lightautoml seaborn matplotlib tqdm scikit-learn pandas","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:10:30.452767Z","iopub.execute_input":"2025-01-21T03:10:30.453038Z","iopub.status.idle":"2025-01-21T03:10:42.121779Z","shell.execute_reply.started":"2025-01-21T03:10:30.453015Z","shell.execute_reply":"2025-01-21T03:10:42.120774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip /kaggle/input/quora-insincere-questions-classification/embeddings.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:10:42.123014Z","iopub.execute_input":"2025-01-21T03:10:42.123272Z","iopub.status.idle":"2025-01-21T03:13:23.535996Z","shell.execute_reply.started":"2025-01-21T03:10:42.123249Z","shell.execute_reply":"2025-01-21T03:13:23.535231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import some libs","metadata":{}},{"cell_type":"code","source":"import re\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:13:27.536682Z","iopub.execute_input":"2025-01-21T03:13:27.53699Z","iopub.status.idle":"2025-01-21T03:14:02.85186Z","shell.execute_reply.started":"2025-01-21T03:13:27.536964Z","shell.execute_reply":"2025-01-21T03:14:02.85123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LAMA params\nN_THREADS = 4\nTIMEOUT = 3600\nTARGET_NAME = \"target\"\n\n# We will use GloVe embeddings (listed on the competition page)\nEMBEDDING_PATH = \"/kaggle/working/glove.840B.300d/glove.840B.300d.txt\"\nEMBED_SIZE = 300\nMAX_LEN = 50\nPAD_TOKEN = \"<PAD>\"\n\n# Reproducibility\nRANDOM_STATE = 42\nnp.random.seed(RANDOM_STATE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:02.852825Z","iopub.execute_input":"2025-01-21T03:14:02.853398Z","iopub.status.idle":"2025-01-21T03:14:02.857321Z","shell.execute_reply.started":"2025-01-21T03:14:02.853374Z","shell.execute_reply":"2025-01-21T03:14:02.8565Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading and preprocessing data","metadata":{}},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:02.859011Z","iopub.execute_input":"2025-01-21T03:14:02.859358Z","iopub.status.idle":"2025-01-21T03:14:07.559175Z","shell.execute_reply.started":"2025-01-21T03:14:02.859311Z","shell.execute_reply":"2025-01-21T03:14:07.558232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class distribution\nprint(train_df[TARGET_NAME].value_counts())\n\n# Class distribution in percentages\nprint(train_df[TARGET_NAME].value_counts(normalize=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:07.560151Z","iopub.execute_input":"2025-01-21T03:14:07.560516Z","iopub.status.idle":"2025-01-21T03:14:07.587328Z","shell.execute_reply.started":"2025-01-21T03:14:07.560482Z","shell.execute_reply":"2025-01-21T03:14:07.58653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class distribution visualization\nplt.figure(figsize=(8, 6))\nsns.countplot(x=TARGET_NAME, data=train_df)\nplt.title(\"Class Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:12.305852Z","iopub.execute_input":"2025-01-21T03:14:12.306227Z","iopub.status.idle":"2025-01-21T03:14:12.612394Z","shell.execute_reply.started":"2025-01-21T03:14:12.306197Z","shell.execute_reply":"2025-01-21T03:14:12.611522Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### There is significant class imbalance in dataset, that will make hard to get high scores on the positive one","metadata":{}},{"cell_type":"code","source":"# Feature types\ntrain_df.info()\n\n# checking for nulls in data\nprint(train_df.isnull().sum())","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate text lengths (number of words)\ntrain_df[\"text_length\"] = train_df[\"question_text\"].apply(lambda x: len(x.split()))\ntest_df[\"text_length\"] = test_df[\"question_text\"].apply(lambda x: len(x.split()))\n\n# Mean and median length\nmean_length = train_df[\"text_length\"].mean()\nmedian_length = train_df[\"text_length\"].median()\n\nprint(f\"Mean text length: {mean_length:.2f}\")\nprint(f\"Median text length: {median_length:.2f}\")\n\n# Quantiles\nquantiles = train_df[\"text_length\"].quantile([0.25, 0.5, 0.75, 0.9, 0.95, 0.99])\n\nprint(\"\\nText Length Quantiles:\")\nprint(quantiles)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:19.969437Z","iopub.execute_input":"2025-01-21T03:14:19.969738Z","iopub.status.idle":"2025-01-21T03:14:21.694802Z","shell.execute_reply.started":"2025-01-21T03:14:19.969716Z","shell.execute_reply":"2025-01-21T03:14:21.694072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Histogram of text lengths\nplt.figure(figsize=(10, 6))\nsns.histplot(train_df[\"text_length\"], bins=50)\nplt.title(\"Distribution of Text Lengths\")\nplt.xlabel(\"Number of Words\")\nplt.ylabel(\"Frequency\")\nplt.axvline(\n    mean_length,\n    color=\"red\",\n    linestyle=\"dashed\",\n    linewidth=1,\n    label=f\"Mean: {mean_length:.2f}\",\n)\nplt.axvline(\n    median_length,\n    color=\"green\",\n    linestyle=\"dashed\",\n    linewidth=1,\n    label=f\"Median: {median_length:.2f}\",\n)\nplt.legend()\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplot of text lengths\nplt.figure(figsize=(10, 6))\nsns.boxplot(x=train_df[\"text_length\"])\nplt.title(\"Boxplot of Text Lengths\")\nplt.xlabel(\"Number of Words\")\nplt.show()\n\n# Identify potential outliers (e.g., using 1.5 * IQR rule)\nQ1 = train_df[\"text_length\"].quantile(0.25)\nQ3 = train_df[\"text_length\"].quantile(0.75)\nIQR = Q3 - Q1\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\noutliers = train_df[\n    (train_df[\"text_length\"] < lower_bound) | (train_df[\"text_length\"] > upper_bound)\n]\n\nprint(f\"\\nNumber of potential outliers: {len(outliers)}\")\n# print(f\"Potential outliers:\\n{outliers[['question_text', 'text_length']]}\") # Print only relevant columns","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### More than 99% of training texts contain less than 39 words with ~81426 potential outliers (which is ~6.2% of the whole dataset), so we set padding and truncatiuon to **50** words","metadata":{}},{"cell_type":"code","source":"# Use some basic preprocessing for real human messages\ndef clean_text(text: str):\n    \"\"\"\n    Cleans the input text by converting to lowercase, replacing contractions,\n    and removing special characters and extra spaces.\n\n    Args:\n        text: The input text string.\n\n    Returns:\n        The cleaned text string.\n    \"\"\"\n    text = text.lower()\n    text = re.sub(r\"what's\", \"what is \", text)\n    text = re.sub(r\"\\'s\", \" \", text)\n    text = re.sub(r\"\\'ve\", \" have \", text)\n    text = re.sub(r\"can't\", \"cannot \", text)\n    text = re.sub(r\"n't\", \" not \", text)\n    text = re.sub(r\"i'm\", \"i am \", text)\n    text = re.sub(r\"\\'re\", \" are \", text)\n    text = re.sub(r\"\\'d\", \" would \", text)\n    text = re.sub(r\"\\'ll\", \" will \", text)\n    text = re.sub(r\"\\'scuse\", \" excuse \", text)\n    text = re.sub(\"\\W\", \" \", text)\n    text = re.sub(\"\\s+\", \" \", text)\n    text = text.strip(\" \")\n    \n    return text","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:49.823345Z","iopub.execute_input":"2025-01-21T03:14:49.823634Z","iopub.status.idle":"2025-01-21T03:14:49.82904Z","shell.execute_reply.started":"2025-01-21T03:14:49.823614Z","shell.execute_reply":"2025-01-21T03:14:49.828207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_glove_embeddings(embedding_path: str):\n    \"\"\"\n    Loads GloVe embeddings from a file.\n\n    Args:\n        embedding_path: The path to the GloVe embedding file.\n\n    Returns:\n        A dictionary where keys are words and values are their embedding vectors.\n    \"\"\"\n    embeddings_index = {}\n    with open(embedding_path, encoding=\"utf8\") as f:\n        for line in tqdm(f):\n            values = line.rstrip().rsplit(' ')\n            word = values[0]\n            coefs = np.asarray(values[1:], dtype='float32')\n            \n            # it's better to norm all the inputs\n            norm = np.linalg.norm(coefs)\n            if norm != 0:\n                coefs = coefs / norm\n\n            embeddings_index[word] = coefs\n\n    # adding CLS token for Transformer solution\n    # it won't make problems for LAMA either\n    cls_embedding = np.random.randn(EMBED_SIZE)\n    cls_embedding /= np.linalg.norm(cls_embedding)\n    embeddings_index[\"<CLS>\"] = cls_embedding\n\n    return embeddings_index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:14:50.042559Z","iopub.execute_input":"2025-01-21T03:14:50.042806Z","iopub.status.idle":"2025-01-21T03:14:50.048092Z","shell.execute_reply.started":"2025-01-21T03:14:50.042786Z","shell.execute_reply":"2025-01-21T03:14:50.047137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### As we restricted to use pre-trained embeddings due to the competition rules, one possible approach to train LAMA models is to compute mean GloVe embedding for every training sentence (padded to fixed len) and use it as table feature (shape of the input = [EMB_DIM] in this case), so one vector represents whole text","metadata":{}},{"cell_type":"code","source":"def pad_sequence(words: list, max_len: int, pad_token: str):\n    \"\"\"\n    Pads a sequence of words to a fixed length.\n\n    Args:\n        words: A list of words.\n        max_len: The desired fixed length.\n        pad_token: The token used for padding.\n\n    Returns:\n        A list of words padded to the fixed length.\n    \"\"\"\n    # CLS token goes first\n    words = [\"<CLS>\"] + words\n    \n    if len(words) >= max_len:\n        return words[:max_len]\n    else:\n        return words + [pad_token] * (max_len - len(words))\n\n\ndef get_sentence_embedding(text: str, embeddings_index: dict, max_len: int, pad_token: str):\n    \"\"\"\n    Generates a sentence embedding by averaging the embeddings of its words,\n    after padding the sentence to a fixed length.\n\n    Args:\n        text: The input sentence.\n        embeddings_index: A dictionary of word embeddings.\n        max_len: The fixed length for padding.\n        pad_token: The padding token.\n\n    Returns:\n        The averaged sentence embedding vector.\n    \"\"\"\n    words = text.split()\n    padded_words = pad_sequence(words, max_len, pad_token)\n    embeddings = [\n        embeddings_index.get(word, np.zeros(EMBED_SIZE)) for word in padded_words\n    ]\n    \n    return np.mean(embeddings, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:15:58.311047Z","iopub.execute_input":"2025-01-21T03:15:58.311391Z","iopub.status.idle":"2025-01-21T03:15:58.316696Z","shell.execute_reply.started":"2025-01-21T03:15:58.311366Z","shell.execute_reply":"2025-01-21T03:15:58.315927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess text\ntrain_df[\"question_text\"] = train_df[\"question_text\"].apply(lambda x: clean_text(x))\ntest_df[\"question_text\"] = test_df[\"question_text\"].apply(lambda x: clean_text(x))\n\n# Looking for duplucates\nprint(f\"Duplicated in train data: {sum(train_df.duplicated(['question_text']))}\")\ntrain_df = train_df.drop_duplicates(['question_text']).reset_index()\n\nprint(f\"Duplicated in test data: {sum(test_df.duplicated(['question_text']))}\")\n\nind = test_df.question_text.isin(train_df.question_text) & train_df.question_text.isin(test_df.question_text)\nprint(f\"Intersection between train and test: {sum(ind)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:16:02.505715Z","iopub.execute_input":"2025-01-21T03:16:02.506011Z","iopub.status.idle":"2025-01-21T03:16:34.169425Z","shell.execute_reply.started":"2025-01-21T03:16:02.505988Z","shell.execute_reply":"2025-01-21T03:16:34.168529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load GloVe embeddings\nglove_embeddings = load_glove_embeddings(EMBEDDING_PATH)\n\n# Generate sentence embeddings with padding\ntrain_df[\"embedding\"] = train_df[\"question_text\"].apply(\n    lambda x: get_sentence_embedding(x, glove_embeddings, MAX_LEN, PAD_TOKEN)\n)\ntest_df[\"embedding\"] = test_df[\"question_text\"].apply(\n    lambda x: get_sentence_embedding(x, glove_embeddings, MAX_LEN, PAD_TOKEN)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:20:14.235332Z","iopub.execute_input":"2025-01-21T03:20:14.23562Z","iopub.status.idle":"2025-01-21T03:25:16.018017Z","shell.execute_reply.started":"2025-01-21T03:20:14.235597Z","shell.execute_reply":"2025-01-21T03:25:16.017074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert embeddings to separate columns\ntrain_embeddings_df = pd.DataFrame(\n    train_df[\"embedding\"].values.tolist(),\n    columns=[f\"emb_{i}\" for i in range(EMBED_SIZE)],\n    index=train_df.index,\n    dtype=\"float32\",\n)\ntest_embeddings_df = pd.DataFrame(\n    test_df[\"embedding\"].values.tolist(),\n    columns=[f\"emb_{i}\" for i in range(EMBED_SIZE)],\n    index=test_df.index,\n    dtype=\"float32\",\n)\n\ntrain_df = pd.concat([train_df, train_embeddings_df], axis=1)\ntest_df = pd.concat([test_df, test_embeddings_df], axis=1)\n\n# Drop original text and embedding columns\ntrain_df.drop([\"embedding\"], axis=1, inplace=True)\ntest_df.drop([\"embedding\"], axis=1, inplace=True)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Some embeddings analysis","metadata":{}},{"cell_type":"code","source":"# Distribution of the first embedding dimension\nplt.figure(figsize=(8, 6))\nsns.histplot(train_df[\"emb_0\"], bins=30)\nplt.title(\"Distribution of Embedding Dimension 0\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplot for the first embedding dimension\nplt.figure(figsize=(8, 6))\nsns.boxplot(x=train_df[\"emb_0\"])\nplt.title(\"Boxplot of Embedding Dimension 0\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# correlation between first 50 embeddings\nplt.figure(figsize=(30, 30))\ncorr_matrix = train_df[[f\"emb_{i}\" for i in range(50)]].corr()\nsns.heatmap(corr_matrix, annot=True, cmap=\"coolwarm\")\nplt.title(\"Correlation Matrix of Embedding Dimensions\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlation with the target variable","metadata":{}},{"cell_type":"code","source":"embedding_correlations = (\n    train_df[[f\"emb_{i}\" for i in range(EMBED_SIZE)] + [TARGET_NAME]]\n    .corr()[TARGET_NAME]\n    .drop(TARGET_NAME)\n)\n\n# Sort by absolute value\nembedding_correlations = embedding_correlations.abs().sort_values(ascending=False)\n\nplt.figure(figsize=(10, 6))\nembedding_correlations.head(20).plot(kind=\"bar\")  # Top 20 most correlated\nplt.title(\"Correlation of Embedding Dimensions with Target (Absolute Value)\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nembedding_correlations.tail(20).plot(kind=\"bar\")  # Top 20 least correlated\nplt.title(\"Correlation of Embedding Dimensions with Target (Absolute Value)\")\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### We defenitely see some corellation between embeddings and target, maybe there is some room for feature engineering here, but don't think it's necessary here","metadata":{}},{"cell_type":"code","source":"# Split data into train and validation sets\n# We are sure there is no data leakage between train and test (checked earlier)\n# 5% of ~1.3kk rows more than enough to be a good val data\ntrain_data, valid_data = train_test_split(\n    train_df, test_size=0.05, random_state=RANDOM_STATE, stratify=train_df[TARGET_NAME]\n)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training LAMA","metadata":{}},{"cell_type":"code","source":"# Define the task (binary classification)\ntask = Task(\"binary\")\n\n# Define roles for LightAutoML\nroles = {\"target\": TARGET_NAME, \"drop\": [\"qid\", \"text_length\", \"question_text\"]}","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create and train the TabularAutoML (Configuration 1)\nautoml_config1 = TabularAutoML(\n    task=task,\n    timeout=TIMEOUT,\n    cpu_limit=N_THREADS,\n    general_params={\"use_algos\": [[\"lgb\"]]},\n)\n\nautoml_config1.fit_predict(train_data, roles=roles, valid_data=valid_data, verbose=10)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create and train the TabularAutoML (Configuration 2)\nautoml_config2 = TabularAutoML(\n    task=task,\n    timeout=TIMEOUT,\n    cpu_limit=N_THREADS,\n    general_params={\"use_algos\": [[\"cb\"]]},\n)\n\nautoml_config2.fit_predict(train_data, roles=roles, valid_data=valid_data, verbose=10)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate performance (example using F1 score for positive class (insincere in our case))\nfrom sklearn.metrics import f1_score\n\n\n# Config 1\nvalid_pred_config1 = automl_config1.predict(valid_data)\nvalid_predictions_config1 = (valid_pred_config1.data[:, 0] > 0.5).astype(int)\nf1_config1 = f1_score(valid_data[TARGET_NAME].values, valid_predictions_config1)\nprint(f\"F1-score (Config 1): {f1_config1}\")\n\n# Config 2\nvalid_pred_config2 = automl_config2.predict(valid_data)\nvalid_predictions_config2 = (valid_pred_config2.data[:, 0] > 0.5).astype(int)\nf1_config2 = f1_score(valid_data[TARGET_NAME].values, valid_predictions_config2)\nprint(f\"F1-score (Config 2): {f1_config2}\")\n\n# Choose best option\nif f1_config1 > f1_config2:\n    best_automl = automl_config1\n    print(\"Config 1 is better\")\nelse:\n    best_automl = automl_config2\n    print(\"Config 2 is better\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## So we got ~0.55 f1 mean on LAMA, let's try PyTorch Transformer","metadata":{}},{"cell_type":"markdown","source":"## Training Torch Transformer","metadata":{}},{"cell_type":"markdown","source":"### Transformer is known for it's exceptional NLP robustness. It can accept 2D input shape ([MAX_LEN, EMB_SIZE], so every input sentence consists of fixed amount of tokens, every one of which represented with it's own vector of EMB_SIZE len). That means we can use full GloVe embedding for every word in sentence instead of computing single mean vector, which for sure should increase quality of the model","metadata":{}},{"cell_type":"code","source":"import math\n\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.metrics import f1_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:16:41.769863Z","iopub.execute_input":"2025-01-21T03:16:41.77028Z","iopub.status.idle":"2025-01-21T03:16:41.775123Z","shell.execute_reply.started":"2025-01-21T03:16:41.770244Z","shell.execute_reply":"2025-01-21T03:16:41.77424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset class\nclass QuoraDataset(Dataset):\n    def __init__(self, texts, targets, embeddings_index, max_len, pad_token):\n        self.texts = texts\n        self.targets = targets\n        self.embeddings_index = embeddings_index\n        self.max_len = max_len\n        self.pad_token = pad_token\n\n    def __len__(self):\n        return len(self.texts)\n\n    def __getitem__(self, idx):\n        text = self.texts[idx]\n        label = np.array(self.targets[idx])\n\n        words = text.split()\n        padded_words = pad_sequence(words, self.max_len, self.pad_token)\n        attention_mask = np.array([1 if el != PAD_TOKEN else 0 for el in padded_words])\n\n        embeddings = np.array([\n            self.embeddings_index.get(word, np.zeros(EMBED_SIZE))\n            for word in padded_words\n        ])\n        \n        return {\n            \"input_ids\": torch.tensor(embeddings, dtype=torch.float),\n            \"attention_mask\": torch.tensor(attention_mask, dtype=torch.float),\n            \"labels\": torch.tensor(label, dtype=torch.long),\n        }\n\n\ndef collate_fn(batch):\n    input_ids = torch.stack([item[\"input_ids\"] for item in batch])\n    attention_mask = torch.stack([item[\"attention_mask\"] for item in batch])\n    labels = torch.stack([item[\"labels\"] for item in batch])\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_mask\": attention_mask,\n        \"labels\": labels,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:16:47.985891Z","iopub.execute_input":"2025-01-21T03:16:47.986266Z","iopub.status.idle":"2025-01-21T03:16:47.993781Z","shell.execute_reply.started":"2025-01-21T03:16:47.986236Z","shell.execute_reply":"2025-01-21T03:16:47.992799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Not-trainable PE\nclass PositionalEncoding(nn.Module):\n    def __init__(self, d_model, dropout=0.1, max_len=5000):\n        super(PositionalEncoding, self).__init__()\n        self.dropout = nn.Dropout(p=dropout)\n\n        pe = torch.zeros(max_len, d_model)\n        position = torch.arange(0, max_len, dtype=torch.float).unsqueeze(1)\n        div_term = torch.exp(\n            torch.arange(0, d_model, 2).float() * (-math.log(10000.0) / d_model)\n        )\n        pe[:, 0::2] = torch.sin(position * div_term)\n        pe[:, 1::2] = torch.cos(position * div_term)\n        pe = pe.unsqueeze(0).transpose(0, 1)\n        self.register_buffer(\"pe\", pe)\n\n    def forward(self, x):\n        x = x + self.pe[: x.size(0), :]\n        return self.dropout(x)\n\n\n# Basic transformer encoder\nclass TransformerClassifier(nn.Module):\n    def __init__(self, embedding_dim, hidden_dim, num_heads, num_layers, output_dim):\n        super(TransformerClassifier, self).__init__()\n        self.embedding = nn.Linear(\n            in_features=embedding_dim, out_features=embedding_dim\n        )\n        self.pe = PositionalEncoding(d_model=embedding_dim)\n        self.encoder = nn.TransformerEncoder(\n            nn.TransformerEncoderLayer(\n                d_model=embedding_dim,\n                nhead=num_heads,\n                dim_feedforward=hidden_dim,\n                batch_first=True,\n                norm_first=True,\n            ),\n            num_layers=num_layers,\n        )\n        self.linear = nn.Linear(in_features=embedding_dim, out_features=output_dim)\n\n    def forward(self, x, att_mask):\n        x = self.embedding(x)\n        x = self.pe(x)\n\n        x = self.encoder(x, src_key_padding_mask=(att_mask == 0))\n\n        # use CLS token as pooled vector of sentence\n        x = x[:, 0, :]\n\n        x = self.linear(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:16:50.204322Z","iopub.execute_input":"2025-01-21T03:16:50.204635Z","iopub.status.idle":"2025-01-21T03:16:50.213262Z","shell.execute_reply.started":"2025-01-21T03:16:50.204613Z","shell.execute_reply":"2025-01-21T03:16:50.212218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_torch(\n    model, train_loader, val_loader, epochs, optimizer, criterion, device, model_path\n):\n    # Training\n    model.to(device)\n    best_val_f1 = float(\"-inf\")\n\n    for epoch in range(epochs):\n        model.train()\n        total_loss = 0\n\n        for batch in tqdm(\n            train_loader,\n            # desc=f\"Epoch {epoch + 1}/{epochs}\",\n            # total=len(train_loader),\n        ):\n            # Getting input\n            input_ids = batch[\"input_ids\"].to(device)\n            attention_mask = batch[\"attention_mask\"].to(device)\n            labels = batch[\"labels\"].to(device)\n\n            # Forward pass\n            optimizer.zero_grad()\n            outputs = model(input_ids, attention_mask)\n\n            # Computing loss\n            loss = criterion(outputs, labels)\n\n            # Backward pass\n            loss.backward()\n            optimizer.step()\n            total_loss += loss.item()\n\n        avg_train_loss = total_loss / len(train_loader)\n        print(f\"Epoch {epoch + 1} - Mean loss: {avg_train_loss:.4f}\")\n\n        # Validation\n        model.eval()\n        val_loss = 0\n        all_preds = []\n        all_labels = []\n        with torch.no_grad():\n            for batch in tqdm(val_loader, desc=\"Validation\"):\n                input_ids = batch[\"input_ids\"].to(device)\n                attention_mask = batch[\"attention_mask\"].to(device)\n                labels = batch[\"labels\"].to(device)\n\n                outputs = model(input_ids, attention_mask)\n                loss = criterion(outputs, labels)\n                val_loss += loss.item()\n                preds = torch.argmax(outputs, dim=1)\n                all_preds.extend(preds.cpu().numpy())\n                all_labels.extend(labels.cpu().numpy())\n\n        avg_val_loss = val_loss / len(val_loader)\n        val_f1 = f1_score(all_labels, all_preds)\n        print(f\"Val loss: {avg_val_loss:.4f}, Val F1-score: {val_f1:.4f}\")\n\n        if val_f1 > best_val_f1:\n            best_val_f1 = val_f1\n            torch.save(model.state_dict(), model_path)\n            print(\n                f\"Checkpoint saved on epoch {epoch + 1} with F1-score: {val_f1:.4f}\"\n            )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:16:52.275186Z","iopub.execute_input":"2025-01-21T03:16:52.275486Z","iopub.status.idle":"2025-01-21T03:16:52.283176Z","shell.execute_reply.started":"2025-01-21T03:16:52.275464Z","shell.execute_reply":"2025-01-21T03:16:52.282259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Inference\ndef predict_torch(model, test_loader, device):\n    model.to(device)\n    model.eval()\n    predictions = []\n    with torch.no_grad():\n        for batch in tqdm(test_loader, desc=\"Preds\"):\n            input_ids = batch['input_ids'].to(device)\n            attention_mask = batch['attention_mask'].to(device)\n            outputs = model(input_ids, attention_mask)\n            \n            preds = torch.argmax(outputs, dim=1)\n            predictions.extend(preds.cpu().numpy())\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:17:55.717381Z","iopub.execute_input":"2025-01-21T03:17:55.717678Z","iopub.status.idle":"2025-01-21T03:17:55.72247Z","shell.execute_reply.started":"2025-01-21T03:17:55.717656Z","shell.execute_reply":"2025-01-21T03:17:55.721476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transformer params\nHIDDEN_DIM = 512\nNUM_LAYERS = 2\nNUM_HEADS = 5\nOUTPUT_DIM = 2\nBATCH_SIZE = 512\nEPOCHS = 15\nLEARNING_RATE = 1e-3\nMODEL_PATH = \"best_transformer_model.pth\"\n\ndevice = \"cuda:0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:19:43.802052Z","iopub.execute_input":"2025-01-21T03:19:43.802406Z","iopub.status.idle":"2025-01-21T03:19:43.807173Z","shell.execute_reply.started":"2025-01-21T03:19:43.80238Z","shell.execute_reply":"2025-01-21T03:19:43.806287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting inputs and labels\ntrain_texts = train_data[\"question_text\"].tolist()\nvalid_texts = valid_data[\"question_text\"].tolist()\n\ntrain_targets = train_data[TARGET_NAME].values\nvalid_targets = valid_data[TARGET_NAME].values","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating train and val dataset objects\ntrain_dataset = QuoraDataset(\n    texts=train_texts,\n    targets=train_targets,\n    embeddings_index=glove_embeddings,\n    max_len=MAX_LEN,\n    pad_token=PAD_TOKEN,\n)\nvalid_dataset = QuoraDataset(\n    texts=valid_texts,\n    targets=valid_targets,\n    embeddings_index=glove_embeddings,\n    max_len=MAX_LEN,\n    pad_token=PAD_TOKEN,\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating train and val dataloader objects\ntrain_loader = DataLoader(\n    dataset=train_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    num_workers=0,\n    collate_fn=collate_fn,\n)\nval_loader = DataLoader(\n    dataset=valid_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=4,\n    collate_fn=collate_fn,\n)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating test dataloader objects\ntest_texts = test_df[\"question_text\"].tolist()\ntest_dataset = QuoraDataset(\n    texts=test_texts,\n    targets=np.zeros(len(test_df)),\n    embeddings_index=glove_embeddings,\n    max_len=MAX_LEN,\n    pad_token=PAD_TOKEN,\n)\ntest_loader = DataLoader(\n    dataset=test_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=4,\n    collate_fn=collate_fn,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:25:56.84968Z","iopub.execute_input":"2025-01-21T03:25:56.849983Z","iopub.status.idle":"2025-01-21T03:25:56.86715Z","shell.execute_reply.started":"2025-01-21T03:25:56.849958Z","shell.execute_reply":"2025-01-21T03:25:56.866254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Init model\nmodel = TransformerClassifier(\n    embedding_dim=EMBED_SIZE,\n    hidden_dim=HIDDEN_DIM,\n    num_heads=NUM_HEADS,\n    num_layers=NUM_LAYERS,\n    output_dim=OUTPUT_DIM,\n)\n\n# Init optimizer and loss func\noptimizer = torch.optim.Adam(params=model.parameters(), lr=LEARNING_RATE)\ncriterion = nn.CrossEntropyLoss()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model training\ntrain_torch(\n    model=model,\n    train_loader=train_loader,\n    val_loader=val_loader,\n    epochs=EPOCHS,\n    optimizer=optimizer,\n    criterion=criterion,\n    device=device,\n    model_path=MODEL_PATH,\n)\n\n# Loading saved best ckpt\nmodel.load_state_dict(torch.load(MODEL_PATH))\n\n# Make sure weight are loaded correctly\nvalid_predictions_torch = predict_torch(\n    model=model, test_loader=val_loader, device=device\n)\nf1_torch = f1_score(valid_data[TARGET_NAME].values, valid_predictions_torch)\nprint(f\"F1-score (PyTorch Transformer): {f1_torch}\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## So with Transformer we've reached ~10% more f1 score than LAMA got!","metadata":{}},{"cell_type":"code","source":"# Test preds\nmodel = TransformerClassifier(\n    embedding_dim=EMBED_SIZE,\n    hidden_dim=HIDDEN_DIM,\n    num_heads=NUM_HEADS,\n    num_layers=NUM_LAYERS,\n    output_dim=OUTPUT_DIM,\n)\nmodel.load_state_dict(torch.load(\"/kaggle/input/best_transformer_model.pth/pytorch/default/1/best_transformer_model.pth\"))\n\ntest_predictions_torch = predict_torch(model, test_loader, device)\n\nsubmission_df = pd.DataFrame(\n    {\"qid\": test_df[\"qid\"], \"prediction\": test_predictions_torch}\n)\n\n# Saving submission\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T03:29:32.596467Z","iopub.execute_input":"2025-01-21T03:29:32.596782Z","iopub.status.idle":"2025-01-21T03:30:19.113989Z","shell.execute_reply.started":"2025-01-21T03:29:32.596758Z","shell.execute_reply":"2025-01-21T03:30:19.113065Z"}},"outputs":[],"execution_count":null}]}