{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10737,"databundleVersionId":290346}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Load necessary libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.pipeline import make_pipeline\nfrom lime.lime_text import LimeTextExplainer\nfrom collections import OrderedDict\n\n# hide warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:48.122943Z","iopub.execute_input":"2026-04-19T09:20:48.123231Z","iopub.status.idle":"2026-04-19T09:20:51.541217Z","shell.execute_reply.started":"2026-04-19T09:20:48.123196Z","shell.execute_reply":"2026-04-19T09:20:51.540345Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading the dataset","metadata":{}},{"cell_type":"code","source":"dataset = pd.read_csv(r\"/kaggle/input/competitions/quora-insincere-questions-classification/train.csv\")\n\n# first few attributes\ndataset.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:51.542760Z","iopub.execute_input":"2026-04-19T09:20:51.543150Z","iopub.status.idle":"2026-04-19T09:20:55.854591Z","shell.execute_reply.started":"2026-04-19T09:20:51.543119Z","shell.execute_reply":"2026-04-19T09:20:55.853872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# basic info of the dataset\nfeatures = list(dataset.columns[:-1])\ntarget = dataset.columns[-1]\nclasses = {\n    0 : \"sincere\",\n    1 : \"insincere\",\n}\n\nprint(f\"1. Features (X) are : {features} \\nqid - unique question identifier \\nquestion_text - Quora question text\\n\")\nprint(f\"2. Target (y) is : {target} a question labeled \\\"insincere\\\" has a value of 1, otherwise 0\\n\")\n      \nprint(f\"3. Shape of the dataset : {dataset.shape}\\n\")\nprint(f\"4. Information about the features\")\ndataset.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:55.855643Z","iopub.execute_input":"2026-04-19T09:20:55.856327Z","iopub.status.idle":"2026-04-19T09:20:55.998824Z","shell.execute_reply.started":"2026-04-19T09:20:55.856300Z","shell.execute_reply":"2026-04-19T09:20:55.998239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check for null values\ndataset.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:55.999751Z","iopub.execute_input":"2026-04-19T09:20:56.000090Z","iopub.status.idle":"2026-04-19T09:20:56.128956Z","shell.execute_reply.started":"2026-04-19T09:20:56.000067Z","shell.execute_reply":"2026-04-19T09:20:56.128233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class split\nsincere = dataset[dataset['target'] == 0]\ninsincere = dataset[dataset['target'] == 1]\n\nprint(f\"Sincere: {len(sincere)}\")\nprint(f\"Insincere: {len(insincere)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:56.130107Z","iopub.execute_input":"2026-04-19T09:20:56.130447Z","iopub.status.idle":"2026-04-19T09:20:56.237554Z","shell.execute_reply.started":"2026-04-19T09:20:56.130411Z","shell.execute_reply":"2026-04-19T09:20:56.236822Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Dataset is highly imbalanced","metadata":{}},{"cell_type":"markdown","source":"## Splitting the data into train and val data","metadata":{}},{"cell_type":"code","source":"train_df, val_df = train_test_split(dataset, test_size=0.2, random_state=42)\n\nprint(f\"Train data : {train_df.head()}\\n\")\nprint(f\"Val data : {val_df.head()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:56.238564Z","iopub.execute_input":"2026-04-19T09:20:56.238840Z","iopub.status.idle":"2026-04-19T09:20:56.800576Z","shell.execute_reply.started":"2026-04-19T09:20:56.238813Z","shell.execute_reply":"2026-04-19T09:20:56.799948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TF-IDF vectorizer","metadata":{}},{"cell_type":"code","source":"# Create a TF-IDF vectorizer and transform the training and validation data\n\n# vectorize to tf-idf vectors\ntfidf_vc = TfidfVectorizer(min_df = 10, max_features = 100000, analyzer = \"word\", ngram_range = (1, 2), stop_words = 'english', lowercase = True)\ntrain_vc = tfidf_vc.fit_transform(train_df[\"question_text\"])\nval_vc = tfidf_vc.transform(val_df[\"question_text\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:20:56.802576Z","iopub.execute_input":"2026-04-19T09:20:56.802874Z","iopub.status.idle":"2026-04-19T09:21:28.351421Z","shell.execute_reply.started":"2026-04-19T09:20:56.802851Z","shell.execute_reply":"2026-04-19T09:21:28.350775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train a Logistic Regression model on the training data\n\nmodel = LogisticRegression(C = 0.5, solver = \"sag\")\nmodel = model.fit(train_vc, train_df.target)\n\n# Predict on the validation data\nval_pred = model.predict(val_vc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:28.352282Z","iopub.execute_input":"2026-04-19T09:21:28.352546Z","iopub.status.idle":"2026-04-19T09:21:38.923278Z","shell.execute_reply.started":"2026-04-19T09:21:28.352515Z","shell.execute_reply":"2026-04-19T09:21:38.922636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate evaluation metrics\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix, classification_report\n\naccuracy = accuracy_score(val_df.target, val_pred)\nprecision = precision_score(val_df.target, val_pred)\nrecall = recall_score(val_df.target, val_pred)\nf1 = f1_score(val_df.target, val_pred)\n\n# Print evaluation metrics\nprint(\"Accuracy:\", accuracy)\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\nprint(\"F1 Score:\", f1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:38.924303Z","iopub.execute_input":"2026-04-19T09:21:38.924596Z","iopub.status.idle":"2026-04-19T09:21:38.977421Z","shell.execute_reply.started":"2026-04-19T09:21:38.924566Z","shell.execute_reply":"2026-04-19T09:21:38.976600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display confusion matrix\nconf_matrix = confusion_matrix(val_df.target, val_pred)\nprint(\"Confusion Matrix:\\n\", conf_matrix)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:38.978378Z","iopub.execute_input":"2026-04-19T09:21:38.979110Z","iopub.status.idle":"2026-04-19T09:21:38.991056Z","shell.execute_reply.started":"2026-04-19T09:21:38.979075Z","shell.execute_reply":"2026-04-19T09:21:38.990299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display classification report\n# Define class names\nclass_names = [\"sincere\", \"insincere\"]\nclass_report = classification_report(val_df.target, val_pred, target_names=class_names)\nprint(\"Classification Report:\\n\", class_report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:38.991990Z","iopub.execute_input":"2026-04-19T09:21:38.992330Z","iopub.status.idle":"2026-04-19T09:21:39.126385Z","shell.execute_reply.started":"2026-04-19T09:21:38.992297Z","shell.execute_reply":"2026-04-19T09:21:39.125717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter the rows where target is 1\ntarget_1_rows = val_df[val_df['target'] == 1]\n\n# Print the filtered rows and their row indices\nprint(\"Rows with target = 1:\")\nprint(target_1_rows)\n\nprint(\"\\nRow indices of rows with target = 1:\")\nprint(target_1_rows.index.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.127205Z","iopub.execute_input":"2026-04-19T09:21:39.127509Z","iopub.status.idle":"2026-04-19T09:21:39.155981Z","shell.execute_reply.started":"2026-04-19T09:21:39.127486Z","shell.execute_reply":"2026-04-19T09:21:39.155233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select a specific instance from the validation set for explanation\nimport numpy as np\nprediction_index = 20\nidx = int(val_df.index[prediction_index])\n# print(idx)\nc = make_pipeline(tfidf_vc, model)\nclass_names = [\"sincere\", \"insincere\"]\n\n# Create a LIME text explainer\nexplainer = LimeTextExplainer(class_names = class_names)\n\n# Explain the prediction for the selected instance\nexp = explainer.explain_instance(val_df[\"question_text\"][idx], c.predict_proba, num_features = 10)\n\n# Print the selected question text and its prediction probabilities\nprint(val_df[\"question_text\"][idx])\nprint(\"Probability (Insincere) =\", c.predict_proba([val_df[\"question_text\"][idx]])[0, 1])\nprint(\"Probability (Sincere) =\", c.predict_proba([val_df[\"question_text\"][idx]])[0, 0])\nprint(\"True Class is:\", class_names[int(val_df[\"target\"][idx])])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.156960Z","iopub.execute_input":"2026-04-19T09:21:39.157166Z","iopub.status.idle":"2026-04-19T09:21:39.431845Z","shell.execute_reply.started":"2026-04-19T09:21:39.157147Z","shell.execute_reply":"2026-04-19T09:21:39.431255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get explanation weights as a list of tuples\nexp.as_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.432696Z","iopub.execute_input":"2026-04-19T09:21:39.433044Z","iopub.status.idle":"2026-04-19T09:21:39.438059Z","shell.execute_reply.started":"2026-04-19T09:21:39.433015Z","shell.execute_reply":"2026-04-19T09:21:39.437221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print original prediction probability\nprint('Original prediction:',  model.predict_proba(val_vc[prediction_index])[0, 1])\n\n# Create a copy of the selected instance's TF-IDF vector and modify specific features\ntmp = val_vc[prediction_index].copy()\ntmp[0, tfidf_vc.vocabulary_['track']] = 0\ntmp[0, tfidf_vc.vocabulary_['college']] = 0\n\n# Print prediction after removing specific features\nprint('Prediction after removing some features:', model.predict_proba(tmp)[0, 1])\n\n# Print the difference in prediction probabilities\nprint('Difference:', model.predict_proba(tmp)[0, 1] - model.predict_proba(val_vc[prediction_index])[0, 1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.439023Z","iopub.execute_input":"2026-04-19T09:21:39.439423Z","iopub.status.idle":"2026-04-19T09:21:39.451581Z","shell.execute_reply.started":"2026-04-19T09:21:39.439401Z","shell.execute_reply":"2026-04-19T09:21:39.450885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display LIME explanation in a notebook\nexp.show_in_notebook(text=val_df[\"question_text\"][idx], labels=(1,))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.452489Z","iopub.execute_input":"2026-04-19T09:21:39.453641Z","iopub.status.idle":"2026-04-19T09:21:39.488983Z","shell.execute_reply.started":"2026-04-19T09:21:39.453610Z","shell.execute_reply":"2026-04-19T09:21:39.488036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract and plot LIME weights\nweights = OrderedDict(exp.as_list())\nlime_weights = pd.DataFrame({\"words\": list(weights.keys()), \"weights\": list(weights.values())})\n\n# Plot the feature weights\nsns.barplot(x = \"words\", y = \"weights\", data = lime_weights, palette=\"viridis\")\nplt.xticks(rotation = 45)\nplt.title(\"Sample {} features weights given by LIME\".format(idx))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T09:21:39.490048Z","iopub.execute_input":"2026-04-19T09:21:39.490320Z","iopub.status.idle":"2026-04-19T09:21:39.757541Z","shell.execute_reply.started":"2026-04-19T09:21:39.490300Z","shell.execute_reply":"2026-04-19T09:21:39.756668Z"}},"outputs":[],"execution_count":null}]}