{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Text Classification with Bag of Words - Natural Language Processing**\n\n### **Outline**\n1. Download and explore a real-world dataset\n2. Apply text preprocessing techniques\n3. Implement the bag of words model\n4. Train ML models for text classification\n5. Make predictions and submit to Kaggle","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:05:57.348239Z","iopub.execute_input":"2025-10-24T06:05:57.348615Z","iopub.status.idle":"2025-10-24T06:05:57.723904Z","shell.execute_reply.started":"2025-10-24T06:05:57.348585Z","shell.execute_reply":"2025-10-24T06:05:57.723056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Download and explore a real-world dataset**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\nraw_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\nraw_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:06:08.758254Z","iopub.execute_input":"2025-10-24T06:06:08.758750Z","iopub.status.idle":"2025-10-24T06:06:13.696253Z","shell.execute_reply.started":"2025-10-24T06:06:08.758722Z","shell.execute_reply":"2025-10-24T06:06:13.695007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sincere_df = raw_df[raw_df.target == 0]\nsincere_df.question_text.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:06:19.950297Z","iopub.execute_input":"2025-10-24T06:06:19.951367Z","iopub.status.idle":"2025-10-24T06:06:20.049509Z","shell.execute_reply.started":"2025-10-24T06:06:19.951333Z","shell.execute_reply":"2025-10-24T06:06:20.048177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"insincere_df = raw_df[raw_df.target == 1]\ninsincere_df.question_text.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:06:33.944144Z","iopub.execute_input":"2025-10-24T06:06:33.944483Z","iopub.status.idle":"2025-10-24T06:06:33.970254Z","shell.execute_reply.started":"2025-10-24T06:06:33.944442Z","shell.execute_reply":"2025-10-24T06:06:33.969107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_df.target.value_counts(normalize=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:06:50.144850Z","iopub.execute_input":"2025-10-24T06:06:50.145191Z","iopub.status.idle":"2025-10-24T06:06:50.166711Z","shell.execute_reply.started":"2025-10-24T06:06:50.145164Z","shell.execute_reply":"2025-10-24T06:06:50.165645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_df.target.value_counts(normalize=True).plot(kind='bar')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:07:07.373902Z","iopub.execute_input":"2025-10-24T06:07:07.374332Z","iopub.status.idle":"2025-10-24T06:07:07.661007Z","shell.execute_reply.started":"2025-10-24T06:07:07.374305Z","shell.execute_reply":"2025-10-24T06:07:07.660060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\ntest_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:07:24.138029Z","iopub.execute_input":"2025-10-24T06:07:24.138376Z","iopub.status.idle":"2025-10-24T06:07:25.597349Z","shell.execute_reply.started":"2025-10-24T06:07:24.138350Z","shell.execute_reply":"2025-10-24T06:07:25.596294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\")\nsub_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:07:29.321900Z","iopub.execute_input":"2025-10-24T06:07:29.322223Z","iopub.status.idle":"2025-10-24T06:07:29.675527Z","shell.execute_reply.started":"2025-10-24T06:07:29.322198Z","shell.execute_reply":"2025-10-24T06:07:29.674541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Create a Working Sample**","metadata":{}},{"cell_type":"code","source":"SAMPLE_SIZE = 100_000\n\nsample_df = raw_df.sample(SAMPLE_SIZE, random_state=42)\n\nsample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:28:25.382417Z","iopub.execute_input":"2025-10-24T07:28:25.383721Z","iopub.status.idle":"2025-10-24T07:28:26.653741Z","shell.execute_reply.started":"2025-10-24T07:28:25.383676Z","shell.execute_reply":"2025-10-24T07:28:26.652645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Apply Text Preprocessing Techniques**\n\nOutline:\n\n1. Understand the bag of words model\n2. Tokenization\n3. Stop word removal\n4. Stemming","metadata":{}},{"cell_type":"markdown","source":"### **Bag of Words Intuition**\n\n1. Create a list of all the words across all the text documents\n2. You convert each document into vector counts of each word\n\n\nLimitations:\n1. There may be too many words in the dataset\n2. Some words may occur too frequently\n3. Some words may occur very rarely or only once\n4. A single word may have many forms (go, gone, going or bird vs. birds)","metadata":{}},{"cell_type":"code","source":"q0 = sincere_df.question_text.values[1]\nq0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:08:34.141523Z","iopub.execute_input":"2025-10-24T06:08:34.142659Z","iopub.status.idle":"2025-10-24T06:08:34.148807Z","shell.execute_reply.started":"2025-10-24T06:08:34.142625Z","shell.execute_reply":"2025-10-24T06:08:34.147715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1 = raw_df[raw_df.target == 1].question_text.values[0]\nq1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:08:34.781617Z","iopub.execute_input":"2025-10-24T06:08:34.781935Z","iopub.status.idle":"2025-10-24T06:08:34.812628Z","shell.execute_reply.started":"2025-10-24T06:08:34.781913Z","shell.execute_reply":"2025-10-24T06:08:34.811259Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Tokenization**\n\nsplitting a document into words and separators using [word_tokenize](https://www.nltk.org/)","metadata":{}},{"cell_type":"code","source":"import nltk\nfrom nltk.tokenize import word_tokenize","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:08:49.461892Z","iopub.execute_input":"2025-10-24T06:08:49.462232Z","iopub.status.idle":"2025-10-24T06:08:50.354129Z","shell.execute_reply.started":"2025-10-24T06:08:49.462210Z","shell.execute_reply":"2025-10-24T06:08:50.353041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:08:50.356007Z","iopub.execute_input":"2025-10-24T06:08:50.356345Z","iopub.status.idle":"2025-10-24T06:08:50.363527Z","shell.execute_reply.started":"2025-10-24T06:08:50.356321Z","shell.execute_reply":"2025-10-24T06:08:50.362371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"word_tokenize(q0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:15:56.738506Z","iopub.execute_input":"2025-10-24T06:15:56.738851Z","iopub.status.idle":"2025-10-24T06:15:56.779877Z","shell.execute_reply.started":"2025-10-24T06:15:56.738827Z","shell.execute_reply":"2025-10-24T06:15:56.778882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:16:00.803981Z","iopub.execute_input":"2025-10-24T06:16:00.804318Z","iopub.status.idle":"2025-10-24T06:16:00.810841Z","shell.execute_reply.started":"2025-10-24T06:16:00.804296Z","shell.execute_reply":"2025-10-24T06:16:00.809539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"word_tokenize(q1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:16:01.252081Z","iopub.execute_input":"2025-10-24T06:16:01.252420Z","iopub.status.idle":"2025-10-24T06:16:01.258691Z","shell.execute_reply.started":"2025-10-24T06:16:01.252390Z","shell.execute_reply":"2025-10-24T06:16:01.257864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_tok = word_tokenize(q0)\nq1_tok = word_tokenize(q1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:16:05.841898Z","iopub.execute_input":"2025-10-24T06:16:05.842236Z","iopub.status.idle":"2025-10-24T06:16:05.847868Z","shell.execute_reply.started":"2025-10-24T06:16:05.842212Z","shell.execute_reply":"2025-10-24T06:16:05.846643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Stop Word Removal**\n\nRemoving commonly occuring words using [stopwords](https://www.geeksforgeeks.org/nlp/removing-stop-words-nltk-python/)","metadata":{}},{"cell_type":"code","source":"q1_tok","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:16:25.522682Z","iopub.execute_input":"2025-10-24T06:16:25.523139Z","iopub.status.idle":"2025-10-24T06:16:25.531634Z","shell.execute_reply.started":"2025-10-24T06:16:25.523104Z","shell.execute_reply":"2025-10-24T06:16:25.530385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.corpus import stopwords","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:16:30.497559Z","iopub.execute_input":"2025-10-24T06:16:30.497844Z","iopub.status.idle":"2025-10-24T06:16:30.503228Z","shell.execute_reply.started":"2025-10-24T06:16:30.497826Z","shell.execute_reply":"2025-10-24T06:16:30.501723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"english_stopwords = stopwords.words('english')\n\n\", \".join(english_stopwords)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:19:08.908111Z","iopub.execute_input":"2025-10-24T06:19:08.908496Z","iopub.status.idle":"2025-10-24T06:19:08.917922Z","shell.execute_reply.started":"2025-10-24T06:19:08.908442Z","shell.execute_reply":"2025-10-24T06:19:08.916875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_stopwords(tokens):\n    return [word for word in tokens if word.lower() not in english_stopwords]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:21:02.747359Z","iopub.execute_input":"2025-10-24T06:21:02.747757Z","iopub.status.idle":"2025-10-24T06:21:02.754330Z","shell.execute_reply.started":"2025-10-24T06:21:02.747730Z","shell.execute_reply":"2025-10-24T06:21:02.752930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_tok","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:21:06.295682Z","iopub.execute_input":"2025-10-24T06:21:06.295974Z","iopub.status.idle":"2025-10-24T06:21:06.303042Z","shell.execute_reply.started":"2025-10-24T06:21:06.295954Z","shell.execute_reply":"2025-10-24T06:21:06.302089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_stp = remove_stopwords(q0_tok)\n\nq0_stp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:21:11.705297Z","iopub.execute_input":"2025-10-24T06:21:11.706021Z","iopub.status.idle":"2025-10-24T06:21:11.712821Z","shell.execute_reply.started":"2025-10-24T06:21:11.705991Z","shell.execute_reply":"2025-10-24T06:21:11.711721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stp = remove_stopwords(q1_tok)\n\nq1_stp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:21:17.161367Z","iopub.execute_input":"2025-10-24T06:21:17.161859Z","iopub.status.idle":"2025-10-24T06:21:17.169376Z","shell.execute_reply.started":"2025-10-24T06:21:17.161822Z","shell.execute_reply":"2025-10-24T06:21:17.168188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Stemming**\n\nStemming is an important text-processing technique that reduces words to their base or root form by removing prefixes and suffixes. This process standardizes words which helps to improve the efficiency and effectiveness of various natural language processing (NLP) tasks.\n\nStemming simplifies words to their most basic form, making it easier to analyze and process text. For example, \"chocolates\" becomes \"chocolate\" and \"retrieval\" becomes \"retrieve\".\n\n* \"go\", \"gone\", \"going\" -> \"go\"\n* \"birds\", \"bird\" -> \"bird\"\n* \"likes\" → \"like\"\n* \"liked\" → \"like\"\n* \"likely\" → \"like\"\n* \"liking\" → \"like\"\n\nmore information - https://www.geeksforgeeks.org/machine-learning/introduction-to-stemming/\n\n![Stemming](https://media.geeksforgeeks.org/wp-content/uploads/20250717111849550734/_What-is-Stemming.webp)","metadata":{}},{"cell_type":"code","source":"from nltk.stem import SnowballStemmer\n\nstemmer = SnowballStemmer(language='english')\n\nwords_to_stem = ['running', 'jumped', 'happily', 'quickly', 'foxes', 'going', 'fighting', 'crying', 'killed']\n\nstemmed_words = [stemmer.stem(word) for word in words_to_stem]\n\nprint(\"Original words:\", words_to_stem)\nprint(\"Stemmed words:\", stemmed_words)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:29:50.524738Z","iopub.execute_input":"2025-10-24T06:29:50.525106Z","iopub.status.idle":"2025-10-24T06:29:50.532308Z","shell.execute_reply.started":"2025-10-24T06:29:50.525069Z","shell.execute_reply":"2025-10-24T06:29:50.531039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_stm = [stemmer.stem(word) for word in q0_stp]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:32:48.027261Z","iopub.execute_input":"2025-10-24T06:32:48.027636Z","iopub.status.idle":"2025-10-24T06:32:48.032892Z","shell.execute_reply.started":"2025-10-24T06:32:48.027610Z","shell.execute_reply":"2025-10-24T06:32:48.031832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_stp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:33:08.345671Z","iopub.execute_input":"2025-10-24T06:33:08.346150Z","iopub.status.idle":"2025-10-24T06:33:08.354073Z","shell.execute_reply.started":"2025-10-24T06:33:08.346120Z","shell.execute_reply":"2025-10-24T06:33:08.352773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q0_stm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:33:21.401120Z","iopub.execute_input":"2025-10-24T06:33:21.401451Z","iopub.status.idle":"2025-10-24T06:33:21.407651Z","shell.execute_reply.started":"2025-10-24T06:33:21.401427Z","shell.execute_reply":"2025-10-24T06:33:21.406670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stm = [stemmer.stem(word) for word in q1_stp]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:35:38.683855Z","iopub.execute_input":"2025-10-24T06:35:38.684259Z","iopub.status.idle":"2025-10-24T06:35:38.689219Z","shell.execute_reply.started":"2025-10-24T06:35:38.684235Z","shell.execute_reply":"2025-10-24T06:35:38.688305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stp","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:35:51.447823Z","iopub.execute_input":"2025-10-24T06:35:51.448147Z","iopub.status.idle":"2025-10-24T06:35:51.455223Z","shell.execute_reply.started":"2025-10-24T06:35:51.448126Z","shell.execute_reply":"2025-10-24T06:35:51.454163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T06:35:56.738930Z","iopub.execute_input":"2025-10-24T06:35:56.739300Z","iopub.status.idle":"2025-10-24T06:35:56.745693Z","shell.execute_reply.started":"2025-10-24T06:35:56.739275Z","shell.execute_reply":"2025-10-24T06:35:56.744755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Lemmatization**\n\nLemmatization is an important text pre-processing technique in Natural Language Processing (NLP) that reduces words to their base form known as a \"lemma.\" For example, the lemma of \"running\" is \"run\" and \"better\" becomes \"good.\" Unlike **stemming** which simply removes **prefixes** or **suffixes**, it considers the word's meaning and **part of speech (POS)** and ensures that the base form is a valid word. This makes lemmatization more accurate as it avoids generating non-dictionary words.\n\n\"love\" -> \"love\"\n\"loving\" -> \"love\"\n\"lovable\" -> \"love\"\n\n![lemmatization](https://media.geeksforgeeks.org/wp-content/uploads/20251004170959676805/lemmatization.webp)","metadata":{}},{"cell_type":"markdown","source":"## **Implement the bag of words model**\n\nIn Natural Language Processing (NLP) text data needs to be converted into numbers so that machine learning algorithms can understand it. One common method to do this is `Bag of Words (BoW) model`. It turns text like `sentence`, `paragraph` or `document` into a collection of words and counts how often each word appears but ignoring the order of the words. It does not consider the order of the words or their grammar but focuses on counting how often each word appears in the text.\n\nOutline:\n\n1. Create a vocabulary using Count Vectorizer\n2. Transform text to vectors using Count Vectorizer\n3. Configure text preprocessing in Count Vectorizer\n\nhttps://www.datacamp.com/tutorial/python-bag-of-words-model","metadata":{}},{"cell_type":"markdown","source":"### **Create a Vocabulary**","metadata":{}},{"cell_type":"code","source":"small_df = sample_df[:5]\nsmall_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:31:21.752816Z","iopub.execute_input":"2025-10-24T07:31:21.753199Z","iopub.status.idle":"2025-10-24T07:31:21.766444Z","shell.execute_reply.started":"2025-10-24T07:31:21.753171Z","shell.execute_reply":"2025-10-24T07:31:21.764989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"small_df.question_text.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:31:22.474785Z","iopub.execute_input":"2025-10-24T07:31:22.475684Z","iopub.status.idle":"2025-10-24T07:31:22.482400Z","shell.execute_reply.started":"2025-10-24T07:31:22.475649Z","shell.execute_reply":"2025-10-24T07:31:22.481234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:31:28.636364Z","iopub.execute_input":"2025-10-24T07:31:28.636803Z","iopub.status.idle":"2025-10-24T07:31:28.642523Z","shell.execute_reply.started":"2025-10-24T07:31:28.636772Z","shell.execute_reply":"2025-10-24T07:31:28.641080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a CountVectorizer Object\nsmall_vect = CountVectorizer()\n\n# Fit \nsmall_vect.fit(small_df.question_text)\n\n# Print the generated vocabulary\nprint(\"Vocabulary:\", small_vect.get_feature_names_out())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:31:43.974885Z","iopub.execute_input":"2025-10-24T07:31:43.975523Z","iopub.status.idle":"2025-10-24T07:31:43.985247Z","shell.execute_reply.started":"2025-10-24T07:31:43.975491Z","shell.execute_reply":"2025-10-24T07:31:43.983779Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Transform documents into Vectors**","metadata":{}},{"cell_type":"code","source":"vectors = small_vect.transform(small_df.question_text)\nvectors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:31:57.032127Z","iopub.execute_input":"2025-10-24T07:31:57.032551Z","iopub.status.idle":"2025-10-24T07:31:57.041309Z","shell.execute_reply.started":"2025-10-24T07:31:57.032520Z","shell.execute_reply":"2025-10-24T07:31:57.039951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"small_df.question_text.values[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:32:03.214212Z","iopub.execute_input":"2025-10-24T07:32:03.214645Z","iopub.status.idle":"2025-10-24T07:32:03.222264Z","shell.execute_reply.started":"2025-10-24T07:32:03.214614Z","shell.execute_reply":"2025-10-24T07:32:03.221153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectors[0].toarray()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:32:07.360534Z","iopub.execute_input":"2025-10-24T07:32:07.360889Z","iopub.status.idle":"2025-10-24T07:32:07.370365Z","shell.execute_reply.started":"2025-10-24T07:32:07.360865Z","shell.execute_reply":"2025-10-24T07:32:07.368769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectors.toarray()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:32:11.910504Z","iopub.execute_input":"2025-10-24T07:32:11.911675Z","iopub.status.idle":"2025-10-24T07:32:11.919727Z","shell.execute_reply.started":"2025-10-24T07:32:11.911631Z","shell.execute_reply":"2025-10-24T07:32:11.918552Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Configure Count Vectorizer Parameters**","metadata":{}},{"cell_type":"code","source":"stemmer = SnowballStemmer(language='english')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:36:55.009358Z","iopub.execute_input":"2025-10-24T07:36:55.009801Z","iopub.status.idle":"2025-10-24T07:36:55.016176Z","shell.execute_reply.started":"2025-10-24T07:36:55.009771Z","shell.execute_reply":"2025-10-24T07:36:55.014904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tokenize(text):\n    return [stemmer.stem(word) for word in word_tokenize(text)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:39:29.787640Z","iopub.execute_input":"2025-10-24T07:39:29.788093Z","iopub.status.idle":"2025-10-24T07:39:29.794227Z","shell.execute_reply.started":"2025-10-24T07:39:29.788057Z","shell.execute_reply":"2025-10-24T07:39:29.793056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenize('What is the really (dealing) here?')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:39:30.980117Z","iopub.execute_input":"2025-10-24T07:39:30.980569Z","iopub.status.idle":"2025-10-24T07:39:30.988447Z","shell.execute_reply.started":"2025-10-24T07:39:30.980537Z","shell.execute_reply":"2025-10-24T07:39:30.987109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectorizer = CountVectorizer(lowercase=True, \n                             tokenizer=tokenize,\n                             stop_words=english_stopwords,\n                             max_features=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:43:04.910433Z","iopub.execute_input":"2025-10-24T07:43:04.910939Z","iopub.status.idle":"2025-10-24T07:43:04.917112Z","shell.execute_reply.started":"2025-10-24T07:43:04.910905Z","shell.execute_reply":"2025-10-24T07:43:04.915743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nvectorizer.fit(sample_df.question_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:43:11.872961Z","iopub.execute_input":"2025-10-24T07:43:11.873385Z","iopub.status.idle":"2025-10-24T07:43:35.990121Z","shell.execute_reply.started":"2025-10-24T07:43:11.873352Z","shell.execute_reply":"2025-10-24T07:43:35.988776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(vectorizer.vocabulary_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:45:24.455564Z","iopub.execute_input":"2025-10-24T07:45:24.456030Z","iopub.status.idle":"2025-10-24T07:45:24.463594Z","shell.execute_reply.started":"2025-10-24T07:45:24.455994Z","shell.execute_reply":"2025-10-24T07:45:24.462320Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:45:38.167274Z","iopub.execute_input":"2025-10-24T07:45:38.168560Z","iopub.status.idle":"2025-10-24T07:45:38.176788Z","shell.execute_reply.started":"2025-10-24T07:45:38.168511Z","shell.execute_reply":"2025-10-24T07:45:38.175601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ninputs = vectorizer.transform(sample_df.question_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:47:13.708088Z","iopub.execute_input":"2025-10-24T07:47:13.709780Z","iopub.status.idle":"2025-10-24T07:47:37.056542Z","shell.execute_reply.started":"2025-10-24T07:47:13.709724Z","shell.execute_reply":"2025-10-24T07:47:37.055258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:48:03.691588Z","iopub.execute_input":"2025-10-24T07:48:03.692522Z","iopub.status.idle":"2025-10-24T07:48:03.699813Z","shell.execute_reply.started":"2025-10-24T07:48:03.692482Z","shell.execute_reply":"2025-10-24T07:48:03.698782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.question_text.values[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:48:09.059758Z","iopub.execute_input":"2025-10-24T07:48:09.060142Z","iopub.status.idle":"2025-10-24T07:48:09.068026Z","shell.execute_reply.started":"2025-10-24T07:48:09.060112Z","shell.execute_reply":"2025-10-24T07:48:09.067030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:48:28.458427Z","iopub.execute_input":"2025-10-24T07:48:28.458846Z","iopub.status.idle":"2025-10-24T07:48:28.472756Z","shell.execute_reply.started":"2025-10-24T07:48:28.458817Z","shell.execute_reply":"2025-10-24T07:48:28.471168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntest_inputs = vectorizer.transform(test_df.question_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:49:06.452621Z","iopub.execute_input":"2025-10-24T07:49:06.453038Z","iopub.status.idle":"2025-10-24T07:50:35.387862Z","shell.execute_reply.started":"2025-10-24T07:49:06.453008Z","shell.execute_reply":"2025-10-24T07:50:35.387024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_inputs.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:51:24.172733Z","iopub.execute_input":"2025-10-24T07:51:24.175907Z","iopub.status.idle":"2025-10-24T07:51:24.186991Z","shell.execute_reply.started":"2025-10-24T07:51:24.175836Z","shell.execute_reply":"2025-10-24T07:51:24.185265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Train ML models for text classification**\n\nOutline:\n\n- Create a training & validation set\n- Train a logistic regression model\n- Make predictions on training, validation & test data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:54:34.970349Z","iopub.execute_input":"2025-10-24T07:54:34.970828Z","iopub.status.idle":"2025-10-24T07:54:34.976142Z","shell.execute_reply.started":"2025-10-24T07:54:34.970797Z","shell.execute_reply":"2025-10-24T07:54:34.974783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_inputs, val_inputs, train_targets, val_targets = train_test_split(inputs, sample_df.target, test_size=0.3, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T07:58:50.488352Z","iopub.execute_input":"2025-10-24T07:58:50.488758Z","iopub.status.idle":"2025-10-24T07:58:50.515693Z","shell.execute_reply.started":"2025-10-24T07:58:50.488731Z","shell.execute_reply":"2025-10-24T07:58:50.514234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train shape: {train_inputs.shape}\")\nprint(f\"train targets: {train_targets.shape}\")\nprint(f\"Validation shape: {val_inputs.shape}\")\nprint(f\"val_targets: {val_targets.shape}\")\nprint(f\"Test shape: {test_inputs.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:02:45.240036Z","iopub.execute_input":"2025-10-24T08:02:45.240656Z","iopub.status.idle":"2025-10-24T08:02:45.249816Z","shell.execute_reply.started":"2025-10-24T08:02:45.240605Z","shell.execute_reply":"2025-10-24T08:02:45.248572Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Baseline logisticRegression model**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:06:40.801519Z","iopub.execute_input":"2025-10-24T08:06:40.803561Z","iopub.status.idle":"2025-10-24T08:06:40.809904Z","shell.execute_reply.started":"2025-10-24T08:06:40.803504Z","shell.execute_reply":"2025-10-24T08:06:40.808040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LogisticRegression(solver='sag',max_iter=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:09:46.168072Z","iopub.execute_input":"2025-10-24T08:09:46.168528Z","iopub.status.idle":"2025-10-24T08:09:46.174918Z","shell.execute_reply.started":"2025-10-24T08:09:46.168497Z","shell.execute_reply":"2025-10-24T08:09:46.173326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel.fit(train_inputs, train_targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:09:49.552616Z","iopub.execute_input":"2025-10-24T08:09:49.553062Z","iopub.status.idle":"2025-10-24T08:10:18.449011Z","shell.execute_reply.started":"2025-10-24T08:09:49.553030Z","shell.execute_reply":"2025-10-24T08:10:18.447900Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Making prediction using the model**","metadata":{}},{"cell_type":"code","source":"train_preds = model.predict(train_inputs)\n\ntrain_preds[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:12:28.414357Z","iopub.execute_input":"2025-10-24T08:12:28.415551Z","iopub.status.idle":"2025-10-24T08:12:28.427665Z","shell.execute_reply.started":"2025-10-24T08:12:28.415512Z","shell.execute_reply":"2025-10-24T08:12:28.426301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_targets[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:12:41.526290Z","iopub.execute_input":"2025-10-24T08:12:41.526680Z","iopub.status.idle":"2025-10-24T08:12:41.534724Z","shell.execute_reply.started":"2025-10-24T08:12:41.526654Z","shell.execute_reply":"2025-10-24T08:12:41.533646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_probs = model.predict_proba(train_inputs)\ntrain_probs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:13:47.865745Z","iopub.execute_input":"2025-10-24T08:13:47.866123Z","iopub.status.idle":"2025-10-24T08:13:47.879033Z","shell.execute_reply.started":"2025-10-24T08:13:47.866098Z","shell.execute_reply":"2025-10-24T08:13:47.877891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.classes_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:14:02.704516Z","iopub.execute_input":"2025-10-24T08:14:02.704937Z","iopub.status.idle":"2025-10-24T08:14:02.715632Z","shell.execute_reply.started":"2025-10-24T08:14:02.704906Z","shell.execute_reply":"2025-10-24T08:14:02.713853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.Series(train_preds).value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:15:10.500449Z","iopub.execute_input":"2025-10-24T08:15:10.501175Z","iopub.status.idle":"2025-10-24T08:15:10.518799Z","shell.execute_reply.started":"2025-10-24T08:15:10.501122Z","shell.execute_reply":"2025-10-24T08:15:10.516188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.Series(train_targets).value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:15:36.952207Z","iopub.execute_input":"2025-10-24T08:15:36.952593Z","iopub.status.idle":"2025-10-24T08:15:36.969347Z","shell.execute_reply.started":"2025-10-24T08:15:36.952565Z","shell.execute_reply":"2025-10-24T08:15:36.967379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Baseline model Evaluation**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:17:12.104847Z","iopub.execute_input":"2025-10-24T08:17:12.105249Z","iopub.status.idle":"2025-10-24T08:17:12.111220Z","shell.execute_reply.started":"2025-10-24T08:17:12.105221Z","shell.execute_reply":"2025-10-24T08:17:12.109713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"accuracy_score(train_targets, train_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:17:36.343544Z","iopub.execute_input":"2025-10-24T08:17:36.343985Z","iopub.status.idle":"2025-10-24T08:17:36.361994Z","shell.execute_reply.started":"2025-10-24T08:17:36.343952Z","shell.execute_reply":"2025-10-24T08:17:36.360800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get probability predictions instead of class predictions\nval_pred_proba = model.predict_proba(val_inputs)[:, 1]  # Probability of class 1\n\n# Calculate ROC-AUC\nfrom sklearn.metrics import roc_auc_score\nauc_score = roc_auc_score(val_targets, val_pred_proba)\nprint(f\"LogisticRegression Model ROC-AUC score: {auc_score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:18:37.614393Z","iopub.execute_input":"2025-10-24T08:18:37.615331Z","iopub.status.idle":"2025-10-24T08:18:37.637447Z","shell.execute_reply.started":"2025-10-24T08:18:37.615292Z","shell.execute_reply":"2025-10-24T08:18:37.636224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for overfitting - compare train vs validation performance\ntrain_accuracy = accuracy_score(train_targets, train_preds)\nval_accuracy = accuracy_score(val_targets, model.predict(val_inputs))\n\nprint(f\"Training Accuracy: {train_accuracy:.4f}\")\nprint(f\"Validation Accuracy: {val_accuracy:.4f}\")\nprint(f\"Validation ROC-AUC: {auc_score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:19:52.433297Z","iopub.execute_input":"2025-10-24T08:19:52.434035Z","iopub.status.idle":"2025-10-24T08:19:52.451952Z","shell.execute_reply.started":"2025-10-24T08:19:52.434002Z","shell.execute_reply":"2025-10-24T08:19:52.450573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:22:11.642381Z","iopub.execute_input":"2025-10-24T08:22:11.642776Z","iopub.status.idle":"2025-10-24T08:22:11.649216Z","shell.execute_reply.started":"2025-10-24T08:22:11.642752Z","shell.execute_reply":"2025-10-24T08:22:11.647431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f1_score(train_targets, train_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:23:16.822449Z","iopub.execute_input":"2025-10-24T08:23:16.822840Z","iopub.status.idle":"2025-10-24T08:23:16.857519Z","shell.execute_reply.started":"2025-10-24T08:23:16.822813Z","shell.execute_reply":"2025-10-24T08:23:16.855908Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Dummy data to check on the model performance**","metadata":{}},{"cell_type":"code","source":"f1_score(train_targets, np.zeros(len(train_targets)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:26:23.399540Z","iopub.execute_input":"2025-10-24T08:26:23.399957Z","iopub.status.idle":"2025-10-24T08:26:23.445028Z","shell.execute_reply.started":"2025-10-24T08:26:23.399932Z","shell.execute_reply":"2025-10-24T08:26:23.444107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_preds = np.random.choice((0, 1), len(train_targets))\nf1_score(train_targets, random_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:26:55.252336Z","iopub.execute_input":"2025-10-24T08:26:55.252940Z","iopub.status.idle":"2025-10-24T08:26:55.294145Z","shell.execute_reply.started":"2025-10-24T08:26:55.252874Z","shell.execute_reply":"2025-10-24T08:26:55.292883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sincere_df.question_text.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:33:21.106881Z","iopub.execute_input":"2025-10-24T08:33:21.107239Z","iopub.status.idle":"2025-10-24T08:33:21.115154Z","shell.execute_reply.started":"2025-10-24T08:33:21.107217Z","shell.execute_reply":"2025-10-24T08:33:21.113768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sincere_df.target.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:33:39.124992Z","iopub.execute_input":"2025-10-24T08:33:39.125398Z","iopub.status.idle":"2025-10-24T08:33:39.134757Z","shell.execute_reply.started":"2025-10-24T08:33:39.125368Z","shell.execute_reply":"2025-10-24T08:33:39.133204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.predict(vectorizer.transform(sincere_df.question_text.values[:10]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:34:00.014846Z","iopub.execute_input":"2025-10-24T08:34:00.015258Z","iopub.status.idle":"2025-10-24T08:34:00.031446Z","shell.execute_reply.started":"2025-10-24T08:34:00.015229Z","shell.execute_reply":"2025-10-24T08:34:00.030264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"insincere_df.question_text.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:34:42.781355Z","iopub.execute_input":"2025-10-24T08:34:42.782290Z","iopub.status.idle":"2025-10-24T08:34:42.790807Z","shell.execute_reply.started":"2025-10-24T08:34:42.782249Z","shell.execute_reply":"2025-10-24T08:34:42.789206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"insincere_df.target.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:35:01.036942Z","iopub.execute_input":"2025-10-24T08:35:01.037313Z","iopub.status.idle":"2025-10-24T08:35:01.046404Z","shell.execute_reply.started":"2025-10-24T08:35:01.037286Z","shell.execute_reply":"2025-10-24T08:35:01.045498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.predict(vectorizer.transform(insincere_df.question_text.values[:10]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:35:21.186634Z","iopub.execute_input":"2025-10-24T08:35:21.188016Z","iopub.status.idle":"2025-10-24T08:35:21.202637Z","shell.execute_reply.started":"2025-10-24T08:35:21.187967Z","shell.execute_reply":"2025-10-24T08:35:21.201397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **RandomForestClassiffier**","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:42:09.813412Z","iopub.execute_input":"2025-10-24T08:42:09.813967Z","iopub.status.idle":"2025-10-24T08:42:10.157950Z","shell.execute_reply.started":"2025-10-24T08:42:09.813938Z","shell.execute_reply":"2025-10-24T08:42:10.156688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_1 = RandomForestClassifier(n_estimators=500, random_state=42, bootstrap=True,max_features=0.7,max_depth=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:50:15.829223Z","iopub.execute_input":"2025-10-24T08:50:15.829567Z","iopub.status.idle":"2025-10-24T08:50:15.835350Z","shell.execute_reply.started":"2025-10-24T08:50:15.829545Z","shell.execute_reply":"2025-10-24T08:50:15.834121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel_1.fit(train_inputs, train_targets)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:50:20.045295Z","iopub.execute_input":"2025-10-24T08:50:20.045625Z","iopub.status.idle":"2025-10-24T08:51:32.043353Z","shell.execute_reply.started":"2025-10-24T08:50:20.045602Z","shell.execute_reply":"2025-10-24T08:51:32.041831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_preds = model_1.predict(train_inputs)\n\ntrain_preds[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:58:49.700675Z","iopub.execute_input":"2025-10-24T08:58:49.701014Z","iopub.status.idle":"2025-10-24T08:58:52.678485Z","shell.execute_reply.started":"2025-10-24T08:58:49.700992Z","shell.execute_reply":"2025-10-24T08:58:52.677405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:58:56.273569Z","iopub.execute_input":"2025-10-24T08:58:56.274280Z","iopub.status.idle":"2025-10-24T08:58:56.279214Z","shell.execute_reply.started":"2025-10-24T08:58:56.274246Z","shell.execute_reply":"2025-10-24T08:58:56.278119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"accuracy_score(train_targets, train_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:58:57.555339Z","iopub.execute_input":"2025-10-24T08:58:57.555658Z","iopub.status.idle":"2025-10-24T08:58:57.569596Z","shell.execute_reply.started":"2025-10-24T08:58:57.555637Z","shell.execute_reply":"2025-10-24T08:58:57.568206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f1_score(train_targets, train_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T08:59:53.362565Z","iopub.execute_input":"2025-10-24T08:59:53.363438Z","iopub.status.idle":"2025-10-24T08:59:53.392511Z","shell.execute_reply.started":"2025-10-24T08:59:53.363406Z","shell.execute_reply":"2025-10-24T08:59:53.391381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sincere_df.target.values[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:01:00.004840Z","iopub.execute_input":"2025-10-24T09:01:00.005490Z","iopub.status.idle":"2025-10-24T09:01:00.015399Z","shell.execute_reply.started":"2025-10-24T09:01:00.005422Z","shell.execute_reply":"2025-10-24T09:01:00.014027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_1.predict(vectorizer.transform(sincere_df.question_text.values[:10]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:01:17.630281Z","iopub.execute_input":"2025-10-24T09:01:17.630919Z","iopub.status.idle":"2025-10-24T09:01:17.673319Z","shell.execute_reply.started":"2025-10-24T09:01:17.630763Z","shell.execute_reply":"2025-10-24T09:01:17.671363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Make predictions and submit to Kaggle**","metadata":{}},{"cell_type":"code","source":"test_preds = model_1.predict(test_inputs)\ntest_preds[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:10:45.971229Z","iopub.execute_input":"2025-10-24T09:10:45.971532Z","iopub.status.idle":"2025-10-24T09:11:01.019448Z","shell.execute_reply.started":"2025-10-24T09:10:45.971512Z","shell.execute_reply":"2025-10-24T09:11:01.018407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:10:19.815562Z","iopub.execute_input":"2025-10-24T09:10:19.816055Z","iopub.status.idle":"2025-10-24T09:10:19.828053Z","shell.execute_reply.started":"2025-10-24T09:10:19.816027Z","shell.execute_reply":"2025-10-24T09:10:19.827013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['prediction'] = test_preds\n\n# Verify the update\nprint(\"Updated submission preview:\")\nprint(sub_df.head())\nprint(f\"\\nSubmission shape: {sub_df.shape}\")\n\n# Save the updated submission\nsub_df.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-24T09:30:16.340799Z","iopub.execute_input":"2025-10-24T09:30:16.341142Z","iopub.status.idle":"2025-10-24T09:30:16.823271Z","shell.execute_reply.started":"2025-10-24T09:30:16.341120Z","shell.execute_reply":"2025-10-24T09:30:16.822007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}