{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[]},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Dataset Link - https://www.kaggle.com/c/quora-insincere-questions-classification","metadata":{"id":"nrQz8kgKhO4K","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.433932Z","iopub.execute_input":"2025-02-12T02:41:11.434256Z","iopub.status.idle":"2025-02-12T02:41:11.439178Z","shell.execute_reply.started":"2025-02-12T02:41:11.434225Z","shell.execute_reply":"2025-02-12T02:41:11.437965Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Text Classification with Bag of words\n\nOutline:\n- Download and explore the data\n- Apply text preprocessing techniques\n- Implement the bag of words model\n- Train ML models for text classification\n- Make Predictions and submit to kaggle","metadata":{"id":"dMU-Zu8wQXSx"}},{"cell_type":"markdown","source":"## Step 1 : Download and Explore the Dataset\n\nOutline:\n1. Downlaod the dataset from Kaggle\n2. Explore the dataset : EDA\n3. Create small working sample","metadata":{"id":"GlMDNfWSQ-TG"}},{"cell_type":"code","source":"# import os\n# os.environ['KAGGLE_CONFIG_DIR'] = '.'","metadata":{"id":"Dilj78XGUEXx","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.440691Z","iopub.execute_input":"2025-02-12T02:41:11.441351Z","iopub.status.idle":"2025-02-12T02:41:11.468224Z","shell.execute_reply.started":"2025-02-12T02:41:11.441308Z","shell.execute_reply":"2025-02-12T02:41:11.466749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Download Data from Kaggle\n# !chmod 600 ./kaggle.json\n# !kaggle competitions download -c quora-insincere-questions-classification -f train.csv -p data","metadata":{"id":"SsTta5HyhPfy","outputId":"88fb1a82-f768-44bf-d506-6162da45027a","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.470337Z","iopub.execute_input":"2025-02-12T02:41:11.470813Z","iopub.status.idle":"2025-02-12T02:41:11.497982Z","shell.execute_reply.started":"2025-02-12T02:41:11.470768Z","shell.execute_reply":"2025-02-12T02:41:11.496811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !kaggle competitions download -c quora-insincere-questions-classification -f test.csv -p data\n# !kaggle competitions download -c quora-insincere-questions-classification -f sample_submission.csv -p data","metadata":{"id":"Y_VO-ThhhPjA","outputId":"75bd97ec-7fe2-4566-d8c2-f0a248b10b13","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.499712Z","iopub.execute_input":"2025-02-12T02:41:11.500115Z","iopub.status.idle":"2025-02-12T02:41:11.522273Z","shell.execute_reply.started":"2025-02-12T02:41:11.500072Z","shell.execute_reply":"2025-02-12T02:41:11.521164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/quora-insincere-questions-classification')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.523394Z","iopub.execute_input":"2025-02-12T02:41:11.523798Z","iopub.status.idle":"2025-02-12T02:41:11.548009Z","shell.execute_reply.started":"2025-02-12T02:41:11.523750Z","shell.execute_reply":"2025-02-12T02:41:11.546791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/quora-insincere-questions-classification'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.549189Z","iopub.execute_input":"2025-02-12T02:41:11.549609Z","iopub.status.idle":"2025-02-12T02:41:11.570962Z","shell.execute_reply.started":"2025-02-12T02:41:11.549568Z","shell.execute_reply":"2025-02-12T02:41:11.569875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Explore the Data Using Pandas\ntrain_fname = data_dir + '/train.csv'\ntest_fname = data_dir + '/test.csv'\nsample_fname = data_dir + '/sample_submission.csv'","metadata":{"id":"mSqnYf1KWyKH","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.572350Z","iopub.execute_input":"2025-02-12T02:41:11.572752Z","iopub.status.idle":"2025-02-12T02:41:11.600707Z","shell.execute_reply.started":"2025-02-12T02:41:11.572706Z","shell.execute_reply":"2025-02-12T02:41:11.599024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"id":"SOrtwU37WyGf","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:11.604105Z","iopub.execute_input":"2025-02-12T02:41:11.604427Z","iopub.status.idle":"2025-02-12T02:41:12.078760Z","shell.execute_reply.started":"2025-02-12T02:41:11.604399Z","shell.execute_reply":"2025-02-12T02:41:12.077030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_df = pd.read_csv(train_fname)\nraw_df.head()","metadata":{"id":"5_rXaJ4qXJJ4","outputId":"5c8ddb6c-2e12-4d73-dcec-f76d34703186","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:12.081644Z","iopub.execute_input":"2025-02-12T02:41:12.082338Z","iopub.status.idle":"2025-02-12T02:41:17.234021Z","shell.execute_reply.started":"2025-02-12T02:41:12.082288Z","shell.execute_reply":"2025-02-12T02:41:17.232754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check the dimension of the data\nprint(f'The shape of the data is {raw_df.shape}')","metadata":{"id":"CiLWy0pRWyCl","outputId":"08b5f919-5bc4-463f-b0fe-a3327cf7cb01","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.235180Z","iopub.execute_input":"2025-02-12T02:41:17.235509Z","iopub.status.idle":"2025-02-12T02:41:17.241405Z","shell.execute_reply.started":"2025-02-12T02:41:17.235479Z","shell.execute_reply":"2025-02-12T02:41:17.239969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check the distribution of target variable\nraw_df['target'].value_counts()","metadata":{"id":"xWv75OgghP18","outputId":"b852000e-59c3-4d5b-96b2-1ce8623fdd3e","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.242868Z","iopub.execute_input":"2025-02-12T02:41:17.243263Z","iopub.status.idle":"2025-02-12T02:41:17.280205Z","shell.execute_reply.started":"2025-02-12T02:41:17.243223Z","shell.execute_reply":"2025-02-12T02:41:17.279201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_df['target'].value_counts(normalize=True).plot(kind='bar')\nplt.show()","metadata":{"id":"mjPc0PFTq_q4","outputId":"32d94163-f19b-4229-8c5f-0f1312f7a8a3","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.281175Z","iopub.execute_input":"2025-02-12T02:41:17.281524Z","iopub.status.idle":"2025-02-12T02:41:17.606091Z","shell.execute_reply.started":"2025-02-12T02:41:17.281475Z","shell.execute_reply":"2025-02-12T02:41:17.604799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check few sincere questions\nraw_df[raw_df['target'] == 0]['question_text'].values[:10]","metadata":{"id":"hpJSgVI4c83l","outputId":"35bde0c6-564c-495d-8458-d45d2908c6dd","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.607085Z","iopub.execute_input":"2025-02-12T02:41:17.607356Z","iopub.status.idle":"2025-02-12T02:41:17.699990Z","shell.execute_reply.started":"2025-02-12T02:41:17.607333Z","shell.execute_reply":"2025-02-12T02:41:17.698827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check few insincere questions\nraw_df[raw_df['target'] == 1]['question_text'].values[:10]","metadata":{"id":"KG7zWb31dLfE","outputId":"759f18a3-3f28-4353-cfd7-e6d07dfc30b4","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.701125Z","iopub.execute_input":"2025-02-12T02:41:17.701453Z","iopub.status.idle":"2025-02-12T02:41:17.724574Z","shell.execute_reply.started":"2025-02-12T02:41:17.701424Z","shell.execute_reply":"2025-02-12T02:41:17.723282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv(test_fname)\ntest_df.head()","metadata":{"id":"VQXqymFKrXtg","outputId":"731d3bf0-af02-4f9b-beaf-fa8294494354","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:17.725862Z","iopub.execute_input":"2025-02-12T02:41:17.726319Z","iopub.status.idle":"2025-02-12T02:41:19.119768Z","shell.execute_reply.started":"2025-02-12T02:41:17.726280Z","shell.execute_reply":"2025-02-12T02:41:19.118658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Shape of the test data is',test_df.shape)","metadata":{"id":"n7XGS6hXr206","outputId":"1e2d8f5e-25ce-405c-c7c1-edd56e8f3c64","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.121036Z","iopub.execute_input":"2025-02-12T02:41:19.121432Z","iopub.status.idle":"2025-02-12T02:41:19.128397Z","shell.execute_reply.started":"2025-02-12T02:41:19.121392Z","shell.execute_reply":"2025-02-12T02:41:19.127156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df = pd.read_csv(sample_fname)\nsub_df.head()","metadata":{"id":"O3P_NiHXrg27","outputId":"c8ee30b3-0166-42c3-9561-7de4b37d1c07","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.130054Z","iopub.execute_input":"2025-02-12T02:41:19.130380Z","iopub.status.idle":"2025-02-12T02:41:19.522898Z","shell.execute_reply.started":"2025-02-12T02:41:19.130352Z","shell.execute_reply":"2025-02-12T02:41:19.521957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Shape of the submission data is',sub_df.shape)","metadata":{"id":"-eOWn3AzsFpn","outputId":"55c502c2-e1bf-4f23-8a7c-f26e537e52a2","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.523855Z","iopub.execute_input":"2025-02-12T02:41:19.524116Z","iopub.status.idle":"2025-02-12T02:41:19.530180Z","shell.execute_reply.started":"2025-02-12T02:41:19.524093Z","shell.execute_reply":"2025-02-12T02:41:19.529011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['prediction'].value_counts()","metadata":{"id":"QnfoVBrrsnLh","outputId":"68b1d2a7-0269-4f2e-8718-96ab95a03e32","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.531326Z","iopub.execute_input":"2025-02-12T02:41:19.531669Z","iopub.status.idle":"2025-02-12T02:41:19.555069Z","shell.execute_reply.started":"2025-02-12T02:41:19.531606Z","shell.execute_reply":"2025-02-12T02:41:19.553918Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 2 : Text Preprocessing Techniques","metadata":{"id":"8DYhOOQoRFim"}},{"cell_type":"code","source":"# Create sample dataset\nsample_size = 100000\nsample_df = raw_df.sample(sample_size,random_state=42)","metadata":{"id":"KeOKK3MuhP8v","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.556093Z","iopub.execute_input":"2025-02-12T02:41:19.556391Z","iopub.status.idle":"2025-02-12T02:41:19.692700Z","shell.execute_reply.started":"2025-02-12T02:41:19.556364Z","shell.execute_reply":"2025-02-12T02:41:19.691587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Text Preprocessing Techniques\n\nOutline:\n1. Understand the bag of words model\n2. Tokenization\n3. Stopwords\n4. Stemming/Lemmatization\n","metadata":{"id":"Mqf8pSCCwF_C"}},{"cell_type":"markdown","source":"### Bag of Word Intuition\n\n1. Create list of all the words across all the text documents\n2. You convert each document into vector containing counts of each word\n\nLimitations:\n1. There may be too many words in the dataset\n2. Some words may occur too frequency\n3. Some words may occur very rarely\n4. A single word may have may forms (go, gone, going)","metadata":{"id":"cs033me1xC9Y"}},{"cell_type":"markdown","source":"### Tokenization\nSplitting text into individual words","metadata":{"id":"8pdDYTeNygG1"}},{"cell_type":"code","source":"q1 = sample_df['question_text'].values[0]\nprint(q1)","metadata":{"id":"ZXtXHy5CyaCy","outputId":"a270060f-1f2b-457f-e746-2bdc6181641d","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.693784Z","iopub.execute_input":"2025-02-12T02:41:19.694164Z","iopub.status.idle":"2025-02-12T02:41:19.700481Z","shell.execute_reply.started":"2025-02-12T02:41:19.694128Z","shell.execute_reply":"2025-02-12T02:41:19.699281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import nltk\n# nltk.download('punkt_tab')","metadata":{"id":"E5JPR3JmzMbF","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.701772Z","iopub.execute_input":"2025-02-12T02:41:19.702169Z","iopub.status.idle":"2025-02-12T02:41:19.719938Z","shell.execute_reply.started":"2025-02-12T02:41:19.702130Z","shell.execute_reply":"2025-02-12T02:41:19.718903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize","metadata":{"id":"-gctX05dyZ-x","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:19.726433Z","iopub.execute_input":"2025-02-12T02:41:19.726807Z","iopub.status.idle":"2025-02-12T02:41:21.143770Z","shell.execute_reply.started":"2025-02-12T02:41:19.726765Z","shell.execute_reply":"2025-02-12T02:41:21.142501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"word_tokenize(q1)","metadata":{"id":"nZ-oWBILvjZv","outputId":"fa127efb-5371-45f7-ba9d-898ad4580520","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:21.147270Z","iopub.execute_input":"2025-02-12T02:41:21.147789Z","iopub.status.idle":"2025-02-12T02:41:21.166969Z","shell.execute_reply.started":"2025-02-12T02:41:21.147755Z","shell.execute_reply":"2025-02-12T02:41:21.165891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_tok = word_tokenize(q1)","metadata":{"id":"9v9Jjx2NzlZz","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:41:21.168022Z","iopub.execute_input":"2025-02-12T02:41:21.168367Z","iopub.status.idle":"2025-02-12T02:41:21.173437Z","shell.execute_reply.started":"2025-02-12T02:41:21.168326Z","shell.execute_reply":"2025-02-12T02:41:21.172183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Stop Word Removal\nRemoving commonly occuring words","metadata":{"id":"5ALDr4G30EH7"}},{"cell_type":"code","source":"import nltk\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')","metadata":{"id":"3NsH7R-lzlWT","outputId":"8c85c954-abc7-4c01-8bd1-d2945cbebad5","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:03.838278Z","iopub.execute_input":"2025-02-12T02:42:03.838867Z","iopub.status.idle":"2025-02-12T02:42:03.920678Z","shell.execute_reply.started":"2025-02-12T02:42:03.838798Z","shell.execute_reply":"2025-02-12T02:42:03.918412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"english_stopwords = stopwords.words('english')","metadata":{"id":"QxbBJIV_zlS_","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:03.922744Z","iopub.execute_input":"2025-02-12T02:42:03.923083Z","iopub.status.idle":"2025-02-12T02:42:03.931514Z","shell.execute_reply.started":"2025-02-12T02:42:03.923055Z","shell.execute_reply":"2025-02-12T02:42:03.929978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Total number of stop words are {len(english_stopwords)}')","metadata":{"id":"VO1T0w4O0bfv","outputId":"a8018058-e938-4097-88f8-420ca07c0b0b","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:03.934369Z","iopub.execute_input":"2025-02-12T02:42:03.934847Z","iopub.status.idle":"2025-02-12T02:42:03.954654Z","shell.execute_reply.started":"2025-02-12T02:42:03.934805Z","shell.execute_reply":"2025-02-12T02:42:03.953417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def remove_stopwords(tokens):\n  return [word for word in tokens if word.lower() not in english_stopwords]","metadata":{"id":"-OLu3L_r7KC-","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:03.956358Z","iopub.execute_input":"2025-02-12T02:42:03.956873Z","iopub.status.idle":"2025-02-12T02:42:03.979412Z","shell.execute_reply.started":"2025-02-12T02:42:03.956828Z","shell.execute_reply":"2025-02-12T02:42:03.977201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stp = remove_stopwords(q1_tok)\nq1_stp","metadata":{"id":"pYfhO9H37J_i","outputId":"f488bb39-577a-49ea-c0f5-89132827cb20","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:03.981004Z","iopub.execute_input":"2025-02-12T02:42:03.981335Z","iopub.status.idle":"2025-02-12T02:42:04.013271Z","shell.execute_reply.started":"2025-02-12T02:42:03.981307Z","shell.execute_reply":"2025-02-12T02:42:04.011791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_tok","metadata":{"id":"RnE5CHL-7J8K","outputId":"5125c0fe-2d02-4449-f5cd-6d5cb5fd4433","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.014906Z","iopub.execute_input":"2025-02-12T02:42:04.015437Z","iopub.status.idle":"2025-02-12T02:42:04.047184Z","shell.execute_reply.started":"2025-02-12T02:42:04.015392Z","shell.execute_reply":"2025-02-12T02:42:04.045751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Stemming\nStemming is process to is a technique that reduces words to their root or base form, essentially removing suffixes and prefixes to normalize different variations of a word, allowing for better analysis and comparison of text data by treating different forms of the same word as essentially the same entity; for example, \"running,\" \"runs,\" and \"runner\" would all be stemmed to \"run\"","metadata":{"id":"c8lS5kh58XEF"}},{"cell_type":"code","source":"from nltk.stem.snowball import SnowballStemmer","metadata":{"id":"Kt0c81Jq7J4h","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.049028Z","iopub.execute_input":"2025-02-12T02:42:04.050052Z","iopub.status.idle":"2025-02-12T02:42:04.084600Z","shell.execute_reply.started":"2025-02-12T02:42:04.049811Z","shell.execute_reply":"2025-02-12T02:42:04.082425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stemmer = SnowballStemmer('english')","metadata":{"id":"Vvks3qBK7Jp_","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.085857Z","iopub.execute_input":"2025-02-12T02:42:04.086284Z","iopub.status.idle":"2025-02-12T02:42:04.122987Z","shell.execute_reply.started":"2025-02-12T02:42:04.086242Z","shell.execute_reply":"2025-02-12T02:42:04.120985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stemmer.stem('running')","metadata":{"id":"YnyB2CtB9H8P","outputId":"63118702-5c91-4786-e385-da8f7e176081","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.128017Z","iopub.execute_input":"2025-02-12T02:42:04.129236Z","iopub.status.idle":"2025-02-12T02:42:04.162427Z","shell.execute_reply.started":"2025-02-12T02:42:04.129190Z","shell.execute_reply":"2025-02-12T02:42:04.160254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stemmer.stem('runs')","metadata":{"id":"gW6nIXGY9UWc","outputId":"e2174850-51e7-4fb1-b3a3-6fc11af874e8","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.164908Z","iopub.execute_input":"2025-02-12T02:42:04.165256Z","iopub.status.idle":"2025-02-12T02:42:04.193490Z","shell.execute_reply.started":"2025-02-12T02:42:04.165229Z","shell.execute_reply":"2025-02-12T02:42:04.190949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stm = [stemmer.stem(word) for word in q1_stp]\nq1_stm","metadata":{"id":"Hv_KosYj9H5K","outputId":"132b94d8-2b2e-4455-eddd-4ebb34130825","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.196242Z","iopub.execute_input":"2025-02-12T02:42:04.196736Z","iopub.status.idle":"2025-02-12T02:42:04.231283Z","shell.execute_reply.started":"2025-02-12T02:42:04.196688Z","shell.execute_reply":"2025-02-12T02:42:04.229672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"q1_stp","metadata":{"id":"zf3MjzPF9H2C","outputId":"cf280c48-3222-4edf-ecd3-cd1d69c69d2d","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.232207Z","iopub.execute_input":"2025-02-12T02:42:04.232664Z","iopub.status.idle":"2025-02-12T02:42:04.265830Z","shell.execute_reply.started":"2025-02-12T02:42:04.232596Z","shell.execute_reply":"2025-02-12T02:42:04.264186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lemmatization\nLemmatization is a text processing technique in NLP that reduces words to their root form. It's used to identify similarities between words and to improve text processing.","metadata":{"id":"b9vZOm9VA3rY"}},{"cell_type":"markdown","source":"## Step 3: Implement Bag of Words\n\nOutline:\n1. Create a vocabulary using Count Vectorizer\n2. Transform text to vectors using Count Vectorizer\n3. Configure text preprocessing in Count Vectorizer","metadata":{"id":"X17JbEu0RMK1"}},{"cell_type":"code","source":"small_df = sample_df[:5]","metadata":{"id":"PrcMl8WrhQGx","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.267104Z","iopub.execute_input":"2025-02-12T02:42:04.267594Z","iopub.status.idle":"2025-02-12T02:42:04.291449Z","shell.execute_reply.started":"2025-02-12T02:42:04.267555Z","shell.execute_reply":"2025-02-12T02:42:04.290194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"small_df['question_text'].values","metadata":{"id":"YQTOkYlLRUwV","outputId":"fd43445b-5183-4ab4-f974-b25fdfa63666","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.293280Z","iopub.execute_input":"2025-02-12T02:42:04.294018Z","iopub.status.idle":"2025-02-12T02:42:04.322047Z","shell.execute_reply.started":"2025-02-12T02:42:04.293967Z","shell.execute_reply":"2025-02-12T02:42:04.320710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Create Vocabulary","metadata":{"id":"0XIyKZbCFESu"}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","metadata":{"id":"N2PhXZOzFCJI","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.323894Z","iopub.execute_input":"2025-02-12T02:42:04.324279Z","iopub.status.idle":"2025-02-12T02:42:04.344931Z","shell.execute_reply.started":"2025-02-12T02:42:04.324248Z","shell.execute_reply":"2025-02-12T02:42:04.343782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"small_vec = CountVectorizer()\nsmall_vec.fit(small_df['question_text'])","metadata":{"id":"rE7IqmkYRUtn","outputId":"724b10e5-4319-4eeb-f29e-88e92561fef0","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.346094Z","iopub.execute_input":"2025-02-12T02:42:04.346397Z","iopub.status.idle":"2025-02-12T02:42:04.384911Z","shell.execute_reply.started":"2025-02-12T02:42:04.346370Z","shell.execute_reply":"2025-02-12T02:42:04.383270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"small_vec.vocabulary_","metadata":{"id":"hUc3S6dLRUqi","outputId":"07b2d605-b9c9-4190-e556-40483797db77","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.385951Z","iopub.execute_input":"2025-02-12T02:42:04.386350Z","iopub.status.idle":"2025-02-12T02:42:04.406380Z","shell.execute_reply.started":"2025-02-12T02:42:04.386306Z","shell.execute_reply":"2025-02-12T02:42:04.404815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check feature names\nsmall_vec.get_feature_names_out()","metadata":{"id":"U_5sI6sPRUnR","outputId":"ce9bf669-281e-476f-ef65-0c35d2781086","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.407323Z","iopub.execute_input":"2025-02-12T02:42:04.407604Z","iopub.status.idle":"2025-02-12T02:42:04.435374Z","shell.execute_reply.started":"2025-02-12T02:42:04.407580Z","shell.execute_reply":"2025-02-12T02:42:04.434114Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Transform Documents into Vectors","metadata":{"id":"KPkPlG6se9l1"}},{"cell_type":"code","source":"## Transform documents into Vectors\nvectors = small_vec.transform(small_df['question_text'])\nvectors.toarray()","metadata":{"id":"I_BsnpivRUkQ","outputId":"86f22b6e-0650-47f3-ebfb-5f4f3c0cf755","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.436593Z","iopub.execute_input":"2025-02-12T02:42:04.437307Z","iopub.status.idle":"2025-02-12T02:42:04.471407Z","shell.execute_reply.started":"2025-02-12T02:42:04.437271Z","shell.execute_reply":"2025-02-12T02:42:04.470025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectors.shape","metadata":{"id":"ncEL2YYMUOl6","outputId":"f8586213-e4de-46f3-b820-3ee9196f71a2","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.473997Z","iopub.execute_input":"2025-02-12T02:42:04.474404Z","iopub.status.idle":"2025-02-12T02:42:04.512527Z","shell.execute_reply.started":"2025-02-12T02:42:04.474362Z","shell.execute_reply":"2025-02-12T02:42:04.510289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nfrom nltk.stem import PorterStemmer\ndef preprocess_text(text):\n    \"\"\"\n    Preprocess the input text by tokenizing, converting to lowercase, removing punctuation,\n    filtering out stopwords, and applying stemming.\n\n    Parameters:\n        text (str): The text to be processed.\n\n    Returns:\n        List[str]: A list of processed tokens.\n    \"\"\"\n    # Step 1: Tokenize the text into individual words\n    tokens = word_tokenize(text)\n\n    # Step 2: Convert each token to lowercase\n    tokens = [token.lower() for token in tokens]\n\n    # Step 3: Remove punctuation and special characters from each token.\n    # The regex pattern '[^a-zA-Z0-9]' matches any character that is not alphanumeric.\n    tokens = [re.sub(r'[^a-zA-Z0-9]', '', token) for token in tokens]\n\n    # Remove any tokens that may have become empty after removing punctuation\n    tokens = [token for token in tokens if token]\n\n    # Step 4: Remove stopwords\n    stop_words = set(stopwords.words('english'))\n    tokens = [token for token in tokens if token not in stop_words]\n\n    # Step 5: Apply stemming to each token using Porter Stemmer\n    stemmer = PorterStemmer()\n    tokens = [stemmer.stem(token) for token in tokens]\n\n    return tokens","metadata":{"id":"adcfc0MVUTT7","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.514336Z","iopub.execute_input":"2025-02-12T02:42:04.514784Z","iopub.status.idle":"2025-02-12T02:42:04.545630Z","shell.execute_reply.started":"2025-02-12T02:42:04.514749Z","shell.execute_reply":"2025-02-12T02:42:04.543916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage:\nsample_text = \"Here's an example sentence, showcasing the functionality: preprocessing text!\"\nprocessed_tokens = preprocess_text(sample_text)\nprint(processed_tokens)","metadata":{"id":"mbgsGhecUTRF","outputId":"78553d55-7234-4009-d5c3-561fa00cfb3b","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.547368Z","iopub.execute_input":"2025-02-12T02:42:04.547813Z","iopub.status.idle":"2025-02-12T02:42:04.586257Z","shell.execute_reply.started":"2025-02-12T02:42:04.547772Z","shell.execute_reply":"2025-02-12T02:42:04.584743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Configure Count Vectorizer Parameters","metadata":{"id":"3y56J--8ez1V"}},{"cell_type":"code","source":"# Configure Count Vectorize Parameters\nvectorizer = CountVectorizer(tokenizer=preprocess_text, max_features=1000)","metadata":{"id":"Rp3ehFBuUTNo","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.587566Z","iopub.execute_input":"2025-02-12T02:42:04.587944Z","iopub.status.idle":"2025-02-12T02:42:04.614800Z","shell.execute_reply.started":"2025-02-12T02:42:04.587906Z","shell.execute_reply":"2025-02-12T02:42:04.613522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nvectorizer.fit(sample_df['question_text'])","metadata":{"id":"nuncMhAYUTKq","outputId":"df3fcaee-4b75-4f5c-d380-dc9734f26c19","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:04.616107Z","iopub.execute_input":"2025-02-12T02:42:04.616427Z","iopub.status.idle":"2025-02-12T02:42:57.034659Z","shell.execute_reply.started":"2025-02-12T02:42:04.616399Z","shell.execute_reply":"2025-02-12T02:42:57.033730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check feature length\nlen(vectorizer.get_feature_names_out())","metadata":{"id":"2ayN_jApUTHi","outputId":"4096aee6-833f-4aa8-8f59-54d690605b2f","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:57.035556Z","iopub.execute_input":"2025-02-12T02:42:57.035936Z","iopub.status.idle":"2025-02-12T02:42:57.042510Z","shell.execute_reply.started":"2025-02-12T02:42:57.035905Z","shell.execute_reply":"2025-02-12T02:42:57.041658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"id":"ko_UDV8EUTEa","outputId":"bed187e6-ef8d-4991-e857-200d4c604ff1","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:57.043565Z","iopub.execute_input":"2025-02-12T02:42:57.043955Z","iopub.status.idle":"2025-02-12T02:42:57.062971Z","shell.execute_reply.started":"2025-02-12T02:42:57.043925Z","shell.execute_reply":"2025-02-12T02:42:57.061954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ninputs = vectorizer.transform(sample_df['question_text'])","metadata":{"id":"jvIfYqvPUTBp","outputId":"6b8fcae0-2924-455d-8d15-4d7a69ed9540","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:42:57.067842Z","iopub.execute_input":"2025-02-12T02:42:57.068183Z","iopub.status.idle":"2025-02-12T02:43:50.396208Z","shell.execute_reply.started":"2025-02-12T02:42:57.068154Z","shell.execute_reply":"2025-02-12T02:43:50.394885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(inputs.shape)","metadata":{"id":"iAdYenfubUbJ","outputId":"6906d271-0d55-45dd-9784-8055581f2eb3","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:43:50.397710Z","iopub.execute_input":"2025-02-12T02:43:50.398054Z","iopub.status.idle":"2025-02-12T02:43:50.403139Z","shell.execute_reply.started":"2025-02-12T02:43:50.398027Z","shell.execute_reply":"2025-02-12T02:43:50.401905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df['question_text'].values[0]","metadata":{"id":"Jw5DIj5HUS-r","outputId":"9ad4f5f4-c1af-4129-c3a0-ee7165cbf916","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:43:50.404283Z","iopub.execute_input":"2025-02-12T02:43:50.404718Z","iopub.status.idle":"2025-02-12T02:43:50.425583Z","shell.execute_reply.started":"2025-02-12T02:43:50.404678Z","shell.execute_reply":"2025-02-12T02:43:50.424504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# display the first row\ninputs[0].toarray()","metadata":{"id":"s5fGxVv0US7v","outputId":"44bee417-5cc6-49a1-9511-c981d090a952","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:43:50.426840Z","iopub.execute_input":"2025-02-12T02:43:50.427167Z","iopub.status.idle":"2025-02-12T02:43:50.451563Z","shell.execute_reply.started":"2025-02-12T02:43:50.427140Z","shell.execute_reply":"2025-02-12T02:43:50.450526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntest_inputs = vectorizer.transform(test_df['question_text'])","metadata":{"id":"qND7QNtSdi0f","outputId":"818f5357-0a93-4556-f071-79b00dcc7eb3","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:43:50.452512Z","iopub.execute_input":"2025-02-12T02:43:50.452856Z","iopub.status.idle":"2025-02-12T02:47:09.758580Z","shell.execute_reply.started":"2025-02-12T02:43:50.452818Z","shell.execute_reply":"2025-02-12T02:47:09.757433Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4: ML Model for Text Classification\n\nOutline:\n- Create a training & validation set\n- Train Logistic Regression Model\n- Make Prediction on traning, validation & test data","metadata":{"id":"UlVM254vuu0z"}},{"cell_type":"markdown","source":"### Split Into Training & Validation Set","metadata":{"id":"UrrE0rQ7vdne"}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(inputs, sample_df['target'], test_size=0.2, random_state=42, stratify=sample_df['target'])","metadata":{"id":"4OFIQbaJdi9d","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:09.759279Z","iopub.execute_input":"2025-02-12T02:47:09.759559Z","iopub.status.idle":"2025-02-12T02:47:09.814555Z","shell.execute_reply.started":"2025-02-12T02:47:09.759535Z","shell.execute_reply":"2025-02-12T02:47:09.813333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'X_train shape is {X_train.shape}')\nprint(f'X_val shape is {X_val.shape}')","metadata":{"id":"seNkq4icdjAw","outputId":"5bd38dfb-f846-4715-dc2f-99436c80c3e7","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:09.815601Z","iopub.execute_input":"2025-02-12T02:47:09.816046Z","iopub.status.idle":"2025-02-12T02:47:09.822182Z","shell.execute_reply.started":"2025-02-12T02:47:09.816009Z","shell.execute_reply":"2025-02-12T02:47:09.820763Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train Logistic Regression Model","metadata":{"id":"SKOWwsQIw5ZL"}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"id":"Yp6WtyIDdjEK","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:09.823354Z","iopub.execute_input":"2025-02-12T02:47:09.823918Z","iopub.status.idle":"2025-02-12T02:47:09.842029Z","shell.execute_reply.started":"2025-02-12T02:47:09.823871Z","shell.execute_reply":"2025-02-12T02:47:09.840509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fit the model\nmodel = LogisticRegression()\nmodel.fit(X_train, y_train)","metadata":{"id":"womiu-7adjHP","outputId":"f2adc191-1838-454f-ccb4-7ab4b0ae42cd","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:09.843297Z","iopub.execute_input":"2025-02-12T02:47:09.843726Z","iopub.status.idle":"2025-02-12T02:47:10.461171Z","shell.execute_reply.started":"2025-02-12T02:47:09.843686Z","shell.execute_reply":"2025-02-12T02:47:10.459919Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluate Model","metadata":{"id":"OAvrTFDV0Kzb"}},{"cell_type":"code","source":"# evaluate on validation data\ny_pred = model.predict(X_val)","metadata":{"id":"I-uQu2s2xJFq","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.462220Z","iopub.execute_input":"2025-02-12T02:47:10.462588Z","iopub.status.idle":"2025-02-12T02:47:10.467925Z","shell.execute_reply.started":"2025-02-12T02:47:10.462558Z","shell.execute_reply":"2025-02-12T02:47:10.466669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Performance Metrics\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\nprint(f'Accuracy: {round(accuracy_score(y_val, y_pred),2)}')\nprint(f'Precision: {round(precision_score(y_val, y_pred),2)}')\nprint(f'Recall: {round(recall_score(y_val, y_pred),2)}')\nprint(f'F1 Score: {round(f1_score(y_val, y_pred),2)}')","metadata":{"id":"fAFssZwgxskb","outputId":"a2bcea3a-9746-4a53-ceb4-9a48706998c5","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.468973Z","iopub.execute_input":"2025-02-12T02:47:10.469297Z","iopub.status.idle":"2025-02-12T02:47:10.518123Z","shell.execute_reply.started":"2025-02-12T02:47:10.469268Z","shell.execute_reply":"2025-02-12T02:47:10.517047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# confusion matrix\nfrom sklearn.metrics import confusion_matrix\nconfusion_matrix(y_val, y_pred)","metadata":{"id":"P0eGwqIZx56Q","outputId":"75a46b4a-fc03-472c-a16a-bb87a50f6648","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.519073Z","iopub.execute_input":"2025-02-12T02:47:10.519373Z","iopub.status.idle":"2025-02-12T02:47:10.532154Z","shell.execute_reply.started":"2025-02-12T02:47:10.519337Z","shell.execute_reply":"2025-02-12T02:47:10.530932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot ROC Curve\nfrom sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\nfpr, tpr, thresholds = roc_curve(y_val, y_pred)\nroc_auc = auc(fpr, tpr)\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=1, label='ROC curve (area = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"id":"NYrVhYfazmtp","outputId":"b32dc156-ac97-49e6-81c7-545dd0b1e199","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.532945Z","iopub.execute_input":"2025-02-12T02:47:10.533256Z","iopub.status.idle":"2025-02-12T02:47:10.758292Z","shell.execute_reply.started":"2025-02-12T02:47:10.533227Z","shell.execute_reply":"2025-02-12T02:47:10.756907Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 5: Make Predictions and Submit to Kaggle","metadata":{"id":"BERQZvHFRVfA"}},{"cell_type":"code","source":"test_inputs.shape","metadata":{"id":"oF-0aEaZRf6O","outputId":"e35bf456-1ccd-408f-a3db-649ecda85e48","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.759348Z","iopub.execute_input":"2025-02-12T02:47:10.759738Z","iopub.status.idle":"2025-02-12T02:47:10.767040Z","shell.execute_reply.started":"2025-02-12T02:47:10.759707Z","shell.execute_reply":"2025-02-12T02:47:10.765694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds = model.predict(test_inputs)","metadata":{"id":"x8ARqbirRf3S","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.767784Z","iopub.execute_input":"2025-02-12T02:47:10.768084Z","iopub.status.idle":"2025-02-12T02:47:10.811689Z","shell.execute_reply.started":"2025-02-12T02:47:10.768058Z","shell.execute_reply":"2025-02-12T02:47:10.810590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['prediction'] = test_preds","metadata":{"id":"8jv2CmFERf0f","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.812785Z","iopub.execute_input":"2025-02-12T02:47:10.813172Z","iopub.status.idle":"2025-02-12T02:47:10.819726Z","shell.execute_reply.started":"2025-02-12T02:47:10.813133Z","shell.execute_reply":"2025-02-12T02:47:10.818151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['prediction'].value_counts()","metadata":{"id":"n63rwvw7Rfxd","outputId":"b4ba3cfb-db96-438f-a2e0-b2355c8d9d21","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.820835Z","iopub.execute_input":"2025-02-12T02:47:10.821236Z","iopub.status.idle":"2025-02-12T02:47:10.850177Z","shell.execute_reply.started":"2025-02-12T02:47:10.821181Z","shell.execute_reply":"2025-02-12T02:47:10.848985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"id":"YSKUn9L17G9O","trusted":true,"execution":{"iopub.status.busy":"2025-02-12T02:47:10.851488Z","iopub.execute_input":"2025-02-12T02:47:10.851946Z","iopub.status.idle":"2025-02-12T02:47:11.257608Z","shell.execute_reply.started":"2025-02-12T02:47:10.851906Z","shell.execute_reply":"2025-02-12T02:47:11.256446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head submission.csv","metadata":{"id":"uTfanWbdRfuT","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"id":"u448zD7ORfrZ","trusted":true},"outputs":[],"execution_count":null}]}