{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Text Classification with bag of words\n    Outline:\n       -Download and explore the data\n       -Apply text preprocessing techniques\n       -Implement the bag of words model\n       -Train ML models for text classification\n       -Make predictions and submit to Kaggle","metadata":{}},{"cell_type":"markdown","source":"# Download and Explore the Data\n     Outline:\n        -Download the dataset from kaggle to jupyter notebook\n        -Explore the data using Pandas\n        -Create a small working sample","metadata":{}},{"cell_type":"code","source":"data_dir='/kaggle/input/quora-insincere-questions-classification'\ntrain_fname=data_dir + '/kaggle/input/quora-insincere-questions-classification/train.csv'\ntest_fname=data_dir + '/kaggle/input/quora-insincere-questions-classification/test.csv'\nsample_fname=data_dir + '/kaggle/input/quora-insincere-questions-classification/sample_submission.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:45.063647Z","iopub.execute_input":"2023-03-27T12:31:45.064037Z","iopub.status.idle":"2023-03-27T12:31:45.086869Z","shell.execute_reply.started":"2023-03-27T12:31:45.064002Z","shell.execute_reply":"2023-03-27T12:31:45.08534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nraw_df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\nraw_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:45.089893Z","iopub.execute_input":"2023-03-27T12:31:45.090415Z","iopub.status.idle":"2023-03-27T12:31:48.191467Z","shell.execute_reply.started":"2023-03-27T12:31:45.090359Z","shell.execute_reply":"2023-03-27T12:31:48.190631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the sincere texts from the given dataset\nsincere_df=raw_df[raw_df.target==0]\nsincere_df.question_text.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.192534Z","iopub.execute_input":"2023-03-27T12:31:48.193648Z","iopub.status.idle":"2023-03-27T12:31:48.280699Z","shell.execute_reply.started":"2023-03-27T12:31:48.193609Z","shell.execute_reply":"2023-03-27T12:31:48.279465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the unsincere texts from the given dataset\nunsincere_df=raw_df[raw_df.target==1]\nunsincere_df.question_text.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.282187Z","iopub.execute_input":"2023-03-27T12:31:48.282577Z","iopub.status.idle":"2023-03-27T12:31:48.309431Z","shell.execute_reply.started":"2023-03-27T12:31:48.282543Z","shell.execute_reply":"2023-03-27T12:31:48.308131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_df.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.31217Z","iopub.execute_input":"2023-03-27T12:31:48.312526Z","iopub.status.idle":"2023-03-27T12:31:48.335339Z","shell.execute_reply.started":"2023-03-27T12:31:48.312495Z","shell.execute_reply":"2023-03-27T12:31:48.334198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_df.target.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.336886Z","iopub.execute_input":"2023-03-27T12:31:48.337267Z","iopub.status.idle":"2023-03-27T12:31:48.361623Z","shell.execute_reply.started":"2023-03-27T12:31:48.337234Z","shell.execute_reply":"2023-03-27T12:31:48.36024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_df.target.value_counts(normalize=True).plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.36274Z","iopub.execute_input":"2023-03-27T12:31:48.363225Z","iopub.status.idle":"2023-03-27T12:31:48.659379Z","shell.execute_reply.started":"2023-03-27T12:31:48.363181Z","shell.execute_reply":"2023-03-27T12:31:48.658269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the length of first 5 questions\nraw_df['Length'] = raw_df['question_text'].apply(len)\nraw_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:48.660637Z","iopub.execute_input":"2023-03-27T12:31:48.66154Z","iopub.status.idle":"2023-03-27T12:31:49.081473Z","shell.execute_reply.started":"2023-03-27T12:31:48.661498Z","shell.execute_reply":"2023-03-27T12:31:49.080096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the length of last 5 questions\nraw_df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:49.085194Z","iopub.execute_input":"2023-03-27T12:31:49.085579Z","iopub.status.idle":"2023-03-27T12:31:49.09959Z","shell.execute_reply.started":"2023-03-27T12:31:49.085543Z","shell.execute_reply":"2023-03-27T12:31:49.098507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Averge lengths of all questions in a Dataframe\nimport numpy as np\navg_lengths = np.mean(raw_df['Length'])\navg_lengths","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:49.101039Z","iopub.execute_input":"2023-03-27T12:31:49.101479Z","iopub.status.idle":"2023-03-27T12:31:49.116671Z","shell.execute_reply.started":"2023-03-27T12:31:49.101429Z","shell.execute_reply":"2023-03-27T12:31:49.115543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Average length of each the question in a DataFrame \navg_lengths_by_question = raw_df.groupby('question_text').mean()\nprint(avg_lengths_by_question)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:49.118488Z","iopub.execute_input":"2023-03-27T12:31:49.118825Z","iopub.status.idle":"2023-03-27T12:31:53.44449Z","shell.execute_reply.started":"2023-03-27T12:31:49.118793Z","shell.execute_reply":"2023-03-27T12:31:53.443179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:53.446148Z","iopub.execute_input":"2023-03-27T12:31:53.446519Z","iopub.status.idle":"2023-03-27T12:31:54.288819Z","shell.execute_reply.started":"2023-03-27T12:31:53.446484Z","shell.execute_reply":"2023-03-27T12:31:54.287699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df=pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\")\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:54.290258Z","iopub.execute_input":"2023-03-27T12:31:54.290592Z","iopub.status.idle":"2023-03-27T12:31:54.645226Z","shell.execute_reply.started":"2023-03-27T12:31:54.29056Z","shell.execute_reply":"2023-03-27T12:31:54.64414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.prediction.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:54.650217Z","iopub.execute_input":"2023-03-27T12:31:54.65059Z","iopub.status.idle":"2023-03-27T12:31:54.663464Z","shell.execute_reply.started":"2023-03-27T12:31:54.650554Z","shell.execute_reply":"2023-03-27T12:31:54.662176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Creating a working sample","metadata":{}},{"cell_type":"code","source":"SAMPLE_SIZE=100_000\nsample_df=raw_df.sample(SAMPLE_SIZE,random_state=42)\nsample_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:54.664967Z","iopub.execute_input":"2023-03-27T12:31:54.665416Z","iopub.status.idle":"2023-03-27T12:31:54.759786Z","shell.execute_reply.started":"2023-03-27T12:31:54.665381Z","shell.execute_reply":"2023-03-27T12:31:54.758766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Preprocessing Techniques\n    Outline:\n        -Understand the bag of words model\n        -Tokenization\n        -Stop word removal\n        -Stemming","metadata":{}},{"cell_type":"markdown","source":"#### Bag of words Intution\n       -Create a list of all the words across all the text documents\n       -Convert each document inot vector counts of each word\n    Limitations:\n       -There may be too many words in the dataset\n       -Some words may occur to frequently\n       -Some words may occur very rarely or only once\n       -A single word may have many forms(go,gone,going or bird vs birds)","metadata":{}},{"cell_type":"markdown","source":"### Tokenization \n    Splitting a document into words and seperators","metadata":{}},{"cell_type":"code","source":"import nltk","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:54.760961Z","iopub.execute_input":"2023-03-27T12:31:54.761274Z","iopub.status.idle":"2023-03-27T12:31:55.919401Z","shell.execute_reply.started":"2023-03-27T12:31:54.761245Z","shell.execute_reply":"2023-03-27T12:31:55.918128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:55.921058Z","iopub.execute_input":"2023-03-27T12:31:55.921432Z","iopub.status.idle":"2023-03-27T12:31:55.926491Z","shell.execute_reply.started":"2023-03-27T12:31:55.921399Z","shell.execute_reply":"2023-03-27T12:31:55.92525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:55.927844Z","iopub.execute_input":"2023-03-27T12:31:55.928364Z","iopub.status.idle":"2023-03-27T12:31:56.100833Z","shell.execute_reply.started":"2023-03-27T12:31:55.92833Z","shell.execute_reply":"2023-03-27T12:31:56.099729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0=sincere_df.question_text.values[1]\nq0","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.102277Z","iopub.execute_input":"2023-03-27T12:31:56.102608Z","iopub.status.idle":"2023-03-27T12:31:56.111266Z","shell.execute_reply.started":"2023-03-27T12:31:56.102577Z","shell.execute_reply":"2023-03-27T12:31:56.109489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1=raw_df[raw_df.target==1].question_text.values[0]\nq1","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.113393Z","iopub.execute_input":"2023-03-27T12:31:56.114159Z","iopub.status.idle":"2023-03-27T12:31:56.143047Z","shell.execute_reply.started":"2023-03-27T12:31:56.114114Z","shell.execute_reply":"2023-03-27T12:31:56.141879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_tokenize(q0)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.144628Z","iopub.execute_input":"2023-03-27T12:31:56.144985Z","iopub.status.idle":"2023-03-27T12:31:56.16458Z","shell.execute_reply.started":"2023-03-27T12:31:56.14495Z","shell.execute_reply":"2023-03-27T12:31:56.163683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_tokenize(q1)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.165503Z","iopub.execute_input":"2023-03-27T12:31:56.165819Z","iopub.status.idle":"2023-03-27T12:31:56.174824Z","shell.execute_reply.started":"2023-03-27T12:31:56.16579Z","shell.execute_reply":"2023-03-27T12:31:56.173635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_tok=word_tokenize(q0)\nq1_tok=word_tokenize(q1)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.176188Z","iopub.execute_input":"2023-03-27T12:31:56.176509Z","iopub.status.idle":"2023-03-27T12:31:56.182852Z","shell.execute_reply.started":"2023-03-27T12:31:56.176479Z","shell.execute_reply":"2023-03-27T12:31:56.18133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stop Word Removal\n    Removing commonly occuring words","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.184165Z","iopub.execute_input":"2023-03-27T12:31:56.184486Z","iopub.status.idle":"2023-03-27T12:31:56.195235Z","shell.execute_reply.started":"2023-03-27T12:31:56.184455Z","shell.execute_reply":"2023-03-27T12:31:56.193975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('stopwords')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.196952Z","iopub.execute_input":"2023-03-27T12:31:56.19731Z","iopub.status.idle":"2023-03-27T12:31:56.214001Z","shell.execute_reply.started":"2023-03-27T12:31:56.197276Z","shell.execute_reply":"2023-03-27T12:31:56.212944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"english_stopwords=stopwords.words('english')\n\", \".join(english_stopwords)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.215266Z","iopub.execute_input":"2023-03-27T12:31:56.215607Z","iopub.status.idle":"2023-03-27T12:31:56.226882Z","shell.execute_reply.started":"2023-03-27T12:31:56.215573Z","shell.execute_reply":"2023-03-27T12:31:56.225857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(tokens):\n    return [word for word in tokens if word.lower() not in english_stopwords]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.228876Z","iopub.execute_input":"2023-03-27T12:31:56.229257Z","iopub.status.idle":"2023-03-27T12:31:56.238837Z","shell.execute_reply.started":"2023-03-27T12:31:56.229221Z","shell.execute_reply":"2023-03-27T12:31:56.237687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_tok","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.240211Z","iopub.execute_input":"2023-03-27T12:31:56.24054Z","iopub.status.idle":"2023-03-27T12:31:56.251063Z","shell.execute_reply.started":"2023-03-27T12:31:56.240509Z","shell.execute_reply":"2023-03-27T12:31:56.249873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stp=remove_stopwords(q0_tok)\nq0_stp","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.252311Z","iopub.execute_input":"2023-03-27T12:31:56.253436Z","iopub.status.idle":"2023-03-27T12:31:56.263309Z","shell.execute_reply.started":"2023-03-27T12:31:56.253382Z","shell.execute_reply":"2023-03-27T12:31:56.262165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_tok","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.264594Z","iopub.execute_input":"2023-03-27T12:31:56.265958Z","iopub.status.idle":"2023-03-27T12:31:56.275729Z","shell.execute_reply.started":"2023-03-27T12:31:56.265903Z","shell.execute_reply":"2023-03-27T12:31:56.274695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_stp=remove_stopwords(q1_tok)\nq1_stp","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.276975Z","iopub.execute_input":"2023-03-27T12:31:56.277937Z","iopub.status.idle":"2023-03-27T12:31:56.287011Z","shell.execute_reply.started":"2023-03-27T12:31:56.27789Z","shell.execute_reply":"2023-03-27T12:31:56.286199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stemming\n    \"go\",\"gone\",\"going\" ->\"go\n    \"birds\",\"bird\" -> \"bird\"","metadata":{}},{"cell_type":"code","source":"from nltk.stem.snowball import SnowballStemmer","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.288616Z","iopub.execute_input":"2023-03-27T12:31:56.289199Z","iopub.status.idle":"2023-03-27T12:31:56.297601Z","shell.execute_reply.started":"2023-03-27T12:31:56.289163Z","shell.execute_reply":"2023-03-27T12:31:56.296592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer=SnowballStemmer(language='english')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.29879Z","iopub.execute_input":"2023-03-27T12:31:56.299378Z","iopub.status.idle":"2023-03-27T12:31:56.308914Z","shell.execute_reply.started":"2023-03-27T12:31:56.299343Z","shell.execute_reply":"2023-03-27T12:31:56.307773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stp","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.310144Z","iopub.execute_input":"2023-03-27T12:31:56.31047Z","iopub.status.idle":"2023-03-27T12:31:56.323414Z","shell.execute_reply.started":"2023-03-27T12:31:56.31044Z","shell.execute_reply":"2023-03-27T12:31:56.322098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stm=[stemmer.stem(word) for word in q0_stp]\nq0_stm","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.324747Z","iopub.execute_input":"2023-03-27T12:31:56.326019Z","iopub.status.idle":"2023-03-27T12:31:56.337138Z","shell.execute_reply.started":"2023-03-27T12:31:56.32597Z","shell.execute_reply":"2023-03-27T12:31:56.336109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_stp","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.33873Z","iopub.execute_input":"2023-03-27T12:31:56.33962Z","iopub.status.idle":"2023-03-27T12:31:56.347956Z","shell.execute_reply.started":"2023-03-27T12:31:56.339584Z","shell.execute_reply":"2023-03-27T12:31:56.346771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_stm=[stemmer.stem(word) for word in q1_stp]\nq1_stm","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:31:56.349043Z","iopub.execute_input":"2023-03-27T12:31:56.349373Z","iopub.status.idle":"2023-03-27T12:31:56.360204Z","shell.execute_reply.started":"2023-03-27T12:31:56.349343Z","shell.execute_reply":"2023-03-27T12:31:56.35895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lemmatization\n    \"Love\" -> \"Love\"\n    \"Loving\" ->\"Love\"\n    \"Lovavle\" ->\"Love\"\nIt generally not used in bag of words because it requires looking up a    dictonary which can be fairly slow especially if we have a large             dictonary and it can also take up a lot of memory\nbecause we are going to turn our documents into vector anyway it doesnt       matter short form is something that its just a root of a word or an           actual word from a dictonary thats why we are not using lemmatization in     this case.But in the case when we want derive the root word we can do         lemmatization.","metadata":{}},{"cell_type":"code","source":"from nltk.stem import WordNetLemmatizer\nnltk.download ('wordnet')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:23.592015Z","iopub.execute_input":"2023-03-27T12:34:23.592812Z","iopub.status.idle":"2023-03-27T12:34:23.603814Z","shell.execute_reply.started":"2023-03-27T12:34:23.592758Z","shell.execute_reply":"2023-03-27T12:34:23.602512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('omw-1.4')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:24.883638Z","iopub.execute_input":"2023-03-27T12:34:24.884091Z","iopub.status.idle":"2023-03-27T12:34:24.954516Z","shell.execute_reply.started":"2023-03-27T12:34:24.88404Z","shell.execute_reply":"2023-03-27T12:34:24.953189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stm","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:27.549024Z","iopub.execute_input":"2023-03-27T12:34:27.549928Z","iopub.status.idle":"2023-03-27T12:34:27.5574Z","shell.execute_reply.started":"2023-03-27T12:34:27.549888Z","shell.execute_reply":"2023-03-27T12:34:27.556058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()\nq0_lemma = [lemmatizer.lemmatize(word) for word in q0_stm]\nq0_lemma ","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:30.068606Z","iopub.execute_input":"2023-03-27T12:34:30.069043Z","iopub.status.idle":"2023-03-27T12:34:30.113314Z","shell.execute_reply.started":"2023-03-27T12:34:30.069006Z","shell.execute_reply":"2023-03-27T12:34:30.110388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Implement Bag of Words\n    Outline:\n        -Create a vocabulary using Count Vectorizer\n        -Transform text to vectors using Count Vectorizer\n        -Configure text preprocessing in Count Vectorizer","metadata":{}},{"cell_type":"markdown","source":"### Create a Vocabulary","metadata":{}},{"cell_type":"code","source":"small_df=sample_df[:5]\nsmall_df.question_text.values","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:35.220315Z","iopub.execute_input":"2023-03-27T12:34:35.220905Z","iopub.status.idle":"2023-03-27T12:34:35.230733Z","shell.execute_reply.started":"2023-03-27T12:34:35.220858Z","shell.execute_reply":"2023-03-27T12:34:35.229624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:37.519632Z","iopub.execute_input":"2023-03-27T12:34:37.520989Z","iopub.status.idle":"2023-03-27T12:34:37.526564Z","shell.execute_reply.started":"2023-03-27T12:34:37.520933Z","shell.execute_reply":"2023-03-27T12:34:37.52511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_vect=CountVectorizer()           #.fit() is used to learn the vocabulary\nsmall_vect.fit(small_df.question_text)   #learned the vocabulary by building countvectorizer","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:39.686816Z","iopub.execute_input":"2023-03-27T12:34:39.687237Z","iopub.status.idle":"2023-03-27T12:34:39.707456Z","shell.execute_reply.started":"2023-03-27T12:34:39.6872Z","shell.execute_reply":"2023-03-27T12:34:39.705995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_vect.vocabulary_","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:40.999615Z","iopub.execute_input":"2023-03-27T12:34:41.000035Z","iopub.status.idle":"2023-03-27T12:34:41.010283Z","shell.execute_reply.started":"2023-03-27T12:34:41Z","shell.execute_reply":"2023-03-27T12:34:41.008842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_vect.get_feature_names_out()   #gives list of words or it gives the output feature names","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:44.219345Z","iopub.execute_input":"2023-03-27T12:34:44.219736Z","iopub.status.idle":"2023-03-27T12:34:44.227104Z","shell.execute_reply.started":"2023-03-27T12:34:44.219703Z","shell.execute_reply":"2023-03-27T12:34:44.226185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transform documents into vectors ","metadata":{}},{"cell_type":"code","source":"vectors=small_vect.transform(small_df.question_text)\nvectors","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:45.516791Z","iopub.execute_input":"2023-03-27T12:34:45.518273Z","iopub.status.idle":"2023-03-27T12:34:45.527153Z","shell.execute_reply.started":"2023-03-27T12:34:45.518227Z","shell.execute_reply":"2023-03-27T12:34:45.525664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectors[0].toarray()    #to view values in sparse matrix ","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:45.886049Z","iopub.execute_input":"2023-03-27T12:34:45.886463Z","iopub.status.idle":"2023-03-27T12:34:45.895739Z","shell.execute_reply.started":"2023-03-27T12:34:45.886427Z","shell.execute_reply":"2023-03-27T12:34:45.894124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectors.shape   ","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:46.074813Z","iopub.execute_input":"2023-03-27T12:34:46.075281Z","iopub.status.idle":"2023-03-27T12:34:46.084372Z","shell.execute_reply.started":"2023-03-27T12:34:46.07524Z","shell.execute_reply":"2023-03-27T12:34:46.083095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_df.question_text.values[0]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:46.224918Z","iopub.execute_input":"2023-03-27T12:34:46.225543Z","iopub.status.idle":"2023-03-27T12:34:46.234268Z","shell.execute_reply.started":"2023-03-27T12:34:46.225499Z","shell.execute_reply":"2023-03-27T12:34:46.232798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configure Count Vectorizer Parameters ","metadata":{}},{"cell_type":"code","source":"stemmer=SnowballStemmer(language='english')","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:46.634895Z","iopub.execute_input":"2023-03-27T12:34:46.635349Z","iopub.status.idle":"2023-03-27T12:34:46.640758Z","shell.execute_reply.started":"2023-03-27T12:34:46.635291Z","shell.execute_reply":"2023-03-27T12:34:46.639492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize(text):\n    return [stemmer.stem(word) for word in word_tokenize(text)]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:46.77259Z","iopub.execute_input":"2023-03-27T12:34:46.772966Z","iopub.status.idle":"2023-03-27T12:34:46.777665Z","shell.execute_reply.started":"2023-03-27T12:34:46.77292Z","shell.execute_reply":"2023-03-27T12:34:46.776734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer=CountVectorizer(lowercase=True,tokenizer=tokenize,stop_words=english_stopwords,max_features=1000)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:46.92373Z","iopub.execute_input":"2023-03-27T12:34:46.924574Z","iopub.status.idle":"2023-03-27T12:34:46.929614Z","shell.execute_reply.started":"2023-03-27T12:34:46.924535Z","shell.execute_reply":"2023-03-27T12:34:46.928521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:47.771193Z","iopub.execute_input":"2023-03-27T12:34:47.771587Z","iopub.status.idle":"2023-03-27T12:34:47.788337Z","shell.execute_reply.started":"2023-03-27T12:34:47.771554Z","shell.execute_reply":"2023-03-27T12:34:47.786929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvectorizer.fit(sample_df.question_text)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:34:48.225923Z","iopub.execute_input":"2023-03-27T12:34:48.226328Z","iopub.status.idle":"2023-03-27T12:35:22.726187Z","shell.execute_reply.started":"2023-03-27T12:34:48.226294Z","shell.execute_reply":"2023-03-27T12:35:22.724787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vectorizer.vocabulary_)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:35:22.728811Z","iopub.execute_input":"2023-03-27T12:35:22.729192Z","iopub.status.idle":"2023-03-27T12:35:22.73702Z","shell.execute_reply.started":"2023-03-27T12:35:22.729156Z","shell.execute_reply":"2023-03-27T12:35:22.735384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:35:22.738959Z","iopub.execute_input":"2023-03-27T12:35:22.739535Z","iopub.status.idle":"2023-03-27T12:35:22.753377Z","shell.execute_reply.started":"2023-03-27T12:35:22.739484Z","shell.execute_reply":"2023-03-27T12:35:22.75239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ninputs=vectorizer.transform(sample_df.question_text)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:35:44.51505Z","iopub.execute_input":"2023-03-27T12:35:44.516093Z","iopub.status.idle":"2023-03-27T12:36:18.688482Z","shell.execute_reply.started":"2023-03-27T12:35:44.516034Z","shell.execute_reply":"2023-03-27T12:36:18.686959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:36:18.690695Z","iopub.execute_input":"2023-03-27T12:36:18.691193Z","iopub.status.idle":"2023-03-27T12:36:18.699156Z","shell.execute_reply.started":"2023-03-27T12:36:18.691146Z","shell.execute_reply":"2023-03-27T12:36:18.697758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:36:18.700622Z","iopub.execute_input":"2023-03-27T12:36:18.700969Z","iopub.status.idle":"2023-03-27T12:36:18.714104Z","shell.execute_reply.started":"2023-03-27T12:36:18.700936Z","shell.execute_reply":"2023-03-27T12:36:18.71277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.question_text.values[0]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:36:18.716646Z","iopub.execute_input":"2023-03-27T12:36:18.717765Z","iopub.status.idle":"2023-03-27T12:36:18.726531Z","shell.execute_reply.started":"2023-03-27T12:36:18.717703Z","shell.execute_reply":"2023-03-27T12:36:18.725171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:36:18.728547Z","iopub.execute_input":"2023-03-27T12:36:18.729041Z","iopub.status.idle":"2023-03-27T12:36:18.752045Z","shell.execute_reply.started":"2023-03-27T12:36:18.728996Z","shell.execute_reply":"2023-03-27T12:36:18.750538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_inputs=vectorizer.transform(test_df.question_text)   #fit vectorizer can be used only once on the training set and the transformation can be done seperatly by trainin the vocabulary in test set it gives meaningless results","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:36:18.75372Z","iopub.execute_input":"2023-03-27T12:36:18.754045Z","iopub.status.idle":"2023-03-27T12:38:26.019109Z","shell.execute_reply.started":"2023-03-27T12:36:18.754015Z","shell.execute_reply":"2023-03-27T12:38:26.018039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.020607Z","iopub.execute_input":"2023-03-27T12:38:26.020932Z","iopub.status.idle":"2023-03-27T12:38:26.026817Z","shell.execute_reply.started":"2023-03-27T12:38:26.020902Z","shell.execute_reply":"2023-03-27T12:38:26.025712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.question_text.values[0]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.02861Z","iopub.execute_input":"2023-03-27T12:38:26.028983Z","iopub.status.idle":"2023-03-27T12:38:26.045258Z","shell.execute_reply.started":"2023-03-27T12:38:26.028949Z","shell.execute_reply":"2023-03-27T12:38:26.043926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_inputs[0].toarray()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.046618Z","iopub.execute_input":"2023-03-27T12:38:26.046972Z","iopub.status.idle":"2023-03-27T12:38:26.062531Z","shell.execute_reply.started":"2023-03-27T12:38:26.046938Z","shell.execute_reply":"2023-03-27T12:38:26.061058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.069652Z","iopub.execute_input":"2023-03-27T12:38:26.070186Z","iopub.status.idle":"2023-03-27T12:38:26.079883Z","shell.execute_reply.started":"2023-03-27T12:38:26.070117Z","shell.execute_reply":"2023-03-27T12:38:26.078594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ML Models for Text Classification\n\n    outline:\n        -Create a training & validation set\n        -Train a logisitic regression model\n        -Make predicions on training,validation & test data","metadata":{}},{"cell_type":"markdown","source":"### Split into training and validation set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.081613Z","iopub.execute_input":"2023-03-27T12:38:26.082074Z","iopub.status.idle":"2023-03-27T12:38:26.091044Z","shell.execute_reply.started":"2023-03-27T12:38:26.082026Z","shell.execute_reply":"2023-03-27T12:38:26.089753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test,y_train,y_test=train_test_split(inputs,sample_df.target,test_size=0.3,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.093044Z","iopub.execute_input":"2023-03-27T12:38:26.093408Z","iopub.status.idle":"2023-03-27T12:38:26.11665Z","shell.execute_reply.started":"2023-03-27T12:38:26.093377Z","shell.execute_reply":"2023-03-27T12:38:26.115753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.117747Z","iopub.execute_input":"2023-03-27T12:38:26.118512Z","iopub.status.idle":"2023-03-27T12:38:26.126899Z","shell.execute_reply.started":"2023-03-27T12:38:26.118462Z","shell.execute_reply":"2023-03-27T12:38:26.125602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.129268Z","iopub.execute_input":"2023-03-27T12:38:26.130388Z","iopub.status.idle":"2023-03-27T12:38:26.139427Z","shell.execute_reply.started":"2023-03-27T12:38:26.130336Z","shell.execute_reply":"2023-03-27T12:38:26.13816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  Train logistic regression model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.141323Z","iopub.execute_input":"2023-03-27T12:38:26.141859Z","iopub.status.idle":"2023-03-27T12:38:26.148166Z","shell.execute_reply.started":"2023-03-27T12:38:26.14181Z","shell.execute_reply":"2023-03-27T12:38:26.147062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_ITER=1000   #setting max iterations to 1000 times","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.14981Z","iopub.execute_input":"2023-03-27T12:38:26.150171Z","iopub.status.idle":"2023-03-27T12:38:26.157828Z","shell.execute_reply.started":"2023-03-27T12:38:26.150138Z","shell.execute_reply":"2023-03-27T12:38:26.156988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model=LogisticRegression(max_iter=1000,solver='sag')  #sag is sochastic gradient descent which is used for the optimization of the model\nmodel.fit(X_train,y_train)  #fitting into train innputs and train targets","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:26.15951Z","iopub.execute_input":"2023-03-27T12:38:26.160612Z","iopub.status.idle":"2023-03-27T12:38:49.668893Z","shell.execute_reply.started":"2023-03-27T12:38:26.160567Z","shell.execute_reply":"2023-03-27T12:38:49.667541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_pred=model.predict(X_train)     #The training inputs gets predicted and stored in training predciton variable\nX_pred","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.670641Z","iopub.execute_input":"2023-03-27T12:38:49.671095Z","iopub.status.idle":"2023-03-27T12:38:49.68299Z","shell.execute_reply.started":"2023-03-27T12:38:49.671032Z","shell.execute_reply":"2023-03-27T12:38:49.681778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_train).value_counts()  #counting the total number of 0's and 1's in training the target values","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.684868Z","iopub.execute_input":"2023-03-27T12:38:49.685337Z","iopub.status.idle":"2023-03-27T12:38:49.69721Z","shell.execute_reply.started":"2023-03-27T12:38:49.685293Z","shell.execute_reply":"2023-03-27T12:38:49.695992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(X_pred).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.69935Z","iopub.execute_input":"2023-03-27T12:38:49.69997Z","iopub.status.idle":"2023-03-27T12:38:49.710415Z","shell.execute_reply.started":"2023-03-27T12:38:49.699921Z","shell.execute_reply":"2023-03-27T12:38:49.709166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_train).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.712138Z","iopub.execute_input":"2023-03-27T12:38:49.712537Z","iopub.status.idle":"2023-03-27T12:38:49.72339Z","shell.execute_reply.started":"2023-03-27T12:38:49.712503Z","shell.execute_reply":"2023-03-27T12:38:49.722166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.725187Z","iopub.execute_input":"2023-03-27T12:38:49.725696Z","iopub.status.idle":"2023-03-27T12:38:49.731395Z","shell.execute_reply.started":"2023-03-27T12:38:49.725659Z","shell.execute_reply":"2023-03-27T12:38:49.730341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_train,X_pred)  #getting the accuracy by comparing train_target and train_predictions","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.733265Z","iopub.execute_input":"2023-03-27T12:38:49.734141Z","iopub.status.idle":"2023-03-27T12:38:49.749042Z","shell.execute_reply.started":"2023-03-27T12:38:49.734094Z","shell.execute_reply":"2023-03-27T12:38:49.748057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\naccuracy_score(y_train,np.zeros(len(y_train)))  #to compare with the fixed values here train_target values are compared with the number of zeros inside the train_target","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.750627Z","iopub.execute_input":"2023-03-27T12:38:49.751459Z","iopub.status.idle":"2023-03-27T12:38:49.766071Z","shell.execute_reply.started":"2023-03-27T12:38:49.751423Z","shell.execute_reply":"2023-03-27T12:38:49.765184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.767437Z","iopub.execute_input":"2023-03-27T12:38:49.768412Z","iopub.status.idle":"2023-03-27T12:38:49.772924Z","shell.execute_reply.started":"2023-03-27T12:38:49.768329Z","shell.execute_reply":"2023-03-27T12:38:49.771858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(y_train,X_pred) #getting the f1 score by comparing by training targets and training input prediction","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.774817Z","iopub.execute_input":"2023-03-27T12:38:49.775196Z","iopub.status.idle":"2023-03-27T12:38:49.810856Z","shell.execute_reply.started":"2023-03-27T12:38:49.775163Z","shell.execute_reply":"2023-03-27T12:38:49.809382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(y_train,np.zeros(len(y_train)))  #This is what happens when we compare a fixed value","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.812349Z","iopub.execute_input":"2023-03-27T12:38:49.812684Z","iopub.status.idle":"2023-03-27T12:38:49.848151Z","shell.execute_reply.started":"2023-03-27T12:38:49.812652Z","shell.execute_reply":"2023-03-27T12:38:49.84723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred=model.predict(X_test)  #predicting the validation data by using the validation inputs","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.853755Z","iopub.execute_input":"2023-03-27T12:38:49.854887Z","iopub.status.idle":"2023-03-27T12:38:49.860658Z","shell.execute_reply.started":"2023-03-27T12:38:49.854836Z","shell.execute_reply":"2023-03-27T12:38:49.859422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test,y_pred)  #compares with the validation target with validation predictions","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.86219Z","iopub.execute_input":"2023-03-27T12:38:49.862643Z","iopub.status.idle":"2023-03-27T12:38:49.877451Z","shell.execute_reply.started":"2023-03-27T12:38:49.862608Z","shell.execute_reply":"2023-03-27T12:38:49.876168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(y_test,y_pred)  #the validation score will be less compared to the training set because the training set is trained by used the labeled data while the validation score doesnt have the labeled data","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.879329Z","iopub.execute_input":"2023-03-27T12:38:49.879833Z","iopub.status.idle":"2023-03-27T12:38:49.89887Z","shell.execute_reply.started":"2023-03-27T12:38:49.879755Z","shell.execute_reply":"2023-03-27T12:38:49.897463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Checking the model predictions with few examples ","metadata":{}},{"cell_type":"markdown","source":"###### Given below shows the example for sincere questions ","metadata":{}},{"cell_type":"code","source":"sincere_df.question_text.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.900234Z","iopub.execute_input":"2023-03-27T12:38:49.901182Z","iopub.status.idle":"2023-03-27T12:38:49.908799Z","shell.execute_reply.started":"2023-03-27T12:38:49.901141Z","shell.execute_reply":"2023-03-27T12:38:49.907611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_df.target.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.910496Z","iopub.execute_input":"2023-03-27T12:38:49.910868Z","iopub.status.idle":"2023-03-27T12:38:49.925692Z","shell.execute_reply.started":"2023-03-27T12:38:49.910831Z","shell.execute_reply":"2023-03-27T12:38:49.924413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict(vectorizer.transform(sincere_df.question_text.values[:10]))","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.926919Z","iopub.execute_input":"2023-03-27T12:38:49.927268Z","iopub.status.idle":"2023-03-27T12:38:49.942186Z","shell.execute_reply.started":"2023-03-27T12:38:49.927236Z","shell.execute_reply":"2023-03-27T12:38:49.940867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### Given below shows the example for unsincere questions ","metadata":{}},{"cell_type":"code","source":"unsincere_df.question_text.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.944036Z","iopub.execute_input":"2023-03-27T12:38:49.944773Z","iopub.status.idle":"2023-03-27T12:38:49.955995Z","shell.execute_reply.started":"2023-03-27T12:38:49.944735Z","shell.execute_reply":"2023-03-27T12:38:49.954866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unsincere_df.target.values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.95807Z","iopub.execute_input":"2023-03-27T12:38:49.958812Z","iopub.status.idle":"2023-03-27T12:38:49.970029Z","shell.execute_reply.started":"2023-03-27T12:38:49.958767Z","shell.execute_reply":"2023-03-27T12:38:49.968818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict(vectorizer.transform(unsincere_df.question_text.values[:10]))","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.97253Z","iopub.execute_input":"2023-03-27T12:38:49.972938Z","iopub.status.idle":"2023-03-27T12:38:49.989555Z","shell.execute_reply.started":"2023-03-27T12:38:49.972889Z","shell.execute_reply":"2023-03-27T12:38:49.988157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Make predictions and Submit to Kaggle","metadata":{}},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:49.990961Z","iopub.execute_input":"2023-03-27T12:38:49.991309Z","iopub.status.idle":"2023-03-27T12:38:50.005355Z","shell.execute_reply.started":"2023-03-27T12:38:49.991277Z","shell.execute_reply":"2023-03-27T12:38:50.004069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_inputs","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.00669Z","iopub.execute_input":"2023-03-27T12:38:50.007786Z","iopub.status.idle":"2023-03-27T12:38:50.019752Z","shell.execute_reply.started":"2023-03-27T12:38:50.007748Z","shell.execute_reply":"2023-03-27T12:38:50.018524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds=model.predict(test_inputs)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.021361Z","iopub.execute_input":"2023-03-27T12:38:50.022489Z","iopub.status.idle":"2023-03-27T12:38:50.043473Z","shell.execute_reply.started":"2023-03-27T12:38:50.02241Z","shell.execute_reply":"2023-03-27T12:38:50.042059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.045152Z","iopub.execute_input":"2023-03-27T12:38:50.04561Z","iopub.status.idle":"2023-03-27T12:38:50.061965Z","shell.execute_reply.started":"2023-03-27T12:38:50.045562Z","shell.execute_reply":"2023-03-27T12:38:50.061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.prediction=test_preds","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.063609Z","iopub.execute_input":"2023-03-27T12:38:50.063998Z","iopub.status.idle":"2023-03-27T12:38:50.07033Z","shell.execute_reply.started":"2023-03-27T12:38:50.063964Z","shell.execute_reply":"2023-03-27T12:38:50.069291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.prediction.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.071791Z","iopub.execute_input":"2023-03-27T12:38:50.072293Z","iopub.status.idle":"2023-03-27T12:38:50.090683Z","shell.execute_reply.started":"2023-03-27T12:38:50.072258Z","shell.execute_reply":"2023-03-27T12:38:50.089226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.092751Z","iopub.execute_input":"2023-03-27T12:38:50.094143Z","iopub.status.idle":"2023-03-27T12:38:50.108917Z","shell.execute_reply.started":"2023-03-27T12:38:50.094073Z","shell.execute_reply":"2023-03-27T12:38:50.107676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=None)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.110238Z","iopub.execute_input":"2023-03-27T12:38:50.110732Z","iopub.status.idle":"2023-03-27T12:38:50.622522Z","shell.execute_reply.started":"2023-03-27T12:38:50.110682Z","shell.execute_reply":"2023-03-27T12:38:50.621187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-03-27T12:38:50.623827Z","iopub.execute_input":"2023-03-27T12:38:50.62494Z","iopub.status.idle":"2023-03-27T12:38:51.75776Z","shell.execute_reply.started":"2023-03-27T12:38:50.624889Z","shell.execute_reply":"2023-03-27T12:38:51.756426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}