{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T18:10:58.601266Z","iopub.execute_input":"2022-08-10T18:10:58.602150Z","iopub.status.idle":"2022-08-10T18:10:58.616844Z","shell.execute_reply.started":"2022-08-10T18:10:58.602097Z","shell.execute_reply":"2022-08-10T18:10:58.615588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\nimport re\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import CountVectorizer\n\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.feature_selection import SelectKBest\nfrom sklearn.feature_selection import chi2\nfrom sklearn.feature_selection import f_classif\n\n\nfrom sklearn.svm import LinearSVC\nfrom sklearn.linear_model import RidgeClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.linear_model import PassiveAggressiveClassifier\n\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.naive_bayes import ComplementNB\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:10:58.903320Z","iopub.execute_input":"2022-08-10T18:10:58.903867Z","iopub.status.idle":"2022-08-10T18:10:58.910621Z","shell.execute_reply.started":"2022-08-10T18:10:58.903827Z","shell.execute_reply":"2022-08-10T18:10:58.909607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load the data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsample_submission_df = pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:10:59.144295Z","iopub.execute_input":"2022-08-10T18:10:59.144610Z","iopub.status.idle":"2022-08-10T18:10:59.183170Z","shell.execute_reply.started":"2022-08-10T18:10:59.144575Z","shell.execute_reply":"2022-08-10T18:10:59.182519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:10:59.388180Z","iopub.execute_input":"2022-08-10T18:10:59.388492Z","iopub.status.idle":"2022-08-10T18:10:59.400013Z","shell.execute_reply.started":"2022-08-10T18:10:59.388460Z","shell.execute_reply":"2022-08-10T18:10:59.399020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# missing values\ntrain_df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:10:59.621377Z","iopub.execute_input":"2022-08-10T18:10:59.621759Z","iopub.status.idle":"2022-08-10T18:10:59.633391Z","shell.execute_reply.started":"2022-08-10T18:10:59.621720Z","shell.execute_reply":"2022-08-10T18:10:59.631922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df['text']\ny = train_df.target\n\nprint('train text shape',X.shape)\nprint('train target shape',y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:10:59.948067Z","iopub.execute_input":"2022-08-10T18:10:59.949032Z","iopub.status.idle":"2022-08-10T18:10:59.955908Z","shell.execute_reply.started":"2022-08-10T18:10:59.948985Z","shell.execute_reply":"2022-08-10T18:10:59.955058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utility\n\nLets create a utility for displaying results of data processing \n\nBy displaying results side by side , it become  more intutive","metadata":{}},{"cell_type":"code","source":"def display_util(method, sample):\n    new_sample = sample.apply(method)\n    frame = pd.DataFrame(data={'before':sample , 'After':new_sample})\n    return frame   ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:00.165047Z","iopub.execute_input":"2022-08-10T18:11:00.165376Z","iopub.status.idle":"2022-08-10T18:11:00.171093Z","shell.execute_reply.started":"2022-08-10T18:11:00.165313Z","shell.execute_reply":"2022-08-10T18:11:00.170066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Text processing\n\n* Remove html elements\n* Remove URLs\n* Remove mentions\n* Remove punctuations\n* Remove numbers\n* Remove stop words\n* Stemming\n* lemmatization\n","metadata":{}},{"cell_type":"code","source":"# to test preprocessing code lets sample 10 rows of text\nsample = X.sample(n=10)\nsample","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:00.385130Z","iopub.execute_input":"2022-08-10T18:11:00.385455Z","iopub.status.idle":"2022-08-10T18:11:00.392877Z","shell.execute_reply.started":"2022-08-10T18:11:00.385417Z","shell.execute_reply":"2022-08-10T18:11:00.392293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove html","metadata":{}},{"cell_type":"code","source":"example = '<p> This is an example <br> hello world <p/>'\n\nre.sub(pattern=r'<\\w+/?>', repl='',string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:00.625872Z","iopub.execute_input":"2022-08-10T18:11:00.626301Z","iopub.status.idle":"2022-08-10T18:11:00.632863Z","shell.execute_reply.started":"2022-08-10T18:11:00.626266Z","shell.execute_reply":"2022-08-10T18:11:00.632222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets apply this to our text data sample","metadata":{}},{"cell_type":"code","source":"def remove_html(text):\n       return re.sub(pattern=r'<\\w+/?>', repl='',string=text)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:00.866028Z","iopub.execute_input":"2022-08-10T18:11:00.866498Z","iopub.status.idle":"2022-08-10T18:11:00.871298Z","shell.execute_reply.started":"2022-08-10T18:11:00.866461Z","shell.execute_reply":"2022-08-10T18:11:00.870671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see this \ndisplay_util(remove_html, sample).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:01.244793Z","iopub.execute_input":"2022-08-10T18:11:01.245219Z","iopub.status.idle":"2022-08-10T18:11:01.256604Z","shell.execute_reply.started":"2022-08-10T18:11:01.245178Z","shell.execute_reply":"2022-08-10T18:11:01.255745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove Mentioning","metadata":{}},{"cell_type":"code","source":"example = '@BritishBakeOff This has opened up a new level of reality show'\n\nre.sub(pattern=r'@[^\\s]*', repl='',string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:01.593524Z","iopub.execute_input":"2022-08-10T18:11:01.594048Z","iopub.status.idle":"2022-08-10T18:11:01.600083Z","shell.execute_reply.started":"2022-08-10T18:11:01.593995Z","shell.execute_reply":"2022-08-10T18:11:01.599493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_mentioning(text):\n    return re.sub(pattern=r'@[^\\s]*', repl='',string=text)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:01.821628Z","iopub.execute_input":"2022-08-10T18:11:01.822065Z","iopub.status.idle":"2022-08-10T18:11:01.826811Z","shell.execute_reply.started":"2022-08-10T18:11:01.822026Z","shell.execute_reply":"2022-08-10T18:11:01.825812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see this in action\ndisplay_util(remove_mentioning, sample).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:01.935724Z","iopub.execute_input":"2022-08-10T18:11:01.936562Z","iopub.status.idle":"2022-08-10T18:11:01.949522Z","shell.execute_reply.started":"2022-08-10T18:11:01.936497Z","shell.execute_reply":"2022-08-10T18:11:01.948427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove URLs","metadata":{}},{"cell_type":"code","source":"example = 'Free Ebay Sniping RT? http://t.co/B231Ul1O1K  get yours now'\n\nre.sub(pattern=r'http[^\\s]*', repl='', string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:02.182747Z","iopub.execute_input":"2022-08-10T18:11:02.183047Z","iopub.status.idle":"2022-08-10T18:11:02.190604Z","shell.execute_reply.started":"2022-08-10T18:11:02.183010Z","shell.execute_reply":"2022-08-10T18:11:02.189681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_url(text):\n    return re.sub(pattern=r'http[^\\s]*', repl='', string=text)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:02.384414Z","iopub.execute_input":"2022-08-10T18:11:02.385010Z","iopub.status.idle":"2022-08-10T18:11:02.390359Z","shell.execute_reply.started":"2022-08-10T18:11:02.384957Z","shell.execute_reply":"2022-08-10T18:11:02.389747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_util(remove_url, sample).head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:02.517847Z","iopub.execute_input":"2022-08-10T18:11:02.518292Z","iopub.status.idle":"2022-08-10T18:11:02.531152Z","shell.execute_reply.started":"2022-08-10T18:11:02.518253Z","shell.execute_reply":"2022-08-10T18:11:02.530060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove punctuations","metadata":{}},{"cell_type":"code","source":"print(string.punctuation)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:02.702780Z","iopub.execute_input":"2022-08-10T18:11:02.703301Z","iopub.status.idle":"2022-08-10T18:11:02.708936Z","shell.execute_reply.started":"2022-08-10T18:11:02.703246Z","shell.execute_reply":"2022-08-10T18:11:02.708048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"re.escape(string.punctuation)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:02.942691Z","iopub.execute_input":"2022-08-10T18:11:02.943119Z","iopub.status.idle":"2022-08-10T18:11:02.950094Z","shell.execute_reply.started":"2022-08-10T18:11:02.943086Z","shell.execute_reply":"2022-08-10T18:11:02.949174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = '@president # obama Healthcare plan is a prioroty #obamacare #us-election '\n\nre.sub(pattern=r\"[{0}]\".format(string.punctuation),repl='',string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:03.180486Z","iopub.execute_input":"2022-08-10T18:11:03.180776Z","iopub.status.idle":"2022-08-10T18:11:03.187464Z","shell.execute_reply.started":"2022-08-10T18:11:03.180746Z","shell.execute_reply":"2022-08-10T18:11:03.186407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punctuation(text):\n    return re.sub(pattern=r\"[{0}]\".format(string.punctuation),repl='',string=text)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:03.414565Z","iopub.execute_input":"2022-08-10T18:11:03.415446Z","iopub.status.idle":"2022-08-10T18:11:03.420078Z","shell.execute_reply.started":"2022-08-10T18:11:03.415400Z","shell.execute_reply":"2022-08-10T18:11:03.419213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see this\ndisplay_util(remove_punctuation, sample).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:03.626508Z","iopub.execute_input":"2022-08-10T18:11:03.626981Z","iopub.status.idle":"2022-08-10T18:11:03.639586Z","shell.execute_reply.started":"2022-08-10T18:11:03.626947Z","shell.execute_reply":"2022-08-10T18:11:03.638422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove numbers","metadata":{}},{"cell_type":"code","source":"example = '13000 people receive wildfires evacuation order'\n\nre.sub(pattern=r'\\d+', repl='', string=example)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:03.964162Z","iopub.execute_input":"2022-08-10T18:11:03.964677Z","iopub.status.idle":"2022-08-10T18:11:03.972651Z","shell.execute_reply.started":"2022-08-10T18:11:03.964639Z","shell.execute_reply":"2022-08-10T18:11:03.971966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_numbers(text):\n    return re.sub(pattern=r'\\d+', repl='', string=text)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:04.330710Z","iopub.execute_input":"2022-08-10T18:11:04.331012Z","iopub.status.idle":"2022-08-10T18:11:04.336233Z","shell.execute_reply.started":"2022-08-10T18:11:04.330976Z","shell.execute_reply":"2022-08-10T18:11:04.335157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_util(remove_numbers, sample)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:04.570939Z","iopub.execute_input":"2022-08-10T18:11:04.571482Z","iopub.status.idle":"2022-08-10T18:11:04.582401Z","shell.execute_reply.started":"2022-08-10T18:11:04.571436Z","shell.execute_reply":"2022-08-10T18:11:04.581414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove Stop words","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords\n\nenglish_stop_words = stopwords.words('english')\nprint('English stop word',english_stop_words[:5])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:04.755025Z","iopub.execute_input":"2022-08-10T18:11:04.756065Z","iopub.status.idle":"2022-08-10T18:11:04.762319Z","shell.execute_reply.started":"2022-08-10T18:11:04.756010Z","shell.execute_reply":"2022-08-10T18:11:04.761252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = 'US inflation eases in July as petrol prices drop'\n\n' '.join([word for word in example.split() if word not in english_stop_words])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:04.941473Z","iopub.execute_input":"2022-08-10T18:11:04.942083Z","iopub.status.idle":"2022-08-10T18:11:04.949342Z","shell.execute_reply.started":"2022-08-10T18:11:04.942032Z","shell.execute_reply":"2022-08-10T18:11:04.948459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize\n\ndef remove_stop_words(text):\n    return ' '.join([word for word in word_tokenize(text) if word not in english_stop_words])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.125739Z","iopub.execute_input":"2022-08-10T18:11:05.126022Z","iopub.status.idle":"2022-08-10T18:11:05.130996Z","shell.execute_reply.started":"2022-08-10T18:11:05.125991Z","shell.execute_reply":"2022-08-10T18:11:05.130318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see this \ndisplay_util(remove_stop_words, sample).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.244209Z","iopub.execute_input":"2022-08-10T18:11:05.245059Z","iopub.status.idle":"2022-08-10T18:11:05.260568Z","shell.execute_reply.started":"2022-08-10T18:11:05.245015Z","shell.execute_reply":"2022-08-10T18:11:05.259526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stemming and lemmatization\n\nFor grammatical reasons, documents are going to use different forms of a word, such as organize, organizes, and organizing. Additionally, there are families of derivationally related words with similar meanings, such as democracy, democratic, and democratization.\n\nThe goal of both stemming and lemmatization is to reduce inflectional forms and sometimes derivationally related forms of a word to a common base form. For instance:\n\nam, are, is $\\Rightarrow$ be\n\ncar, cars, car's, cars' $\\Rightarrow$ car\n\nThe result of this mapping of text will be something like:\n\nthe boy's cars are different colors $\\Rightarrow$\n\nthe boy car be differ color\n\n\n\n### Stemming\n*  keep the stem\n*  process of reducing a word to its word stem that affixes to suffixes and prefixes or to the roots of words known as a lemma.\n\nroot word \"like\"\n* \"likes\"\n* \"liked\"\n* \"likely\"\n* \"liking\"\n\n","metadata":{}},{"cell_type":"code","source":"# Stemming\n# keep the stem\n\n\n\n\n# import these modules\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import PorterStemmer\n\n  \nstemmer = PorterStemmer()\n\nexample = \"the boy's cars are different colors\"\nprint([stemmer.stem(ex) for ex in example.split()])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.418039Z","iopub.execute_input":"2022-08-10T18:11:05.418606Z","iopub.status.idle":"2022-08-10T18:11:05.424439Z","shell.execute_reply.started":"2022-08-10T18:11:05.418559Z","shell.execute_reply":"2022-08-10T18:11:05.423532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stemming \ndef stemming(text):\n    return ' '.join([stemmer.stem(word) for word in word_tokenize(text)])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.617409Z","iopub.execute_input":"2022-08-10T18:11:05.618036Z","iopub.status.idle":"2022-08-10T18:11:05.622299Z","shell.execute_reply.started":"2022-08-10T18:11:05.617991Z","shell.execute_reply":"2022-08-10T18:11:05.621621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see this in action\ndisplay_util(stemming, sample).head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.784184Z","iopub.execute_input":"2022-08-10T18:11:05.784786Z","iopub.status.idle":"2022-08-10T18:11:05.803645Z","shell.execute_reply.started":"2022-08-10T18:11:05.784746Z","shell.execute_reply":"2022-08-10T18:11:05.802684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n**lemmatization:**\n\nThe process of grouping together the different inflected forms of a word so they can be analyzed as a single item. \n\n* rocks : rock\n* corpora : corpus\n* better : good","metadata":{}},{"cell_type":"code","source":"#  lemmatization\n\n# import these modules\nfrom nltk.stem import WordNetLemmatizer\n \nlemmatizer = WordNetLemmatizer()\n\nexample = \"the boy's cars are different colors\"\n\nprint([lemmatizer.lemmatize(ex, pos='v') for ex in example.split()])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:05.976187Z","iopub.execute_input":"2022-08-10T18:11:05.976971Z","iopub.status.idle":"2022-08-10T18:11:05.984256Z","shell.execute_reply.started":"2022-08-10T18:11:05.976922Z","shell.execute_reply":"2022-08-10T18:11:05.983197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lemmatize\ndef lemmatize(text, pos='v'):\n    return ' '.join([lemmatizer.lemmatize(word, pos=pos) for word in word_tokenize(text)])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:06.167999Z","iopub.execute_input":"2022-08-10T18:11:06.168830Z","iopub.status.idle":"2022-08-10T18:11:06.172931Z","shell.execute_reply.started":"2022-08-10T18:11:06.168789Z","shell.execute_reply":"2022-08-10T18:11:06.172187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let see this \ndisplay_util(lemmatize, sample).head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:06.366412Z","iopub.execute_input":"2022-08-10T18:11:06.366707Z","iopub.status.idle":"2022-08-10T18:11:06.383292Z","shell.execute_reply.started":"2022-08-10T18:11:06.366676Z","shell.execute_reply":"2022-08-10T18:11:06.382396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processed Data","metadata":{}},{"cell_type":"code","source":"%time\n\ndef process_data(X):\n    return (X\n          .apply(lambda text: text.lower())\n          .apply(remove_html)\n          .apply(remove_url)\n          .apply(remove_mentioning)\n          .apply(remove_punctuation)\n          .apply(remove_numbers)\n          .apply(remove_stop_words)\n          )\n  \n    \nX_new = process_data(X)\nlemmatized = X_new.apply(lemmatize)\nstem = X_new.apply(stemming)\nframe = pd.DataFrame(data={'Raw':X, 'Processed':X_new, 'Lemmatized':lemmatized, 'Stemming':stem})\n\nframe.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:06.630901Z","iopub.execute_input":"2022-08-10T18:11:06.631393Z","iopub.status.idle":"2022-08-10T18:11:12.125628Z","shell.execute_reply.started":"2022-08-10T18:11:06.631315Z","shell.execute_reply":"2022-08-10T18:11:12.124743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling","metadata":{}},{"cell_type":"code","source":"count_vectorizer = CountVectorizer(strip_accents='unicode',\n                                   stop_words='english',\n                                   ngram_range=(1,2),\n                                   max_df=0.9)\n\n\ntfidf_vectorizer = TfidfVectorizer(strip_accents='unicode',stop_words='english', max_df=0.9)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:12.127293Z","iopub.execute_input":"2022-08-10T18:11:12.127565Z","iopub.status.idle":"2022-08-10T18:11:12.132094Z","shell.execute_reply.started":"2022-08-10T18:11:12.127533Z","shell.execute_reply":"2022-08-10T18:11:12.131130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_pipeline(vectorizer=CountVectorizer(),\n                   score_fn=chi2,\n                   topK=10000,\n                   model=MultinomialNB(),\n                   feature='Processed'):\n    \n    feature = feature\n    \n    feature_transformations = Pipeline(steps=[\n        ('vectorizer', vectorizer),\n        ('select',SelectKBest(score_func=score_fn, k=topK))\n    ])\n\n    transformer = ColumnTransformer(transformers=[\n        ('text', feature_transformations, feature)\n    ])\n    \n    pipe = Pipeline(steps=[\n        ('preprocessing',transformer),('modeling',model)\n    ])\n    \n    return pipe","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:12.133536Z","iopub.execute_input":"2022-08-10T18:11:12.133991Z","iopub.status.idle":"2022-08-10T18:11:12.151250Z","shell.execute_reply.started":"2022-08-10T18:11:12.133950Z","shell.execute_reply":"2022-08-10T18:11:12.150102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_list = [\n    (MultinomialNB(alpha=0.01), 'Multinomial Naive Bayes'),\n    (BernoulliNB(alpha=0.01), 'Bernoulli Naive Bayes'),\n    (ComplementNB(alpha=0.1), 'Complement Naive Bayes'),\n    \n    (PassiveAggressiveClassifier(max_iter=100, early_stopping=True), \"Passive-Aggressive\"),\n    \n    (LinearSVC(penalty='l1', dual=False, tol=1e-3),'LinearSVC with penality = l1'),\n    (LinearSVC(penalty='l2', dual=False, tol=1e-3),'LinearSVC with penality = l2'),\n    \n    (SGDClassifier(alpha=0.0001, max_iter=100, penalty='l1', early_stopping=True),'SGD Classifier with penality = l1'),\n    (SGDClassifier(alpha=0.0001, max_iter=100, penalty='l2', early_stopping=True),'SGD Classifier with penality = l2'),\n    (SGDClassifier(alpha=0.0001, max_iter=100, penalty='elasticnet',early_stopping=True),'SGD Classifier with penality = elasticnet'),\n    \n]\n\ndef score(X,y,**kwargs):\n    for model, name in model_list:\n        pipe = build_pipeline(model=model,**kwargs)\n        score = cross_val_score(pipe, X, y, cv=5, scoring='accuracy')\n        print(f'{name :35} : {score.mean()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:12.153090Z","iopub.execute_input":"2022-08-10T18:11:12.153625Z","iopub.status.idle":"2022-08-10T18:11:12.164665Z","shell.execute_reply.started":"2022-08-10T18:11:12.153588Z","shell.execute_reply":"2022-08-10T18:11:12.163836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model performance","metadata":{}},{"cell_type":"code","source":"# let check the performance of different models on raw data\n\nscore(X=frame, y=y, feature='Raw')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:12.166058Z","iopub.execute_input":"2022-08-10T18:11:12.166291Z","iopub.status.idle":"2022-08-10T18:11:25.661405Z","shell.execute_reply.started":"2022-08-10T18:11:12.166260Z","shell.execute_reply":"2022-08-10T18:11:25.660405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### check the performance of different models using preprocessed data","metadata":{}},{"cell_type":"code","source":"# processed data : remove numbers, html , url, punctuations etc..\nscore(X=frame, y=y, feature='Processed', topK=10000)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:25.662793Z","iopub.execute_input":"2022-08-10T18:11:25.663504Z","iopub.status.idle":"2022-08-10T18:11:34.985505Z","shell.execute_reply.started":"2022-08-10T18:11:25.663465Z","shell.execute_reply":"2022-08-10T18:11:34.984504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# performance on lemmatization\nscore(X=frame, y=y, feature='Lemmatized', topK=10000)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:34.987118Z","iopub.execute_input":"2022-08-10T18:11:34.987600Z","iopub.status.idle":"2022-08-10T18:11:43.478793Z","shell.execute_reply.started":"2022-08-10T18:11:34.987555Z","shell.execute_reply":"2022-08-10T18:11:43.477960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score(X=frame, y=y, feature='Stemming', topK=8000)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:43.480188Z","iopub.execute_input":"2022-08-10T18:11:43.480421Z","iopub.status.idle":"2022-08-10T18:11:50.971354Z","shell.execute_reply.started":"2022-08-10T18:11:43.480393Z","shell.execute_reply":"2022-08-10T18:11:50.970465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction and Submission","metadata":{}},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:50.972704Z","iopub.execute_input":"2022-08-10T18:11:50.974516Z","iopub.status.idle":"2022-08-10T18:11:50.986300Z","shell.execute_reply.started":"2022-08-10T18:11:50.974467Z","shell.execute_reply":"2022-08-10T18:11:50.985307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# processing the test set \n\ntest_df['Processed'] = process_data(test_df['text'])\ntest_df['Lemmatized'] = test_df['Processed'].apply(lemmatize)\ntest_df['Stemming'] = test_df['Processed'].apply(stemming)\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:50.988435Z","iopub.execute_input":"2022-08-10T18:11:50.988686Z","iopub.status.idle":"2022-08-10T18:11:53.330721Z","shell.execute_reply.started":"2022-08-10T18:11:50.988656Z","shell.execute_reply":"2022-08-10T18:11:53.329752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature = 'Lemmatized'\n\n# Pipeline\nmodel = ComplementNB()\nvectors = count_vectorizer.fit_transform(frame[feature])\nmodel.fit(vectors, y)\n\ntest_vector = count_vectorizer.transform(test_df[feature])\npreds = model.predict(test_vector)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:20:40.646060Z","iopub.execute_input":"2022-08-10T18:20:40.646344Z","iopub.status.idle":"2022-08-10T18:20:41.057383Z","shell.execute_reply.started":"2022-08-10T18:20:40.646299Z","shell.execute_reply":"2022-08-10T18:20:41.056708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df['target']=preds\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:20:47.698996Z","iopub.execute_input":"2022-08-10T18:20:47.699576Z","iopub.status.idle":"2022-08-10T18:20:47.711573Z","shell.execute_reply.started":"2022-08-10T18:20:47.699536Z","shell.execute_reply":"2022-08-10T18:20:47.710431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T18:11:53.542156Z","iopub.execute_input":"2022-08-10T18:11:53.542570Z","iopub.status.idle":"2022-08-10T18:11:53.557540Z","shell.execute_reply.started":"2022-08-10T18:11:53.542527Z","shell.execute_reply":"2022-08-10T18:11:53.556869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}