{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## <a id=\"3\"></a>\n\n### In this notebook, we are going to make a simple email filtering system. During the course of this notebook, we will use some common NLP techniques.","metadata":{}},{"cell_type":"markdown","source":"### <center>Give me some confidence by upvoting this notebook!🙂<center>","metadata":{}},{"cell_type":"markdown","source":"## Apparatus Required","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn import metrics\nimport seaborn as sns\nfrom wordcloud import WordCloud\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import MultinomialNB, GaussianNB\nfrom sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:13.913592Z","iopub.execute_input":"2022-10-15T10:57:13.914496Z","iopub.status.idle":"2022-10-15T10:57:15.257873Z","shell.execute_reply.started":"2022-10-15T10:57:13.914398Z","shell.execute_reply":"2022-10-15T10:57:15.256911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/sms-spam-collection-dataset/spam.csv', encoding='latin-1')","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:15.259609Z","iopub.execute_input":"2022-10-15T10:57:15.260151Z","iopub.status.idle":"2022-10-15T10:57:15.299197Z","shell.execute_reply.started":"2022-10-15T10:57:15.260119Z","shell.execute_reply":"2022-10-15T10:57:15.298339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Basic Data Analysis","metadata":{}},{"cell_type":"code","source":"print(df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:15.300696Z","iopub.execute_input":"2022-10-15T10:57:15.301275Z","iopub.status.idle":"2022-10-15T10:57:15.306955Z","shell.execute_reply.started":"2022-10-15T10:57:15.301230Z","shell.execute_reply":"2022-10-15T10:57:15.305600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:15.310071Z","iopub.execute_input":"2022-10-15T10:57:15.310737Z","iopub.status.idle":"2022-10-15T10:57:15.336215Z","shell.execute_reply.started":"2022-10-15T10:57:15.310705Z","shell.execute_reply":"2022-10-15T10:57:15.335073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Columns Meaning\n\n* v1 is our label column. Which represent wether the email is spam or not spam.\n* v2 column contains the text of the emails.\n* Other columns are not important.","metadata":{}},{"cell_type":"markdown","source":"### Dropping the unnecessary columns.","metadata":{}},{"cell_type":"code","source":"df = df.drop(columns = ['Unnamed: 2', 'Unnamed: 3', 'Unnamed: 4'])","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:15.337439Z","iopub.execute_input":"2022-10-15T10:57:15.337764Z","iopub.status.idle":"2022-10-15T10:57:15.348734Z","shell.execute_reply.started":"2022-10-15T10:57:15.337738Z","shell.execute_reply":"2022-10-15T10:57:15.347544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordcloud = WordCloud(background_color='white', width=800, height=400).generate(''.join(df.v2))\n\nplt.figure(figsize=(20, 5))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:15.349978Z","iopub.execute_input":"2022-10-15T10:57:15.350458Z","iopub.status.idle":"2022-10-15T10:57:16.693829Z","shell.execute_reply.started":"2022-10-15T10:57:15.350422Z","shell.execute_reply":"2022-10-15T10:57:16.692726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = df[\"v1\"], data = df)","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.695546Z","iopub.execute_input":"2022-10-15T10:57:16.696238Z","iopub.status.idle":"2022-10-15T10:57:16.902946Z","shell.execute_reply.started":"2022-10-15T10:57:16.696181Z","shell.execute_reply":"2022-10-15T10:57:16.901765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"v1\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.904742Z","iopub.execute_input":"2022-10-15T10:57:16.905489Z","iopub.status.idle":"2022-10-15T10:57:16.916554Z","shell.execute_reply.started":"2022-10-15T10:57:16.905444Z","shell.execute_reply":"2022-10-15T10:57:16.915220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# It seems that the given dataset is imbalanced.\n\n4825 // 747","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.917995Z","iopub.execute_input":"2022-10-15T10:57:16.918393Z","iopub.status.idle":"2022-10-15T10:57:16.928911Z","shell.execute_reply.started":"2022-10-15T10:57:16.918361Z","shell.execute_reply":"2022-10-15T10:57:16.927655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Transform the values of the output variable into 0 and 1.","metadata":{}},{"cell_type":"code","source":"df['v1'] = df[\"v1\"].map({'spam':1,'ham':0})","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.932632Z","iopub.execute_input":"2022-10-15T10:57:16.932995Z","iopub.status.idle":"2022-10-15T10:57:16.940517Z","shell.execute_reply.started":"2022-10-15T10:57:16.932966Z","shell.execute_reply":"2022-10-15T10:57:16.939397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let us balance the dataset first.","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\n# create two different dataframe of majority and minority class \n\ndf_majority = df[(df['v1'] == 0)] \n\ndf_minority = df[(df['v1'] == 1)] \n\n# upsample minority class\ndf_minority_upsampled = resample(df_minority,\n                                 \nreplace = True,    # sample with replacement  \n                                 \n n_samples = 4825, # to match majority class     \n                                 \n random_state = 42) \n\n# reproducible results\n    \n# Combine majority class with upsampled minority class\ndf_upsampled = pd.concat([df_majority, df_minority_upsampled])","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.942411Z","iopub.execute_input":"2022-10-15T10:57:16.943346Z","iopub.status.idle":"2022-10-15T10:57:16.955673Z","shell.execute_reply.started":"2022-10-15T10:57:16.943294Z","shell.execute_reply":"2022-10-15T10:57:16.954622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The given dataset doesn't contain any missing values as shown below.","metadata":{}},{"cell_type":"code","source":"df_upsampled.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.957077Z","iopub.execute_input":"2022-10-15T10:57:16.957510Z","iopub.status.idle":"2022-10-15T10:57:16.969612Z","shell.execute_reply.started":"2022-10-15T10:57:16.957473Z","shell.execute_reply":"2022-10-15T10:57:16.968318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Converting text data into vectors.\n\nMachine learning models don't understand textual data. So we have to convert it into numerical form. There are many methods for doing this. For instance, Bag of words, TF-IDF, One-Hot encoding, Word Embedding, etc. We will use a bag of words because it is very intuitive.","metadata":{}},{"cell_type":"markdown","source":"## Bag of words:\n\n In this model, a text (such as a sentence or a document) is represented as the bag (multiset) of its words, disregarding grammar and even word order but keeping multiplicity.The bag-of-words model is commonly used in methods of document classification where the (frequency of) occurrence of each word is used as a feature for training a classifier.\n \n ## Example Below","metadata":{}},{"cell_type":"code","source":"# Let's try to convert these text into vectors using bag of words\n\ntext = ['Hello my name is james', 'james this is my python notebook', 'james trying to create a big dataset', 'james of words to try differnt', 'features of count vectorizer']\n\nvectorizer = CountVectorizer(stop_words='english')\n\ncount_matrix = vectorizer.fit_transform(text)\n\ncount_array = count_matrix.toarray()\n\ndf1 = pd.DataFrame(data = count_array,columns = vectorizer.get_feature_names())\n\ndf1","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.971240Z","iopub.execute_input":"2022-10-15T10:57:16.972400Z","iopub.status.idle":"2022-10-15T10:57:16.994544Z","shell.execute_reply.started":"2022-10-15T10:57:16.972336Z","shell.execute_reply":"2022-10-15T10:57:16.993337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Converting email texts into vectors.","metadata":{}},{"cell_type":"code","source":"# Remove the stop words and transform the texts into the vectorized input variables X\n\nX = vectorizer.fit_transform(df[\"v2\"])\n\ny = df[\"v1\"]\n\n# Split the data into train and test sets\n\nX_train, X_test, y_train, y_test = train_test_split(X.toarray(), y, test_size=0.3, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:16.996254Z","iopub.execute_input":"2022-10-15T10:57:16.996998Z","iopub.status.idle":"2022-10-15T10:57:17.439284Z","shell.execute_reply.started":"2022-10-15T10:57:16.996955Z","shell.execute_reply":"2022-10-15T10:57:17.438062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model","metadata":{}},{"cell_type":"code","source":"clf = GaussianNB()\n\nclf.fit(X_train, y_train)\n\nclf.score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-15T10:57:17.441006Z","iopub.execute_input":"2022-10-15T10:57:17.441479Z","iopub.status.idle":"2022-10-15T10:57:18.257160Z","shell.execute_reply.started":"2022-10-15T10:57:17.441425Z","shell.execute_reply":"2022-10-15T10:57:18.256319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The accuracy of the model is pretty good. It is almsot 89% 🔥.\n\n### <center>Please upvote the notebook if you found it usefull.<center>","metadata":{}}]}