{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The objective of this project is to classify the data into Sincere and Insincere questions\n## Feature Extraction Data Techniques Used\n### Count Vectorizer with Logistic Regression and Naive Bayes\n### Tfidf Vectorizer with Logistic Regression and Naive Bayes\n### HashingVectorizer with Logistic Regression\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Importing the libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport string\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette\n\n%matplotlib inline\n\nfrom plotly import tools, subplots\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:54:37.606791Z","iopub.execute_input":"2021-09-10T13:54:37.607163Z","iopub.status.idle":"2021-09-10T13:54:37.615934Z","shell.execute_reply.started":"2021-09-10T13:54:37.607133Z","shell.execute_reply":"2021-09-10T13:54:37.615143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Overview","metadata":{}},{"cell_type":"markdown","source":"### Available input files","metadata":{}},{"cell_type":"code","source":"ls ../input/quora-insincere-questions-classification","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:20:49.558085Z","iopub.execute_input":"2021-09-10T13:20:49.558477Z","iopub.status.idle":"2021-09-10T13:20:50.202881Z","shell.execute_reply.started":"2021-09-10T13:20:49.558440Z","shell.execute_reply":"2021-09-10T13:20:50.201807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Unzipping the embeddings","metadata":{}},{"cell_type":"code","source":"from zipfile import ZipFile \n!unzip ../input/quora-insincere-questions-classification/embeddings","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:20:50.206873Z","iopub.execute_input":"2021-09-10T13:20:50.207165Z","iopub.status.idle":"2021-09-10T13:24:18.696910Z","shell.execute_reply.started":"2021-09-10T13:20:50.207137Z","shell.execute_reply":"2021-09-10T13:24:18.696025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:18.698808Z","iopub.execute_input":"2021-09-10T13:24:18.699247Z","iopub.status.idle":"2021-09-10T13:24:19.352159Z","shell.execute_reply.started":"2021-09-10T13:24:18.699204Z","shell.execute_reply":"2021-09-10T13:24:19.351161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Loading the data into training data and test data","metadata":{}},{"cell_type":"code","source":"train_path = \"../input/quora-insincere-questions-classification/train.csv\"\ntest_path = \"../input/quora-insincere-questions-classification/test.csv\"\ntrain_data = pd.read_csv(train_path)\ntest_data = pd.read_csv(test_path)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:19.353705Z","iopub.execute_input":"2021-09-10T13:24:19.354082Z","iopub.status.idle":"2021-09-10T13:24:27.028224Z","shell.execute_reply.started":"2021-09-10T13:24:19.354034Z","shell.execute_reply":"2021-09-10T13:24:27.027290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get the rows and columns of the training and test dataset","metadata":{}},{"cell_type":"code","source":"print(f\"There are {train_data.shape[0]} Rows and {train_data.shape[1]} Columns inside train data\")\nprint(f\"There are {train_data.shape[0]} questions in total in the training dataset\")\nprint(f\"There are {test_data.shape[0]} Rows and {test_data.shape[1]} Columns inside test data\")\nprint(f\"There are {test_data.shape[0]} questions in total in the test dataset\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:27.029534Z","iopub.execute_input":"2021-09-10T13:24:27.029902Z","iopub.status.idle":"2021-09-10T13:24:27.038268Z","shell.execute_reply.started":"2021-09-10T13:24:27.029866Z","shell.execute_reply":"2021-09-10T13:24:27.035155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Info about the training data and test data","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:27.039395Z","iopub.execute_input":"2021-09-10T13:24:27.039868Z","iopub.status.idle":"2021-09-10T13:24:30.211732Z","shell.execute_reply.started":"2021-09-10T13:24:27.039833Z","shell.execute_reply":"2021-09-10T13:24:30.210885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:30.214676Z","iopub.execute_input":"2021-09-10T13:24:30.215052Z","iopub.status.idle":"2021-09-10T13:24:30.293208Z","shell.execute_reply.started":"2021-09-10T13:24:30.214993Z","shell.execute_reply":"2021-09-10T13:24:30.292205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Total number of sincere (0) and Insincere Questions (1)","metadata":{}},{"cell_type":"code","source":"target_count = train_data['target'].value_counts()\nprint(target_count)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:30.295449Z","iopub.execute_input":"2021-09-10T13:24:30.295831Z","iopub.status.idle":"2021-09-10T13:24:31.690662Z","shell.execute_reply.started":"2021-09-10T13:24:30.295793Z","shell.execute_reply":"2021-09-10T13:24:31.689641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bar chart to plot the target count ","metadata":{}},{"cell_type":"code","source":"target_count = train_data['target'].value_counts()\n\nbarchart_data = go.Bar(\n    x=target_count.index,\n    y=target_count.values,\n    marker=dict(\n        color=target_count.values,\n        colorscale = 'Picnic',\n        reversescale = True\n    ),\n)\n\nlayout = go.Layout(\n    title='Target Count',\n    font=dict(size=18)\n)\n\ndata = [barchart_data]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"TargetCount\")\n\n","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:31.692334Z","iopub.execute_input":"2021-09-10T13:24:31.694175Z","iopub.status.idle":"2021-09-10T13:24:32.958842Z","shell.execute_reply.started":"2021-09-10T13:24:31.694130Z","shell.execute_reply":"2021-09-10T13:24:32.957966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bar chart for target distribution","metadata":{}},{"cell_type":"code","source":"# target distribution\nlabels = (np.array(target_count.index))\nsizes = (np.array((target_count / target_count.sum())*100))\n\npiechart_trace = go.Pie(labels=labels, values=sizes)\nlayout = go.Layout(\n    title='Target distribution',\n    font=dict(size=18),\n    width=600,\n    height=600,\n)\ndata = [piechart_trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"target_distribution\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:32.960122Z","iopub.execute_input":"2021-09-10T13:24:32.960453Z","iopub.status.idle":"2021-09-10T13:24:33.906792Z","shell.execute_reply.started":"2021-09-10T13:24:32.960423Z","shell.execute_reply":"2021-09-10T13:24:33.905960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Word Cloud before Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Word Cloud for all training questions\nfrom wordcloud import WordCloud, STOPWORDS\n\n# custom function for plotting the word cloud\ndef plot_wordcloud(text, mask=None, max_words=200, max_font_size=100, figure_size=(24.0,16.0), \n                   title = None, title_size=40, image_color=False):\n    stopwords = set(STOPWORDS)\n    more_stopwords = {'one', 'br', 'Po', 'th', 'sayi', 'fo', 'Unknown'}\n    stopwords = stopwords.union(more_stopwords)\n\n    wordcloud = WordCloud(background_color='black',\n                    stopwords = stopwords,\n                    max_words = max_words,\n                    max_font_size = max_font_size, \n                    random_state = 42,\n                    width=800, \n                    height=400,\n                    mask = mask)\n    wordcloud.generate(str(text))\n    \n    plt.figure(figsize=figure_size)\n    \n    plt.imshow(wordcloud);\n    plt.title(title, fontdict={'size': title_size, 'color': 'black', \n                                  'verticalalignment': 'bottom'})\n    plt.axis('off');\n    plt.tight_layout()  \n    \nplot_wordcloud(train_data[\"question_text\"], title=\"Word Cloud of Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:33.908073Z","iopub.execute_input":"2021-09-10T13:24:33.908575Z","iopub.status.idle":"2021-09-10T13:24:35.344296Z","shell.execute_reply.started":"2021-09-10T13:24:33.908529Z","shell.execute_reply":"2021-09-10T13:24:35.340084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Word cloud for sincere questions Before Data Preprocessing\nplot_wordcloud(train_data[train_data[\"target\"] == 0][\"question_text\"], title=\"Word Cloud of Sincere Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:35.345544Z","iopub.execute_input":"2021-09-10T13:24:35.345856Z","iopub.status.idle":"2021-09-10T13:24:36.210485Z","shell.execute_reply.started":"2021-09-10T13:24:35.345825Z","shell.execute_reply":"2021-09-10T13:24:36.203985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Word cloud for insincere questions before  Data Preprocessing\nplot_wordcloud(train_data[train_data[\"target\"] == 1][\"question_text\"], title=\"Word Cloud of Insincere Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:36.211912Z","iopub.execute_input":"2021-09-10T13:24:36.212294Z","iopub.status.idle":"2021-09-10T13:24:36.990098Z","shell.execute_reply.started":"2021-09-10T13:24:36.212258Z","shell.execute_reply":"2021-09-10T13:24:36.988994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Horizontal bar chart of frequently asked questions on both classes\n","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\n# Separate the sincere questions from training dataset\ntrain_sincere_data = train_data[train_data[\"target\"] == 0]\ntrain_insincere_data = train_data[train_data[\"target\"] == 1]\n\n# Next step is generating barchart for both the classes\ndef generate_ngrams(text, n_gram=1):\n    token = [token for token in text.lower().split(\" \") if token != \"\" if token not in STOPWORDS]\n    ngrams = zip(*[token[i:] for i in range(n_gram)])\n    return [\" \".join(ngram) for ngram in ngrams]\n\n# custom function for horizontal bar chart\ndef horizontal_bar_chart(df, color):\n    trace = go.Bar(\n        y=df[\"word\"].values[::-1],\n        x=df[\"wordcount\"].values[::-1],\n        showlegend=False,\n        orientation = 'h',\n        marker=dict(\n            color=color,\n        ),\n    )\n    return trace\n\n# Bar chart for frequent words sincere questions\nsincere_dict = defaultdict(int)\nfor text in train_sincere_data[\"question_text\"]:\n    for word in generate_ngrams(text):\n        sincere_dict[word] += 1\n        \nsincere_dict_sorted = pd.DataFrame(sorted(sincere_dict.items(), key=lambda x: x[1])[::-1])\nsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_sincere = horizontal_bar_chart(sincere_dict_sorted.head(50), 'red')\n\n# Bar chart for frequent words insincere questions\ninsincere_dict = defaultdict(int)\nfor text in train_insincere_data[\"question_text\"]:\n    for word in generate_ngrams(text):\n        insincere_dict[word] += 1\ninsincere_dict_sorted = pd.DataFrame(sorted(insincere_dict.items(), key=lambda x: x[1])[::-1])\ninsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_insincere = horizontal_bar_chart(insincere_dict_sorted.head(50), 'red')\n\n# Creating two subplots\nfig = subplots.make_subplots(rows=1, cols=2, vertical_spacing=0.04,\n                          subplot_titles=[\"Frequent words of sincere questions\", \n                                          \"Frequent words of insincere questions\"])\nfig.append_trace(trace_sincere, 1, 1)\nfig.append_trace(trace_insincere, 1, 2)\nfig['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\npy.iplot(fig, filename='word-plots')","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:36.991626Z","iopub.execute_input":"2021-09-10T13:24:36.991969Z","iopub.status.idle":"2021-09-10T13:24:48.640561Z","shell.execute_reply.started":"2021-09-10T13:24:36.991927Z","shell.execute_reply":"2021-09-10T13:24:48.639711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bigram Bar chart plot","metadata":{}},{"cell_type":"code","source":"# Bar chart for frequent words sincere questions\nsincere_dict = defaultdict(int)\nfor text in train_sincere_data[\"question_text\"]:\n    for word in generate_ngrams(text, 2):\n        sincere_dict[word] += 1\n        \nsincere_dict_sorted = pd.DataFrame(sorted(sincere_dict.items(), key=lambda x: x[1])[::-1])\nsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_sincere = horizontal_bar_chart(sincere_dict_sorted.head(50), 'blue')\n\n# Bar chart for frequent words insincere questions\ninsincere_dict = defaultdict(int)\nfor text in train_insincere_data[\"question_text\"]:\n    for word in generate_ngrams(text, 2):\n        insincere_dict[word] += 1\ninsincere_dict_sorted = pd.DataFrame(sorted(insincere_dict.items(), key=lambda x: x[1])[::-1])\ninsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_insincere = horizontal_bar_chart(insincere_dict_sorted.head(50), 'blue')\n\n# Creating two subplots\nfig = subplots.make_subplots(rows=1, cols=2, vertical_spacing=0.04,\n                          subplot_titles=[\"Frequent words of sincere questions\", \n                                          \"Frequent words of insincere questions\"])\nfig.append_trace(trace_sincere, 1, 1)\nfig.append_trace(trace_insincere, 1, 2)\nfig['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\npy.iplot(fig, filename='word-plots')","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:24:48.642013Z","iopub.execute_input":"2021-09-10T13:24:48.642398Z","iopub.status.idle":"2021-09-10T13:25:05.342790Z","shell.execute_reply.started":"2021-09-10T13:24:48.642362Z","shell.execute_reply":"2021-09-10T13:25:05.341929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Trigram","metadata":{}},{"cell_type":"code","source":"# Bar chart for frequent words sincere questions\nsincere_dict = defaultdict(int)\nfor text in train_sincere_data[\"question_text\"]:\n    for word in generate_ngrams(text, 3):\n        sincere_dict[word] += 1\n        \nsincere_dict_sorted = pd.DataFrame(sorted(sincere_dict.items(), key=lambda x: x[1])[::-1])\nsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_sincere = horizontal_bar_chart(sincere_dict_sorted.head(50), 'green')\n\n# Bar chart for frequent words insincere questions\ninsincere_dict = defaultdict(int)\nfor text in train_insincere_data[\"question_text\"]:\n    for word in generate_ngrams(text, 3):\n        insincere_dict[word] += 1\ninsincere_dict_sorted = pd.DataFrame(sorted(insincere_dict.items(), key=lambda x: x[1])[::-1])\ninsincere_dict_sorted.columns = [\"word\", \"wordcount\"]\ntrace_insincere = horizontal_bar_chart(insincere_dict_sorted.head(50), 'green')\n\n# Creating two subplots\nfig = subplots.make_subplots(rows=1, cols=2, vertical_spacing=0.04, horizontal_spacing=0.2,\n                          subplot_titles=[\"Frequent words of sincere questions\", \n                                          \"Frequent words of insincere questions\"])\nfig.append_trace(trace_sincere, 1, 1)\nfig.append_trace(trace_insincere, 1, 2)\nfig['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\npy.iplot(fig, filename='word-plots')","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:25:05.344193Z","iopub.execute_input":"2021-09-10T13:25:05.344558Z","iopub.status.idle":"2021-09-10T13:25:22.264284Z","shell.execute_reply.started":"2021-09-10T13:25:05.344522Z","shell.execute_reply":"2021-09-10T13:25:22.262722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering\n### Features that is added\n1. Number of words in the text\n2. Number of unique words in the text\n3. Number of characters in the text\n4. Number of stopwords\n5. Number of punctuations\n6. Number of upper case words\n7. Number of title case words\n8. Average length of the words","metadata":{}},{"cell_type":"code","source":"# Number of words in the text\ntrain_data[\"num_words\"] = train_data[\"question_text\"].apply(lambda x: len(str(x).split()))\ntest_data[\"num_words\"] = test_data[\"question_text\"].apply(lambda x: len(str(x).split()))\n\n# Number of unique words in the text\ntrain_data[\"num_unique_words\"] = train_data[\"question_text\"].apply(lambda x: len(set(str(x).split())))\ntest_data[\"num_unique_words\"] = test_data[\"question_text\"].apply(lambda x: len(set(str(x).split())))\n\n# Number of characters in the text\ntrain_data[\"num_chars\"] = train_data[\"question_text\"].apply(lambda x: len(str(x)))\ntest_data[\"num_chars\"] = test_data[\"question_text\"].apply(lambda x: len(str(x)))\n\n# Number of stopwords in the text\ntrain_data[\"num_stopwords\"] = train_data[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\ntest_data[\"num_stopwords\"] = test_data[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\n\n# Number of punctuations in the text\ntrain_data[\"num_punctuations\"] = train_data['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )\ntest_data[\"num_punctuations\"] = test_data['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )\n\n# Number of title case words in the text\ntrain_data[\"num_words_upper\"] = train_data[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\ntest_data[\"num_words_upper\"] = test_data[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\n\n# Number of title case words in the text\ntrain_data[\"num_words_title\"] = train_data[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\ntest_data[\"num_words_title\"] = test_data[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n\n# Average length of the words in the text\ntrain_data[\"mean_word_len\"] = train_data[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\ntest_data[\"mean_word_len\"] = test_data[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:25:22.266481Z","iopub.execute_input":"2021-09-10T13:25:22.266849Z","iopub.status.idle":"2021-09-10T13:26:26.359138Z","shell.execute_reply.started":"2021-09-10T13:25:22.266812Z","shell.execute_reply":"2021-09-10T13:26:26.358273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Box plot for truncated features","metadata":{}},{"cell_type":"code","source":"## Truncate some extreme values for better visuals ##\ntrain_data['num_words'].loc[train_data['num_words']>60] = 60 #truncation for better visuals\ntrain_data['num_punctuations'].loc[train_data['num_punctuations']>10] = 10 #truncation for better visuals\ntrain_data['num_chars'].loc[train_data['num_chars']>350] = 350 #truncation for better visuals\n\nf, axes = plt.subplots(3, 1, figsize=(10,20))\nsns.boxplot(x='target', y='num_words', data=train_data, ax=axes[0])\naxes[0].set_xlabel('Target', fontsize=12)\naxes[0].set_title(\"Number of words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_chars', data=train_data, ax=axes[1])\naxes[1].set_xlabel('Target', fontsize=12)\naxes[1].set_title(\"Number of characters in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_punctuations', data=train_data, ax=axes[2])\naxes[2].set_xlabel('Target', fontsize=12)\n#plt.ylabel('Number of punctuations in text', fontsize=12)\naxes[2].set_title(\"Number of punctuations in each class\", fontsize=15)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:54:52.356867Z","iopub.execute_input":"2021-09-10T13:54:52.357266Z","iopub.status.idle":"2021-09-10T13:54:53.568667Z","shell.execute_reply.started":"2021-09-10T13:54:52.357234Z","shell.execute_reply":"2021-09-10T13:54:53.563902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SHowing that the features are added\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.433939Z","iopub.execute_input":"2021-09-10T13:26:27.434292Z","iopub.status.idle":"2021-09-10T13:26:27.453672Z","shell.execute_reply.started":"2021-09-10T13:26:27.434257Z","shell.execute_reply":"2021-09-10T13:26:27.452927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the training data after feature scaling\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.454902Z","iopub.execute_input":"2021-09-10T13:26:27.455264Z","iopub.status.idle":"2021-09-10T13:26:27.707407Z","shell.execute_reply.started":"2021-09-10T13:26:27.455228Z","shell.execute_reply":"2021-09-10T13:26:27.706378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing\n### Agenda\n1. Converting questions to lower case\n2. Removing the punctuation marks\n3. Cleaning numbers\n4. Correcting misspelled words\n5. removing contractions\n6. Removing stop words","metadata":{}},{"cell_type":"markdown","source":"### Removing the punctuation marks","metadata":{}},{"cell_type":"code","source":"# Removing punctuations\npunctuation_list =[',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.709077Z","iopub.execute_input":"2021-09-10T13:26:27.709478Z","iopub.status.idle":"2021-09-10T13:26:27.743429Z","shell.execute_reply.started":"2021-09-10T13:26:27.709436Z","shell.execute_reply":"2021-09-10T13:26:27.742305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_punctuation(text):\n    for punctuation in punctuation_list:\n        if punctuation in text:\n            text = text.replace(punctuation, '{}' .format(punctuation))\n    return text","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.748191Z","iopub.execute_input":"2021-09-10T13:26:27.748528Z","iopub.status.idle":"2021-09-10T13:26:27.755240Z","shell.execute_reply.started":"2021-09-10T13:26:27.748491Z","shell.execute_reply":"2021-09-10T13:26:27.754494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cleaning numbers","metadata":{}},{"cell_type":"code","source":"def clean_numbers(text):\n    if bool(re.search(r'\\d', text)):\n        text = re.sub('[0-9]{5,}', '#####', text)\n        text = re.sub('[0-9]{4}', '####', text)\n        text = re.sub('[0-9]{3}', '###', text)\n        text = re.sub('[0-9]{2}', '##', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.757558Z","iopub.execute_input":"2021-09-10T13:26:27.758063Z","iopub.status.idle":"2021-09-10T13:26:27.765692Z","shell.execute_reply.started":"2021-09-10T13:26:27.757995Z","shell.execute_reply":"2021-09-10T13:26:27.764858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correcting misspelled words","metadata":{}},{"cell_type":"code","source":"mispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.766892Z","iopub.execute_input":"2021-09-10T13:26:27.767468Z","iopub.status.idle":"2021-09-10T13:26:27.782843Z","shell.execute_reply.started":"2021-09-10T13:26:27.767430Z","shell.execute_reply":"2021-09-10T13:26:27.782029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef get_misspelled_dict_and_regex(mispell_dict):\n    mispell_re = re.compile('(%s)' % '|'.join(mispell_dict.keys()))\n    return mispell_dict, mispell_re\n\nmispellings, mispellings_re = get_misspelled_dict_and_regex(mispell_dict)\ndef replace_typical_misspell(text):\n    def replace(match):\n        return mispellings[match.group(0)]\n    return mispellings_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.784336Z","iopub.execute_input":"2021-09-10T13:26:27.784836Z","iopub.status.idle":"2021-09-10T13:26:27.798701Z","shell.execute_reply.started":"2021-09-10T13:26:27.784711Z","shell.execute_reply":"2021-09-10T13:26:27.797952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Removing Contractions","metadata":{}},{"cell_type":"code","source":"contraction_dict = {\n    \"ain't\": \"is not\", \n    \"aren't\": \"are not\",\n    \"can't\": \"cannot\", \n    \"'cause\": \"because\", \n    \"could've\": \"could have\", \n    \"couldn't\": \"could not\", \n    \"didn't\": \"did not\",  \n    \"doesn't\": \"does not\", \n    \"don't\": \"do not\", \n    \"hadn't\": \"had not\", \n    \"hasn't\": \"has not\", \n    \"haven't\": \"have not\", \n    \"he'd\": \"he would\",\n    \"he'll\": \"he will\", \n    \"he's\": \"he is\", \n    \"how'd\": \"how did\", \n    \"how'd'y\": \"how do you\", \n    \"how'll\": \"how will\", \n    \"how's\": \"how is\",  \n    \"I'd\": \"I would\", \n    \"I'd've\": \"I would have\",\n    \"I'll\": \"I will\", \n    \"I'll've\": \"I will have\",\n    \"I'm\": \"I am\", \n    \"I've\": \"I have\", \n    \"i'd\": \"i would\", \n    \"i'd've\": \"i would have\", \n    \"i'll\": \"i will\",  \n    \"i'll've\": \"i will have\",\n    \"i'm\": \"i am\", \n    \"i've\": \"i have\", \n    \"isn't\": \"is not\", \n    \"it'd\": \"it would\", \n    \"it'd've\": \"it would have\", \n    \"it'll\": \"it will\", \n    \"it'll've\": \"it will have\",\n    \"it's\": \"it is\", \n    \"let's\": \"let us\", \n    \"ma'am\": \"madam\", \n    \"mayn't\": \"may not\", \n    \"might've\": \"might have\",\n    \"mightn't\": \"might not\",\n    \"mightn't've\": \"might not have\", \n    \"must've\": \"must have\", \n    \"mustn't\": \"must not\", \n    \"mustn't've\": \"must not have\", \n    \"needn't\": \"need not\", \n    \"needn't've\": \"need not have\",\n    \"o'clock\": \"of the clock\", \n    \"oughtn't\": \"ought not\", \n    \"oughtn't've\": \"ought not have\", \n    \"shan't\": \"shall not\", \n    \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \n    \"she'd\": \"she would\", \"she'd've\": \"she would have\", \n    \"she'll\": \"she will\", \"she'll've\": \"she will have\", \n    \"she's\": \"she is\", \"should've\": \"should have\", \n    \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \n    \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\n    \"that'd\": \"that would\", \"that'd've\": \"that would have\", \n    \"that's\": \"that is\", \"there'd\": \"there would\", \n    \"there'd've\": \"there would have\", \"there's\": \"there is\", \n    \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \n    \"they'll\": \"they will\", \"they'll've\": \"they will have\", \n    \"they're\": \"they are\", \"they've\": \"they have\", \n    \"to've\": \"to have\", \"wasn't\": \"was not\", \n    \"we'd\": \"we would\", \"we'd've\": \"we would have\", \n    \"we'll\": \"we will\", \"we'll've\": \"we will have\", \n    \"we're\": \"we are\", \"we've\": \"we have\", \n    \"weren't\": \"were not\", \"what'll\": \"what will\", \n    \"what'll've\": \"what will have\", \"what're\": \"what are\",  \n    \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \n    \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \n    \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \n    \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \n    \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \n    \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \n    \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\n    \"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \n    \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \n    \"you're\": \"you are\", \"you've\": \"you have\"}","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.800210Z","iopub.execute_input":"2021-09-10T13:26:27.800596Z","iopub.status.idle":"2021-09-10T13:26:27.815618Z","shell.execute_reply.started":"2021-09-10T13:26:27.800562Z","shell.execute_reply":"2021-09-10T13:26:27.814630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_contractions_dict_and_regex(contraction_dict):\n    contraction_re = re.compile('(%s)' % '|'.join(contraction_dict.keys()))\n    return contraction_dict, contraction_re\n\ncontractions, contractions_re = get_contractions_dict_and_regex(contraction_dict)\n\ndef replace_contractions(text):\n    def replace(match):\n        return contractions[match.group(0)]\n    return contractions_re.sub(replace, text)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.817259Z","iopub.execute_input":"2021-09-10T13:26:27.817691Z","iopub.status.idle":"2021-09-10T13:26:27.829697Z","shell.execute_reply.started":"2021-09-10T13:26:27.817627Z","shell.execute_reply":"2021-09-10T13:26:27.828825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Removing stopwords","metadata":{}},{"cell_type":"code","source":"import nltk\nfrom nltk.tokenize.toktok import ToktokTokenizer\nstopword_list = nltk.corpus.stopwords.words('english')\ndef remove_stopwords(text, is_lower_case=True):\n    tokenizer = ToktokTokenizer()\n    tokens = tokenizer.tokenize(text)\n    tokens = [token.strip() for token in tokens]\n    if is_lower_case:\n        filtered_tokens = [token for token in tokens if token not in stopword_list]\n    else:\n        filtered_tokens = [token for token in tokens if token.lower() not in stopword_list]\n    filtered_text = ' '.join(filtered_tokens)\n    return filtered_text","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:27.830895Z","iopub.execute_input":"2021-09-10T13:26:27.831281Z","iopub.status.idle":"2021-09-10T13:26:28.508680Z","shell.execute_reply.started":"2021-09-10T13:26:27.831244Z","shell.execute_reply":"2021-09-10T13:26:28.507562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Applying all the preprocessing techniques discussed\ndef clean_questions(x):\n    x = x.lower()\n    x = remove_punctuation(x)\n    x = clean_numbers(x)\n    x = replace_typical_misspell(x)\n    x = remove_stopwords(x)\n    x = replace_contractions(x)\n    x = x.replace(\"'\",\"\")\n    return x","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:28.510221Z","iopub.execute_input":"2021-09-10T13:26:28.511420Z","iopub.status.idle":"2021-09-10T13:26:28.519774Z","shell.execute_reply.started":"2021-09-10T13:26:28.511380Z","shell.execute_reply":"2021-09-10T13:26:28.518111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Doing feature engineering again after data preprocessing","metadata":{}},{"cell_type":"code","source":"train_data['preprocessed_question_text'] = train_data['question_text'].apply(lambda x: clean_questions(x))\ntest_data['preprocessed_question_text'] = test_data['question_text'].apply(lambda x: clean_questions(x))","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:26:28.521061Z","iopub.execute_input":"2021-09-10T13:26:28.521404Z","iopub.status.idle":"2021-09-10T13:31:27.666855Z","shell.execute_reply.started":"2021-09-10T13:26:28.521367Z","shell.execute_reply":"2021-09-10T13:31:27.665990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:31:27.668226Z","iopub.execute_input":"2021-09-10T13:31:27.668586Z","iopub.status.idle":"2021-09-10T13:31:28.036372Z","shell.execute_reply.started":"2021-09-10T13:31:27.668551Z","shell.execute_reply":"2021-09-10T13:31:28.035328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Word Cloud Comparison of Insincere Questions (Before and after preprocessing)","metadata":{}},{"cell_type":"code","source":"# word cloud before preprocessing\nplot_wordcloud(train_data[train_data[\"target\"] == 1][\"question_text\"], title=\"Word Cloud of Insincere Questions Before Preprocessing\")\n\n# word cloud after preprocessing\nplot_wordcloud(train_data[train_data[\"target\"] == 1][\"preprocessed_question_text\"], title=\"Word Cloud of Insincere Questions After Preprocessing\")","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:31:28.037821Z","iopub.execute_input":"2021-09-10T13:31:28.038188Z","iopub.status.idle":"2021-09-10T13:31:29.867361Z","shell.execute_reply.started":"2021-09-10T13:31:28.038150Z","shell.execute_reply":"2021-09-10T13:31:29.866502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Building Vectorizers and models","metadata":{}},{"cell_type":"code","source":"import copy\nimport time\nfrom sklearn.metrics.classification import log_loss\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer, HashingVectorizer \nfrom sklearn.naive_bayes import MultinomialNB\n\nfrom sklearn import model_selection\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import f1_score\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:35:07.881591Z","iopub.execute_input":"2021-09-10T13:35:07.881917Z","iopub.status.idle":"2021-09-10T13:35:07.889781Z","shell.execute_reply.started":"2021-09-10T13:35:07.881888Z","shell.execute_reply":"2021-09-10T13:35:07.888766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count Vectorizer","metadata":{}},{"cell_type":"code","source":"# Creating CountVectorizer object\nvectorizer = CountVectorizer(\n    dtype=np.float32, \n    strip_accents='unicode', \n    analyzer='word',\n    token_pattern=r'\\w{1,}',\n    ngram_range=(1, 3),\n    min_df=3\n)\n# Fit the vectorizer on training data after preprocessing\nvectorizer.fit_transform(train_data['preprocessed_question_text'].values.tolist() + test_data['preprocessed_question_text'].values.tolist())\ntrain_vectorizer = vectorizer.transform(train_data['preprocessed_question_text'].values.tolist())\ntest_vectorizer = vectorizer.transform(test_data['preprocessed_question_text'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:31:29.880675Z","iopub.execute_input":"2021-09-10T13:31:29.881298Z","iopub.status.idle":"2021-09-10T13:33:16.708930Z","shell.execute_reply.started":"2021-09-10T13:31:29.881260Z","shell.execute_reply":"2021-09-10T13:33:16.708094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For storing the threshold values and f1 score\nthreshold_list = []\nbest_f1_score_list = []","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:33:16.710140Z","iopub.execute_input":"2021-09-10T13:33:16.710468Z","iopub.status.idle":"2021-09-10T13:33:16.717104Z","shell.execute_reply.started":"2021-09-10T13:33:16.710419Z","shell.execute_reply":"2021-09-10T13:33:16.716381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Custom function for building model and finding f1 score","metadata":{}},{"cell_type":"code","source":"train_y = train_data[\"target\"].values\n\ndef buildModel(train_X, train_y, test_X, test_y, test_X2, model_obj):\n    model = copy.deepcopy(model_obj)\n    model.fit(train_X, train_y)\n    pred_test_y = model.predict_proba(test_X)[:,1]\n    pred_test_y2 = model.predict_proba(test_X2)[:,1]\n    return pred_test_y, pred_test_y2, model\n\ndef best_threshold_function(val_y, pred_val_y):\n    threshold_dict = {}\n    for thresh in np.arange(0.1, 0.201, 0.01):\n        thresh = np.round(thresh, 2)\n        # Updating the dict with threshold as key and f1 score as value\n        threshold_dict[thresh] =  metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n        \n    # Finding the max key\n    best_threshold = max(threshold_dict, key=threshold_dict.get)\n    \n    # finding the max value\n    best_f1_score = max(threshold_dict.values())\n    \n    print(f\"Best F1 Score: {best_f1_score} for threshold {best_threshold}\")\n    # Appending the f1 score and threshold for count vectorizer\n    threshold_list.append(best_threshold)\n    best_f1_score_list.append(best_f1_score)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:33:16.718357Z","iopub.execute_input":"2021-09-10T13:33:16.718715Z","iopub.status.idle":"2021-09-10T13:33:16.729627Z","shell.execute_reply.started":"2021-09-10T13:33:16.718681Z","shell.execute_reply":"2021-09-10T13:33:16.728862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\n# Creating a zero list equal to the shape of training data\npred_train = np.zeros([train_data.shape[0]])\n\n# kfold with 5 n_splits\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\n\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, LogisticRegression(C=5., solver='sag'))\n    pred_full_test = pred_full_test + pred_test_y\n    \n    # Updating the pred_train list with prediction value\n    pred_train[val_index] = pred_val_y\n    \n    # appending the cv scores\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:35:19.219759Z","iopub.execute_input":"2021-09-10T13:35:19.220118Z","iopub.status.idle":"2021-09-10T13:36:36.792896Z","shell.execute_reply.started":"2021-09-10T13:35:19.220086Z","shell.execute_reply":"2021-09-10T13:36:36.792018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Naive Bayes","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\n# Creating a zero list equal to the shape of training data\npred_train = np.zeros([train_data.shape[0]])\n\n# kfold with 5 n_splits\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, MultinomialNB())\n    pred_full_test = pred_full_test + pred_test_y\n    \n    # Updating the pred_train list with prediction value\n    pred_train[val_index] = pred_val_y\n    \n    # appending the cv scores\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:37:56.478611Z","iopub.execute_input":"2021-09-10T13:37:56.478954Z","iopub.status.idle":"2021-09-10T13:37:58.027590Z","shell.execute_reply.started":"2021-09-10T13:37:56.478922Z","shell.execute_reply":"2021-09-10T13:37:58.026526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TFIDF Vextorizer","metadata":{}},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(stop_words='english', ngram_range=(1,3))\nvectorizer.fit_transform(train_data['preprocessed_question_text'].values.tolist() + test_data['preprocessed_question_text'].values.tolist())\ntrain_vectorizer = vectorizer.transform(train_data['preprocessed_question_text'].values.tolist())\ntest_vectorizer = vectorizer.transform(test_data['preprocessed_question_text'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:38:10.675621Z","iopub.execute_input":"2021-09-10T13:38:10.675948Z","iopub.status.idle":"2021-09-10T13:40:48.862313Z","shell.execute_reply.started":"2021-09-10T13:38:10.675919Z","shell.execute_reply":"2021-09-10T13:40:48.861336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\n\n# Creating a zero list equal to the shape of training data\npred_train = np.zeros([train_data.shape[0]])\n\n# kfold with 5 n_splits\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, LogisticRegression(C=5., solver='sag'))\n    pred_full_test = pred_full_test + pred_test_y\n    \n    # Updating the pred_train list with prediction value\n    pred_train[val_index] = pred_val_y\n    \n     # appending the cv scores\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:40:48.863720Z","iopub.execute_input":"2021-09-10T13:40:48.864077Z","iopub.status.idle":"2021-09-10T13:41:32.749820Z","shell.execute_reply.started":"2021-09-10T13:40:48.864034Z","shell.execute_reply":"2021-09-10T13:41:32.748905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Naive Bayes","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\npred_train = np.zeros([train_data.shape[0]])\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, MultinomialNB())\n    pred_full_test = pred_full_test + pred_test_y\n    pred_train[val_index] = pred_val_y\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:41:52.481419Z","iopub.execute_input":"2021-09-10T13:41:52.481738Z","iopub.status.idle":"2021-09-10T13:41:55.290168Z","shell.execute_reply.started":"2021-09-10T13:41:52.481711Z","shell.execute_reply":"2021-09-10T13:41:55.289252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hashing Vectorizer","metadata":{}},{"cell_type":"code","source":"vectorizer = HashingVectorizer(\n    dtype=np.float32,\n    strip_accents='unicode', \n    analyzer='word',\n    ngram_range=(1, 3),\n    n_features=2**10\n)\nvectorizer.fit_transform(train_data['preprocessed_question_text'].values.tolist() + test_data['preprocessed_question_text'].values.tolist())\ntrain_vectorizer = vectorizer.transform(train_data['preprocessed_question_text'].values.tolist())\ntest_vectorizer = vectorizer.transform(test_data['preprocessed_question_text'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:42:59.953493Z","iopub.execute_input":"2021-09-10T13:42:59.953846Z","iopub.status.idle":"2021-09-10T13:44:04.374438Z","shell.execute_reply.started":"2021-09-10T13:42:59.953815Z","shell.execute_reply":"2021-09-10T13:44:04.373421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\npred_train = np.zeros([train_data.shape[0]])\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, LogisticRegression(C=5., solver='sag'))\n    pred_full_test = pred_full_test + pred_test_y\n    pred_train[val_index] = pred_val_y\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:44:38.457431Z","iopub.execute_input":"2021-09-10T13:44:38.457773Z","iopub.status.idle":"2021-09-10T13:45:38.507203Z","shell.execute_reply.started":"2021-09-10T13:44:38.457744Z","shell.execute_reply":"2021-09-10T13:45:38.506285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Comparing all the models","metadata":{}},{"cell_type":"code","source":"\nfrom prettytable import PrettyTable\n    \ntable = PrettyTable()\nvect = ([\"CountVectorizer\"] * 2) + ([\"TFIDFVectorizer\"] * 2) + ([\"HashingVectorizer\"])\nmodel = ([\"Logistic Regression\", \"Naive Bayes\"] * 2) + ([\"Logistic Regression\"])\ntable.add_column(\"Model\", model)\ntable.add_column(\"Vectorizer\", vect)\ntable.add_column(\"Test F1-Score\", best_f1_score_list)\ntable.add_column(\"Best Threshold\", threshold_list)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:47:32.120055Z","iopub.execute_input":"2021-09-10T13:47:32.120392Z","iopub.status.idle":"2021-09-10T13:47:32.127754Z","shell.execute_reply.started":"2021-09-10T13:47:32.120365Z","shell.execute_reply":"2021-09-10T13:47:32.126786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(table)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:47:37.561914Z","iopub.execute_input":"2021-09-10T13:47:37.562273Z","iopub.status.idle":"2021-09-10T13:47:37.570962Z","shell.execute_reply.started":"2021-09-10T13:47:37.562242Z","shell.execute_reply":"2021-09-10T13:47:37.570076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training for best f1 score\n### Creating Count vectorizer and model","metadata":{}},{"cell_type":"code","source":"# Creating CountVectorizer object\nvectorizer = CountVectorizer(\n    dtype=np.float32, \n    strip_accents='unicode', \n    analyzer='word',\n    token_pattern=r'\\w{1,}',\n    ngram_range=(1, 3),\n    min_df=3\n)\n# Fit the vectorizer on training data after preprocessing\nvectorizer.fit_transform(train_data['preprocessed_question_text'].values.tolist() + test_data['preprocessed_question_text'].values.tolist())\ntrain_vectorizer = vectorizer.transform(train_data['preprocessed_question_text'].values.tolist())\ntest_vectorizer = vectorizer.transform(test_data['preprocessed_question_text'].values.tolist())","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:47:44.428268Z","iopub.execute_input":"2021-09-10T13:47:44.428858Z","iopub.status.idle":"2021-09-10T13:49:30.774687Z","shell.execute_reply.started":"2021-09-10T13:47:44.428822Z","shell.execute_reply":"2021-09-10T13:49:30.773789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Building and running model","metadata":{}},{"cell_type":"code","source":"cv_scores = []\npred_full_test = 0\n# Creating a zero list equal to the shape of training data\npred_train = np.zeros([train_data.shape[0]])\n\n# kfold with 5 n_splits\nkf = model_selection.KFold(n_splits=5, shuffle=True, random_state=2017)\n\nfor dev_index, val_index in kf.split(train_data):\n    dev_X, val_X = train_vectorizer[dev_index], train_vectorizer[val_index]\n    dev_y, val_y = train_y[dev_index], train_y[val_index]\n    pred_val_y, pred_test_y, model = buildModel(dev_X, dev_y, val_X, val_y, test_vectorizer, LogisticRegression(C=5., solver='sag'))\n    pred_full_test = pred_full_test + pred_test_y\n    \n    # Updating the pred_train list with prediction value\n    pred_train[val_index] = pred_val_y\n    \n    # appending the cv scores\n    cv_scores.append(metrics.log_loss(val_y, pred_val_y))\n    break\n    \nbest_threshold_function(val_y, pred_val_y)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:49:55.481919Z","iopub.execute_input":"2021-09-10T13:49:55.482301Z","iopub.status.idle":"2021-09-10T13:50:59.398090Z","shell.execute_reply.started":"2021-09-10T13:49:55.482270Z","shell.execute_reply":"2021-09-10T13:50:59.397165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Result - Output to csv file","metadata":{}},{"cell_type":"code","source":"pred_full_test = (pred_full_test > 0.2).astype(int)\noutput = pd.DataFrame({\n    \"qid\":test_data[\"qid\"].values, \n    \"prediction\": pred_full_test\n})\noutput.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-10T13:52:35.927410Z","iopub.execute_input":"2021-09-10T13:52:35.927735Z","iopub.status.idle":"2021-09-10T13:52:36.707640Z","shell.execute_reply.started":"2021-09-10T13:52:35.927706Z","shell.execute_reply":"2021-09-10T13:52:36.706793Z"},"trusted":true},"execution_count":null,"outputs":[]}]}