{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette\n%matplotlib inline\n\nfrom plotly import tools, subplots\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn import model_selection, metrics, linear_model\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-08T13:34:36.978135Z","iopub.execute_input":"2021-09-08T13:34:36.978482Z","iopub.status.idle":"2021-09-08T13:34:38.002811Z","shell.execute_reply.started":"2021-09-08T13:34:36.978398Z","shell.execute_reply":"2021-09-08T13:34:38.002022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls ../input/quora-insincere-questions-classification","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:38.004235Z","iopub.execute_input":"2021-09-08T13:34:38.004573Z","iopub.status.idle":"2021-09-08T13:34:38.669147Z","shell.execute_reply.started":"2021-09-08T13:34:38.004536Z","shell.execute_reply":"2021-09-08T13:34:38.668034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"../input/quora-insincere-questions-classification/train.csv\"\ntest_path = \"../input/quora-insincere-questions-classification/test.csv\"\ntrain_data = pd.read_csv(train_path)\ntest_data = pd.read_csv(test_path)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:38.671382Z","iopub.execute_input":"2021-09-08T13:34:38.671754Z","iopub.status.idle":"2021-09-08T13:34:44.263153Z","shell.execute_reply.started":"2021-09-08T13:34:38.671705Z","shell.execute_reply":"2021-09-08T13:34:44.262212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"There are {train_data.shape[0]} Rows and {train_data.shape[1]} Columns inside train data\")\nprint(f\"There are {train_data.shape[0]} questions in total in the training dataset\")\nprint(f\"There are {test_data.shape[0]} Rows and {test_data.shape[1]} Columns inside test data\")\nprint(f\"There are {test_data.shape[0]} questions in total in the test dataset\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:44.264842Z","iopub.execute_input":"2021-09-08T13:34:44.265260Z","iopub.status.idle":"2021-09-08T13:34:44.272626Z","shell.execute_reply.started":"2021-09-08T13:34:44.265222Z","shell.execute_reply":"2021-09-08T13:34:44.271687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(30)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:44.274254Z","iopub.execute_input":"2021-09-08T13:34:44.274671Z","iopub.status.idle":"2021-09-08T13:34:44.306367Z","shell.execute_reply.started":"2021-09-08T13:34:44.274631Z","shell.execute_reply":"2021-09-08T13:34:44.305612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:44.307660Z","iopub.execute_input":"2021-09-08T13:34:44.308008Z","iopub.status.idle":"2021-09-08T13:34:44.316988Z","shell.execute_reply.started":"2021-09-08T13:34:44.307972Z","shell.execute_reply":"2021-09-08T13:34:44.315948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_count = train_data['target'].value_counts()\nprint(target_count)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:44.318551Z","iopub.execute_input":"2021-09-08T13:34:44.319321Z","iopub.status.idle":"2021-09-08T13:34:44.339508Z","shell.execute_reply.started":"2021-09-08T13:34:44.319198Z","shell.execute_reply":"2021-09-08T13:34:44.338534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data for barchart\nbarchart_data = go.Bar(\n    x=target_count.index,\n    y=target_count.values,\n    marker=dict(\n        color=target_count.values,\n        colorscale = 'Picnic',\n        reversescale = True\n    ),\n)\n# Layout with title\nlayout = go.Layout(\n    title='Target Count',\n    font=dict(size=18)\n)\n\nfig = go.Figure(data=[barchart_data], layout=layout)\npy.iplot(fig, filename=\"TargetCount\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:44.342863Z","iopub.execute_input":"2021-09-08T13:34:44.343226Z","iopub.status.idle":"2021-09-08T13:34:45.231054Z","shell.execute_reply.started":"2021-09-08T13:34:44.343200Z","shell.execute_reply":"2021-09-08T13:34:45.230274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = (np.array(target_count.index))\nsizes = (np.array((target_count / target_count.sum())*100))\n\npiechart_trace = go.Pie(labels=labels, values=sizes)\nlayout = go.Layout(\n    title='Target Distribution',\n    font=dict(size=18),\n    width=600,\n    height=600,\n)\ndata = [piechart_trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"target_distribution\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:45.232983Z","iopub.execute_input":"2021-09-08T13:34:45.233408Z","iopub.status.idle":"2021-09-08T13:34:45.284311Z","shell.execute_reply.started":"2021-09-08T13:34:45.233352Z","shell.execute_reply":"2021-09-08T13:34:45.283490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference\n#### From here we can say that there is only 6.19 % of insincere questions\n#### This clearly tells us that the samples to predict from is pretty low i.e. a case of undersampling\n","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\nsincere_data = train_data[train_data[\"target\"] == 0]\ninsincere_data = train_data[train_data[\"target\"] == 1]\ntrain_sampled = pd.concat([resample(sincere_data, replace = True, n_samples = len(insincere_data)*4), insincere_data])\ntrain_sampled","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:45.285511Z","iopub.execute_input":"2021-09-08T13:34:45.285843Z","iopub.status.idle":"2021-09-08T13:34:45.568085Z","shell.execute_reply.started":"2021-09-08T13:34:45.285807Z","shell.execute_reply":"2021-09-08T13:34:45.567128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_sampled['target']\ny.value_counts().plot(kind='bar', rot=0)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:45.569484Z","iopub.execute_input":"2021-09-08T13:34:45.569833Z","iopub.status.idle":"2021-09-08T13:34:45.739468Z","shell.execute_reply.started":"2021-09-08T13:34:45.569796Z","shell.execute_reply":"2021-09-08T13:34:45.737961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Word cloud\nfrom wordcloud import WordCloud, STOPWORDS\n\ndef plot_wordcloud(text, mask=None, max_words=200, max_font_size=100, figure_size=(24.0,16.0), \n                   title = None, title_size=40, image_color=False):\n    stopwords = set(STOPWORDS)\n    more_stopwords = {'one', 'br', 'Po', 'th', 'sayi', 'fo', 'Unknown'}\n    stopwords = stopwords.union(more_stopwords)\n\n    wordcloud = WordCloud(\n        background_color='black',\n        stopwords=stopwords,\n        max_words=max_words,\n        max_font_size=max_font_size, \n        random_state=42,\n        width=800, \n        height=400,\n        mask=mask\n    )\n    wordcloud.generate(str(text))\n    \n    plt.figure(figsize=figure_size)\n    plt.imshow(wordcloud)\n    plt.title(title, fontdict={'size': title_size, 'color': 'black', \n                                  'verticalalignment': 'bottom'})\n    plt.axis('off');\n    plt.tight_layout()  \n    \nplot_wordcloud(train_data[\"question_text\"], title=\"Word Cloud of Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:45.740760Z","iopub.execute_input":"2021-09-08T13:34:45.741121Z","iopub.status.idle":"2021-09-08T13:34:46.596013Z","shell.execute_reply.started":"2021-09-08T13:34:45.741084Z","shell.execute_reply":"2021-09-08T13:34:46.595122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Word cloud for sincere questions\nplot_wordcloud(train_data[train_data[\"target\"] == 0][\"question_text\"], title=\"Word Cloud of Sincere Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:46.597081Z","iopub.execute_input":"2021-09-08T13:34:46.597370Z","iopub.status.idle":"2021-09-08T13:34:47.507829Z","shell.execute_reply.started":"2021-09-08T13:34:46.597340Z","shell.execute_reply":"2021-09-08T13:34:47.506922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Word cloud for insincere questions\nplot_wordcloud(train_data[train_data[\"target\"] == 1][\"question_text\"], title=\"Word Cloud of Insincere Questions\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:47.509358Z","iopub.execute_input":"2021-09-08T13:34:47.509711Z","iopub.status.idle":"2021-09-08T13:34:48.674860Z","shell.execute_reply.started":"2021-09-08T13:34:47.509676Z","shell.execute_reply":"2021-09-08T13:34:48.674055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing\nCleaning the questions\n","metadata":{}},{"cell_type":"code","source":"import re\n\ndef clean_text(text):\n\n  # Remove HTML Tags\n  text = re.sub(re.compile('<.*?>'), '', text)\n\n  # Remove [\\], ['], [\"]\n  text = re.sub(r'\\\\', '', text)\n  text = re.sub(r'\\\"', '', text)\n  text = re.sub(r'\\'', '', text)\n\n  # Remove number\n  text = re.sub('[0-9]{5,}','#####', text)\n  text = re.sub('[0-9]{4,}','####', text)\n  text = re.sub('[0-9]{3,}','###', text)\n  text = re.sub('[0-9]{2,}','##', text)\n\n  ## Remove Roman words\n  roman = re.compile(r'^M{0,4}(CM|CD|D?C{0,3})(XC|XL|L?X{0,3})(IX|IV|V?I{0,3})$')\n  text = roman.sub(r'', text)\n\n  # Convert all text to lowercase\n  text = text.strip().lower()\n\n  # Replace punctuation chars with spaces\n  filters = '!\"\\'#$%@&*()+_-;:<=>.?{}|`\\\\^\\t\\n'\n  translate_dict = dict((c, \" \") for c in filters)\n  translate_map = str.maketrans(translate_dict)\n  text = text.translate(translate_map)\n\n  return text","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:48.675960Z","iopub.execute_input":"2021-09-08T13:34:48.676347Z","iopub.status.idle":"2021-09-08T13:34:48.695537Z","shell.execute_reply.started":"2021-09-08T13:34:48.676300Z","shell.execute_reply":"2021-09-08T13:34:48.694706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nvectorizer = TfidfVectorizer(stop_words=\"english\",\n                             preprocessor=clean_text,\n                             ngram_range=(1, 3))\n\nX = vectorizer.fit_transform(train_sampled['question_text'])\nx = vectorizer.transform(test_data['question_text'])","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:34:48.697193Z","iopub.execute_input":"2021-09-08T13:34:48.697799Z","iopub.status.idle":"2021-09-08T13:35:51.218659Z","shell.execute_reply.started":"2021-09-08T13:34:48.697759Z","shell.execute_reply":"2021-09-08T13:35:51.217791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:35:51.220064Z","iopub.execute_input":"2021-09-08T13:35:51.220428Z","iopub.status.idle":"2021-09-08T13:35:51.297167Z","shell.execute_reply.started":"2021-09-08T13:35:51.220390Z","shell.execute_reply":"2021-09-08T13:35:51.296318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logistic = linear_model.LogisticRegression(solver='sag')\nlogistic.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:35:51.298414Z","iopub.execute_input":"2021-09-08T13:35:51.298921Z","iopub.status.idle":"2021-09-08T13:35:58.463196Z","shell.execute_reply.started":"2021-09-08T13:35:51.298866Z","shell.execute_reply":"2021-09-08T13:35:58.462122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score, accuracy_score, classification_report\n\ndef get_f1(model, name):\n  y_train_pred, y_pred = model.predict(X_train), model.predict(X_test)\n  print(classification_report(y_test, y_pred), '\\n')\n\n  print('{} model with F1 score = {}'.format(name, f1_score(y_test, y_pred)))\n\nget_f1(logistic, 'LogisticRegression')\n","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:35:58.464639Z","iopub.execute_input":"2021-09-08T13:35:58.465019Z","iopub.status.idle":"2021-09-08T13:35:58.672093Z","shell.execute_reply.started":"2021-09-08T13:35:58.464969Z","shell.execute_reply":"2021-09-08T13:35:58.671245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prdiction on test data \ntest_preds = logistic.predict(x)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:35:58.673290Z","iopub.execute_input":"2021-09-08T13:35:58.673763Z","iopub.status.idle":"2021-09-08T13:35:58.704589Z","shell.execute_reply.started":"2021-09-08T13:35:58.673726Z","shell.execute_reply":"2021-09-08T13:35:58.703638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xgboost\nimport xgboost as xgb\nxgb = xgb.XGBClassifier()\nxgb.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:35:58.706169Z","iopub.execute_input":"2021-09-08T13:35:58.706532Z","iopub.status.idle":"2021-09-08T13:40:38.145157Z","shell.execute_reply.started":"2021-09-08T13:35:58.706494Z","shell.execute_reply":"2021-09-08T13:40:38.144375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_f1(xgb, 'XGBClassifier')","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:40:38.146463Z","iopub.execute_input":"2021-09-08T13:40:38.146792Z","iopub.status.idle":"2021-09-08T13:40:43.934065Z","shell.execute_reply.started":"2021-09-08T13:40:38.146755Z","shell.execute_reply":"2021-09-08T13:40:43.932556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb\ntest_preds = model.predict(x)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:40:43.937122Z","iopub.execute_input":"2021-09-08T13:40:43.937386Z","iopub.status.idle":"2021-09-08T13:40:44.159639Z","shell.execute_reply.started":"2021-09-08T13:40:43.937359Z","shell.execute_reply":"2021-09-08T13:40:44.158153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Result","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({\n    \"qid\":test_data[\"qid\"].values, \n    \"prediction\": test_preds\n}) \noutput.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T13:40:44.161093Z","iopub.status.idle":"2021-09-08T13:40:44.161670Z"},"trusted":true},"execution_count":null,"outputs":[]}]}