{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import *\nfrom wordcloud import *\nfrom tqdm import tqdm\n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn import model_selection, preprocessing, metrics, ensemble, naive_bayes, linear_model\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.decomposition import TruncatedSVD\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nimport nltk\nfrom nltk.tokenize import word_tokenize\nfrom nltk.tokenize import sent_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.stem import PorterStemmer\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"#### Read the files"},{"metadata":{"trusted":true,"_uuid":"4b0024a94399f6cb9ec6d2e4037c54c173db21bd"},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \", train_df.shape)\nprint(\"Test shape : \", test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ba0ae69f8696440f44ee2d460f47222bb989c8a","scrolled":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c095c8fda2eada2879d248e5cd3ae94130249956"},"cell_type":"markdown","source":"#### Target Variable distribution:\nCheck if the data set is imbalanced"},{"metadata":{"trusted":true,"_uuid":"4c4984665b994c71762dd8316c51849c247abf84"},"cell_type":"code","source":"train_df['target'].value_counts().plot(kind='bar')\ntarget_table = pd.crosstab(index = train_df['target'], columns='count')\nprint(target_table/target_table.sum())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"87ad9b278d6dcb7877992453d747b5531771ffcb"},"cell_type":"markdown","source":"##### Looks like we have Sincere:Insincere == 93%:7%\n\n## Explore the data\n\n#### N-gram function from the text\n\nReference: http://www.albertauyeung.com/post/generating-ngrams-python/"},{"metadata":{"trusted":true,"_uuid":"fb87282d4db2bd8cb023a712483495385f8c6699"},"cell_type":"code","source":"def generate_ngrams(text, n_gram=1):\n    token = [token for token in text.lower().split(\" \") if token != \"\" if token not in STOPWORDS]\n    ngrams = zip(*[token[i:] for i in range(n_gram)])\n    return [\" \".join(ngram) for ngram in ngrams]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d9752e7d1019c982d197f3ef67009614e247480"},"cell_type":"code","source":"train0_df = train_df[train_df['target']==0]\ntrain1_df = train_df[train_df['target']==1]\n\n#For all sincere questions\nfrom collections import defaultdict\nsinc_dict = defaultdict(int)\nfor sent in train0_df[\"question_text\"]:\n    for word in generate_ngrams(sent):\n        sinc_dict[word] += 1\nsinc_sorted = pd.DataFrame(sorted(sinc_dict.items(), key=lambda x: x[1])[::-1])\nsinc_sorted.columns = [\"word\", \"wordcount\"]\n\n#For all insincere questions\nfrom collections import defaultdict\ninsinc_dict = defaultdict(int)\nfor sent in train1_df[\"question_text\"]:\n    for word in generate_ngrams(sent):\n        insinc_dict[word] += 1\ninsinc_sorted = pd.DataFrame(sorted(insinc_dict.items(), key=lambda x: x[1])[::-1])\ninsinc_sorted.columns = [\"word\", \"wordcount\"]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"91fb361cc8bfef831a91e134538ee91278d5fb3f"},"cell_type":"markdown","source":"#### 1 gram - Most common words in both the types of questions"},{"metadata":{"trusted":true,"_uuid":"295f5839d288556d51e92fe915541b82d31babaf"},"cell_type":"code","source":"gram1_0 = go.Bar(y = sinc_sorted[\"word\"].head(20),x = sinc_sorted[\"wordcount\"].head(20),orientation=\"h\")\ngram1_1 = go.Bar(y = insinc_sorted[\"word\"].head(20),x = insinc_sorted[\"wordcount\"].head(20),orientation=\"h\")\n\nfig = tools.make_subplots(rows=1, cols=2, vertical_spacing=0.04,\n                          subplot_titles=[\"Frequent words of sincere questions\", \n                                          \"Frequent words of insincere questions\"])\nfig.append_trace(gram1_0, 1, 1)\nfig.append_trace(gram1_1, 1, 2)\nfig['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\npy.iplot(fig, filename='word-plots')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"38fb8615b05f11aefa20a94d591165789ffb78ae"},"cell_type":"markdown","source":"Words like best, people, good are highlighted in sincere questions.\nWords like trump, women, people, white, muslims are highlighed in insincere questions\nSome words are stop words and also some words are common between them"},{"metadata":{"_uuid":"0ed5af539417474fcfb32772817b88e8a7172faf"},"cell_type":"markdown","source":"#### 2 gram - Most common words in both the types of questions"},{"metadata":{"trusted":true,"_uuid":"178e6d6fcb738f55812a7899a866f1016253dec6"},"cell_type":"code","source":"sinc_dict2 = defaultdict(int)\nfor sent in train0_df[\"question_text\"]:\n    for word in generate_ngrams(sent,n_gram=2):\n        sinc_dict2[word] += 1\nsinc_sorted_2 = pd.DataFrame(sorted(sinc_dict2.items(), key=lambda x: x[1])[::-1])\nsinc_sorted_2.columns = [\"word\", \"wordcount\"]\n\n#For all insincere questions\nfrom collections import defaultdict\ninsinc_dict_2 = defaultdict(int)\nfor sent in train1_df[\"question_text\"]:\n    for word in generate_ngrams(sent,n_gram=2):\n        insinc_dict_2[word] += 1\ninsinc_sorted_2 = pd.DataFrame(sorted(insinc_dict_2.items(), key=lambda x: x[1])[::-1])\ninsinc_sorted_2.columns = [\"word\", \"wordcount\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d363bb10ffe6a696b90b8b1f9ebeb1acfc0bf43"},"cell_type":"code","source":"gram2_0 = go.Bar(y = sinc_sorted_2[\"word\"].head(20),x = sinc_sorted_2[\"wordcount\"].head(20),orientation=\"h\")\ngram2_1 = go.Bar(y = insinc_sorted_2[\"word\"].head(20),x = insinc_sorted_2[\"wordcount\"].head(20),orientation=\"h\")\n\nfig2 = tools.make_subplots(rows=1, cols=2, vertical_spacing=0.04,\n                          subplot_titles=[\"Frequent words of sincere questions\", \n                                          \"Frequent words of insincere questions\"])\nfig2.append_trace(gram2_0, 1, 1)\nfig2.append_trace(gram2_1, 1, 2)\nfig2['layout'].update(height=1200, width=900, paper_bgcolor='rgb(233,233,233)', title=\"Word Count Plots\")\npy.iplot(fig2, filename='word-plots')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7435bf645bbc1f796546dff322c1905df7716fbb"},"cell_type":"markdown","source":"Again we see some common words popping up, somewords that stand out though might be able to help classify the sincere ones from the insincere ones."},{"metadata":{"_uuid":"c1579f58b1f5b5c745ca1876d7d7b7029e33b133"},"cell_type":"markdown","source":"#### Meta Features:\n\nNow let us create some meta features and then look at how they are distributed between the classes. The ones that we will create are\n\nNumber of words in the text\nNumber of unique words in the text\nNumber of characters in the text\nNumber of stopwords\nNumber of punctuations\nNumber of upper case words\nNumber of title case words\nAverage length of the words\n\nReference: SRK's code"},{"metadata":{"trusted":true,"_uuid":"30e429b3a4efce61587630dd48cc202438fcfa41"},"cell_type":"code","source":"from wordcloud import WordCloud, STOPWORDS\n\n## Number of words in the text ##\ntrain_df[\"num_words\"] = train_df[\"question_text\"].apply(lambda x: len(str(x).split()))\ntest_df[\"num_words\"] = test_df[\"question_text\"].apply(lambda x: len(str(x).split()))\n\n## Number of unique words in the text ##\ntrain_df[\"num_unique_words\"] = train_df[\"question_text\"].apply(lambda x: len(set(str(x).split())))\ntest_df[\"num_unique_words\"] = test_df[\"question_text\"].apply(lambda x: len(set(str(x).split())))\n\n## Number of characters in the text ##\ntrain_df[\"num_chars\"] = train_df[\"question_text\"].apply(lambda x: len(str(x)))\ntest_df[\"num_chars\"] = test_df[\"question_text\"].apply(lambda x: len(str(x)))\n\n## Number of stopwords in the text ##\ntrain_df[\"num_stopwords\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\ntest_df[\"num_stopwords\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\n\n## Number of punctuations in the text ##\ntrain_df[\"num_punctuations\"] =train_df['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )\ntest_df[\"num_punctuations\"] =test_df['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )\n\n## Number of title case words in the text ##\ntrain_df[\"num_words_upper\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\ntest_df[\"num_words_upper\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\n\n## Number of title case words in the text ##\ntrain_df[\"num_words_title\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\ntest_df[\"num_words_title\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n\n## Average length of the words in the text ##\ntrain_df[\"mean_word_len\"] = train_df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\ntest_df[\"mean_word_len\"] = test_df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1cf7ba9e197eb59d70980c7bd5e91a6eb6f064e"},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fed06066cc55c98c3cd8a9cba07dec3cdda3b069"},"cell_type":"markdown","source":"Let's explore this dataset. "},{"metadata":{"trusted":true,"_uuid":"4e7f462c4e2541d73aa06d0ba570d2643beb2166"},"cell_type":"code","source":"## Truncate some extreme values for better visuals ##\ntrain_df['num_words'].loc[train_df['num_words']>60] = 60 #truncation for better visuals\ntrain_df['num_punctuations'].loc[train_df['num_punctuations']>10] = 10 #truncation for better visuals\ntrain_df['num_chars'].loc[train_df['num_chars']>350] = 350 #truncation for better visuals\n\nf, axes = plt.subplots(3, 1, figsize=(10,20))\nsns.boxplot(x='target', y='num_words', data=train_df, ax=axes[0])\naxes[0].set_xlabel('Target', fontsize=12)\naxes[0].set_title(\"Number of words in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_chars', data=train_df, ax=axes[1])\naxes[1].set_xlabel('Target', fontsize=12)\naxes[1].set_title(\"Number of characters in each class\", fontsize=15)\n\nsns.boxplot(x='target', y='num_punctuations', data=train_df, ax=axes[2])\naxes[2].set_xlabel('Target', fontsize=12)\n#plt.ylabel('Number of punctuations in text', fontsize=12)\naxes[2].set_title(\"Number of punctuations in each class\", fontsize=15)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e62465956e60a62d3ff857e2a67498f1ebc07f07"},"cell_type":"markdown","source":"In general, looks like insincere questions have higher number of punctuations, words and charecters"},{"metadata":{"_uuid":"ccab622b4cf5ed7b786a11560a9596871e67acaa"},"cell_type":"markdown","source":"### BASE MODEL"},{"metadata":{"_uuid":"af4ca47a5247d74c51a3797fc48ec77da43951a4"},"cell_type":"markdown","source":"#### Define a tokenize function"},{"metadata":{"trusted":true,"_uuid":"e79b495f41fa129f9bb613764a8b1d817192df09"},"cell_type":"code","source":"#def tokenize(data):\n#    tokenized_docs = [word_tokenize(doc.lower()) for doc in data]\n#    alpha_tokens = [[t for t in doc if t.isalpha() == True] for doc in tokenized_docs]\n#    stemmer = PorterStemmer ()\n#    stemmed_tokens = [[stemmer.stem(alpha) for alpha in doc] for doc in alpha_tokens]\n#    X_stem_as_string = [\" \".join(x_t) for x_t in stemmed_tokens]\n#    return X_stem_as_string","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"111b4950489301c4bd1cb88b06817b4890090429"},"cell_type":"code","source":"X = train_df['question_text']\ny = train_df['target']\nX_test = test_df['question_text']\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=.2, random_state=42, stratify=y)\nX_train.shape, y_train.shape, X_val.shape, y_val.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19c6df9e0dd3412b55ef36f5b85774ac92ebdfe2"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.pipeline import Pipeline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f52fd720b8db0233a17589d927cc24d08c04c2a0"},"cell_type":"markdown","source":"#### Random Forest Model"},{"metadata":{"trusted":true,"_uuid":"3051b400202d38e4242cf84abe6f0317b0fc4aff"},"cell_type":"code","source":"rf = ensemble.RandomForestClassifier(class_weight='balanced_subsample')\ntfvec = TfidfVectorizer(stop_words='english', lowercase=False)\npipe = Pipeline([\n    ('vectorizer', tfvec),\n    ('rf', rf )\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98225aa2eaa37454b0b0e0a03ab7155bc70f6d99"},"cell_type":"code","source":"pipe.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"674c1ee62e8cb4cc10fe6b79d2e5056dd9064e20"},"cell_type":"code","source":"y_pred = pipe.predict(X_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"778b538e8512b47d23549e7b2887519c40f9247c"},"cell_type":"code","source":"cm = metrics.confusion_matrix(y_val, y_pred)\n\nax = plt.gca()\nsns.heatmap(cm, cmap='Blues', cbar=False, annot=True, xticklabels=y_val.unique(), yticklabels=y_val.unique(), ax=ax);\nax.set_xlabel('y_pred');\nax.set_ylabel('y_true');\nax.set_title('Confusion Matrix');\n\ncr = metrics.classification_report(y_val, y_pred)\nprint(cr)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dc1377e43e5ebf38a14e89e8c405a4cb583d5c53"},"cell_type":"markdown","source":"### Logistic Regression Model"},{"metadata":{"trusted":true,"_uuid":"78b1ca48e5c844c0be7725ba9ba1dd7d32b05be5"},"cell_type":"code","source":"lr = linear_model.LogisticRegression()\npipe_lr = Pipeline([\n    ('vectorizer', tfvec),\n    ('lr', lr )\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea5038d20b3db8c8fbdc0048cabb8ab45f3dd2a1"},"cell_type":"code","source":"pipe_lr.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0ab5b1c01dd20e2f3185a3453089c9b91fd73b3"},"cell_type":"code","source":"y_pred_lr = pipe_lr.predict(X_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dbc7e36f850362571c985817f1355c26ba7147f1"},"cell_type":"code","source":"cm_lr = metrics.confusion_matrix(y_val, y_pred_lr)\n\nax = plt.gca()\nsns.heatmap(cm_lr, cmap='Blues', cbar=False, annot=True, xticklabels=y_val.unique(), yticklabels=y_val.unique(), ax=ax);\nax.set_xlabel('y_pred');\nax.set_ylabel('y_true');\nax.set_title('Confusion Matrix');\n\ncr = metrics.classification_report(y_val, y_pred_lr)\nprint(cr)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"11ef8be63185db62ed6c9d27e9b41d71aef6b780"},"cell_type":"markdown","source":"### Play around with the Threshold to see if f1_score can be increased"},{"metadata":{"trusted":true,"_uuid":"2abb56e77937e5f9f18570bf304b9c49accac87f"},"cell_type":"code","source":"metrics.f1_score(y_pred=y_pred_lr,y_true=y_val)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ce828d1f39f01c13328657ed4bc32573ba171a2"},"cell_type":"markdown","source":"Logistic regression seems to do a better job here. We can further optimize the f1-score by tuning the threshold."},{"metadata":{"trusted":true,"_uuid":"c71ba3f037d088031a11c1ec974461d01fc4a52c"},"cell_type":"code","source":"y_prob_lr = pipe_lr.predict_proba(X_val)\nbest_threshold = 0\nf1=0\nfor i in np.arange(.1, .51, 0.01):\n    y_pred2_lr = [1 if proba>i else 0 for proba in y_prob_lr[:, 1]]\n    f1score = metrics.f1_score(y_pred=y_pred2_lr, y_true=y_val)\n    if f1score>f1:\n        best_threshold = i\n        f1=f1score\n        \ny_pred2_lr = [1 if proba>best_threshold else 0 for proba in y_prob_lr[:, 1]]\nf1 = metrics.f1_score(y_pred2_lr, y_val)\nprint('The best threshold is {}, with an f1_score of {}'.format(best_threshold, f1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7fd730d8677dea8d58684eeca3b878d0a3347900"},"cell_type":"markdown","source":"Looks like we have a decent model with F1 of 0.607"},{"metadata":{"trusted":true,"_uuid":"be465eef2b17d7f68005095f27204e04a5681457"},"cell_type":"code","source":"y_pred_sub = pipe_lr.predict(X_test) \n\nsub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = (y_pred_sub > best_threshold).astype(int)\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"09d3f1f2bef32cac251bdabf5b3482e58d94e7b6"},"cell_type":"markdown","source":"# Tune Logistic regression - CV"},{"metadata":{"trusted":true,"_uuid":"2355c1364b5a1de7e2385f59e734bd164dffd2b1"},"cell_type":"code","source":"lr = linear_model.LogisticRegression(penalty='l2',solver='sag')\npipe_cv = Pipeline([\n    ('vectorizer', tfvec),\n    ('lr', lr )\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"152dc53aa2e339f5161da69b0eae8037cece996e"},"cell_type":"code","source":"param_grid = {'lr__C': [0.001, 0.01, 0.1, 1, 10, 100, 1000] }\n\nclf = model_selection.GridSearchCV(pipe_cv, param_grid,cv=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a92a2a28dd887224d9935ce6ee94031b771eb36c"},"cell_type":"code","source":"clf.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"abb5ea251994e6309914c4f2e914c152544a7f17"},"cell_type":"code","source":"y_pred_lrcv = clf.best_estimator_.predict(X_val)\nprint(metrics.f1_score(y_val, y_pred_lrcv))\nprint(metrics.classification_report(y_val, y_pred_lrcv))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b2f8b2535f39f4ca2142ce6ce7127defe802057"},"cell_type":"code","source":"### Lets find the best threshold for cut-off\ny_prob_lrcv = clf.best_estimator_.predict_proba(X_val)\nbest_threshold = 0\nf1=0\nfor i in np.arange(.1, .51, 0.01):\n    y_pred2_lrcv = [1 if proba>i else 0 for proba in y_prob_lrcv[:, 1]]\n    f1score = metrics.f1_score(y_pred=y_pred2_lrcv, y_true=y_val)\n    if f1score>f1:\n        best_threshold = i\n        f1=f1score\n        \ny_pred2_lrcv = [1 if proba>best_threshold else 0 for proba in y_prob_lrcv[:, 1]]\nf1 = metrics.f1_score(y_pred2_lrcv, y_val)\nprint('The best threshold is {}, with an f1_score of {}'.format(best_threshold, f1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9e3cf43098b0c9a3f76a9fdd405448c054ef14a"},"cell_type":"code","source":"y_pred_sub = pipe_lr.predict(X_test) \n\nsub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = (y_pred_sub > best_threshold).astype(int)\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf7d88cbae808662744afe08e21f2153dfebfbe9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}