{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm_notebook as tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer,CountVectorizer\nfrom wordcloud import WordCloud \nimport string\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn import metrics\nfrom sklearn.metrics import confusion_matrix, classification_report\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"088f477aab5d71de9e9a357f564eec7f337b0ea7","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\n\nprint(train.shape)\nprint(test.shape)\n\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21ea61af58ea42538ef4190deeac1af0c449c097","trusted":true},"cell_type":"code","source":"train['target'].value_counts()\nprint('Train data : ')\nprint(\"% of sincere questions : {:.2f}\".format(train.target.value_counts()[0] / len(train)))\nprint(\"% of insincere questions : {:.2f}\".format(train.target.value_counts()[1] / len(train)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"65602bd9720642110714a2c245ee1ce5422a79d4","trusted":true},"cell_type":"code","source":"no_insincere = len(train[train.target == 1])\ninsincere_index = train[train.target == 1].index\nsincere_index = train[train.target == 0].index\nchosen_sincere = np.random.choice(sincere_index, no_insincere, replace = False)\nmix_index = np.concatenate([chosen_sincere, insincere_index])\nsampled_df = train.loc[mix_index]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fd783a92dc4e6ac23589d74373594b0f3dc06690","trusted":true},"cell_type":"code","source":"del train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44ae84188479eb5e41592be22062b8cb4d78c767"},"cell_type":"code","source":"contractions_dict = {\n    \"ain't\": \"are not\",\n    \"aren't\": \"are not\",\n    \"can't\": \"cannot\",\n    \"can't've\": \"cannot have\",\n    \"'cause\": \"because\",\n    \"could've\": \"could have\",\n    \"couldn't\": \"could not\",\n    \"couldn't've\": \"could not have\",\n    \"didn't\": \"did not\",\n    \"doesn't\": \"does not\",\n    \"don't\": \"do not\",\n    \"hadn't\": \"had not\",\n    \"hadn't've\": \"had not have\",\n    \"hasn't\": \"has not\",\n    \"haven't\": \"have not\",\n    \"he'd\": \"he would\",\n    \"he'd've\": \"he would have\",\n    \"he'll\": \"he will\",\n    \"he'll've\": \"he will have\",\n    \"he's\": \"he is\",\n    \"how'd\": \"how did\",\n    \"how'd'y\": \"how do you\",\n    \"how'll\": \"how will\",\n    \"how's\": \"how is\",\n    \"i'd\": \"i would\",\n    \"i'd've\": \"i would have\",\n    \"i'll\": \"i will\",\n    \"i'll've\": \"i will have\",\n    \"i'm\": \"i am\",\n    \"i've\": \"i have\",\n    \"isn't\": \"is not\",\n    \"it'd\": \"it would\",\n    \"it'd've\": \"it would have\",\n    \"it'll\": \"it will\",\n    \"it'll've\": \"it will have\",\n    \"it's\": \"it is\",\n    \"let's\": \"let us\",\n    \"ma'am\": \"madam\",\n    \"mayn't\": \"may not\",\n    \"might've\": \"might have\",\n    \"mightn't\": \"might not\",\n    \"mightn't've\": \"might not have\",\n    \"must've\": \"must have\",\n    \"mustn't\": \"must not\",\n    \"mustn't've\": \"must not have\",\n    \"needn't\": \"need not\",\n    \"needn't've\": \"need not have\",\n    \"o'clock\": \"of the clock\",\n    \"oughtn't\": \"ought not\",\n    \"oughtn't've\": \"ought not have\",\n    \"shan't\": \"shall not\",\n    \"sha'n't\": \"shall not\",\n    \"shan't've\": \"shall not have\",\n    \"she'd\": \"she would\",\n    \"she'd've\": \"she would have\",\n    \"she'll\": \"she will\",\n    \"she'll've\": \"she will have\",\n    \"she's\": \"she is\",\n    \"should've\": \"should have\",\n    \"shouldn't\": \"should not\",\n    \"shouldn't've\": \"should not have\",\n    \"so've\": \"so have\",\n    \"so's\": \"so is\",\n    \"that'd\": \"that would\",\n    \"that'd've\": \"that would have\",\n    \"that's\": \"that is\",\n    \"there'd\": \"there would\",\n    \"there'd've\": \"there would have\",\n    \"there's\": \"there is\",\n    \"they'd\": \"they would\",\n    \"they'd've\": \"they would have\",\n    \"they'll\": \"they will\",\n    \"they'll've\": \"they will have\",\n    \"they're\": \"they are\",\n    \"they've\": \"they have\",\n    \"to've\": \"to have\",\n    \"wasn't\": \"was not\",\n    \"we'd\": \"we would\",\n    \"we'd've\": \"we would have\",\n    \"we'll\": \"we will\",\n    \"we'll've\": \"we will have\",\n    \"we're\": \"we are\",\n    \"we've\": \"we have\",\n    \"weren't\": \"were not\",\n    \"what'll\": \"what will\",\n    \"what'll've\": \"what will have\",\n    \"what're\": \"what are\",\n    \"what's\": \"what is\",\n    \"what've\": \"what have\",\n    \"when's\": \"when is\",\n    \"when've\": \"when have\",\n    \"where'd\": \"where did\",\n    \"where's\": \"where is\",\n    \"where've\": \"where have\",\n    \"who'll\": \"who will\",\n    \"who'll've\": \"who will have\",\n    \"who's\": \"who is\",\n    \"who've\": \"who have\",\n    \"why's\": \"why is\",\n    \"why've\": \"why have\",\n    \"will've\": \"will have\",\n    \"won't\": \"will not\",\n    \"won't've\": \"will not have\",\n    \"would've\": \"would have\",\n    \"wouldn't\": \"would not\",\n    \"wouldn't've\": \"would not have\",\n    \"y'all\": \"you all\",\n    \"y'all'd\": \"you all would\",\n    \"y'all'd've\": \"you all would have\",\n    \"y'all're\": \"you all are\",\n    \"y'all've\": \"you all have\",\n    \"you'd\": \"you would\",\n    \"you'd've\": \"you would have\",\n    \"you'll\": \"you will\",\n    \"you'll've\": \"you shall have\",\n    \"you're\": \"you are\",\n    \"you've\": \"you have\",\n    \"doin'\": \"doing\",\n    \"goin'\": \"going\",\n    \"nothin'\": \"nothing\",\n    \"somethin'\": \"something\",\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26c48621b899bf3663e45c2ef2f3d9da0c9e4780"},"cell_type":"code","source":"sampled_df.question_text = sampled_df.question_text.apply(lambda question: question.strip().lower())\ntest.question_text = test.question_text.apply(lambda question: question.strip().lower())\n\nsampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub(\"\\s{2,}\", \" \",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub(\"\\s{2,}\", \" \",question))\n\nsampled_df.question_text = sampled_df.question_text.apply(lambda question: question.replace(\"’\", \"'\"))\ntest.question_text = test.question_text.apply(lambda question: question.replace(\"’\", \"'\"))\n\nsampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub(\"'s\", \"\",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub(\"'s\", \"\",question))\n\nsampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub('\"', \"\",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub('\"', \"\",question))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"223bee97b64dbc6301f62c3cf6c0c4df1e81650b"},"cell_type":"code","source":"def fix(word):\n    val = ''\n    if word != '':\n        val = contractions_dict.get(word,word)\n    return val\n\ndef remove_contractions(question):\n    words = question.split(\" \")\n    new_words = [fix(word) for word in words]\n    return \" \".join(new_words)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f0299697c77ded88658cb20098b18a43c139bb2"},"cell_type":"code","source":"sampled_df.question_text = sampled_df.question_text.apply(lambda question: remove_contractions(question))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e22549b004d3985a819eb1df87830486881ee7d"},"cell_type":"code","source":"sampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub(\"'\", \"\",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub(\"'\", \"\",question))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ef14b128054588c02d0b75882bb369479100fc1"},"cell_type":"code","source":"sampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub(\"/\", \" \",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub(\"/\", \" \",question))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"763a107ad900a0ee47e4107de8b9ce7e6b909758"},"cell_type":"code","source":"sampled_df.question_text = sampled_df.question_text.apply(lambda question: re.sub(\"-\", \" \",question))\ntest.question_text = test.question_text.apply(lambda question: re.sub(\"-\", \" \",question))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df7bf99e16879ff05e5248df082d3f2ede781bb4","trusted":true},"cell_type":"code","source":"string.punctuation+= '“”’-‘'+ \"``\" + \"''\"\nstring.punctuation","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"13622414a5eca085102739f3a7aab2bb8eaf02e6","trusted":true},"cell_type":"code","source":"stop_words = set(stopwords.words('english'))\nlem = WordNetLemmatizer()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6436379e91ff621eac037d2e5b4ca611c81ded57","trusted":true},"cell_type":"code","source":"def preprocess_question(question):\n    words = word_tokenize(question)\n    words = [lem.lemmatize(word) for word in words if (not word in stop_words) and (not word in string.punctuation)]\n    return re.sub('\\d+','',\" \".join(words))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"86a4d2da31c3f84d25534b29015639f78eda8f4d","trusted":true},"cell_type":"code","source":"sampled_df['question_text'] = sampled_df.question_text.apply(lambda question: preprocess_question(question))\ntest['question_text'] = test.question_text.apply(lambda question: preprocess_question(question))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f0c34ccd365a362993bf3a146e1b37d1adf894e4","trusted":true},"cell_type":"code","source":"def plot_wordcloud(words, ttle):\n    wordcloud = WordCloud(width = 800, height = 400, background_color ='black', \n                min_font_size = 6, max_font_size = 150).generate(str(words))  \n    plt.figure(figsize = (18, 12), facecolor = None) \n    plt.imshow(wordcloud)\n    plt.title(ttle, fontsize = 40)\n    plt.axis(\"off\") \n    plt.tight_layout(pad = 0) \n    plt.show() ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3595789f4c0f585f3fd6a6b4e58e9684c2ca2de7","trusted":true},"cell_type":"code","source":"plot_wordcloud(sampled_df['question_text'],'Word Cloud of Questions')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9014eb13ae11a823f58022851a2970efeb7605a6","trusted":true},"cell_type":"code","source":"plot_wordcloud(sampled_df.question_text[sampled_df.target == 1],'Word Cloud of Insincere Questions')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8ac1553d9179c18cd231a97f4cc6e973273a0f44","trusted":true},"cell_type":"code","source":"plot_wordcloud(sampled_df.question_text[sampled_df.target == 0],'Word Cloud of Sincere Questions')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6561c2b140f06b7cddf7bfcce27614873380915c","trusted":true},"cell_type":"code","source":"def bar_plot(data, ttle, ntw = 15, fig_size = (12,8), ttle_size = 30):\n    plt.figure(figsize = fig_size)\n    sns.barplot(y = np.array(list(dict(data[:ntw]).keys()),dtype = object), x = np.array(list(dict(data[:ntw]).values())).astype(float))\n    plt.title(ttle, fontsize = ttle_size)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"19151b346bf573f4fb17c34fc04705bde87cfd2e","trusted":true},"cell_type":"code","source":"def vocabulary(questions, Verbose = True):\n    vocab = dict()\n    for question in tqdm(questions, disable = (not Verbose)):\n        for word in word_tokenize(question):\n            if (word != \" \") and (word not in stop_words) and (word not in '?!.,\"#$%\\'()*+-/:;<=>@[\\\\]^_`{|}~' + '“”’`\"' + \"'``\"):\n                vocab[word] = vocab.get(word,0) + 1\n    return vocab\n\nvocab = vocabulary(sampled_df.question_text)\n\nsorted_vocab = sorted(vocab.items(), key=lambda kv: kv[1],reverse = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"be8d5f966189e83ff98e149f05e7d3c78dc998fa","trusted":true},"cell_type":"code","source":"bar_plot(sorted_vocab, 'Most frequent words in all the Questions', 50,(15,12))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"17a34a6a60b055434b5356c3a9170f23c5fa0054","trusted":true},"cell_type":"code","source":"vocab = vocabulary(sampled_df.question_text[sampled_df.target == 1])\nsorted_vocab = sorted(vocab.items(), key=lambda kv: kv[1],reverse = True)\nbar_plot(sorted_vocab, 'Most frequent words in Insincere Questions', 50,(8,12))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"50d95dee8be32cebc0f3ff4bdce155121a9ec122","trusted":true},"cell_type":"code","source":"vocab = vocabulary(sampled_df.question_text[sampled_df.target == 0])\nsorted_vocab = sorted(vocab.items(), key=lambda kv: kv[1],reverse = True)\nbar_plot(sorted_vocab, 'Most frequent words in Sincere Questions', 50,(8,12))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"13a0952f01bed04b35836d6353d81557da30b707","trusted":true},"cell_type":"code","source":"def ngram(questions, n):\n    counts_ng = dict()\n    for questions in tqdm(questions, disable = True):\n        words = word_tokenize(questions)\n        ngram_tuples = list(nltk.ngrams(words,n))\n        for ngram_tuple in ngram_tuples:\n            counts_ng[ngram_tuple] = counts_ng.get(ngram_tuple, 0) + 1\n    return counts_ng","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dbae06132cdf8d8b46adc68231a6d00c7164d7b4","trusted":true},"cell_type":"code","source":"def plot_ngrams(questions, n,ttle, ntw = 25, ttle_size = 30, fig_size = (8,14)):\n    counts_ng = ngram(questions, n)\n    sorted_by_value_ng = sorted(np.array(list(counts_ng.items())), key=lambda kv: kv[-1],reverse = True)\n    ng = sorted_by_value_ng[:ntw]\n    ng_indx = [str(ng[i][0]) for i in range(ntw)] \n    ng_val = [ng[i][1] for i in range(ntw)] \n    plt.figure(figsize = fig_size)\n    ax = sns.barplot(y = ng_indx, x = ng_val)\n    plt.title(ttle, fontsize = ttle_size )","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6bd09ca689a395f92cd3605dba7cc19da29e08df","trusted":true},"cell_type":"code","source":"plot_ngrams(sampled_df.question_text[sampled_df.target == 1], 3, 'Frequent trigrams in Insincere questions', 50,30,(8,14))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"50224e556319dbecdcfb646f6c04c95c2636257d","trusted":true},"cell_type":"code","source":"plot_ngrams(sampled_df.question_text[sampled_df.target == 0], 3, 'Frequent trigrams in Sincere questions', 50,30,(8,14))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c6ca8b58b6898b8ccfd644aa6a4f88c20c707363","trusted":true},"cell_type":"code","source":"cv = TfidfVectorizer(sublinear_tf = True,stop_words = 'english', ngram_range = (1,2), max_features = 5000, token_pattern = '(\\S+)') ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d830733eaea5e3345aaf6f51ae8b069d803dfab4","trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(sampled_df['question_text'], sampled_df['target'], test_size = 0.25)\ncvec = cv.fit(X_train)\ndf_train = pd.DataFrame(cvec.transform(X_train).todense(),columns = cvec.get_feature_names())\ndf_val = pd.DataFrame(cvec.transform(X_val).todense(), columns = cvec.get_feature_names())\nloreg = LinearSVC()\nloreg.fit(df_train, y_train)\ny_pred = loreg.predict(df_val)\ncm = confusion_matrix(y_val, y_pred)\nlabels = ['sincere', 'unsincere']\nprint(pd.DataFrame(cm, columns=labels, index=labels))\nprint(classification_report(y_val, y_pred))\nprint(\"Accuracy : {:.2f}\".format(metrics.accuracy_score(y_val, y_pred)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f24df17d895cb2fc0db1959070390d02dfe52a04","trusted":true},"cell_type":"code","source":"df_test = pd.DataFrame(cvec.transform(test.question_text).todense(),columns = cvec.get_feature_names())\ntest_pred = loreg.predict(df_test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5c986092460760d4f9389505b74d96d236d6fc3f","trusted":true},"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['qid'] = test.qid\nsubmission['prediction'] = test_pred","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4a3f5586d73652d7d316c900b7968e1efc9741fc","trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"80ead5bcb383f57cc5da9e757b7e306b89b9bede","trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}