{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nfrom pandas.io.json import json_normalize\nfrom gensim.models import word2vec\n\nfrom sklearn.manifold import TSNE\nimport time\nfrom tqdm import tqdm\n\nimport lightgbm as lgb\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.naive_bayes import GaussianNB, MultinomialNB, BernoulliNB\n\nimport nltk\nfrom nltk.corpus import stopwords\nimport string\n\nfrom scipy.sparse import hstack\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette()\n\n%matplotlib inline\n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn import model_selection, preprocessing, metrics\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2b147aefe5ceda0eeb4a274b77b5c05ba3834f9"},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54bedc30a79b9bea8fa0767943459596d42459ce"},"cell_type":"code","source":"print(os.listdir(\"../input/embeddings\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58bb8988b21be8a7af5acf9a5af17178760ebe62"},"cell_type":"code","source":"#Size of datasets\nprint(\"Size of train data  (Rows, Columns): \",train_df.shape)\nprint(\"Size of test data  (Rows, Columns): \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"876000201a0629b29782cd3f3653a818db89f543"},"cell_type":"code","source":"#Snapshot of train data\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e12c14ef537cf81da73bd10937e24b5ceb0e750d"},"cell_type":"code","source":"#Sanapshot of test data\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"12f2a86f4f9d165f45733ff5853c20082b94ff3f"},"cell_type":"markdown","source":"Target Variable Distribution"},{"metadata":{"trusted":true,"_uuid":"2e7b8f920bc51479768ab3e56adfff4970286d4d"},"cell_type":"code","source":"np.unique(train_df['target'].values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"08e7c053cd1da4f76e27bcb04ea664669505e860"},"cell_type":"code","source":"np.mean(train_df['target'].values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8cc04ca974957870728bcf1aee97b592b036576e"},"cell_type":"code","source":"print('---------------------------------------------------------------------------------')\nprint(train_df['target'].value_counts())\nprint('---------------------------------------------------------------------------------')\nprint(train_df['target'].value_counts()/train_df['target'].shape[0])\nprint('---------------------------------------------------------------------------------')\n#sns.set(style=\"darkgrid\")\nax = sns.countplot(x=train_df['target'], data=train_df)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dda59a47253570e3c85284145a1b8fb95b69ff36"},"cell_type":"markdown","source":"__Insight:__ 6.1% of the training data are insincere questions and rest of them are sincere."},{"metadata":{"trusted":true,"_uuid":"e7d8e2e63adf50990fb98e455160f8d2712a8565"},"cell_type":"code","source":"eng_stopwords = set(stopwords.words(\"english\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c398323d2a80bae961e94cc81fe6180051f19a4"},"cell_type":"code","source":"## Number of words in the text ##\ntrain_df[\"num_words\"] = train_df[\"question_text\"].apply(lambda x: len(str(x).split()))\ntest_df[\"num_words\"] = test_df[\"question_text\"].apply(lambda x: len(str(x).split()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9cd957dce5f9f482167fee723474ea3de422863"},"cell_type":"code","source":"cnt_srs = train_df[\"num_words\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of words in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34e92c8b30d2178597af54820ec091005299636a"},"cell_type":"code","source":"## Number of unique words in the text ##\ntrain_df[\"num_unique_words\"] = train_df[\"question_text\"].apply(lambda x: len(set(str(x).split())))\ntest_df[\"num_unique_words\"] = test_df[\"question_text\"].apply(lambda x: len(set(str(x).split())))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d2ffaba35f5e35db5efc9a92bab40a006d5159e"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_unique_words\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of num_unique_words in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"beffc389b40f690a54b3863d58bbf46e1001d594"},"cell_type":"code","source":"## Number of characters in the text ##\ntrain_df[\"num_chars\"] = train_df[\"question_text\"].apply(lambda x: len(str(x)))\ntest_df[\"num_chars\"] = test_df[\"question_text\"].apply(lambda x: len(str(x)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4034bfbe857b6c430fd67451356f5d029e3debf4"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_chars\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of char in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9782ce18986804447adbd50b23ecc4a9b54bb49a"},"cell_type":"code","source":"## Number of stopwords in the text ##\ntrain_df[\"num_stopwords\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in eng_stopwords]))\ntest_df[\"num_stopwords\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in eng_stopwords]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42100086f7abb87b39ffce42d57acbf12c483ec5"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_stopwords\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of stopwords in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87dd2933177df1f011cee0bd57f08f614084f689"},"cell_type":"code","source":"## Number of punctuations in the text ##\ntrain_df[\"num_punctuations\"] =train_df['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )\ntest_df[\"num_punctuations\"] =test_df['question_text'].apply(lambda x: len([c for c in str(x) if c in string.punctuation]) )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4e3a5ad553875fadf2c6ec8382fba809c997a3e"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_punctuations\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of punctuations in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bbc35942951ae749fe200bee918fbab9ca410f82"},"cell_type":"code","source":"## Number of title case words in the text ##\ntrain_df[\"num_words_upper\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\ntest_df[\"num_words_upper\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82574bb96007f8631b3e2875ac1a581b0054ae9e"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_words_upper\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of words_upper in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b532efda675c74ca0a9189050a2de29ad8b68e59"},"cell_type":"code","source":"## Number of title case words in the text ##\ntrain_df[\"num_words_title\"] = train_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\ntest_df[\"num_words_title\"] = test_df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f52bf3755ba91f9115de3b20c9e6308f2a7ea8b7"},"cell_type":"code","source":"\ncnt_srs = train_df[\"num_words_title\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of words_title in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1ee2f02dcff7d6e01c4ef88b020255aa3ec9b39"},"cell_type":"code","source":"## Average length of the words in the text ##\ntrain_df[\"mean_word_len\"] = train_df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\ntest_df[\"mean_word_len\"] = test_df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c3c0b7d165fef238996e61cb9b48c57a4525f35"},"cell_type":"code","source":"\ncnt_srs = train_df[\"mean_word_len\"].value_counts()\n\nplt.figure(figsize=(12,6))\nsns.barplot(cnt_srs.index, cnt_srs.values, alpha=0.8, color=color[0])\nplt.ylabel('Number of Occurrences', fontsize=12)\nplt.xlabel('Number of mean_word_len in the question', fontsize=12)\nplt.xticks(rotation='vertical')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"480bbb88a65c0bab3001380e3f28a164b04491f7"},"cell_type":"code","source":"features = ['num_words', 'num_unique_words', 'num_chars', \n                'num_stopwords', 'num_punctuations', 'num_words_upper', \n                'num_words_title', 'mean_word_len']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af3cf0e25d855af84a261a85cfa66a8367ca07e0"},"cell_type":"code","source":"for i in features:\n    plt.figure(figsize=(8,4))\n    sns.set(style=\"whitegrid\")\n    sns.violinplot(data=train_df[i])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bec8adbe8671f261c246a51b4123952dea9a4539"},"cell_type":"code","source":"def missing_check(df):\n    total = df.isnull().sum().sort_values(ascending=False)\n    percent = (df.isnull().sum()/df.isnull().count()).sort_values(ascending=False)\n    missing_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\n    return missing_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4e3d9cf1e8309d141498da7250a61a2ed5c99718"},"cell_type":"code","source":"missing_check(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f853583cc4bf2a21f1874ce1e6b411a9a275b0cf"},"cell_type":"code","source":"missing_check(test_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a31bcfb9cca15b3a22b9d6fd1816d00e4246a39"},"cell_type":"code","source":"from bs4 import BeautifulSoup\ndef strip_html(text):\n    soup = BeautifulSoup(text, \"html.parser\")\n    return soup.get_text()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1002415f87972c0d64c9e624361b33dc9fc92dc0"},"cell_type":"code","source":"train_df['question_text'] = train_df['question_text'].apply(strip_html)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ddd91e72e964edc64b839c05c1eacaba7364fd2d"},"cell_type":"code","source":"STOP_WORDS = nltk.corpus.stopwords.words()\n\ndef clean_sentence(val):\n    \"remove chars that are not letters or numbers, downcase, then remove stop words\"\n    regex = re.compile('([^\\s\\w]|_)+')\n    sentence = regex.sub('', val).lower()\n    sentence = sentence.split(\" \")\n    \n    for word in list(sentence):\n        if word in STOP_WORDS:\n            sentence.remove(word)  \n            \n    sentence = \" \".join(sentence)\n    return sentence\n\ndef clean_dataframe(data):\n    \"drop nans, then apply 'clean_sentence' function to question1 and 2\"\n    data = data.dropna(how=\"any\")\n    data[\"question_text\"] = data[\"question_text\"].apply(clean_sentence)\n    \n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"502704af848f30a1b737f33dda79c1d44a3f5515"},"cell_type":"markdown","source":"#Work_in_Progress"},{"metadata":{"trusted":true,"_uuid":"4132fe56c5b8a28ac50677c7196a51f39a0d69ad"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}