{"cells":[{"metadata":{"_uuid":"a21b1a9185f4634d8ae72aff10b2679a9285bdde"},"cell_type":"markdown","source":"# Quora Insincere Questions Exploratory Data Analysis\n\nWe will begin exploring the training data in order to come up with insights and a plan for modeling."},{"metadata":{"trusted":true,"_uuid":"f627f68db31821e36f9b39736702956d0ba0b5f6"},"cell_type":"code","source":"# import packages\nimport numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport spacy\nimport nltk\nimport re\n\nimport string\nfrom nltk.corpus import stopwords\nfrom collections import Counter\n\n# print any variable/statement on its own line (not just the last one!)\n#from IPython.core.interactiveshell import InteractiveShell\n#InteractiveShell.ast_node_interactivity = \"all\"\n\nnp.random.seed(27)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba862b614139fd9f6578d383124d4cb1ad6b8e30"},"cell_type":"code","source":"# setting up default plotting parameters\n%matplotlib inline\n\nplt.rcParams['figure.figsize'] = [20.0, 7.0]\nplt.rcParams.update({'font.size': 22,})\n\nsns.set_palette('viridis')\nsns.set_style('white')\nsns.set_context('talk', font_scale=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4a047a5616fbc049af4482381a1fba57efe5db3"},"cell_type":"code","source":"# load in data and print shape/head/tail\nraw_data = pd.read_csv('../input/train.csv')\nprint(raw_data.shape)\nraw_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b93c8f21a0d0fe8f670336e5ff37416aeb707e28"},"cell_type":"code","source":"# using seaborns countplot to show distribution of questions in dataset\nfig, ax = plt.subplots()\ng = sns.countplot(raw_data.target, palette='viridis')\ng.set_xticklabels(['Sincere', 'Insincere'])\ng.set_yticklabels([])\n\n# function to show values on bars\ndef show_values_on_bars(axs):\n    def _show_on_single_plot(ax):        \n        for p in ax.patches:\n            _x = p.get_x() + p.get_width() / 2\n            _y = p.get_y() + p.get_height()\n            value = '{:.0f}'.format(p.get_height())\n            ax.text(_x, _y, value, ha=\"center\") \n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)\nshow_values_on_bars(ax)\n\nsns.despine(left=True, bottom=True)\nplt.xlabel('')\nplt.ylabel('')\nplt.title('Distribution of Questions', fontsize=30)\nplt.tick_params(axis='x', which='major', labelsize=15)\nfig.savefig('classes.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76bd58ce0b3c817a51d9611a3fcbc3f772a1c3e0"},"cell_type":"code","source":"# print percentage of questions where target == 1\n(len(raw_data.loc[raw_data.target==1])) / (len(raw_data.loc[raw_data.target == 0])) * 100","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dd36f29f3902d9f356b369ec5e5d6dba5830aebf"},"cell_type":"markdown","source":"### Class Imbalance\nImbalanced classes are a common problem in machine learning classification where there are a disproportionate ratio of observations in each class.  With just 6.6% of our dataset belonging to the target class, we can definitely have an imbalanced class!\n\nThis is a problem because many machine learning models are designed to maximize overall accuracy, which especially with imbalanced classes may not be the best metric to use.  Classification accuracy is defined as the number of correct predictions divided by total predictions times 100.  For example, if we simply predicted that all questions are sincere, we would get a classification acuracy score of 93%!\n\nThis competition uses the F1 score which balances precision and recall.\n - Precision is the number of true positives divided by all positive predictions.  Precision is also called Positive Predictive Value.  It is a measure of a classifier's exactness.  Low precision indicates a high number of false positives.\n - Recall is the number of true positives divided by the number of positive values in the test data.  Recall is also called Sensitivity or the True Positive Rate.  It is a measure of a classifier's completeness.  Low recall indicates a high number of false negatives."},{"metadata":{"trusted":true,"_uuid":"edd79a10a092344ce409c06047d53e62c220674f"},"cell_type":"code","source":"# printing out a random sample of questions labeled insincere\nimport random\n\nindex = random.sample(raw_data.index[raw_data.target == 1].tolist(), 5)\nfor i in index:\n    print(raw_data.iloc[i, 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5967982e9ff239cb7d49d7829a22146005a408a4"},"cell_type":"code","source":"# taking a sample of the training data to speed up processing\ndf = raw_data.sample(frac=0.3)\ndf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5fc27688403760452c669e2c0db47440344909c"},"cell_type":"code","source":"# tokenize with spacy\nnlp = spacy.load('en')\n\ndf['tokens'] = [nlp(text, # disable parts of the language processing pipeline we don't need here to speed up processing\n                    disable=['ner', # named entity recognition\n                                   'tagger', # part-of-speech tagger\n                                   'textcat', # document label categorizer\n                                  ]) for text in df.question_text]\ndf.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f6c7e3243e453595ef7bdae68a39e2e69ea5d9a9"},"cell_type":"code","source":"df['num_tokens'] = [len(token) for token in df.tokens]\ndf.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c34b01e6ce6291675ad5242631374b5c0633ae2a"},"cell_type":"code","source":"# using seaborns boxplot to visualize number of tokens per question\nfig, ax = plt.subplots()\ng = sns.boxplot(x=df.target, y=df.num_tokens, palette='viridis')\ng.set_xticklabels(['Sincere', 'Insincere'])\ng.set_yticklabels([])\n\nsns.despine(left=True, bottom=True)\nplt.xlabel('')\nplt.ylabel('')\nplt.title('Number of Tokens per Question', fontsize=30)\nplt.tick_params(axis='x', which='major', labelsize=15)\nfig.savefig('tokens.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d658ecbb578a31dcdd3365b33b9669a25a8888d4"},"cell_type":"code","source":"# get number of sentences per question\nprint(list(df.iloc[0,3].sents))\n\nsents = [list(x.sents) for x in df.tokens]\ndf['num_sents'] = [len(sent) for sent in sents]\ndf.sample(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8043a18f27857dfba831fe8d090ec5f1bffc927"},"cell_type":"code","source":"# plotting number of sentences per question\nfig, ax = plt.subplots()\ng = sns.countplot(df.num_sents, hue=df.target, palette='viridis')\n#g.set_xticklabels(['Sincere', 'Insincere'])\ng.set_yticklabels([])\n\n# using log scale on y-axis so we can better see the questions with more sentences\nax.set(yscale='log')\n\nsns.despine(left=True, bottom=True)\nplt.xlabel('')\nplt.ylabel('')\nplt.title('Number of Sentences per Question', fontsize=30)\nplt.tick_params(axis='x', which='major', labelsize=15)\nfig.savefig('sentences.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a094cda6b83f0385526aa2460693e5d8c2859874"},"cell_type":"code","source":"# Finding most common words\n# Define function to cleanup text by removing personal pronouns, stopwords, and puncuation\npunctuations = string.punctuation\nstop_words = set(stopwords.words(\"english\"))\n\ndef cleanup_text(docs):\n    texts = []\n    for doc in docs:\n        doc = re.sub(r'[^a-zA-Z\\s]', '', doc, re.I|re.A)\n        doc = nlp(doc, disable=['ner'])\n        tokens = [tok.lemma_.lower().strip() for tok in doc if tok.lemma_ != '-PRON-']\n        tokens = [tok for tok in tokens if tok not in stop_words and tok not in punctuations]\n        tokens = ' '.join(tokens)\n        texts.append(tokens)\n    return pd.Series(texts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd669d3b16208535ee597d5c527e4a46c0e91cc6"},"cell_type":"code","source":"# Grab all text associated with insincere questions\ninsincere_text = [text for text in df[df['target'] == 1]['question_text']]\ninsincere_clean = cleanup_text(insincere_text)\ninsincere_clean = ' '.join(insincere_clean).split()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97b964439b1f0e0b8bbf4b75836aebd0ce6955db"},"cell_type":"code","source":"# Count all unique words\ninsincere_counts = Counter(insincere_clean)\n# get words and word counts\ninsincere_common_words = [word[0] for word in insincere_counts.most_common(20)]\ninsincere_common_counts = [word[1] for word in insincere_counts.most_common(20)]\n\n# plot 20 most common words in insincere questions\nsns.barplot(insincere_common_words, insincere_common_counts, palette='viridis')\nsns.despine(left=True, bottom=True)\nplt.xlabel('')\nplt.ylabel('')\nplt.title('Insincere Questions Common Words', fontsize=30)\nplt.tick_params(axis='x', which='major', labelsize=15)\nfig.savefig('insincere_words.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c32e41120eb9871970cb51c04e268e76b45246f"},"cell_type":"code","source":"# Grab all text associated with sincere questions\nsincere_text = [text for text in df[df['target'] == 0]['question_text']]\nsincere_clean = cleanup_text(sincere_text)\nsincere_clean = ' '.join(sincere_clean).split()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99f55a49dfabae2ee708aca1bf5fcedb47825b5a"},"cell_type":"code","source":"# Count all unique words\nsincere_counts = Counter(sincere_clean)\n# get words and word counts\nsincere_common_words = [word[0] for word in sincere_counts.most_common(20)]\nsincere_common_counts = [word[1] for word in sincere_counts.most_common(20)]\n\n# plot 20 most common words in sincere questions\nsns.barplot(sincere_common_words, sincere_common_counts, palette='viridis')\nsns.despine(left=True, bottom=True)\nplt.xlabel('')\nplt.ylabel('')\nplt.title('Sincere Questions Common Words', fontsize=30)\nplt.tick_params(axis='x', which='major', labelsize=15)\nfig.savefig('sincere.png')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49dda8f4cdb2f1d2c1801de60164c0e688636456"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}