{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \n\nfrom nltk.corpus import stopwords, words, qc, sentiwordnet as swn\nfrom nltk.stem.porter import PorterStemmer\nfrom nltk import FreqDist, WordNetLemmatizer\nfrom nltk import help, pos_tag, pos_tag_sents, word_tokenize\n\nimport unicodedata\nfrom collections import defaultdict\nimport string\nimport re\nimport os\n#print(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"**Table of Contents**\n* Load Data\n* Clean Data\n* Basic Text Features\n* Data Exploration\n\n**Load Data**\n"},{"metadata":{"trusted":true,"_uuid":"0a327b27ffa80d1045bea0b972f30b71dd11be3d"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\n\ntrain.set_index('qid',inplace=True,drop=True)\ntest.set_index('qid',inplace=True,drop=True)\n\ntrain.info()\ntrain.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd8552752a69f13ce0160980e5a86e88865d4ffe"},"cell_type":"code","source":"# Work on subset of data for now\n\n# Downsampled sincere q's\nsincere = train[train.target==0].sample(frac=0.1)\n\n# All insincere q's\ninsincere = train[train.target==1]\n\ntrain = sincere.append(insincere)\n\ntrain.info()\n\n# Target distribution of subset\nprint(train.target.value_counts(normalize=True))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"83dfd204d050c60f7ef65f25e64b74ab7285a263"},"cell_type":"markdown","source":"**Clean Data**\n* Check for distinct values in ID and target columns\n* Missing values\n* Non-ASCII characters\n* Correct accented characters\n* Future steps: Expand contractions (\"isn't\", \"don't\", etc.)"},{"metadata":{"trusted":true,"_uuid":"83e4e7df113c144abcb1458f053c34a0a121e841"},"cell_type":"code","source":"# Concatenate train and test questions\nX = pd.concat([train.drop('target', axis=1), test])\n\n# Check for distinct values\nprint(\"Total Rows:\",X.index.size)\nprint(\"Distinct Rows:\",X.index.nunique())\n\n# Check for missing values\nnan_cols = list(X.columns[X.isnull().any()])\nprint(\"Number of columns with NaN values:\",len(nan_cols))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e436c8906ccdb88102c6f1094af24f7fe06516e1"},"cell_type":"code","source":"# Check for non-ascii characters\nX['non_ascii'] = X.question_text.apply(lambda x: len(x) != len(x.encode()))\n\nprint(X['non_ascii'].value_counts(normalize=True))\n\nX[X['non_ascii']==1].sample(2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f8a1f9c5628cb92a836a12ea6c6558f886fe756"},"cell_type":"markdown","source":"We see 2% of the questions containing non-ASCII characters. Let's examine which characters we are appearing."},{"metadata":{"trusted":true,"_uuid":"90c6f956f8ef8785865140c41aff9cfca0edce6b"},"cell_type":"code","source":"# Concatente all questions that contain non-ASCII chars\nnon_ascii_questions = X['question_text'][X['non_ascii']].str.cat()\n\n# Dictionary to store frequency distribution\nchar_count = defaultdict(int)\n\nfor char in non_ascii_questions:\n    try:\n        char.encode('ascii')\n    except UnicodeEncodeError:\n        char_count[char] += 1\n\n# Convert dictionary to DataFrame\nchar_count_df = pd.DataFrame(data=list(char_count.items()),\n                             columns=['Char','Count'])\n\nchar_count_df['Percent of Total'] = char_count_df['Count'] / char_count_df['Count'].sum()\n\n# Character code (for use with chr())\nchar_count_df['Char Code'] = char_count_df['Char'].apply(lambda x: ord(x))\n\nchar_count_df.sort_values(by='Count',\n                          inplace=True,\n                          ascending=False)\n\nprint('Distinct non-ASCII chars:',len(char_count.keys()))\nprint(\"Total non-ASCII chars:\",char_count_df['Count'].sum())\n\nchar_count_df.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a29fd8481116f21b98e43a041300c629f4908c63"},"cell_type":"markdown","source":"Looking at these values, we can fix many of these issues by converting some of the chars to similar ASCII values. "},{"metadata":{"trusted":true,"_uuid":"7f9d93627f7ac2e53ed0ab6778a3b28f6c247ed8"},"cell_type":"code","source":"# Manually map corrected characters\ncorrections = {chr(8217):'\\'',\n               chr(8221):'\"',\n               chr(8220):'\"',\n               chr(8230):'...',\n               chr(8216):'\\'',\n               chr(247):'/',\n               chr(960):'pi',\n               chr(215):'x',\n               chr(8211):'-',\n               chr(180):'\\'',\n               chr(65311):'?'}\n\n# Map to manually corrected chars\ndef correct_non_ascii(question):\n    result = ''\n    for char in question:\n        if char in corrections.keys():\n            result += corrections[char]\n        else:\n            result += char\n    return result\n\nX.loc[X['non_ascii'],'question_text'] = X.loc[X['non_ascii'],'question_text'].apply(correct_non_ascii) \n\n# Correct Accents and remove other non-ASCII chars\ndef remove_accented_chars(text):\n    text = unicodedata.normalize('NFKD', text).encode('ascii', 'ignore').decode('utf-8', 'ignore')\n    return text\n\nX.loc[X['non_ascii'],'question_text'] = X.loc[X['non_ascii'],'question_text'].apply(remove_accented_chars)\n\n# Check for non-ascii characters again\nprint(\"Number of non-ascii characters:\",X.question_text.apply(lambda x: len(x) != len(x.encode())).sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7d722a8101e5b4409954dbd677a7362df68923b3"},"cell_type":"code","source":"X.drop('non_ascii',axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c714e434c0117782f3d571974eae38dead3a684f"},"cell_type":"markdown","source":"**Basic Feature Extraction**\n\nThese features will be generated from the raw text (before preprocessing). We will extract the following features:\n\n* Number of words\n* Number of characters\n* Average word length (chars)\n* Number of numerical characters\n* Number of punctuation marks\n* Number of stopwords\n* Number of non-stopwords\n* Lexical diversity\n* Number of Uppercase words\n* POS Tagging\n"},{"metadata":{"trusted":true,"_uuid":"fd3c69559ec853106972ab97ffe435ee3b7e8b3a"},"cell_type":"code","source":"# Number of words\nX['word_count'] = X.question_text.apply(lambda x: len(x.split()))\n\n# Number of characters\nX['char_count'] = X.question_text.apply(lambda x: len(x))\n\n# Average word length (chars)\nX['avg_word_len'] = X['char_count'] / X['word_count']\n\n# Number of numerical characters\nX['numerics_count'] = X['question_text'].apply(lambda x: len([char for char in x if char.isnumeric()]))\n\n# Number of punctuation characters\n# EOS punctuation: .?!\nX['punct_count'] = X['question_text'].apply(lambda x: len([char for char in x if char in string.punctuation]))\n\n# Number of stopwords\nstop_words = stopwords.words('english')\nX['stopword_count'] = X['question_text'].apply(lambda x: len([word for word in re.split(r'\\W+', x) if word.lower() in stop_words]))\n\n# Number of non-stopwords\nX['non_stopword_count'] = X['word_count'] - X['stopword_count']\n\n# Lexical diversity - word count / number of distinct words\ndef lexical_diversity(question):\n    word_count = len(re.findall(r'\\W+',question))\n    vocab_size = len(set([word.lower() for word in question.split()]))\n    return float(word_count/vocab_size)\n\nX['lex_diversity'] = X['question_text'].apply(lexical_diversity)\n\n# Number of uppercase words\ndef uppercase_count(question):\n    uppercase_words = []\n    for word in question.split():\n        if word.isupper() and len(word) > 3:\n            uppercase_words.append(word)\n    return len(uppercase_words)\n\nX['uppercase_count'] = X['question_text'].apply(uppercase_count)\n\n#X['sentiment']\n\nX.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"8f07edaead07ef89e41797101945879b49eef83b"},"cell_type":"code","source":"# POS Tagging (from https://www.nltk.org/book/ch05.html)\n\n# Tag    Meaning              English Examples\n# ADJ    adjective            new, good, high, special, big, local\n# ADP    adposition           on, of, at, with, by, into, under\n# ADV    adverb               really, already, still, early, now\n# CONJ   conjunction          and, or, but, if, while, although\n# DET    determiner, article  the, a, some, most, every, no, which\n# NOUN   noun                 year, home, costs, time, Africa\n# NUM    numeral              twenty-four, fourth, 1991, 14:24\n# PRT    particle             at, on, out, over per, that, up, with\n# PRON   pronoun              he, their, her, its, my, I, us\n# VERB   verb                 is, say, told, given, playing, would\n# .      punctuation marks    . , ; !\n# X      other                ersatz, esprit, dunno, gr8, univeristy\n\n# Sentence tag descriptions\nhelp.upenn_tagset()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75f531bed3008ef13beed8a45b4c2a46b07c5b73"},"cell_type":"code","source":"X['pos_tag_tuples'] = pos_tag_sents(X['question_text'].apply(word_tokenize))\n\nX['pos_tags'] = X['pos_tag_tuples'].apply(lambda x: [tup[1] for tup in x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"793a9e49fad4537d22003646c2776beb8546ed3d"},"cell_type":"code","source":"X.head(2)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f8d040000a3067110d4eb949cc0d7c938ca2e4f6"},"cell_type":"markdown","source":"**Exploratory Data Analysis**\n\nNow that we've gathered some basic features, let's plot some values to help develop a hypothesis."},{"metadata":{"trusted":true,"_uuid":"13012710d18a7ca104b05913735e36f9b6b59995"},"cell_type":"code","source":"X_graph = pd.concat(objs=[X, train[['target']]],\n                    axis=1,\n                    join_axes=[train.index])\n\n# target distribution\nprint(\"Number of rows\",X_graph.size)\nprint(X_graph.target.value_counts())\nprint(X_graph.target.value_counts(normalize=True))\n\nX_graph.target.value_counts(normalize=True).plot(kind='bar',\n                                                 title='Distribution of Target (%)');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49245f6dc2dc46621c7be880f54a21e4b31d7630"},"cell_type":"code","source":"# Sample insincere questions\nX_graph.loc[X_graph.target==1,'question_text'].sample(5).values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"867512767e28ad4745dc714b387257bfdf1bbc5e"},"cell_type":"markdown","source":"After seeing some insincere questions, my impression is to focus on negative adjectves that could be pulled from POS tagging. To get a clearer hypothesis, let's extract some basic features. For now, let's work on a subset of the sincere questions to speed things up."},{"metadata":{"trusted":true,"_uuid":"182f8f5b9996981e3dcc50c752c58e1f5c20e7c2","scrolled":false},"cell_type":"code","source":"# Percentiles for numerical columns\nX_graph = X.merge(train[['target']],left_index=True, right_index=True)\n\nbounds = X_graph.describe(percentiles=[.25, .5, .75, .999]).T\n\n# Dictionary of data types and column names\ndtypes_dict = X_graph.columns.to_series().groupby(X_graph.dtypes).groups\ncol_dtypes = {key.name:set(value) for key, value in dtypes_dict.items()}\n\ngraph_cols = (col_dtypes['int64'] - {'target'}) | col_dtypes['float64']\n\nfor col in graph_cols:\n    X_graph.groupby('target')[col].plot(kind='kde',\n                                        legend=True,\n                                        title=col)\n    plt.xlim(bounds.loc[col]['min'],bounds.loc[col]['99.9%']) # exclude outliers from density plot\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9aefb23a7f558fb47e3859d5ceff8b337f6e92f7"},"cell_type":"code","source":"def preprocessing(question):\n    # Convert to lower case\n    cleaned = \" \".join(word.lower() for word in re.split(r'\\W+',question))\n\n    # Remove punctuation\n    cleaned = re.sub(r'[^\\w\\s]',' ',cleaned) # do we want to keep website url's?\n\n    # Keep only alphabetical characters\n    cleaned = \" \".join(word for word in cleaned.split() if word.isalpha())\n    \n    # Remove stop words\n    cleaned = \" \".join(word for word in cleaned.split() if word not in stop_words)\n\n    return cleaned\n\nX['clean_text'] = X['question_text'].apply(preprocessing)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c1d186e5f3405a01504165ee96f2052a7687d19"},"cell_type":"code","source":"# # Stemming\n# ps = PorterStemmer()\n\n# def stemmer(question):\n#     text = ' '.join([ps.stem(word) for word in question.split()])\n#     return text\n\n# Lemmatization\nwnl = WordNetLemmatizer()\n\ndef lemmatizer(question):\n    text = ' '.join([wnl.lemmatize(word) for word in question.split()])\n    return text\n\nX['clean_text'] = X['clean_text'].apply(lemmatizer)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fef31ba8fbb48ca78e664accc35e25aed2eb746d"},"cell_type":"code","source":"# Remove frequent words and rare words\n\ndef cound_words(df,col):\n    # Return dataframe containing frequency count of each word\n    word_bag = df[col].str.cat(sep=' ')\n\n    fdist = FreqDist([word for word in word_bag.split()])\n\n    print('Number of distinct words:',len(fdist.keys()))\n\n    word_count_df = pd.DataFrame(data=list(fdist.items()),\n                                 columns=['Word','Count'])\n    \n    word_count_df.set_index(keys='Word',\n                            drop=True,\n                            inplace=True)\n\n    word_count_df.sort_values(by='Count',\n                              inplace=True,\n                              ascending=False)\n    \n    return word_count_df\n\nword_count = cound_words(X,'clean_text')\n\nword_count.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"38c08d593fe0fa82cb3f47dd611b89b4d037e53b"},"cell_type":"code","source":"# Remove frequent and rare words\nMAX_WORD_COUNT = 8000\nMIN_WORD_COUNT = 10\n\ndef keep_word(word):\n    wc = word_count['Count'].loc[word]\n    if MIN_WORD_COUNT < wc < MAX_WORD_COUNT:\n        return True\n    else:\n        return False\n\ndef remove_words(question):\n    text = ' '.join([word for word in question.split() if keep_word(word)])\n    return text\n\nX['clean_text'] = X['clean_text'].apply(remove_words)\n\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a74e251df21c9618aba4f6500c189a35a3849f2b"},"cell_type":"markdown","source":"**Next Steps**\n* Spell check, Count Vectorizer, TF-IDF"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}