{"cells":[{"metadata":{"_uuid":"526bf9b0f4233a95504c95b82874c4e6f61ee8d6"},"cell_type":"markdown","source":"# Import"},{"metadata":{"trusted":false,"_uuid":"21c57d57414f6d479779227645dfdbed29da7856"},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport spacy\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"02b0fe574769da262d118870389fee16805e3788"},"cell_type":"markdown","source":"# Load the dataset"},{"metadata":{"trusted":false,"_uuid":"da09f4f19dccab4fa1c6959736eaa87c373ba0d8"},"cell_type":"code","source":"df_train = pd.read_csv(os.path.join('..', 'input', 'train.csv'))\ndf_test = pd.read_csv(os.path.join('..', 'input', 'test.csv'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cdc243c55b42d338f2ed7e15f1f1001bb209a377"},"cell_type":"markdown","source":"# Look at the dataset"},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"9987ad7fa1dde65e1ecc333b19e99f2cb05acf8b"},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"559dd4047537429703edf9643edf9cfeb8547790"},"cell_type":"code","source":"df_train.shape, df_test.shape","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"89368f85065d5bbdc189f7ebb449877a8af707cc"},"cell_type":"code","source":"df_train.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"46f93c652cadda4bea0e721471503b8202c3c503"},"cell_type":"markdown","source":"The data is clean, there is no Naan values"},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"5920aeec1d1c9b492a836607b1d3f9ec6415fab5"},"cell_type":"code","source":"df_train['target'].value_counts().plot(kind='bar');","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"27ab27a98c4d4bed834d51e0023e0cdad8293d7a"},"cell_type":"code","source":"insincere_ratio = (80810 / 1225312) * 100\ninsincere_ratio","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"74e71dbeb373e77113e084aabbc77d19c5454c73"},"cell_type":"code","source":"y = df_train['target']\nX = df_train['question_text']","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"803885305300df5118a495d6e4c0118e39a9a9db"},"cell_type":"code","source":"X_insincere = X[y == 1]\nX_insincere.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf53a32dc481a6237f9b5c716d02b0cdebc7579d"},"cell_type":"markdown","source":"We can already notice the troll content within the questions."},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"45ea845343b9def80ae240a6665b052462320b9f"},"cell_type":"code","source":"X_sincere = X[y == 0]\nX_sincere.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"93192bcab41fe2c65110d242b9b0ed70fd82d259"},"cell_type":"markdown","source":"Whereas within the sincere question, the questions are legit."},{"metadata":{"_uuid":"7b864d39631da43a7f6f74a560d8989b1aeb46bf"},"cell_type":"markdown","source":"# Preprocessing"},{"metadata":{"_uuid":"f38640a9ae108f31aa17663020dadd5d1e2ea4e5"},"cell_type":"markdown","source":"## on X"},{"metadata":{"trusted":false,"_uuid":"216bbe31c5d63e3d3d6939d5916dcf83b26d6a98"},"cell_type":"code","source":"from nltk.tokenize import word_tokenize","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"6fcb277d88d4951b2523c27986ffb29aea6138f5"},"cell_type":"code","source":"corpus = [word_tokenize(token) for token in X]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8d3791f0e014e1005c53e45000a40b36a280d3a0"},"cell_type":"code","source":"lowercase_train = [[token.lower() for token in doc] for doc in corpus]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1dba88d57a50edc04217ffae3fdc643542c8836e"},"cell_type":"code","source":"alphas = [[token for token in doc if token.isalpha()] for doc in lowercase_train]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c4e1342a868531e5ca41c49de333c5195d3e95b2"},"cell_type":"code","source":"from nltk.corpus import stopwords\nstop_words = stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"0f18fa8eb02e101289daf66d6175f1e0aa2cd4fd"},"cell_type":"code","source":"train_no_stop = [[token for token in doc if token not in stop_words] for doc in alphas]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8f9f3c620e657cc0a1d4b3bfb6618ef0fb8fdc97"},"cell_type":"code","source":"from nltk.stem import PorterStemmer\nstemmer = PorterStemmer()\nstemmed = [[stemmer.stem(token) for token in doc] for doc in train_no_stop]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d261d3789b4fc58357c9af09be4bf4789068a95e"},"cell_type":"code","source":"train_clean_str = [ ' '.join(doc) for doc in stemmed]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5eb17f1a0dff684fd31cb4780b0d576d62255241"},"cell_type":"markdown","source":"## Features "},{"metadata":{"_uuid":"ba8e62c688fe9f8a9239fe06a7d6d7393bd2bf6b"},"cell_type":"markdown","source":"**1)** Number of words"},{"metadata":{"trusted":false,"_uuid":"903336840c05ecbb780e755a09ae30f48caed08e"},"cell_type":"code","source":"nb_words = [len(tokens) for tokens in alphas]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d125361f2a1dccbff4e1989a7f44c44452df6ab0"},"cell_type":"markdown","source":"**2)** Number of unique words"},{"metadata":{"trusted":false,"_uuid":"20d798fb3f22eabd054c3f49adc1f95dca535c15"},"cell_type":"code","source":"alphas_unique = [set(doc) for doc in alphas]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"27a81d9d9d5f887ebc96d635fea615b62c5b8f92"},"cell_type":"code","source":"nb_words_unique = [len(doc) for doc in alphas_unique]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ebd98655d22d97965769eb5aed4ba449f98ad185"},"cell_type":"markdown","source":"**3)** Number of characters"},{"metadata":{"trusted":false,"_uuid":"2df5d1d2c2ec485fac0cc006ab3a043219314c1f"},"cell_type":"code","source":"train_str = [ ' '.join(doc) for doc in lowercase_train]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"6cd596c23c8b1d01409a8480ea86b2afd59bd43d"},"cell_type":"code","source":"nb_characters = [len(doc) for doc in train_str]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"43ce9749b3e73626f6f2898ff6550e87eb3474b7"},"cell_type":"markdown","source":"**4)** Number of stopwords"},{"metadata":{"trusted":false,"_uuid":"de15d981ef7d1bb33c75422270571766c55dd952"},"cell_type":"code","source":"train_stopwords = [[token for token in doc if token in stop_words] for doc in alphas]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"956a758d34c84d532c98ccff0959f1d0e20ff124"},"cell_type":"code","source":"nb_stopwords = [len(doc) for doc in train_stopwords]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c6a89803c39c0323679dfa3b115ac1a358a18b31"},"cell_type":"markdown","source":"**5)** Number of punctuations"},{"metadata":{"trusted":false,"_uuid":"7404274a22583bb8e90aedcd3dfd67a08a91e7cc"},"cell_type":"code","source":"non_alphas = [[token for token in doc if token.isalpha() == False] for doc in lowercase_train]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"55e3c75cabdb877e0bedf3c2f0bea02d6dd0885e"},"cell_type":"code","source":"nb_punctuation = [len(doc) for doc in non_alphas]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d2cdbbbdec0d69e6a37af07f96946c9d955451fb"},"cell_type":"markdown","source":"**6)** Number of title case words"},{"metadata":{"trusted":false,"_uuid":"930f2b46c3d775fb6637164955656761e97505e5"},"cell_type":"code","source":"train_title = [[token for token in doc if token.istitle() == True] for doc in corpus]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c5cdd2be7b0de5a89f11c02ab61c38604031323a"},"cell_type":"code","source":"nb_title = [len(doc) for doc in train_title]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"02f5b67ca26c6233e9cc655106a99c2630319684"},"cell_type":"markdown","source":"# New Dataframe with features"},{"metadata":{"trusted":false,"_uuid":"9f6e07fcd823730870fd1a5278a235d39558f22d"},"cell_type":"code","source":"df_clean = pd.DataFrame(data={'text_clean': train_clean_str})\ndf_clean.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"6ad1a69d49cd35bb6152d056921c0a654eacbbc0"},"cell_type":"code","source":"nb_words = pd.Series(nb_words)\nnb_words_unique = pd.Series(nb_words_unique)\nnb_characters = pd.Series(nb_characters)\nnb_stopwords = pd.Series(nb_stopwords)\nnb_punctuation = pd.Series(nb_punctuation)\nnb_title = pd.Series(nb_title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"e19cddcbc4e9376d14a06f29dd614c402520e365"},"cell_type":"code","source":"df_show = pd.concat([df_clean, nb_words, nb_words_unique, nb_characters, nb_stopwords, nb_punctuation, nb_title], axis=1).rename(columns={\n    0: \"Number of words\", 1: 'Number of unique words', 2: 'Number of characters', 3: 'Number of stopwords', 4: 'Number of punctuations',\n    5: 'Number of titlecase words'\n})\ndf_show.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"56a57aa80d30cd5fcae288e6e71f74f9e39cd5dc"},"cell_type":"code","source":"df_feat = df_show.drop(['text_clean'], axis=1)\ndf_feat.head()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"25623399152540b34ec7de2267dddf9ee9ae93b5"},"cell_type":"code","source":"df_feat.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"88b5b7bdd3fc9e0910c78f2afe917d935c3ad78e"},"cell_type":"markdown","source":"For now, this represents too much data to visualise. We'll start with the insincere one."},{"metadata":{"_uuid":"54a38bb71b8ac72af3af3aed1e9a5c64f50305e5"},"cell_type":"markdown","source":"## EDA on the X_insincere"},{"metadata":{"trusted":false,"_uuid":"fe3a185b32a9e23438e17435789ddaf5aac7a0ee"},"cell_type":"code","source":"from nltk.tokenize import word_tokenize","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"01005217fd92eea296f2f3910d8c63c17cdc1999"},"cell_type":"markdown","source":"Let's tokenize our document"},{"metadata":{"trusted":false,"_uuid":"df5aff8652ce2cbd3d745f92a1dc9e2e6495b248"},"cell_type":"code","source":"%%time\ncorpus_insincere = [word_tokenize(t) for t in X_insincere]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df35a7bb5e7509eff475f4daec682aaee0773841"},"cell_type":"markdown","source":"Lowercase all the words"},{"metadata":{"trusted":false,"_uuid":"6f15148c8f74480ba012d1de5f523e89f4b31332"},"cell_type":"code","source":"lowercase = [[t.lower() for t in doc] for doc in corpus_insincere]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a43596c0d331565594ea61f4202b8f1824a43dc"},"cell_type":"markdown","source":"Remove stop words"},{"metadata":{"trusted":false,"_uuid":"f7baed35ae1a578f91328292f057e4b3cccac98c"},"cell_type":"code","source":"from nltk.corpus import stopwords\nstop_words = stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"712f26f58cea387790a2ecade26594346b6449ef"},"cell_type":"code","source":"no_stops = [[t for t in doc if t not in stop_words] for doc in lowercase]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"817b053ec51bf1133f3ab7d5317bf0119e21f2ff"},"cell_type":"markdown","source":"Remove non-alpha-numerical caracters"},{"metadata":{"trusted":false,"_uuid":"b4015c3bfb021ed0a591b7438463fd5b83677c99"},"cell_type":"code","source":"alphas_insincere = [[token for token in doc if token.isalpha()] for doc in no_stops]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df16dd7e69225fe0d236274f20ae71b3bb94adf5"},"cell_type":"markdown","source":"Stem the words"},{"metadata":{"trusted":false,"_uuid":"09d75b852d107b3236baf5641f89a9d74360eaf9"},"cell_type":"code","source":"from nltk.stem import PorterStemmer\nstemmer = PorterStemmer()\nstemmed_insincere = [[stemmer.stem(token) for token in doc] for doc in alphas_insincere]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"23eee785226fc464161ccf76848d38739c4b166a"},"cell_type":"markdown","source":"Show the length of \"insincere\" sentence "},{"metadata":{"trusted":false,"_uuid":"28d8f4c0bb363b687d37c78570b480565dc18282"},"cell_type":"code","source":"nb_words_insincere_nostop = [len(tokens) for tokens in no_stops]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0a4e9b960e744075f8d6a14b1446887c5e262313"},"cell_type":"markdown","source":"Average number of words per insincere question "},{"metadata":{"trusted":false,"_uuid":"8bb2f9d23bd02a87323797083465723a2021f436"},"cell_type":"code","source":"avg_nostop = np.mean(nb_words_insincere_nostop)\navg_nostop","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"be3b14160d9186ce301d4a290d124f09f6d18410"},"cell_type":"code","source":"nb_words_insincere_stop = [len(tokens) for tokens in lowercase]\navg_stop = np.mean(nb_words_insincere_stop)\navg_stop","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1beb89efe3e7925b17d6d0a99b1efebddd742513"},"cell_type":"code","source":"np.median(nb_words_insincere_nostop)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"83f063cfc83a3d14d6144cea0fe99e0c1f1ec45c"},"cell_type":"code","source":"np.median(nb_words_insincere_stop)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"860615a02e6352ae4ed6c49e9356dbec609e2f1f"},"cell_type":"code","source":"nb_words_insincere_stop = pd.Series(nb_words_insincere_stop)\nnb_words_insincere_nostop = pd.Series(nb_words_insincere_nostop)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"47aee6cc88af04390e80acd0247c7ba4c6da1959"},"cell_type":"code","source":"df_insincere =  pd.DataFrame(X_insincere)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"051ccd809768071552559dd12396c045b3c15442"},"cell_type":"code","source":"df_insincere.info()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"36746bb0b3728b4452d585d2391b835cd0f0c0da"},"cell_type":"code","source":"df_insincere = pd.concat([X_insincere.reset_index(), nb_words_insincere_nostop, nb_words_insincere_stop], axis=1).set_index('index').rename(columns={\n    0: \"nb_words_no_stop\", 1: 'nb_words_stop'\n})\ndf_insincere.head()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"834ad8c3f4c58e4a8c316962ac3851378b3a4d08"},"cell_type":"code","source":"#plt.hist(nb_words_insincere_nostop, bins=30)\n#plt.hist(nb_words_insincere_stop, bins=30)\nsns.distplot(np.log1p(nb_words_insincere_nostop), kde=False, label=\"No stop\")\nsns.distplot(np.log1p(nb_words_insincere_stop), kde=False, label=\"Stop\")\nplt.legend();","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"dc403552f10a30260dde72ee9e20e257fde68e4b"},"cell_type":"code","source":"sns.distplot(nb_words_insincere_stop, hist=False, color='red', label='Stop')\nsns.distplot(nb_words_insincere_nostop, hist=False, color='blue', label='No stop')\nplt.legend();","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b9110c579403955310d03edb8892cd0b97af5fe3"},"cell_type":"markdown","source":"The average number of words within the insincere questions is around **11 words** (without the stop_words) and **19 words** with the stop_words.\nLet's compare it with the proper questions."},{"metadata":{"_uuid":"4f38717033a86b60674c6591f22b6173825f9359"},"cell_type":"markdown","source":"**Counting the ten most common words in the insincere questiosn**"},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"e2ac158b3263cb32bb5b9316ed2df3014e05785a"},"cell_type":"code","source":"from collections import defaultdict\n\ncounter = defaultdict(int)\nfor doc in alphas_insincere:\n    for token in doc:\n        counter[token] += 1\n\nfrom collections import Counter\n\nc = Counter(counter)\n\nc.most_common(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7889d4c3e7cbd74daa413df044fa0e1ac7bdfc92"},"cell_type":"markdown","source":"## EDA on the X_sincere"},{"metadata":{"trusted":false,"_uuid":"5642296c8e883d161a88654339e7bf7beecf9fe3"},"cell_type":"code","source":"corpus_sincere = [word_tokenize(t) for t in X_sincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b9d0a096958603b79a0709ea88474e44b7cbc7b3"},"cell_type":"code","source":"lowercase_sincere = [[t.lower() for t in doc] for doc in corpus_sincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"071cf9ad25348a9d2011c1d246d54baebc391e72"},"cell_type":"code","source":"no_stop_sincere = [[t for t in doc if t not in stop_words] for doc in lowercase_sincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"09c5e2f8bd9fcfc25470824812035535e61e73f9"},"cell_type":"code","source":"alphas_sincere = [[token for token in doc if token.isalpha()] for doc in no_stop_sincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"49e46651d4a4cb74ec172d0947f9b3c6c60be911"},"cell_type":"code","source":"nb_words_sincere_nostop = [len(t) for t in no_stop_sincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1e722c2d163ede74c19194f5c17d0fd904ec8c7a"},"cell_type":"code","source":"avg_words_sincere_nostop = np.mean(nb_words_sincere_nostop)\navg_words_sincere_nostop","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7859e8fa46817c99812cb799dc7119e2325c309a"},"cell_type":"code","source":"nb_words_sincere_stop = [len(t) for t in lowercase_sincere]\navg_words_sincere = np.mean(nb_words_sincere_stop)\navg_words_sincere","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c615625b63e50ce4c4ff4fd5e59a812e2852fe95"},"cell_type":"code","source":"np.median(nb_words_sincere_nostop)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ce5ea3264801cce0b4409b4ab2c3f4351db70f82"},"cell_type":"code","source":"np.median(nb_words_sincere_stop)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0be96edbea4d7022a9079f2683c3cbaa920e8604"},"cell_type":"markdown","source":"The average number of words within the sincere questions is around **8 words** (without the stop_words) and **14 words** with the stop_words. Let's compare it with the proper questions."},{"metadata":{"trusted":false,"_uuid":"737c2ed5272e4c58d3e6aebc9d8ca8ca6da3ec0e"},"cell_type":"code","source":"nb_words_sincere_stop = pd.Series(nb_words_sincere_stop)\nnb_words_sincere_nostop = pd.Series(nb_words_sincere_nostop)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5cb10cfead95a2d807344ef2e5f5b05b698171d4"},"cell_type":"code","source":"df_sincere =  pd.DataFrame(X_sincere)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c5735e565cec05eb76d5b2b1d6ea20de8837af07"},"cell_type":"code","source":"df_sincere.info()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"0fc26e54aa34c40036aba1a5f27dbaa3453eed94"},"cell_type":"code","source":"df_sincere = pd.concat([X_sincere.reset_index(), nb_words_sincere_nostop, nb_words_sincere_stop], axis=1).set_index('index').rename(columns={\n    0: \"nb_words_no_stop\", 1: 'nb_words_stop'\n})\ndf_sincere.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ed86bd89b1948304e77ce44fba02996fac3606d"},"cell_type":"markdown","source":"**Counting the ten most common words in the sincere questiosn**"},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"19ae907ff3fd4f250aa28cb71025d8424bcbfa8b"},"cell_type":"code","source":"from collections import defaultdict\n\ncounter_sincere = defaultdict(int)\nfor doc in alphas_sincere:\n    for token in doc:\n        counter[token] += 1\n\nfrom collections import Counter\n\nc_sincere = Counter(counter)\n\nc_sincere.most_common(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a288d11611d6db3c92388e6ea00cec35dcb7beda"},"cell_type":"markdown","source":"# Topic Modeling"},{"metadata":{"_uuid":"beaafa405edab7cfea0a2ba74b0ebca2e5a7798e"},"cell_type":"markdown","source":"## Latent Semantic Analysis"},{"metadata":{"_uuid":"8cbe3a4744f565f39ebd6104984af51b6460975d"},"cell_type":"markdown","source":"Our goal is to determine most common topics among the insincere questions.\n\nFirst, we use our tokenized document that has been preprocessed."},{"metadata":{"_uuid":"901f82a420944f68c7e47cdf037001c0aba3a8e5"},"cell_type":"markdown","source":"Then we use Gensim to achieve our LSA."},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"3451fdc2c6a3f7287fa969aebbfdc6c45c57dbeb"},"cell_type":"code","source":"from gensim import corpora\ndictionary = corpora.Dictionary(alphas_insincere)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"5e5c0b51572f9117ec5a7147886b3e3a4cd19aea"},"cell_type":"code","source":"corpus_1 = [dictionary.doc2bow(t) for t in alphas_insincere]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"7e4933139e2c9920e2b8fce976eb1d50be7d3fa5"},"cell_type":"code","source":"from gensim.models.ldamodel import LdaModel","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"527a9623830480a414816feb894b6726294b0b8d"},"cell_type":"code","source":"%%time\nlda_model = LdaModel(\n    corpus=corpus_1, id2word=dictionary, num_topics=4, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"667c776f59e342f690a38fdbd06ccf967b19a5b6"},"cell_type":"code","source":"from pprint import pprint\npprint(lda_model.print_topics())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"38ea0cc7d79053aa0908bb2393bc9ef40417eb55"},"cell_type":"code","source":"%%time\nlda_model_1 = LdaModel(\n    corpus=corpus_1, id2word=dictionary, num_topics=4, random_state=42, iterations=10)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"519791b61ea0376a3e87d83c2c7f1020655b2000"},"cell_type":"code","source":"from pprint import pprint\npprint(lda_model_1.print_topics())","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d2fb58e6774c7e3339850ada04fc4895ffcfc4f3"},"cell_type":"code","source":"import pyLDAvis.gensim","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d6561b81e51a92e0531531acd91be43666740b73"},"cell_type":"code","source":"pyLDAvis.enable_notebook()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"f3295fb0e9dc24cf82885d0b2840d8790ee4cede"},"cell_type":"code","source":"pyLDAvis.gensim.prepare(lda_model, corpus_1, dictionary)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"6a50d0278c2c773d5f272730be014ada9047ee6c"},"cell_type":"code","source":"pyLDAvis.gensim.prepare(lda_model_1, corpus_1, dictionary)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"1f3b0e084ec638303edc53336e93b37f327e50e9"},"cell_type":"code","source":"weight_topic = lda_model_1.top_topics(corpus=corpus_1, dictionary=dictionary, topn=30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"6e81098b9833814e91f4aba2107c132844d55ba6"},"cell_type":"code","source":"politic, religion, sex, america = weight_topic","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"aadbcf92b3260ffed1337dc0cd721b46976c9d53"},"cell_type":"code","source":"politic = politic[0]\npolitic = [tup[1] for tup in politic]\npolitic","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"bffd5036d91b3834800555f7e2246fd7f889d99c"},"cell_type":"code","source":"religion = religion[0]\nreligion = [tup[1] for tup in religion]\nreligion","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"4cdc1eaf644d36229d1c76d86522fd102840de7a"},"cell_type":"code","source":"sex = sex[0]\nsex = [tup[1] for tup in sex]\nsex","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"7124a4aa7dc530c2e58383665909e46f960d4eb8"},"cell_type":"code","source":"america = america[0]\namerica = [tup[1] for tup in america]\namerica","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1553a2e9baaa9093115684b0717ff3d10887b532"},"cell_type":"markdown","source":" We will now add  a new feature to check whether a question contains at least 2 words of a topic to tag it insincere."},{"metadata":{"trusted":false,"_uuid":"2602a0116b9395ec2cf2a05591ac2b0f86410833"},"cell_type":"code","source":"y_labeled = []","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"0df9591ff37874e5b7a2e9b59fa647eb3b9f11d2"},"cell_type":"code","source":"for doc in train_no_stop:\n    counter = 0\n    for word in doc:\n        if word in politic or word in religion or word in sex or word in america:\n            counter += 1\n    if counter >= 3:\n        y_labeled.append(1)\n    else:\n        y_labeled.append(0)\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d1b7b297907feec03f6bcc86a8d52f045f9a5686"},"cell_type":"code","source":"y_labeled[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"dd12755cea47ded5f4868334ca5429e97346b305"},"cell_type":"code","source":"y_labeled = pd.Series(y_labeled)\ny_labeled[:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"92d7bfed63665f6143dad8a8259387335d8ac62f"},"cell_type":"code","source":"df_feat['y_topic_labeled'] = y_labeled","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ae694e985d35857cf1606f99477cc126d812cf43"},"cell_type":"code","source":"df_show['y_topic_labeled'] = y_labeled","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"fa691ff1f939f88a22352a0b1c8a501f7f50d979"},"cell_type":"code","source":"df_feat.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"12e06e2403234bb8e0d172e7eca848a079e457d5"},"cell_type":"code","source":"df_feat['y_topic_labeled'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"837f1bbc0e1604041f3ee519ebe047acb919d12c"},"cell_type":"code","source":"df_show.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de2f55c4be656510753b06af76bec777c1ff9972"},"cell_type":"markdown","source":"# Machine Learning"},{"metadata":{"_uuid":"9cc7126b6c3c601958dc1b383c410f0594a5c66e"},"cell_type":"markdown","source":"We'll try to combine two models : \n\n1) First, we will process over our raw document as intermediate predictions.\n\n\n2) Secondly, we will add our predictions as a new feature in our dataframe features and try ML models over them."},{"metadata":{"trusted":false,"_uuid":"e5cc143e15c2fd5904eb7999e5dceb42bb2337f6"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f020a13f1202f281a1846edc1ef3b4425b919042"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"fba2acab548b3a8f4d50ace95e72ed48306b23be"},"cell_type":"code","source":"X_train.shape, X_test.shape, y_train.shape, y_test.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"267d20e82a52fe48aab545e091f31468928fd4d0"},"cell_type":"markdown","source":"## Over our raw questions"},{"metadata":{"_uuid":"7c2603876b75fda07be0f17057a9c4656cbd7788"},"cell_type":"markdown","source":"## Preprocessing"},{"metadata":{"_uuid":"6010eb1af9ea3fa516b1948465d311a844276483"},"cell_type":"markdown","source":"### TfidfVectorizer"},{"metadata":{"trusted":false,"_uuid":"da07d03c50291ea4f4b60a225e12aef8d8f716b9"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer, TfidfTransformer","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"e6a9d56177fdbcc6e4f8f86f1db0d7ace3d3401b"},"cell_type":"code","source":"tvec = TfidfVectorizer(stop_words='english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"909a77ea852b7245b9fe9928ef216984e3b594b7"},"cell_type":"code","source":"tf = tvec.fit_transform(X_train)\ntf","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6d378574450554791f4fae1a995058bedd85e7fc"},"cell_type":"markdown","source":"### CountVectorizer"},{"metadata":{"trusted":false,"_uuid":"74247a47674328a05c96d743cd9a2dc83994e8d2"},"cell_type":"code","source":"cvec = CountVectorizer(stop_words='english')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"288685412b3de9f4ca6f5add808611d15c3dde86"},"cell_type":"markdown","source":"### Truncated SVD"},{"metadata":{"trusted":false,"_uuid":"0772f6f8fad7d359ccbbbda251072e78de5b37f2"},"cell_type":"code","source":"from sklearn.decomposition import TruncatedSVD","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"e233fd4232e5a9f0612b05c0733875f0a46458a6"},"cell_type":"code","source":"svd = TruncatedSVD(n_components=100, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f5f1e1f0b76c2215cdc61df3ff76a3fa1092bd6e"},"cell_type":"markdown","source":"### Preprocessing pipeline"},{"metadata":{"trusted":false,"_uuid":"beb7b018eb6ba2b9b280e63cc22b0db7c21fcb19"},"cell_type":"code","source":"from sklearn.pipeline import Pipeline","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"bc56ded0dedd9e61367b627474d8efe048714827"},"cell_type":"code","source":"preprocessing_pipeline = Pipeline([('tvec', tvec), ('svd', svd)])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"211650766ce285c0aed3b746a9606591e1310de3"},"cell_type":"code","source":"preprocessing_pipeline.fit_transform(X_train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f0c2ddf062479710d3177975a213ac9fec8b7437"},"cell_type":"markdown","source":"## Machine learning models"},{"metadata":{"trusted":false,"_uuid":"e5aa9a6ea0b4990f0abaea037a7e35d9205bdf57"},"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"52c04454c6e77e885a581f1c5803166b286e8a7f"},"cell_type":"markdown","source":"### MultinomialNB"},{"metadata":{"trusted":false,"_uuid":"83f2824bb1dfede5619ca34e73615e3eb30e3deb"},"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"3d6f1d6e9add85f8413ee1a5cb1eaa0ab663119c"},"cell_type":"code","source":"mnb = MultinomialNB()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"df1abcfdcd156166e11614145116ce9297984829"},"cell_type":"code","source":"pipe_mnb = Pipeline([('vectorizer', cvec), ('mnb', mnb)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f4b46262850a473b6f0fa4939090f395ed9dc547"},"cell_type":"code","source":"pipe_mnb.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8c090ebb8082d9dd8a75ff41286dc6b781c2bb8b"},"cell_type":"code","source":"y_pred_mnb = pipe_mnb.predict(X_test)\ny_pred_mnb","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"cc998ac7399ce79e661d6b1d93a244aa3ce0263a"},"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred_mnb)\ncm","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"ae6e0decb7db407edf6bde3dee28b5ab256dd7e3"},"cell_type":"code","source":"cr = classification_report(y_test, y_pred_mnb)\nprint(cr)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"759403c02de1ad2318b49901918aedd46462c076"},"cell_type":"markdown","source":"### Random Forest"},{"metadata":{"trusted":false,"_uuid":"8e5882254bcda8f010a214bb07f3dfdf8e410a94"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"79a25c266593403368ea846dddeb153150bbbeff"},"cell_type":"code","source":"rf = RandomForestClassifier()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"78072ce614ac6dbce8e7ee46e8be3ab3187d3fd8"},"cell_type":"code","source":"#pipe_rf = Pipeline([('vectorizer', tvec), ('rf', rf)])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"6fc404fbba8cf59036769eb0970aa28670f70592"},"cell_type":"code","source":"#pipe_rf.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"dd01a5ac3e3bf0fb1b0b93df5d02e8afa8ce8a8f"},"cell_type":"code","source":"#y_pred = pipe_rf.predict(X_test)\n#y_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"150741a0b4d25868bfb531bdd4abc5fc1e90c5d5"},"cell_type":"code","source":"#cm = confusion_matrix(y_test, y_pred)\n#cm","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f94196de3a8ce8b03df03ca59f61ec370408ccc5"},"cell_type":"code","source":"#cr = classification_report(y_test, y_pred)\n#print(cr)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"489ba1f37e1da5e2b0e8e7bb050b590cc189edea"},"cell_type":"markdown","source":"### Logistic Regression"},{"metadata":{"trusted":false,"_uuid":"115f7f3859f136fb7769e6457ce158107f1e2623"},"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"2b7a1e1111b424529184a15c09aed3b23fb18da0"},"cell_type":"code","source":"lr = LogisticRegression()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"662197f08f6aae2ce182dbcf061c590985012037"},"cell_type":"code","source":"pipe_lr = Pipeline([('vectorizer', cvec), ('lr', lr)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"dd89a0b6b21f16158d2a996081bcdb322764a605"},"cell_type":"code","source":"pipe_lr.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a6292076f86942690f2d38f2e7905ce2cc5e45c5"},"cell_type":"code","source":"y_pred_lr = pipe_lr.predict(X_test)\ny_pred_lr","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"a7f8283912a2714a0e7b897161b83af4da8d0aa7"},"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred_lr)\ncm","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"42359a3f7c8e7c1e3272cee4c618326f4ce3d61b"},"cell_type":"code","source":"cr = classification_report(y_test, y_pred_lr)\nprint(cr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5851ff04b2594c0bb7e60a63b23bbef43d45764f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}