{"cells":[{"metadata":{"_uuid":"d121597a55b4845a3806fb4f1ac133b3e9ba5cd7"},"cell_type":"markdown","source":"# Spooky Author Identification"},{"metadata":{"_uuid":"8d846409ee1cde2b725f0fffcd6e3da1e87a712e"},"cell_type":"markdown","source":"# Load data"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:31:53.714506Z","start_time":"2019-02-01T07:31:53.703505Z"},"trusted":true,"_uuid":"6a0c62800d1e6dd980a876ff59a941b2b871cd82"},"cell_type":"code","source":"# import libraries\nimport pandas as pd\nimport numpy as np\n\n# to make this notebook's output stable across runs\nnp.random.seed(42)\nimport seaborn as sns\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nimport re\n\n# Gensim\nimport gensim\nimport gensim.corpora as corpora\nfrom gensim.utils import simple_preprocess\nfrom gensim.models import CoherenceModel\n\n# Plotting tools\nimport pyLDAvis\nimport pyLDAvis.gensim  \n\n# set options for rendering plots\n%matplotlib inline\n\n# display multiple outputs within a cell\nfrom IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\";\n\nimport warnings\nwarnings.filterwarnings('ignore');","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:31:53.881523Z","start_time":"2019-02-01T07:31:53.718506Z"},"trusted":true,"_uuid":"307c1a6167f05d1e1980e07037ff528ccf326904"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"75f92e3f15b232eae84671c21fb03485d77b8aad"},"cell_type":"markdown","source":"# Data exploration"},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:54.873539Z","start_time":"2019-01-31T23:04:54.846539Z"},"trusted":true,"_uuid":"c3868a9540709ad0d354dc68ea8642658ce1aadc"},"cell_type":"code","source":"train.head()\ntrain.shape\ntrain[\"author\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-29T11:08:32.234000Z","start_time":"2019-01-29T11:08:32.219000Z"},"_uuid":"0d119e8e51f3504d2e6c56dc2009b42e78b0688a"},"cell_type":"markdown","source":"**Three Authors** <br>\n\nEdgar Allen Poe <br>\nMary Wollstonecraft Shelley <br>\nH.P Lovecraft"},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.017539Z","start_time":"2019-01-31T23:04:54.883539Z"},"trusted":true,"_uuid":"8d6ba117ae4a216a04814147b23e612aff7f4664"},"cell_type":"code","source":"sns.countplot('author', data = train, palette='dark');","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.032539Z","start_time":"2019-01-31T23:04:55.021539Z"},"trusted":true,"_uuid":"66966d9e62d141e5d19fff5a8761b20860ad33e7"},"cell_type":"code","source":"# look at some of the writing from edgar allen poe\ntrain[train[\"author\"] == \"EAP\"][\"text\"][0]","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.113539Z","start_time":"2019-01-31T23:04:55.036539Z"},"trusted":true,"_uuid":"4694f88fa1c5e68d0c4e97981b3897b1bce5f11e"},"cell_type":"code","source":"# add a rough count of words in the sentences as a feature\ntrain['word_length'] = train.text.str.count(' ')\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.132539Z","start_time":"2019-01-31T23:04:55.116539Z"},"trusted":true,"_uuid":"39608fc8b5ed86c4d84c5b40ab814713a547df4a"},"cell_type":"code","source":"# look at some of the writing from edgar allen poe\ntrain[train[\"author\"] == \"EAP\"][\"word_length\"].describe()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.332539Z","start_time":"2019-01-31T23:04:55.136539Z"},"trusted":true,"_uuid":"328f9928a507f36088c227c71d408c78f6904062"},"cell_type":"code","source":"plt.figure(figsize=(8,8))\nsns.boxplot(x=\"author\", y=\"word_length\", data=train, palette='bright');","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:55.587539Z","start_time":"2019-01-31T23:04:55.336539Z"},"trusted":true,"_uuid":"4e5dc48a1a8a1468435278ea71fadf7312b7bff0"},"cell_type":"code","source":"plt.figure(figsize=(10,5))\nsns.set_context('talk')\nsns.distplot(train['word_length'], color='black');","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:56.094539Z","start_time":"2019-01-31T23:04:55.590539Z"},"trusted":true,"_uuid":"aeb54aa3544f7d59cf282c25a2ab393de231a167"},"cell_type":"code","source":"f,ax=plt.subplots(1,3,figsize=(16,6));\nsns.distplot(train[train['author']=='EAP'].word_length,ax=ax[0]);\nax[0].set_title('Edgar Allen Poe');\nsns.distplot(train[train['author']=='HPL'].word_length,ax=ax[1], color='r')\nax[1].set_title('H.P. Lovecraft');\nsns.distplot(train[train['author']=='MWS'].word_length,ax=ax[2], color='g')\nax[2].set_title('Mary Shelley');\nplt.show();","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:56.115539Z","start_time":"2019-01-31T23:04:56.097539Z"},"trusted":true,"_uuid":"4c664c92b0f89414abcd3746ab3afc26d56123d1"},"cell_type":"code","source":"train[train[\"author\"] == \"HPL\"][\"word_length\"].describe()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-01-31T23:04:56.141539Z","start_time":"2019-01-31T23:04:56.118539Z"},"trusted":true,"_uuid":"4fcbf9abf10e905cc57971859baaa4ed65f9ef13"},"cell_type":"code","source":"train[train[\"author\"] == \"MWS\"][\"word_length\"].describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"933ed57f5f82b590e460b0aae37afd358699e1b6"},"cell_type":"markdown","source":"# Preprocessing and feature extraction"},{"metadata":{"_uuid":"75abb591f4a4fdafe21eac76153f87b129c94457"},"cell_type":"markdown","source":"## stop words and tokens"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:31:54.290564Z","start_time":"2019-02-01T07:31:53.884523Z"},"trusted":false,"_uuid":"45e23dc54c74bb8dd2a8f4ba70eb0d81f7798015"},"cell_type":"code","source":"from nltk.corpus import stopwords \nfrom nltk.tokenize import word_tokenize \n  \ndata = train[\"text\"]\n  \nstop_words = set(stopwords.words('english')) \n  \nword_tokens = word_tokenize(data[1]) \n  \nfiltered_sentence = [w for w in word_tokens if not w in stop_words] \n  \nprint(word_tokens) \nprint(filtered_sentence)","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:31:55.854720Z","start_time":"2019-02-01T07:31:54.293564Z"},"trusted":false,"_uuid":"6f3ca77329fedf0cd78c96aae2868312084b7c53"},"cell_type":"code","source":"def sent_to_words(sentences):\n    for sentence in sentences:\n        yield(gensim.utils.simple_preprocess(str(sentence), deacc=True))  # deacc=True removes punctuations\n\ndata_words = list(sent_to_words(data))\n\nprint(data_words[:1])","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:31:55.962731Z","start_time":"2019-02-01T07:31:55.857720Z"},"trusted":false,"_uuid":"608a1f158c279a4656e48a6018e946d014a796f3"},"cell_type":"code","source":"# flatten list and join together as a string\nflat_list = [item for sublist in data_words for item in sublist]\nstr1 = ' '.join(flat_list)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5c23257651cc2e46e67ad699bf62b7a7fd2d0359"},"cell_type":"markdown","source":"## bigram, trigram"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:06.641799Z","start_time":"2019-02-01T07:31:55.965731Z"},"trusted":false,"_uuid":"fa7024e64af850d1e348abb34007905dae68d60a"},"cell_type":"code","source":"# Build the bigram and trigram models\nbigram = gensim.models.Phrases(data_words, min_count=5, threshold=100) # higher threshold fewer phrases.\ntrigram = gensim.models.Phrases(bigram[data_words], threshold=100)  \n\n# Faster way to get a sentence clubbed as a trigram/bigram\nbigram_mod = gensim.models.phrases.Phraser(bigram)\ntrigram_mod = gensim.models.phrases.Phraser(trigram)\n\n# See trigram example\nprint(trigram_mod[bigram_mod[data_words[0]]])","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:06.665801Z","start_time":"2019-02-01T07:32:06.644799Z"},"trusted":false,"_uuid":"2f6e1615882c1a407d02f86e43d83b2fe323eea4"},"cell_type":"code","source":"bigrams = []\nfor phrase in bigram.export_phrases(data_words[:100]):\n    bigrams.append(phrase)\nbigrams[:10]","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:08.136948Z","start_time":"2019-02-01T07:32:06.672802Z"},"trusted":false,"_uuid":"4381ea4de2ae5573d719409366d821096d8a179f"},"cell_type":"code","source":"# Define functions for stopwords, bigrams, trigrams and lemmatization\ndef remove_stopwords(texts):\n    return [[word for word in simple_preprocess(str(doc)) if word not in stop_words] for doc in texts]\n\ntrain_text = remove_stopwords(train['text'])\ntest_text = remove_stopwords(test['text'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a756f94e9d115a837054e99dbf6efb37b17d1f85"},"cell_type":"markdown","source":"## bag of words"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:08.170952Z","start_time":"2019-02-01T07:32:08.138948Z"},"trusted":false,"_uuid":"4ad11b49385eb240fdf0c7a937d7857e81400dec"},"cell_type":"code","source":"train_text = [' '.join(sent) for sent in train_text]\ntest_text = [' '.join(sent) for sent in test_text]","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:08.600995Z","start_time":"2019-02-01T07:32:08.177952Z"},"trusted":false,"_uuid":"9bd9f66fafc4872154197f741c287cc62871f587"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\n# Initialize the \"CountVectorizer\" object, which is scikit-learn's\n# bag of words tool.  \nvectorizer = CountVectorizer(analyzer = \"word\",   \\\n                             tokenizer = None,    \\\n                             preprocessor = None, \\\n                             stop_words = None,   \\\n                             max_features = 6000)\n\nfeature_vec = vectorizer.fit_transform(train_text)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"63183cab639d1bf24c3ff2286368b873b916563c"},"cell_type":"markdown","source":"## tf-idf"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:08.625997Z","start_time":"2019-02-01T07:32:08.603995Z"},"trusted":false,"_uuid":"c24d8cdd242983b826db16af359c69de12859550"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfTransformer\ntf_transformer = TfidfTransformer(use_idf=True).fit(feature_vec)\nX_train_tf = tf_transformer.transform(feature_vec)\nX_train_tf.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dabd3b3cd90b0b7a873a968b8b94dd765b8778f3"},"cell_type":"markdown","source":"# Training Classifier Models"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:08.643999Z","start_time":"2019-02-01T07:32:08.628997Z"},"trusted":false,"_uuid":"7e9eadfe639954e78ae4909e29a31364eb973a3a"},"cell_type":"code","source":"# create train/test set\ntrain_data = train_text\ntrain_labels = train[\"author\"]\nfrom sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test=train_test_split(train_data,train_labels,test_size=0.20,random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:09.112046Z","start_time":"2019-02-01T07:32:08.646999Z"},"trusted":false,"_uuid":"0172f24db358a2fadbed06de9f6d0042e891eb31"},"cell_type":"code","source":"# feature processing pipeline\nfrom sklearn.pipeline import Pipeline\n\ntext_features = Pipeline([\n    ('vect', CountVectorizer()),\n    ('tfidf', TfidfTransformer()),\n])\n\ntext_features.fit_transform(X_train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f492cc9787309c01f245e79cd494b87bd84162c"},"cell_type":"markdown","source":"# Multinomial Naive Bayes"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:58:02.353000Z","start_time":"2019-02-01T07:58:01.823000Z"},"trusted":false,"_uuid":"8c9c8fbf1b38a6b374fd1a9418d073d4ba0bf4f1"},"cell_type":"code","source":"from sklearn.naive_bayes import MultinomialNB\nfrom sklearn.metrics import accuracy_score, log_loss\n\npipe = Pipeline([\n    ('features', text_features),\n    ('clf', MultinomialNB()),\n])\n\npipe.fit(X_train, y_train)\n\nnb_pred = pipe.predict(X_test)\nnb_probs = pipe.predict_proba(X_test)\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, nb_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, nb_probs)));","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:32:09.765111Z","start_time":"2019-02-01T07:32:09.758110Z"},"trusted":false,"_uuid":"8db17b575dd471aa7bde88332d0695499c477166"},"cell_type":"code","source":"pipe.get_params().keys()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:59:46.266000Z","start_time":"2019-02-01T07:58:05.136000Z"},"trusted":false,"_uuid":"72a1c583aa1b1f703ce3c3a0bc3d89dc3e2240d1"},"cell_type":"code","source":"# Grid Search\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nlog_loss_build = make_scorer(log_loss, greater_is_better=False, needs_proba=True)\n\nparameters = {'features__vect__max_features': [10000, 12000, 15000],\n              'features__vect__ngram_range': [(1,1), (1,2)],\n              'features__tfidf__use_idf': [True, False],\n              'clf__alpha': [0.01, 0.1, 1]\n             }\n\ngs = GridSearchCV(pipe, parameters, cv=5, n_jobs=-1, scoring=log_loss_build)\n \n# Fit and tune model\ngs.fit(X_train, y_train);","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:59:52.557000Z","start_time":"2019-02-01T07:59:52.001000Z"},"trusted":false,"_uuid":"4ad4bb0fe78e0e4665aacf5eeb6a706e4dc05bc7"},"cell_type":"code","source":"gs.best_params_\ngs.best_score_\nfinal_model = gs.best_estimator_\n\nfinal_model.fit(X_train, y_train)\nfinal_pred = final_model.predict(X_test)\nfinal_probs = final_model.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, final_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, final_probs)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f8f0a5afb20790a8f7dc1838c8d6d0f1da8d795"},"cell_type":"markdown","source":"# Complement NB"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:35:27.553000Z","start_time":"2019-02-01T07:35:26.215000Z"},"trusted":false,"_uuid":"c5185859ae095f7c8f38ed8e9d7eaf3401e4cf7a"},"cell_type":"code","source":"from sklearn.naive_bayes import ComplementNB\n\npipe = Pipeline([\n    ('vect', CountVectorizer(max_features=10000, ngram_range=(1,2))),\n    ('tfidf', TfidfTransformer(use_idf=False)),\n    ('clf', ComplementNB(alpha=0.01))\n])\n\npipe.fit(X_train, y_train)\ncnb_pred = pipe.predict(X_test)\ncnb_probs = pipe.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, cnb_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, cnb_probs)));","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cc1189e8ff2c9390e04b397e8b30f4dbbc17d0cd"},"cell_type":"markdown","source":"# Random Forest"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:35:48.845000Z","start_time":"2019-02-01T07:35:37.418000Z"},"trusted":false,"_uuid":"3177309680d6690797656e9700370b58a0394a19"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\npipe = Pipeline([\n    ('vect', CountVectorizer(max_features=10000)),\n    ('tfidf', TfidfTransformer(use_idf=False)),\n    ('clf', RandomForestClassifier(n_estimators = 50)) \n])\n\npipe.fit(X_train, y_train)\nrf_pred = pipe.predict(X_test)\nrf_probs = pipe.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, rf_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, rf_probs)));","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T03:31:11.859000Z","start_time":"2019-02-01T03:31:11.851000Z"},"trusted":false,"_uuid":"30ba538f0d86dd95752d055574599eb7b5ffed58"},"cell_type":"code","source":"pipe.get_params().keys()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:37:16.021000Z","start_time":"2019-02-01T07:36:12.754000Z"},"trusted":false,"_uuid":"543214bbb47e4ccf6868b31043218683dd48682a"},"cell_type":"code","source":"# Grid Search\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nlog_loss_build = make_scorer(log_loss, greater_is_better=False, needs_proba=True)\n\nparameters = {'clf__max_depth': [16, 32],\n              'clf__max_leaf_nodes': [24, 36],\n              'clf__n_estimators': [250, 500]\n             }\n\ngs = GridSearchCV(pipe, parameters, cv=5, n_jobs=-1, scoring=log_loss_build)\n \n# Fit and tune model\ngs.fit(X_train, y_train);","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:37:33.925000Z","start_time":"2019-02-01T07:37:29.122000Z"},"trusted":false,"_uuid":"8b660da31a644d327a9154f03455e2cf13c2e018"},"cell_type":"code","source":"gs.best_params_\ngs.best_score_\nfinal_rf = gs.best_estimator_\n\nfinal_rf.fit(X_train, y_train)\nrf_pred = final_rf.predict(X_test)\nrf_probs = final_rf.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, rf_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, rf_probs)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b0ec49d61c975dd69ffd1b1f69150f1ee74c1ec8"},"cell_type":"markdown","source":"# Logistic Regression with Stochastic Gradient Descent"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:37:51.698000Z","start_time":"2019-02-01T07:37:51.155000Z"},"trusted":false,"_uuid":"438407804eb6cf11e2c43aef127cfa82d219d513"},"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier\n\npipe = Pipeline([\n    ('vect', CountVectorizer(max_features=10000)),\n    ('tfidf', TfidfTransformer(use_idf=False)),\n    ('clf', SGDClassifier(loss='log', penalty='l2')) \n])\n\npipe.fit(X_train, y_train)\nlr_pred = pipe.predict(X_test)\nlr_probs = pipe.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, rf_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, rf_probs)));","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T03:48:40.522000Z","start_time":"2019-02-01T03:48:40.514000Z"},"trusted":false,"_uuid":"0965cf98b78dc7c2e82b4131839be25e62abcbd7"},"cell_type":"code","source":"pipe.get_params().keys()","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:46:19.366000Z","start_time":"2019-02-01T07:45:19.493000Z"},"trusted":false,"_uuid":"cf92709b7d7328e174b2347c1a2a9518df45db8b"},"cell_type":"code","source":"# Grid search\nalpha_range = 10.0**-np.arange(1,7)\n\nparameters = {'clf__alpha': alpha_range,\n              'clf__penalty': ['l1', 'l2'],\n              'clf__max_iter': [10, 50]\n             }\n\ngs = GridSearchCV(pipe, parameters, cv=5, n_jobs=-1, scoring=log_loss_build)\n \n# Fit and tune model\ngs.fit(X_train, y_train);\n\ngs.best_params_\nlr_model = gs.best_estimator_","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:46:20.456000Z","start_time":"2019-02-01T07:46:19.370000Z"},"trusted":false,"_uuid":"229331a5c3409d600890c8ff7b748103958fb5c8"},"cell_type":"code","source":"gs.best_params_\ngs.best_score_\nlr_model = gs.best_estimator_\n\nlr_model.fit(X_train, y_train)\nlr_pred = lr_model.predict(X_test)\nlr_probs = lr_model.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, lr_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, lr_probs)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c18a0c9da9ecfebf35b34d7ed91bfefce179b23f"},"cell_type":"markdown","source":"# Ensemble Classifiers"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T07:52:34.543000Z","start_time":"2019-02-01T07:52:03.369000Z"},"trusted":false,"_uuid":"eccc9f3a3ec770e5f114e003774a0305004b500d"},"cell_type":"code","source":"from sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nada = AdaBoostClassifier(DecisionTreeClassifier(max_depth=2),\n                         algorithm=\"SAMME\",\n                         n_estimators=600)\n\npipe = Pipeline([\n    ('vect', CountVectorizer(max_features=10000)),\n    ('tfidf', TfidfTransformer(use_idf=False)),\n    ('clf', ada) \n])\n\npipe.fit(X_train, y_train)\nada_pred = pipe.predict(X_test)\nada_probs = pipe.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, ada_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, ada_probs)));","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T08:00:44.412000Z","start_time":"2019-02-01T08:00:39.403000Z"},"trusted":false,"_uuid":"9d1dbb63111fc651ac2711b1de536d1b7f9e060c"},"cell_type":"code","source":"from sklearn.ensemble import VotingClassifier\neclf1 = VotingClassifier(estimators=[\n        ('rf', final_rf), ('lr', lr_model), ('nb', final_model)], voting='soft')\n\npipe = Pipeline([\n    ('vect', CountVectorizer(max_features=10000, ngram_range=(1,2))),\n    ('tfidf', TfidfTransformer(use_idf=False))\n])\n\npipe.fit_transform(X_train)\neclf1.fit(X_train, y_train)\nvote_pred = eclf1.predict(X_test)\nvote_probs = eclf1.predict_proba(X_test);\n\nprint(\"Accuracy score: \" + str(accuracy_score(y_test, vote_pred)))\nprint(\"Log loss: \" + str(log_loss(y_test, vote_probs)));","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c5e3149510733907c49e457378e768e7700b86a3"},"cell_type":"markdown","source":"# Final submission"},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T08:10:46.509000Z","start_time":"2019-02-01T08:10:46.342000Z"},"trusted":false,"_uuid":"f994e521390e0b2dc7e830bd1cce617949dfa450"},"cell_type":"code","source":"X_test = test_text\npredictions = final_model.predict_proba(X_test)","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T08:11:56.087000Z","start_time":"2019-02-01T08:11:56.080000Z"},"trusted":false,"_uuid":"3362db7f96f0f15db4b8690b6ca46d9119aad83e"},"cell_type":"code","source":"preds = pd.DataFrame(data=predictions, columns = final_model.named_steps['clf'].classes_)","execution_count":null,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2019-02-01T08:13:35.178000Z","start_time":"2019-02-01T08:13:35.161000Z"},"trusted":false,"_uuid":"3fbaa1980e18eaebf0ecac293c6ab9d99e52baf8"},"cell_type":"code","source":"# generating a submission file\nresult = pd.concat([test[['id']], preds], axis=1)\nresult.set_index('id', inplace = True)\nresult.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"toc":{"base_numbering":1,"nav_menu":{"height":"39px","width":"298px"},"number_sections":true,"sideBar":true,"skip_h1_title":false,"title_cell":"Table of Contents","title_sidebar":"Contents","toc_cell":false,"toc_position":{},"toc_section_display":true,"toc_window_display":false}},"nbformat":4,"nbformat_minor":1}