{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn import naive_bayes, metrics, preprocessing, model_selection\nfrom sklearn import feature_extraction, feature_selection\nfrom sklearn import linear_model\nimport matplotlib.pyplot as plt\nimport itertools\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_data = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7792a8a43007ef904fa5f382e4bc3b62e2455e3"},"cell_type":"code","source":"train_data.sample(30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42d980aa634d998d84d0646f4193ecce76ee8b24"},"cell_type":"code","source":"train_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0e8e82abe0cde297b6b45c348d86a65f9ec53f2"},"cell_type":"code","source":"train_df, test_df = model_selection.train_test_split(train_data, test_size=0.1, stratify=train_data['target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eadb2682f32b679e869d5166d4e7f6dd1b201361"},"cell_type":"code","source":"train_df.sample(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f542746a51285c65be73b4507990d1777a694c7f"},"cell_type":"code","source":"test_df.sample(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76caee78834963cdbec113cf58b26ff803e2d109"},"cell_type":"code","source":"count_vectorizer = feature_extraction.text.TfidfVectorizer(max_features=50000, stop_words='english')\ntrain_vectors = count_vectorizer.fit_transform(train_df.question_text)\ntest_vectors = count_vectorizer.transform(test_df.question_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94b2c96c05ead3cc66a90c003e415fd79157ad73"},"cell_type":"code","source":"feature_names = count_vectorizer.get_feature_names()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a63a6559a8b3f5d8dcb175ea70b08c83722ff8e"},"cell_type":"code","source":"y_train, y_test = train_df['target'], test_df['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9db4cc8bc0e14a388a0b6db6f395d3f93783bbbf"},"cell_type":"code","source":"lreg = linear_model.LogisticRegression(penalty='l2',solver='lbfgs', verbose=1, n_jobs=3, class_weight='balanced')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52ad20cf795cb895807e74e8e0c780c9af84a065"},"cell_type":"code","source":"lreg.fit(train_vectors, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70fcd175020abd7497c017a06bdd11b839952f8a"},"cell_type":"code","source":"lr_pred = lreg.predict(test_vectors)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b83b9739971793a4115cb23c30b2126b6e863483"},"cell_type":"code","source":"print(metrics.classification_report(y_test, lr_pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"804364a00121cd19f96ef9bb5f3262d4bb0bddf2"},"cell_type":"code","source":"print(lreg.coef_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d518fa19f6fcfedeb39e2e662b43643b26f1527"},"cell_type":"code","source":"weights = lreg.coef_[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce2a5485529a57428e473115d4ed088e49cfd02f"},"cell_type":"code","source":"weights_ordering = weights.argsort()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"674638e4c3b98d048b9d678f5ee30176cccbdfa1"},"cell_type":"code","source":"print('-')\nprint(*[feature_names[i] for i in weights_ordering[:100]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d925fa70f34332904925fa212825e9f0dd8a1c7"},"cell_type":"code","source":"print('+')\nprint(*[feature_names[i] for i in weights_ordering[::-1][:100]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49d4d7248d412b74e7b4f7dfb8a5854532523a81"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}