{"cells":[{"metadata":{"_uuid":"fb0bfb35d12daf8787d4aeed3d645dc953cedca5"},"cell_type":"markdown","source":"**Load required libraries**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom wordcloud import WordCloud, STOPWORDS\nfrom nltk.corpus import stopwords\nimport string\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn import linear_model\nimport eli5\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9cb73db50001e300de7967746d517475e6a59029"},"cell_type":"markdown","source":"**Read data**"},{"metadata":{"trusted":true,"_uuid":"c2c47c03f9e72e54b0928c9bfb3d7eedeedd6166"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\nsub = pd.read_csv('../input/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"63c555277adf79acd65e02bfb9c00e1edfaf8922"},"cell_type":"markdown","source":"**Generate Features**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"def generate_features(df):\n    df[\"word_count\"] = df[\"question_text\"].apply(lambda x: len(str(x).split()))\n    df[\"unique_word_count\"] = df[\"question_text\"].apply(lambda x: len(set(str(x).split())))\n    df[\"char_length\"] = df[\"question_text\"].apply(lambda x: len(str(x)))\n    df[\"stop_words_count\"] = df[\"question_text\"].apply(lambda x: len([w for w in str(x).lower().split() if w in STOPWORDS]))\n    df[\"punc_count\"] = df[\"question_text\"].apply(lambda x: len([c for c in str(x) if c in string.punctuation]))\n    df[\"upper_words\"] = df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.isupper()]))\n    df[\"title_words\"] = df[\"question_text\"].apply(lambda x: len([w for w in str(x).split() if w.istitle()]))\n    df[\"word_length\"] = df[\"question_text\"].apply(lambda x: np.mean([len(w) for w in str(x).split()]))\n    return df\n\ntrain = generate_features(train)\ntest = generate_features(test)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0d1002aed5c7605c95fc3c8bcad97fa4491e8540"},"cell_type":"markdown","source":"**Generate vectors for train and test data**"},{"metadata":{"trusted":true,"_uuid":"4463470099f1bb56055150e8cbdff626ac92e795"},"cell_type":"code","source":"# Get the tfidf vectors\ntfidf_vec = TfidfVectorizer(stop_words='english', ngram_range=(1,3))\ntfidf_vec.fit_transform(train['question_text'].values.tolist() + test['question_text'].values.tolist())\ntrain_tfidf = tfidf_vec.transform(train['question_text'].values.tolist())\ntest_tfidf = tfidf_vec.transform(test['question_text'].values.tolist())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb1cfbcf2f23e0e53a6c00f677966d61e63a0091"},"cell_type":"markdown","source":"**Logistic Regression model**"},{"metadata":{"trusted":true,"_uuid":"ff2b3fbbc80741093dc02a645062bd40080e8cdb"},"cell_type":"code","source":"y_train = train[\"target\"].values\n\nx_train = train_tfidf\nx_test = test_tfidf\n\nmodel = linear_model.LogisticRegression(C=5., solver='sag')\nmodel.fit(x_train, y_train)\ny_test = model.predict_proba(x_test)[:,1]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"86db8018f1fa27c272aab06ffba39aef9a3b7c02"},"cell_type":"markdown","source":"**Important words**"},{"metadata":{"trusted":true,"_uuid":"9c697155582ae2d1856746fe628f8375825e2e70"},"cell_type":"code","source":"eli5.show_weights(model, vec=tfidf_vec, top=100, feature_filter=lambda x: x != '<BIAS>')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"af335ce6993e92b57b6ab7eeea4c7a63cc8a25f3"},"cell_type":"markdown","source":"**Output data to file**"},{"metadata":{"trusted":true,"_uuid":"c109cd85761cf50527291ff32709c5d3a78e2aba"},"cell_type":"code","source":"sub['prediction'] = y_test\nsub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}