{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport lightgbm as lgb\n\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfTransformer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_val_score\n\nfrom tqdm import tqdm\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"raw_train = pd.read_csv('../input/train.csv')\nraw_test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef1ebc80adeff0f22be5519e7951c2cf625a1c9c"},"cell_type":"code","source":"# print(raw_train.info())\n# print(raw_test.info())\n\npNum = (raw_train.target==1).sum()\nnNum = (raw_train.target==0).sum()\nprint('pNum:\\t', pNum)\nprint('nNum:\\t', nNum)\nprint('ratio:\\t', pNum/(pNum+nNum))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40f7d087ef80cac936aafab446738d105b383ceb"},"cell_type":"code","source":"train_text = raw_train['question_text']\ntest_text = raw_test['question_text']\nall_text = pd.concat([train_text, test_text])\n\ncount_vectorizer = CountVectorizer(\n#     sublinear_tf=True,\n    strip_accents='unicode',\n#     analyzer='word',\n#     token_pattern=r'\\w{1,}',\n#     stop_words='english',\n#     ngram_range=(1, 1),\n    max_features=20000)\n# )\ntfidf_transformer = TfidfTransformer(\n#     sublinear_tf=True,\n)\n\ncount_vectorizer.fit(all_text)\nprint('CountVectorizer has been fitted.')\n\ntfidf_transformer.fit(count_vectorizer.transform(all_text))\nprint('TfidfTransformer has been fitted.')\n\ntrain_text_counts = count_vectorizer.transform(train_text)\ntest_text_counts = count_vectorizer.transform(test_text)\n\n# train_text_tfidf = tfidf_transformer.transform(train_text_counts)\n# test_text_tfidf = tfidf_transformer.transform(test_text_counts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60fe44379141dee18698aa21b928b5038d439566"},"cell_type":"code","source":"countsP = train_text_counts[raw_train.loc[raw_train.target==1].index.tolist()].sum(axis=0)\ncountsN = train_text_counts[raw_train.loc[raw_train.target==0].index.tolist()].sum(axis=0)\nimportances = (countsP/(countsN+1)).tolist()[0]\nimporttancesDf = pd.DataFrame({'fn':count_vectorizer.get_feature_names(), 'importances':importances})\n\nselectedFeaturesNames = importtancesDf.sort_values(by='importances', ascending=False).head(20000).fn\nselectedFeaturesIndex = selectedFeaturesNames.index.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"876a1e4e03c384bd79ef119a3f5b52dedb438608"},"cell_type":"code","source":"train_text_tfidf = tfidf_transformer.transform(train_text_counts)\ntest_text_tfidf = tfidf_transformer.transform(test_text_counts)\n\ntrain_text_tfidf = train_text_tfidf[:, selectedFeaturesIndex]\ntest_text_tfidf = test_text_tfidf[:, selectedFeaturesIndex]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e44460b894e4bedcb45b7bf2200a76d668611f2d","_kg_hide-output":true,"_kg_hide-input":false},"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=2018)\noutTest = np.zeros((test_text_tfidf.shape[0]))\nvalidPre = np.zeros((train_text_tfidf.shape[0]))\nfor train_index, valid_index in kfold.split(train_text_tfidf, raw_train.target):\n    X_train, X_valid = train_text_tfidf[train_index], train_text_tfidf[valid_index]\n    y_train, y_valid = raw_train.target[train_index], raw_train.target[valid_index]\n    \n#     model = LogisticRegression(solver='liblinear', C=1)\n    model = lgb.LGBMClassifier(random_state=2018,\n                               max_depth=-1,\n                               num_leaves=63,\n                               n_estimators=300,\n                               min_child_samples=30,\n                               learning_rate=0.2)\n    \n    model.fit(X_train, y_train,\n              eval_set=[(X_valid, y_valid)],\n              eval_metric='binary_error',\n              early_stopping_rounds=20)\n    \n    pre = model.predict_proba(X_valid)[:, 1]\n    validPre[valid_index] = pre\n    \n#     score = f1_score(y_valid, pre>0.5)\n#     print('score:\\t', score)\n    \n    outTest += model.predict_proba(test_text_tfidf)[:, 1]/5","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1214f162e17f567f79d5f39b15d0bb58d13a23f"},"cell_type":"code","source":"best_score = 0\nbest_threshold = 0\nfor i in tqdm(range(100)):\n    Threshold = float(i)/100\n    score = f1_score(raw_train.target, validPre>Threshold)\n    if score > best_score:\n        best_score = score\n        best_threshold = Threshold\n\nprint('threshold:\\t', best_threshold, ', score:\\t', best_score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"618549ebd1f4408baf5588ef302e2d4b9f34c0f1"},"cell_type":"code","source":"outDf = pd.DataFrame({'qid': raw_test.qid, 'prediction': outTest>best_threshold})\noutDf.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}