{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport time\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a5b3515aae7cc8c794777431cbd900ea6459ee2"},"cell_type":"code","source":"from skopt import gp_minimize, forest_minimize\nfrom skopt.space import Real, Categorical, Integer\nfrom skopt.plots import plot_convergence\nfrom skopt.plots import plot_objective, plot_evaluations\nfrom skopt.utils import use_named_args\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler\n\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_auc_score","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf3beae0afa49cdc4e0ddc4c996867e7f2d80521"},"cell_type":"code","source":"train.groupby('target').count()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"759ac2e405e7ae50a300cfbd71f9a00538640c73"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2445a7a9326516003dc54ee23f422a27c5480fe"},"cell_type":"code","source":"vec = TfidfVectorizer(min_df=3, stop_words='english', ngram_range=(1,2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9536f932da061a3a59665b27b285813d3b30d9b3"},"cell_type":"code","source":"X = vec.fit_transform(train['question_text'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a8366680af3fa0954629cef3f7711951b814a01"},"cell_type":"code","source":"Y = train['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed7f2f836260750fa8c5c4df57e8c84c961ddb27"},"cell_type":"code","source":"spliter = StratifiedKFold(n_splits=5,shuffle=True, random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bbd9d6972537e27fe713ce8c27b5a5bb5a3b0483"},"cell_type":"code","source":"FOLD_LIST = list(spliter.split(Y,Y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"991d323df051ee7b9e89002b92210889edd7ae6d"},"cell_type":"code","source":"lgbm_params =  {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'binary',\n    'metric': 'auc',\n    'max_depth': 15,\n    'num_leaves': 64,  # 63, 127, 255\n    'feature_fraction': 0.05, # 0.1, 0.01\n    'bagging_fraction': 0.8,\n    'verbose': 1\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b7a689be574ec3e147d79c00546fa80742bcf44"},"cell_type":"code","source":"# dim_learning_rate = Real(low=1e-3, high=0.99, prior='log-uniform',name='learning_rate')\n# dim_estimators = Integer(low=50, high=6000,name='n_estimators')\n# dim_max_depth = Integer(low=1, high=15,name='max_depth')\n\n# dimensions = [dim_learning_rate,\n#               dim_estimators,\n#               dim_max_depth]\n\n# default_parameters = [0.3,100,3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c1edd0173f7b8efcbea5f33b9dab0f82d215be7"},"cell_type":"code","source":"# def createModel(learning_rate,n_estimators,max_depth):       \n\n#     oof_preds = np.zeros(shape=(Y.shape[0], 2))\n#     for fold_, (trn_, val_) in enumerate(FOLD_LIST):\n#         trn_x, trn_y = X[trn_], Y[trn_]\n#         val_x, val_y = X[val_], Y[val_]\n\n#         clf = lgb.LGBMClassifier(**lgb_params,learning_rate=learning_rate,\n#                                 n_estimators=n_estimators,max_depth=max_depth)\n#         clf.fit(\n#             trn_x, trn_y,\n#             eval_set=[(trn_x, trn_y), (val_x, val_y)],\n#             verbose=False,\n#             early_stopping_rounds=50\n#         )\n#         oof_preds[val_, :] = clf.predict_proba(val_x, num_iteration=clf.best_iteration_)\n#         print('fold',fold_+1, \n#               roc_auc_score(\n#                   val_y,\n#                   np.where(clf.predict_proba(val_x, num_iteration=clf.best_iteration_)>0.5, 1,0)[:,0]\n#               )\n#         )\n# #         clfs.append(clf)\n#     loss = roc_auc_score(y_true=Y,  y_score=oof_preds[:,0])    \n#     return loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"411f6caf9c67e923bb760233b5cf4671c445e00f"},"cell_type":"code","source":"# iter_count = 0\n# @use_named_args(dimensions=dimensions)\n# def fitness(learning_rate,n_estimators,max_depth):\n#     \"\"\"\n#     Hyper-parameters:\n#     learning_rate:     Learning-rate for the optimizer.\n#     n_estimators:      Number of estimators.\n#     max_depth:         Maximum Depth of tree.\n#     \"\"\"\n#     global iter_count\n#     iter_count+=1\n#     # Print the hyper-parameters.\n#     print('iteration:', iter_count)\n#     print('learning rate: {0:.2e}'.format(learning_rate), 'estimators:', n_estimators, 'max depth:', max_depth)\n    \n#     lv= createModel(learning_rate=learning_rate,\n#                     n_estimators=n_estimators,\n#                     max_depth = max_depth)\n#     return lv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb12da6a3f25bb253f43fef072067de1c475ea7a"},"cell_type":"code","source":"# error = fitness(default_parameters)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb04b552e46a9caf34c7f5c9981c52e1583f97ce"},"cell_type":"code","source":"# # use only if you haven't found out the optimal parameters for xgb. else comment this block.\n# search_result = gp_minimize(func=fitness,\n#                             dimensions=dimensions,\n#                             acq_func='EI', # Expected Improvement.\n#                             n_calls=70,\n#                            x0=default_parameters)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e92c5fe7c3039c39d949fa4ffc3248e78914678"},"cell_type":"code","source":"# plot_convergence(search_result)\n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bab30857e6d9676dd67a5ce3255e5967a7bebcc6"},"cell_type":"code","source":"# print(search_result.x)\n# learning_rate = search_result.x[0]\n# n_estimators = search_result.x[1]\n# max_depth = search_result.x[2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d24af02b0d6ca091f727aa654e3e9b77ebd2806"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a5b388907faa7d264a511fc6feaff16244c83d5"},"cell_type":"code","source":"X_target = vec.transform(test.question_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71ddf7a7d10233b033a34b4d10c1409452523b8a"},"cell_type":"code","source":"Y_target = []\noff_predict = np.zeros(shape=Y.shape)\nfor fold_id,(train_idx, val_idx) in enumerate(FOLD_LIST):\n    print('FOLD:',fold_id)\n    X_train = X[train_idx]\n    y_train = Y[train_idx]\n    X_valid = X[val_idx]\n    y_valid = Y[val_idx]\n    \n    lgtrain = lgb.Dataset(X_train, y_train,\n                feature_name=vec.get_feature_names(),\n    #             categorical_feature = categorical\n                         )\n\n    lgvalid = lgb.Dataset(X_valid, y_valid,\n                feature_name=vec.get_feature_names(),\n    #             categorical_feature = categorical\n                         )\n\n    modelstart = time.time()\n    lgb_clf = lgb.train(\n        lgbm_params,\n        lgtrain,\n        num_boost_round=36000,\n        valid_sets=[lgtrain, lgvalid],\n        valid_names=['train','valid'],\n        early_stopping_rounds=200,\n        verbose_eval=1000\n    )\n    off_predict[val_idx] = lgb_clf.predict(X_valid)\n    test_pred = lgb_clf.predict(X_target)\n    Y_target.append(np.log1p(test_pred))\n    print('fold finish after', time.time()-modelstart)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45178333e1b2a6816a701e25d3c13a46ee3eb60d"},"cell_type":"code","source":"off_predict.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca33a3d74959167d0b9bb749c7fc433f8c957ea2"},"cell_type":"code","source":"Y_target = np.array(Y_target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26fc479adce61e43baeeafa8f1919b15ab858d05"},"cell_type":"code","source":"from sklearn import  metrics","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d42f35697396981ababd5702755c3356095ca9dd"},"cell_type":"code","source":"\n_thresh = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    _thresh.append([thresh, metrics.f1_score(Y, (off_predict>thresh).astype(int))])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(Y, (off_predict>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4dec9da0ff567994d277e718b19b0d4f05a25eea"},"cell_type":"code","source":"_thresh = np.array(_thresh)\nbest_id = _thresh[:,1].argmax()\nbest_thresh = _thresh[best_id][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ecc334b66fb89ea14edd8490ae88ebb0b93b4daa"},"cell_type":"code","source":"best_thresh","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4bcb3d825f2f285e799790ac06b6173f12f91407"},"cell_type":"code","source":"# 0.11, 0.13, 1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b113e3b7d81f138942b0e0b70b134a17af285de"},"cell_type":"code","source":"_thresh = []\nfor thresh in np.linspace(best_thresh-0.05, best_thresh+0.05, 100):\n    _thresh.append([thresh, metrics.f1_score(Y, (off_predict>thresh).astype(int))])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(Y, (off_predict>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35c751b4ce93775a2d430ac2dd64684ba571f612"},"cell_type":"code","source":"_thresh = np.array(_thresh)\nbest_id = _thresh[:,1].argmax()\nbest_thresh = _thresh[best_id][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb0d882f76a10cd493009428dd05a6371f2dd061"},"cell_type":"code","source":"!head ./../input/sample_submission.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"658a3a28afab2af9c022055921a1588b1a0bb268"},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub['prediction'] = np.where(np.expm1(Y_target.mean(axis=0))>best_thresh, 1,0,)\nsub.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc07fec2ab7653df993a361b63d720d24715855d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}