{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":false,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\npd.set_option('precision', 5)\npd.set_option('display.float_format', lambda x: '%.5f' % x)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"89a65825-2a6c-4bd3-815f-7a910dbbf134","_uuid":"08e0db4a8b3fabefad38bafc0abf84940865ebe6","trusted":false,"collapsed":true},"cell_type":"code","source":"tr = pd.read_csv('../input/train.csv')\nte = pd.read_csv('../input/test.csv')\nprint('train data shape is :', tr.shape)\nprint('test data shape is :', te.shape)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"94b3355a-de55-474c-b764-5743883fb7e4","collapsed":true,"_uuid":"094959c486491e29f1761316225c2eed0eac16e2","trusted":false},"cell_type":"code","source":"data = pd.concat([tr, te], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"0d2c59ec-6a22-458e-9bfb-5ab081adc809","_uuid":"eb3cc99363f3c8062279084bea3de77887d8c3e3","trusted":false,"collapsed":true},"cell_type":"code","source":"tr.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"76d993e1-6f15-4a27-bd7b-3b2c634cc45e","_uuid":"e8ece2933c47e2c44fba25ef8acd13ea83e250e5","trusted":false,"collapsed":true},"cell_type":"code","source":"data.shape","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4671806d-ba50-428b-8d3b-57f1d9203662","collapsed":true,"_uuid":"39f0896c7c9736e1d02b4d2397feb756ee2ffe0f","trusted":false},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ab2325b1-74c6-411f-8260-5440ada76695","collapsed":true,"_uuid":"53cdd14f4c307a62afc5897a446e6fe7fe3c0857","trusted":false},"cell_type":"code","source":"data.activation_date = pd.to_datetime(data.activation_date)\ntr.activation_date = pd.to_datetime(tr.activation_date)\n\ndata['day_of_month'] = data.activation_date.apply(lambda x: x.day)\ndata['day_of_week'] = data.activation_date.apply(lambda x: x.weekday())\n\ntr['day_of_month'] = tr.activation_date.apply(lambda x: x.day)\ntr['day_of_week'] = tr.activation_date.apply(lambda x: x.weekday())","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e172151f-747a-433c-bb88-6ce8b5f76715","collapsed":true,"_uuid":"db384cd32f47b9cd9f66695caa98e1542c3d2048","trusted":false},"cell_type":"code","source":"data['char_len_title'] = data.title.apply(lambda x: len(str(x)))\ndata['char_len_desc'] = data.description.apply(lambda x: len(str(x)))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"966bdc71-c635-4b68-a48a-6137a7b4e2a2","_uuid":"46569af93baa463cf39de9328cb7df8554d7cc79","trusted":false,"collapsed":true},"cell_type":"code","source":"agg_cols = ['region', 'city', 'parent_category_name', 'category_name',\n            'image_top_1', 'user_type','item_seq_number','day_of_month','day_of_week'];\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['deal_probability']\n    mean = gp.mean()\n    std  = gp.std()\n    data[c + '_deal_probability_avg'] = data[c].map(mean)\n    data[c + '_deal_probability_std'] = data[c].map(std)\n\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['price']\n    mean = gp.mean()\n    data[c + '_price_avg'] = data[c].map(mean)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3f195c23-c3e7-45b2-b52f-3ebd70b5797c","_uuid":"1d15decf6011d6614cbc6913ed9e64f3327b946e","trusted":false,"collapsed":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"49990709-fcc5-40f7-aa95-6ac0245d7e26","collapsed":true,"_uuid":"486a3a58e1a73a8f9854400f045aa8ddbaa66130","trusted":false},"cell_type":"code","source":"cate_cols = ['city',  'category_name', 'user_type',]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9dd73adc-b798-4864-8e10-d7ddfca00ff4","collapsed":true,"_uuid":"baedd144853f7bf9ca5a1111a02c23d38761d06a","trusted":false},"cell_type":"code","source":"for c in cate_cols:\n    data[c] = LabelEncoder().fit_transform(data[c].values)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6b794a84-b2e9-46fd-a988-c1a95a1821e7","collapsed":true,"_uuid":"2265bef44d0d86b7df191332389e93c3415da8a6","trusted":false},"cell_type":"code","source":"from nltk.corpus import stopwords\nstopWords = stopwords.words('russian')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"194ea227-8fa4-4891-bc14-a306f3718ee3","_uuid":"bd4c522653bfdab4860ab513a8a4525582e3cdd8"},"cell_type":"markdown","source":"Set different max_feature and experiment"},{"metadata":{"_cell_guid":"23191d4d-91f4-42df-9e92-5f82e061b23e","collapsed":true,"_uuid":"b26aba1303e2decd281c92d14c9bab7d7c1cd21d","trusted":false},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\ndata['description'] = data['description'].fillna(' ')\ntfidf = TfidfVectorizer(max_features=100, stop_words = stopWords)\ntfidf_train = np.array(tfidf.fit_transform(data['description']).todense(), dtype=np.float16)\nfor i in range(100):\n    data['tfidf_' + str(i)] = tfidf_train[:, i]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fdc73ac7-eb7a-40d1-bbe7-9e5f77b66c2c","collapsed":true,"_uuid":"afb496cc51c4af71b864d97cc1156c37febb82f3","trusted":false},"cell_type":"code","source":"new_data = data.drop(['user_id','description','image','parent_category_name','region',\n                      'item_id','param_1','param_2','param_3','title'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fd965e4d-61aa-4cbf-ad8a-15d5255f3a81","_uuid":"f23e3585a0011cae3b74e54359dd94ad4b82e27c","trusted":false,"collapsed":true},"cell_type":"code","source":"import gc\ndel data\ndel tr\ndel te\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"08d59f78-7ba8-4df0-8b95-a577258b2071","collapsed":true,"_uuid":"3a49c4023e047db65ff1a28882a37026b90fd956","trusted":false},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c379cc48-0039-47a6-88bd-408ad926152d","_uuid":"45613787120d257ebe5b0c3b5624403c5b137526","trusted":false,"collapsed":true},"cell_type":"code","source":"X = new_data.loc[new_data.activation_date<=pd.to_datetime('2017-04-07')]\nX_te = new_data.loc[new_data.activation_date>=pd.to_datetime('2017-04-08')]\n\ny = X['deal_probability']\nX = X.drop(['deal_probability','activation_date'],axis=1)\nX_tr, X_va, y_tr, y_va = train_test_split(X, y, test_size=0.2, random_state=2018)\nX_te = X_te.drop(['deal_probability','activation_date'],axis=1)\n\nprint(X_tr.shape, X_va.shape, X_te.shape)\n\n\n#del X\n#del y\n#gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f9df2d40-e16a-420a-a37d-f4009910d3c4","_uuid":"4f8ba6fc86c9525a659f23c72e8a036f858d7a19","trusted":false,"collapsed":true},"cell_type":"code","source":"'''\n# Classifier\nbayes_cv_tuner = BayesSearchCV(\n    estimator = xgb.XGBRegressor(\n        n_jobs = 1,\n        objective = 'regression',\n        eval_metric = 'rmse',\n        silent=1,\n        tree_method='approx'\n    ),\n    search_spaces = {\n        'learning_rate': (0.01, 1.0, 'log-uniform'),\n        'min_child_weight': (0, 10),\n        'max_depth': (0, 50),\n        'max_delta_step': (0, 20),\n        'subsample': (0.01, 1.0, 'uniform'),\n        'colsample_bytree': (0.01, 1.0, 'uniform'),\n        'colsample_bylevel': (0.01, 1.0, 'uniform'),\n        'reg_lambda': (1e-9, 1000, 'log-uniform'),\n        'reg_alpha': (1e-9, 1.0, 'log-uniform'),\n        'gamma': (1e-9, 0.5, 'log-uniform'),\n        'min_child_weight': (0, 5),\n        'n_estimators': (50, 100),\n        'scale_pos_weight': (1e-6, 500, 'log-uniform')\n    },    \n    scoring = 'roc_auc',\n    cv = StratifiedKFold(\n        n_splits=3,\n        shuffle=True,\n        random_state=42\n    ),\n    n_jobs = 3,\n    n_iter = ITERATIONS,   \n    verbose = 0,\n    refit = True,\n    random_state = 42\n)\n\ndef status_print(optim_result):\n    \"\"\"Status callback durring bayesian hyperparameter search\"\"\"\n    \n    # Get all the models tested so far in DataFrame format\n    all_models = pd.DataFrame(bayes_cv_tuner.cv_results_)    \n    \n    # Get current parameters and the best parameters    \n    best_params = pd.Series(bayes_cv_tuner.best_params_)\n    print('Model #{}\\nBest ROC-AUC: {}\\nBest params: {}\\n'.format(\n        len(all_models),\n        np.round(bayes_cv_tuner.best_score_, 4),\n        bayes_cv_tuner.best_params_\n    ))\n    \n    # Save all model results\n    clf_name = bayes_cv_tuner.estimator.__class__.__name__\n    all_models.to_csv(clf_name+\"_cv_results.csv\")\n    \n'''","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e5e8c02f-cac8-4caf-97d3-823a47a8c833","_uuid":"b8bd692eacd4840ffc2d04b08fe8060aa4b2df76","trusted":false,"collapsed":true},"cell_type":"code","source":"import xgboost as xgb\n\nparams = {'eta': 0.3,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'reg:logistic', \n          'eval_metric': 'rmse', \n          'random_state': 99, \n          'silent': True}\n\ntr_data = xgb.DMatrix(X_tr, y_tr)\nva_data = xgb.DMatrix(X_va, y_va)\ndel X_tr\ndel X_va\ndel y_tr\ndel y_va\ngc.collect()\n\nwatchlist = [(tr_data, 'train'), (va_data, 'valid')]\n\nmodel = xgb.train(params, tr_data, 1000, watchlist, maximize=False, early_stopping_rounds = 25, verbose_eval=5)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8fc101d8-a0ab-4840-8f5e-9298afbdc9e5","_uuid":"45958a2257a4afbaac6c1bbdf7c024904437813f","trusted":false,"collapsed":true},"cell_type":"code","source":"X_te = xgb.DMatrix(X_te)\ny_pred = model.predict(X_te, ntree_limit=model.best_ntree_limit)\nsub = pd.read_csv('../input/sample_submission.csv')\nsub['deal_probability'] = y_pred\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)\nsub.to_csv('xgb_with_mean_encode_and_nlp.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e27cea07-ee45-4a74-85f7-e56ebff5b5a2","_uuid":"fd28e7029c02cf2a21296c319dc751cdb95857fd","trusted":false,"collapsed":true},"cell_type":"code","source":"from xgboost import plot_importance\nimport matplotlib.pyplot as plt\nplot_importance(model)\nplt.gcf().savefig('feature_importance_xgb.png')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4ee683bc-5078-43df-b2d0-724ab7181e81","collapsed":true,"_uuid":"7fa8f21e37b9b420a6dd891d2d5b36a9e5fa6b0b","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8fb48890-9d39-4d24-a902-92595a03d85b","collapsed":true,"_uuid":"357c7293d2e9142597e808c11cf99767460d1f0f","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b84bcca3-b248-40ad-b532-108d299c3725","collapsed":true,"_uuid":"9cf6e108ce69f229a8aaf8db2517bf87cd6ca8cb","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"pygments_lexer":"ipython3","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python","nbconvert_exporter":"python","file_extension":".py","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}