{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\npd.set_option('precision', 5)\npd.set_option('display.float_format', lambda x: '%.5f' % x)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_uuid":"08e0db4a8b3fabefad38bafc0abf84940865ebe6","_cell_guid":"89a65825-2a6c-4bd3-815f-7a910dbbf134","trusted":true,"collapsed":true},"cell_type":"code","source":"tr = pd.read_csv('../input/train.csv')\nte = pd.read_csv('../input/test.csv')\nprint('train data shape is :', tr.shape)\nprint('test data shape is :', te.shape)","execution_count":2,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"094959c486491e29f1761316225c2eed0eac16e2","_cell_guid":"94b3355a-de55-474c-b764-5743883fb7e4","trusted":true},"cell_type":"code","source":"data = pd.concat([tr, te], axis=0)","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"e8ece2933c47e2c44fba25ef8acd13ea83e250e5","_cell_guid":"76d993e1-6f15-4a27-bd7b-3b2c634cc45e","trusted":true,"collapsed":true},"cell_type":"code","source":"data.shape","execution_count":4,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"39f0896c7c9736e1d02b4d2397feb756ee2ffe0f","_cell_guid":"4671806d-ba50-428b-8d3b-57f1d9203662","trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom tqdm import tqdm","execution_count":5,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"53cdd14f4c307a62afc5897a446e6fe7fe3c0857","_cell_guid":"ab2325b1-74c6-411f-8260-5440ada76695","trusted":true},"cell_type":"code","source":"data.activation_date = pd.to_datetime(data.activation_date)\ntr.activation_date = pd.to_datetime(tr.activation_date)\n\ndata['day_of_month'] = data.activation_date.apply(lambda x: x.day)\ndata['day_of_week'] = data.activation_date.apply(lambda x: x.weekday())\n\ntr['day_of_month'] = tr.activation_date.apply(lambda x: x.day)\ntr['day_of_week'] = tr.activation_date.apply(lambda x: x.weekday())","execution_count":6,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"db384cd32f47b9cd9f66695caa98e1542c3d2048"},"cell_type":"code","source":"data['char_len_title'] = data.title.apply(lambda x: len(str(x)))\ndata['char_len_desc'] = data.description.apply(lambda x: len(str(x)))","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"46569af93baa463cf39de9328cb7df8554d7cc79","_cell_guid":"966bdc71-c635-4b68-a48a-6137a7b4e2a2","trusted":true,"collapsed":true},"cell_type":"code","source":"agg_cols = ['region', 'city', 'parent_category_name', 'category_name',\n            'image_top_1', 'user_type','item_seq_number','day_of_month','day_of_week'];\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['deal_probability']\n    mean = gp.mean()\n    std  = gp.std()\n    data[c + '_deal_probability_avg'] = data[c].map(mean)\n    data[c + '_deal_probability_std'] = data[c].map(std)\n\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['price']\n    mean = gp.mean()\n    data[c + '_price_avg'] = data[c].map(mean)","execution_count":8,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"486a3a58e1a73a8f9854400f045aa8ddbaa66130","_cell_guid":"49990709-fcc5-40f7-aa95-6ac0245d7e26","trusted":true},"cell_type":"code","source":"cate_cols = ['city',  'category_name', 'user_type',]","execution_count":9,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"baedd144853f7bf9ca5a1111a02c23d38761d06a","_cell_guid":"9dd73adc-b798-4864-8e10-d7ddfca00ff4","trusted":true},"cell_type":"code","source":"for c in cate_cols:\n    data[c] = LabelEncoder().fit_transform(data[c].values)","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"afb496cc51c4af71b864d97cc1156c37febb82f3","_cell_guid":"fdc73ac7-eb7a-40d1-bbe7-9e5f77b66c2c","trusted":true,"collapsed":true},"cell_type":"code","source":"new_data = data.drop(['user_id','description','image','parent_category_name','region',\n                      'item_id','param_1','param_2','param_3','title'], axis=1)","execution_count":11,"outputs":[]},{"metadata":{"_uuid":"f23e3585a0011cae3b74e54359dd94ad4b82e27c","_cell_guid":"fd965e4d-61aa-4cbf-ad8a-15d5255f3a81","trusted":true,"collapsed":true},"cell_type":"code","source":"import gc\ndel data\ndel tr\ndel te\ngc.collect()","execution_count":12,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"3a49c4023e047db65ff1a28882a37026b90fd956","_cell_guid":"08d59f78-7ba8-4df0-8b95-a577258b2071","trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":13,"outputs":[]},{"metadata":{"_uuid":"45613787120d257ebe5b0c3b5624403c5b137526","_cell_guid":"c379cc48-0039-47a6-88bd-408ad926152d","trusted":true,"collapsed":true},"cell_type":"code","source":"X = new_data.loc[new_data.activation_date<=pd.to_datetime('2017-04-07')]\nX_te = new_data.loc[new_data.activation_date>=pd.to_datetime('2017-04-08')]\n\ny = X['deal_probability']\nX = X.drop(['deal_probability','activation_date'],axis=1)\nX_tr, X_va, y_tr, y_va = train_test_split(X, y, test_size=0.2, random_state=2018)\nX_te = X_te.drop(['deal_probability','activation_date'],axis=1)\n\nprint(X_tr.shape, X_va.shape, X_te.shape)\n\n\ndel X\ndel y\ngc.collect()","execution_count":14,"outputs":[]},{"metadata":{"_uuid":"f65805c5975359243199e6457ed2da210720e7de","_cell_guid":"cc10a918-1f36-46a2-a306-37ce5c01d33c","trusted":true,"collapsed":true},"cell_type":"code","source":"# Create the LightGBM data containers\ntr_data = lgb.Dataset(X_tr, label=y_tr, categorical_feature=cate_cols)\nva_data = lgb.Dataset(X_va, label=y_va, categorical_feature=cate_cols, reference=tr_data)\ndel X_tr\ndel X_va\ndel y_tr\ndel y_va\ngc.collect()\n\n# Train the model\nparameters = {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5,\n    'verbose': 50\n}\n\n\nmodel = lgb.train(parameters,\n                  tr_data,\n                  valid_sets=va_data,\n                  num_boost_round=300,\n                  early_stopping_rounds=120,\n                  verbose_eval=50)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45958a2257a4afbaac6c1bbdf7c024904437813f","_cell_guid":"8fc101d8-a0ab-4840-8f5e-9298afbdc9e5","trusted":true,"collapsed":true},"cell_type":"code","source":"y_pred = model.predict(X_te)\nsub = pd.read_csv('../input/sample_submission.csv')\nsub['deal_probability'] = y_pred\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)\nsub.to_csv('lgb_with_mean_encode.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d74a761634e6172f09c6933ef87570b68f61f0ee","_cell_guid":"c68a20dc-70f6-414a-bd45-6930f9ce2174","trusted":true,"collapsed":true},"cell_type":"code","source":"lgb.plot_importance(model, importance_type='gain', figsize=(10,20))","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}