{"cells":[{"metadata":{"_uuid":"751a147979cceeb83185a0eff91befe38a2b825b"},"cell_type":"markdown","source":"This notebook is a reimplementation of this [kernel](!https://www.kaggle.com/classtag/lightgbm-with-mean-encode-feature-0-233), after fixing the data leakage.<br>\nIn brief, the original kernel computed statistics of the \"deal_probability\" and \"price\" for groups of features (e.g. \"category\"). The problem is that the statistics were computed on both the training and validation set. This is now ifxed in this kernel, which computes the mean and std on the training set only."},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:33:39.294372Z","start_time":"2018-04-30T14:33:38.909984Z"},"collapsed":true,"trusted":false,"_uuid":"dcbf4ed5b364a993c80a72522209026339b5bd75"},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\npd.set_option('precision', 5)\npd.set_option('display.float_format', lambda x: '%.5f' % x)\n\nimport os\n\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom tqdm import tqdm","execution_count":1,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:33:39.831257Z","start_time":"2018-04-30T14:33:39.828039Z"},"trusted":false,"collapsed":true,"_uuid":"7b9a3b994b4558e1313df960fde945ace6c136c7"},"cell_type":"code","source":"DATA_PATH = '/media/florian/8fd68a96-fc7e-47a1-a13e-4c4f910f6e51/ML_Data/avito/'\n\nprint(os.listdir(DATA_PATH))","execution_count":2,"outputs":[]},{"metadata":{"_uuid":"1331242d31bc9cb78e743e362005194e41a05640"},"cell_type":"markdown","source":"# Prepare the data"},{"metadata":{"_uuid":"d207bc8382adc9abc4dea73f72f2b0cc1fffbfbc"},"cell_type":"markdown","source":"## Load the data "},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:33:56.061055Z","start_time":"2018-04-30T14:33:40.916778Z"},"trusted":false,"collapsed":true,"_uuid":"663a61d49faafaf72734b8d624ff66f4eb6e4fed"},"cell_type":"code","source":"data_tr = pd.read_csv(DATA_PATH+'/train.csv')\ndata_te = pd.read_csv(DATA_PATH+'/test.csv')\nprint('train data shape is :', data_tr.shape)\nprint('test data shape is :', data_te.shape)","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"c2a27bcff684ac0ebd8256c791b85e4eaae8b415"},"cell_type":"markdown","source":"## Split training and validation sets "},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:11:10.542083Z","start_time":"2018-04-30T14:11:10.536898Z"},"collapsed":true,"_uuid":"686e9b217af84ccec89ceb5d1e7e24849145946c"},"cell_type":"markdown","source":"Before any feature aggregation, we first split the training set between training and validation sets."},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:33:56.755111Z","start_time":"2018-04-30T14:33:56.062293Z"},"collapsed":true,"trusted":false,"_uuid":"7bd47e254bd9ae3fc8d5b2b8f1065ece9a443369"},"cell_type":"code","source":"data_tr, data_va = train_test_split(data_tr, shuffle=True, \n                              test_size=0.05, random_state=42)","execution_count":4,"outputs":[]},{"metadata":{"_uuid":"56e793f566fc01478c8fab61348894a3de5ebbae"},"cell_type":"markdown","source":"## Preprocess date and text features"},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:33:56.763056Z","start_time":"2018-04-30T14:33:56.756429Z"},"collapsed":true,"trusted":false,"_uuid":"3c9148771cce89925fde6adc3bfac7affd6ec7b7"},"cell_type":"code","source":"def preprocessData(data):\n\n    # Extract date info from the activate date\n    data.activation_date    = pd.to_datetime(data.activation_date)\n\n    data['day_of_month']    = data.activation_date.apply(lambda x: x.day)\n    data['day_of_week']     = data.activation_date.apply(lambda x: x.weekday())\n\n    # Extract info from the title\n    data['char_len_title']  = data.title.apply(lambda x: len(str(x)))\n    data['char_len_desc']   = data.description.apply(lambda x: len(str(x)))","execution_count":5,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:34:11.506054Z","start_time":"2018-04-30T14:33:56.76415Z"},"collapsed":true,"trusted":false,"_uuid":"aa6c892164882fc2d7565428c9d26d866a6cf922"},"cell_type":"code","source":"preprocessData(data_te)\npreprocessData(data_tr)\npreprocessData(data_va)\n","execution_count":6,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:34:25.622088Z","start_time":"2018-04-30T14:34:11.507338Z"},"collapsed":true,"trusted":false,"_uuid":"b3d370974f4e639a84c52a5d293672ae471d7408"},"cell_type":"code","source":"# Encore the city, category_name and user_type labels\ncate_cols = ['city',  'category_name', 'user_type']\n\nfor c in cate_cols:\n    le = LabelEncoder()\n    allvalues = np.unique(data_tr[c].values).tolist() \\\n                + np.unique(data_va[c].values).tolist() \\\n                + np.unique(data_te[c].values).tolist()\n    le.fit(allvalues)\n    \n    for d in [data_tr, data_va, data_te]:\n        d[c] = le.transform(d[c].values)\ndel d","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"62f7d02e881b61843b3be47246502257a12ffe0e"},"cell_type":"markdown","source":"## Extract aggregated features "},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:34:25.657091Z","start_time":"2018-04-30T14:34:25.623394Z"},"collapsed":true,"trusted":false,"_uuid":"192da2490018bc1fa129b61ce4ff290c83815141"},"cell_type":"code","source":"class FeaturesStatistics():\n    def __init__(self):\n        self._stats = None\n        self._agg_cols  = ['region', 'city', 'parent_category_name', 'category_name',\n            'image_top_1', 'user_type','item_seq_number','day_of_month','day_of_week']\n    \n    def fit(self,df):\n        '''\n        Compute the mean and std of some features from a given data frame\n        '''\n        self._stats             = {}\n        \n        # For each feature to be aggregated\n        for c in tqdm(self._agg_cols,total=len(self._agg_cols)):\n            # Compute the mean and std of the deal prob and the price.\n            gp              = df.groupby(c)[['deal_probability','price']]\n            desc            = gp.describe()\n            self._stats[c]  = desc[ [('deal_probability','mean'),('deal_probability','std'),\n                                     ('price','mean')] ]\n\n    def transform(self,df):\n        '''\n        Add the mean features statistics computed from another dataset.\n        '''\n        # For each feature to be aggregated\n        for c in tqdm(self._agg_cols,total=len(self._agg_cols)):\n            # Add the deal proba and price statistics corrresponding to the feature\n            df[c+'_deal_probability_mean']  = df[c].map(self._stats[c][('deal_probability','mean')])\n            df[c+'_deal_probability_std']   = df[c].map(self._stats[c][('deal_probability','std')])\n            df[c+'_price_mean']             = df[c].map(self._stats[c][('price','mean')])\n            \n        \n    def fit_transform(self,df):\n        '''\n        First learn the feature statistics, then add them to the dataframe.\n        '''\n        self.fit(df)\n        self.transform(df)","execution_count":8,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:34:25.779878Z","start_time":"2018-04-30T14:34:25.658376Z"},"collapsed":true,"trusted":false,"_uuid":"1f5257a947d83938185d76edf84e686d259e6e09"},"cell_type":"code","source":"fStats = FeaturesStatistics()","execution_count":9,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:31.479562Z","start_time":"2018-04-30T14:34:25.781134Z"},"trusted":false,"collapsed":true,"_uuid":"3824014d46f1077173a1b00dad7ec0474aedc1a0"},"cell_type":"code","source":"fStats.fit_transform(data_tr)","execution_count":10,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:31.626716Z","start_time":"2018-04-30T14:36:31.480719Z"},"trusted":false,"collapsed":true,"_uuid":"2894ead8b50b875083e210c14c340b1a6d1b570c"},"cell_type":"code","source":"fStats.transform(data_va)","execution_count":11,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:32.568954Z","start_time":"2018-04-30T14:36:31.627924Z"},"trusted":false,"collapsed":true,"_uuid":"23c6f8d9ef171d9a97bf0bcba0980892829eac61"},"cell_type":"code","source":"fStats.transform(data_te)","execution_count":12,"outputs":[]},{"metadata":{"_uuid":"8f15f32ae98920397a52e037c0f4b2080004e42a"},"cell_type":"markdown","source":"### Drop some features "},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:33.081114Z","start_time":"2018-04-30T14:36:32.570162Z"},"trusted":false,"collapsed":true,"_uuid":"0ba4eeede8a3a742a3a7670717c3cbe020b5affe"},"cell_type":"code","source":"col_to_drops = ['activation_date','user_id','description',\n                'image','parent_category_name','region',\n                'item_id','param_1','param_2','param_3','title']\n\ny_tr = data_tr['deal_probability']\nX_tr = data_tr.drop(col_to_drops+['deal_probability'],axis=1)\n\n#y_te = data_te['deal_probability']\nX_te = data_te.drop(col_to_drops,axis=1)\n\ny_va = data_va['deal_probability']\nX_va = data_va.drop(col_to_drops+['deal_probability'],axis=1)\n\nprint(X_tr.shape, X_va.shape, X_te.shape)","execution_count":13,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:33.803597Z","start_time":"2018-04-30T14:36:33.082898Z"},"_cell_guid":"c379cc48-0039-47a6-88bd-408ad926152d","_uuid":"45613787120d257ebe5b0c3b5624403c5b137526","trusted":false},"cell_type":"code","source":"import gc\ndel data_tr, data_te, data_va\ngc.collect()","execution_count":14,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T13:54:48.663707Z","start_time":"2018-04-30T13:54:48.489786Z"},"_cell_guid":"fdc73ac7-eb7a-40d1-bbe7-9e5f77b66c2c","_uuid":"afb496cc51c4af71b864d97cc1156c37febb82f3","collapsed":true},"cell_type":"markdown","source":"# Train LightGBM"},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:33.897488Z","start_time":"2018-04-30T14:36:33.804787Z"},"collapsed":true,"trusted":false,"_uuid":"04a3cb4e7f4ee369af21979b79efc448fc291133"},"cell_type":"code","source":"# Create the LightGBM data containers\ntr_data = lgb.Dataset(X_tr, label=y_tr, categorical_feature=cate_cols)\nva_data = lgb.Dataset(X_va, label=y_va, categorical_feature=cate_cols, reference=tr_data)","execution_count":15,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:36:34.047548Z","start_time":"2018-04-30T14:36:33.89863Z"},"trusted":false,"_uuid":"2fd6b24c0b3634912c7f107f9a8194fd0a3e9b2d"},"cell_type":"code","source":"del X_tr\ndel X_va\ndel y_tr\ndel y_va\ngc.collect()","execution_count":16,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:38:27.474313Z","start_time":"2018-04-30T14:36:34.04888Z"},"_cell_guid":"cc10a918-1f36-46a2-a306-37ce5c01d33c","_uuid":"f65805c5975359243199e6457ed2da210720e7de","scrolled":true,"trusted":false,"collapsed":true},"cell_type":"code","source":"# Train the model\nparameters = {\n    'task':             'train',\n    'boosting_type':    'gbdt',\n    'objective':        'regression',\n    'metric':           'rmse',\n    'num_leaves':       31,\n    'learning_rate':    0.05,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq':     5,\n    'verbose':          50\n}\n\n\nmodel = lgb.train(parameters,\n                  tr_data,\n                  valid_sets=va_data,\n                  num_boost_round=2000,\n                  early_stopping_rounds=120,\n                  verbose_eval=50)","execution_count":17,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:38:41.633494Z","start_time":"2018-04-30T14:38:27.479545Z"},"_cell_guid":"8fc101d8-a0ab-4840-8f5e-9298afbdc9e5","_uuid":"45958a2257a4afbaac6c1bbdf7c024904437813f","trusted":false},"cell_type":"code","source":"y_pred = model.predict(X_te)\nsub = pd.read_csv(DATA_PATH+'/sample_submission.csv')\nsub['deal_probability'] = y_pred\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)\nsub.to_csv('lgb_with_mean_encode.csv', index=False)\nsub.head()","execution_count":18,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:38:41.712412Z","start_time":"2018-04-30T14:38:41.634708Z"},"collapsed":true,"trusted":false,"_uuid":"b67f0deed7e9653682ccb0ba9a852facb36f18d0"},"cell_type":"code","source":"%matplotlib inline","execution_count":19,"outputs":[]},{"metadata":{"ExecuteTime":{"end_time":"2018-04-30T14:38:42.471312Z","start_time":"2018-04-30T14:38:41.713604Z"},"_cell_guid":"c68a20dc-70f6-414a-bd45-6930f9ce2174","_uuid":"d74a761634e6172f09c6933ef87570b68f61f0ee","trusted":false},"cell_type":"code","source":"lgb.plot_importance(model, importance_type='gain', figsize=(10,20))","execution_count":20,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"54ab6b9db6a74841349cb0388f51e62139dea430"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"toc":{"toc_cell":false,"toc_number_sections":true,"toc_threshold":4,"toc_window_display":false}},"nbformat":4,"nbformat_minor":1}