{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# import lightgbm as lgb\nfrom tqdm import tqdm\n\npd.set_option('precision', 5)\npd.set_option('display.float_format', lambda x: '%.5f' % x)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"scrolled":true,"_cell_guid":"89a65825-2a6c-4bd3-815f-7a910dbbf134","_uuid":"08e0db4a8b3fabefad38bafc0abf84940865ebe6","trusted":true},"cell_type":"code","source":"tr = pd.read_csv('../input/train.csv',parse_dates=[\"activation_date\"],infer_datetime_format=True)\nte = pd.read_csv('../input/test.csv',parse_dates=[\"activation_date\"],infer_datetime_format=True)\nprint('train data shape is :', tr.shape)\nprint('test data shape is :', te.shape)\ntr.head()","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"9352ac60-e164-4e21-91d3-0e47797cfe2d","_uuid":"a59d51a5cfdca330d4c955602732b658d318c609","trusted":true},"cell_type":"code","source":"tr.describe()","execution_count":3,"outputs":[]},{"metadata":{"scrolled":true,"_cell_guid":"5e7c8a4a-fd3d-43f9-8947-d5711565fb76","_uuid":"24e54cf6c15874638d7fe93bdbb168ec9eaa0e31","trusted":true},"cell_type":"code","source":"tr.describe(include=[\"O\"])","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"94b3355a-de55-474c-b764-5743883fb7e4","_uuid":"094959c486491e29f1761316225c2eed0eac16e2","trusted":true},"cell_type":"code","source":"data = pd.concat([tr, te], axis=0)\nprint(\"merged shape:\",data.shape)","execution_count":5,"outputs":[]},{"metadata":{"_cell_guid":"3c8cd84f-0776-4690-ab70-99293ae1ae46","_uuid":"f44b41aa8875842d1b94a2cc6d7ffe50fef4fbc3","trusted":true},"cell_type":"code","source":"print(\"unique item_seq_number:\", len(set(data.item_seq_number)))\nprint(\"unique image_top_1:\", len(set(data.image_top_1)))","execution_count":6,"outputs":[]},{"metadata":{"_cell_guid":"ab2325b1-74c6-411f-8260-5440ada76695","collapsed":true,"_uuid":"53cdd14f4c307a62afc5897a446e6fe7fe3c0857","trusted":true},"cell_type":"code","source":"# data['day_of_month'] = data.activation_date.apply(lambda x: x.day)\n# data['day_of_week'] = data.activation_date.apply(lambda x: x.weekday())\n\n# tr['day_of_month'] = tr.activation_date.apply(lambda x: x.day)\n# tr['day_of_week'] = tr.activation_date.apply(lambda x: x.weekday())","execution_count":7,"outputs":[]},{"metadata":{"_cell_guid":"60e31d9b-9b95-4310-ac1e-9a56790746ba","_uuid":"cdcd853f66080f61cc4fc38475557e713bc8c973"},"cell_type":"markdown","source":"### Get target encoding for some variables\n* This is currently leaky: should add prior odds +- smoothing and/or only get odds for occurences with freq> (some threshhold, e.g. 3)"},{"metadata":{"_cell_guid":"966bdc71-c635-4b68-a48a-6137a7b4e2a2","_uuid":"46569af93baa463cf39de9328cb7df8554d7cc79","trusted":true},"cell_type":"code","source":"agg_cols = ['region', 'city', 'parent_category_name', 'category_name',\n            \"param_1\",\"param_2\",'image_top_1', 'user_type','item_seq_number']\nmax_cols = [\"category_name\", 'region',\"city\", \"param_1\",\"param_2\"\n            ,'item_seq_number'\n#             'image_top_1'\n           ]\n\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['deal_probability']\n    mean = gp.mean()\n    std  = gp.std()\n    maximum = gp.max()\n    data[c + '_deal_proba_avg'] = data[c].map(mean)\n    data[c + '_deal_proba_std'] = data[c].map(std)\n    if c in max_cols:\n        data[c + '_deal_proba_max_one'] = data[c].map(maximum) # added , may easily overfit!\n        data[c + '_deal_proba_max_one'] = data[c + '_deal_proba_max_one']>0.6\n\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['price']\n    mean = gp.mean()\n    data[c + '_price_avg'] = data[c].map(mean)","execution_count":10,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afa439675c9d090ced15ceaffde5b1368b182b5f"},"cell_type":"code","source":"data.shape","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7989a05f837fac0ce9ed1c37419b2e1dc5312374"},"cell_type":"code","source":"print(\"tr (orig)\",tr.shape , \"\\n te (orig)\",te.shape)","execution_count":15,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ff1d7f128f49f7fc75c290d84486a217252c4b52"},"cell_type":"code","source":"tr = data.loc[~data.deal_probability.isnull()]\nte = data.loc[data.deal_probability.isnull()].drop(\"deal_probability\",axis=1)","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfd362eeeb17c0ed740564fd5548e74ac6482d39"},"cell_type":"code","source":"print(\"tr (new)\",tr.shape , \"\\n te (new)\",te.shape)","execution_count":17,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"632b0ae9c6034f932beba178ea96bd56036105dc"},"cell_type":"code","source":"tr.info()","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"99f908e0883fcd2c3610777e459b0127139348e8"},"cell_type":"code","source":"tr.to_csv(\"avitoAdDemand-train-augV1.csv.gz\",index=False,compression=\"gzip\")\nte.to_csv(\"avitoAdDemand-test-augV1.csv.gz\",index=False,compression=\"gzip\")","execution_count":19,"outputs":[]},{"metadata":{"_cell_guid":"fdc73ac7-eb7a-40d1-bbe7-9e5f77b66c2c","collapsed":true,"_uuid":"afb496cc51c4af71b864d97cc1156c37febb82f3","trusted":false},"cell_type":"code","source":"# new_data = data.drop(['user_id','description','image',\n#                       'item_id','param_1','param_2','param_3','title'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"fd965e4d-61aa-4cbf-ad8a-15d5255f3a81","collapsed":true,"_uuid":"f23e3585a0011cae3b74e54359dd94ad4b82e27c","trusted":false},"cell_type":"code","source":"# import gc\n# del data\n# del tr\n# del te\n# gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"08d59f78-7ba8-4df0-8b95-a577258b2071","collapsed":true,"_uuid":"3a49c4023e047db65ff1a28882a37026b90fd956","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c379cc48-0039-47a6-88bd-408ad926152d","collapsed":true,"_uuid":"45613787120d257ebe5b0c3b5624403c5b137526","trusted":false},"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# X = new_data.loc[new_data.activation_date<=pd.to_datetime('2017-04-07')]\n# X_te = new_data.loc[new_data.activation_date>=pd.to_datetime('2017-04-08')]\n\n# y = X['deal_probability']\n# X = X.drop(['deal_probability','activation_date'],axis=1)\n# X_tr, X_va, y_tr, y_va = train_test_split(X, y, test_size=0.2, random_state=2018)\n# X_te = X_te.drop(['deal_probability','activation_date'],axis=1)\n\n# print(X_tr.shape, X_va.shape, X_te.shape)\n\n\n# del X\n# del y\n# gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"cc10a918-1f36-46a2-a306-37ce5c01d33c","collapsed":true,"_uuid":"f65805c5975359243199e6457ed2da210720e7de","trusted":false},"cell_type":"code","source":"# # Create the LightGBM data containers\n# tr_data = lgb.Dataset(X_tr, label=y_tr, categorical_feature=cate_cols)\n# va_data = lgb.Dataset(X_va, label=y_va, categorical_feature=cate_cols, reference=tr_data)\n# del X_tr\n# del X_va\n# del y_tr\n# del y_va\n# gc.collect()\n\n# # Train the model\n# parameters = {\n#     'task': 'train',\n#     'boosting_type': 'gbdt',\n#     'objective': 'regression',\n#     'metric': 'rmse',\n#     'num_leaves': 31,\n#     'learning_rate': 0.05,\n#     'feature_fraction': 0.9,\n#     'bagging_fraction': 0.8,\n#     'bagging_freq': 5,\n#     'verbose': 50\n# }\n\n\n# model = lgb.train(parameters,\n#                   tr_data,\n#                   valid_sets=va_data,\n#                   num_boost_round=2000,\n#                   early_stopping_rounds=120,\n#                   verbose_eval=50)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8fc101d8-a0ab-4840-8f5e-9298afbdc9e5","collapsed":true,"_uuid":"45958a2257a4afbaac6c1bbdf7c024904437813f","trusted":false},"cell_type":"code","source":"# y_pred = model.predict(X_te)\n# sub = pd.read_csv('../input/sample_submission.csv')\n# sub['deal_probability'] = y_pred\n# sub['deal_probability'].clip(0.0, 1.0, inplace=True)\n# sub.to_csv('lgb_with_mean_encode.csv', index=False)\n# sub.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c68a20dc-70f6-414a-bd45-6930f9ce2174","collapsed":true,"_uuid":"d74a761634e6172f09c6933ef87570b68f61f0ee","trusted":false},"cell_type":"code","source":"# lgb.plot_importance(model, importance_type='gain', figsize=(10,20))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d15f0cce-203d-4367-9f2f-0ead56048d7b","collapsed":true,"_uuid":"ad0bee2a0e546bd3a86de686d2e70b986659962f","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}