{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":false,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\npd.set_option('precision', 5)\npd.set_option('display.float_format', lambda x: '%.5f' % x)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"08e0db4a8b3fabefad38bafc0abf84940865ebe6","_cell_guid":"89a65825-2a6c-4bd3-815f-7a910dbbf134","trusted":false,"collapsed":true},"cell_type":"code","source":"tr = pd.read_csv('../input/train.csv')\nte = pd.read_csv('../input/test.csv')\nprint('train data shape is :', tr.shape)\nprint('test data shape is :', te.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"094959c486491e29f1761316225c2eed0eac16e2","_cell_guid":"94b3355a-de55-474c-b764-5743883fb7e4","collapsed":true,"trusted":false},"cell_type":"code","source":"data = pd.concat([tr, te], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"82403489-7429-403f-9354-1ab2642244d7","_uuid":"48a159aae7db065602a7315cb4c0ef5986fea951","trusted":false,"collapsed":true},"cell_type":"code","source":"tr.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e8ece2933c47e2c44fba25ef8acd13ea83e250e5","_cell_guid":"76d993e1-6f15-4a27-bd7b-3b2c634cc45e","trusted":false,"collapsed":true},"cell_type":"code","source":"data.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"39f0896c7c9736e1d02b4d2397feb756ee2ffe0f","_cell_guid":"4671806d-ba50-428b-8d3b-57f1d9203662","collapsed":true,"trusted":false},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53cdd14f4c307a62afc5897a446e6fe7fe3c0857","_cell_guid":"ab2325b1-74c6-411f-8260-5440ada76695","collapsed":true,"trusted":false},"cell_type":"code","source":"data.activation_date = pd.to_datetime(data.activation_date)\ntr.activation_date = pd.to_datetime(tr.activation_date)\n\ndata['day_of_month'] = data.activation_date.apply(lambda x: x.day)\ndata['day_of_week'] = data.activation_date.apply(lambda x: x.weekday())\n\ntr['day_of_month'] = tr.activation_date.apply(lambda x: x.day)\ntr['day_of_week'] = tr.activation_date.apply(lambda x: x.weekday())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db384cd32f47b9cd9f66695caa98e1542c3d2048","_cell_guid":"e172151f-747a-433c-bb88-6ce8b5f76715","collapsed":true,"trusted":false},"cell_type":"code","source":"data['char_len_title'] = data.title.apply(lambda x: len(str(x)))\ndata['char_len_desc'] = data.description.apply(lambda x: len(str(x)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"46569af93baa463cf39de9328cb7df8554d7cc79","_cell_guid":"966bdc71-c635-4b68-a48a-6137a7b4e2a2","trusted":false,"collapsed":true},"cell_type":"code","source":"agg_cols = ['region', 'city', 'parent_category_name', 'category_name',\n            'image_top_1', 'user_type','item_seq_number','day_of_month','day_of_week'];\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['deal_probability']\n    mean = gp.mean()\n    std  = gp.std()\n    data[c + '_deal_probability_avg'] = data[c].map(mean)\n    data[c + '_deal_probability_std'] = data[c].map(std)\n\nfor c in tqdm(agg_cols):\n    gp = tr.groupby(c)['price']\n    mean = gp.mean()\n    data[c + '_price_avg'] = data[c].map(mean)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b588050a-35bc-4a06-af99-530521b7909f","_uuid":"e4059216f58113624ab80ff3e64c9b935958a2d2","trusted":false,"collapsed":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"486a3a58e1a73a8f9854400f045aa8ddbaa66130","_cell_guid":"49990709-fcc5-40f7-aa95-6ac0245d7e26","collapsed":true,"trusted":false},"cell_type":"code","source":"cate_cols = ['city',  'category_name', 'user_type',]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"baedd144853f7bf9ca5a1111a02c23d38761d06a","_cell_guid":"9dd73adc-b798-4864-8e10-d7ddfca00ff4","collapsed":true,"trusted":false},"cell_type":"code","source":"for c in cate_cols:\n    data[c] = LabelEncoder().fit_transform(data[c].values)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"806f67e7-fd61-41d6-a5a7-1a6a2210c504","_uuid":"56543681d7361836f587d22fc5155271e3845049","collapsed":true,"trusted":false},"cell_type":"code","source":"from nltk.corpus import stopwords\nstopWords = stopwords.words('russian')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f2475b4d-456e-4ef1-9d99-96c805a7b7d7","_uuid":"4b4a4ecc5253206a3eecccd3c899008619cd67e3"},"cell_type":"markdown","source":"Set different max_feature and experiment"},{"metadata":{"_cell_guid":"4690e6ee-348d-481d-8172-e9aad0166775","_uuid":"25533352c446d7fac5ade044b1ad35fa59af302e","collapsed":true,"trusted":false},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\ndata['description'] = data['description'].fillna(' ')\ntfidf = TfidfVectorizer(max_features=100, stop_words = stopWords)\ntfidf_train = np.array(tfidf.fit_transform(data['description']).todense(), dtype=np.float16)\nfor i in range(100):\n    data['tfidf_' + str(i)] = tfidf_train[:, i]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afb496cc51c4af71b864d97cc1156c37febb82f3","_cell_guid":"fdc73ac7-eb7a-40d1-bbe7-9e5f77b66c2c","collapsed":true,"trusted":false},"cell_type":"code","source":"new_data = data.drop(['user_id','description','image','parent_category_name','region',\n                      'item_id','param_1','param_2','param_3','title'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f23e3585a0011cae3b74e54359dd94ad4b82e27c","_cell_guid":"fd965e4d-61aa-4cbf-ad8a-15d5255f3a81","trusted":false,"collapsed":true},"cell_type":"code","source":"import gc\ndel data\ndel tr\ndel te\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a49c4023e047db65ff1a28882a37026b90fd956","_cell_guid":"08d59f78-7ba8-4df0-8b95-a577258b2071","collapsed":true,"trusted":false},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45613787120d257ebe5b0c3b5624403c5b137526","_cell_guid":"c379cc48-0039-47a6-88bd-408ad926152d","trusted":false,"collapsed":true},"cell_type":"code","source":"X = new_data.loc[new_data.activation_date<=pd.to_datetime('2017-04-07')]\nX_te = new_data.loc[new_data.activation_date>=pd.to_datetime('2017-04-08')]\n\ny = X['deal_probability']\nX = X.drop(['deal_probability','activation_date'],axis=1)\nX_tr, X_va, y_tr, y_va = train_test_split(X, y, test_size=0.2, random_state=2018)\nX_te = X_te.drop(['deal_probability','activation_date'],axis=1)\n\nprint(X_tr.shape, X_va.shape, X_te.shape)\n\n\ndel X\ndel y\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f65805c5975359243199e6457ed2da210720e7de","_cell_guid":"cc10a918-1f36-46a2-a306-37ce5c01d33c","trusted":false,"collapsed":true},"cell_type":"code","source":"# Create the LightGBM data containers\ntr_data = lgb.Dataset(X_tr, label=y_tr, categorical_feature=cate_cols)\nva_data = lgb.Dataset(X_va, label=y_va, categorical_feature=cate_cols, reference=tr_data)\ndel X_tr\ndel X_va\ndel y_tr\ndel y_va\ngc.collect()\n\n# Train the model\nparameters = {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5,\n    'verbose': 50\n}\n\n\nmodel = lgb.train(parameters,\n                  tr_data,\n                  valid_sets=va_data,\n                  num_boost_round=2000,\n                  early_stopping_rounds=120,\n                  verbose_eval=50)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"45958a2257a4afbaac6c1bbdf7c024904437813f","_cell_guid":"8fc101d8-a0ab-4840-8f5e-9298afbdc9e5","trusted":false,"collapsed":true},"cell_type":"code","source":"y_pred = model.predict(X_te)\nsub = pd.read_csv('../input/sample_submission.csv')\nsub['deal_probability'] = y_pred\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)\nsub.to_csv('lgb_with_mean_encode_and_nlp.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d74a761634e6172f09c6933ef87570b68f61f0ee","_cell_guid":"c68a20dc-70f6-414a-bd45-6930f9ce2174","trusted":false,"collapsed":true},"cell_type":"code","source":"lgb.plot_importance(model, importance_type='gain', figsize=(10,20))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a5711421-81d8-46ac-8310-1e5075381dfe","_uuid":"011d96cae627f4bdb6b3a36dd5992263564c2f26","collapsed":true,"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","mimetype":"text/x-python","version":"3.6.5","pygments_lexer":"ipython3","file_extension":".py","nbconvert_exporter":"python","codemirror_mode":{"name":"ipython","version":3}}},"nbformat":4,"nbformat_minor":1}