{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from sklearn.cross_validation import train_test_split\nimport lightgbm as lgb\nimport time","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"3dcf1d67-72f7-403c-b789-46bb9f669511","collapsed":true,"_uuid":"2de05569080818f0130763114031b0df7c6ecd9c","trusted":true},"cell_type":"code","source":"def lgb_modelfit_nocv(params,dtrain,dvalid,predictors,target='target',objective='binary',metrics='auc',feval=None,early_stopping_rounds=20,\n                     num_boost_round=3000,verbose_eval=10,categorical_features=None):\n    lgb_params={\n        'boosting_type':'gbdt',\n        'objective':objective,\n        'metric':metrics,\n        'learning_rate':0.01,\n        #'is_unbalance':'true',这里数据不平衡\n        'num_leaves':31,#需要让他小于 2^(max_depth)\n        'max_depth':-1,\n        'min_child_samples':20,#在子集中所需最小数据量\n        'max_bin':255,#Number of bucketed bin for feature values\n        'subsample':0.6,\n        'subsample_freq':0,#子采样频率\n        'colsample_bytree':0.3,#列采样比率\n        'min_child_weight':5,#孩子需要的实例权重（hessian）的最小总和（叶子）\n        'subsample_for_bin':200000,#构建垃圾桶的样本数量\n        'min_split_gain':0,\n        'reg_alpha':0,#L1 regularization term on weights\n        'reg_lambda':0,#L2 regularization term on weights\n        'nthread':8,\n        'verbose':0\n    }\n    lgb_params.update(params)\n    print('preparing valildation datasets')\n    \n    xgtrain=lgb.Dataset(dtrain[predictors].values,label=dtrain[target].values,feature_name=predictors,\n                        categorical_feature=categorical_features)\n    xgvalid=lgb.Dataset(dvalid[predictors].values,label=dvalid[target].values,feature_name=predictors,\n                       categorical_feature=categorical_features)\n    \n    evals_results={}\n    \n    bst1=lgb.train(lgb_params,xgtrain,valid_sets=[xgtrain,xgvalid],valid_names=['trian','valid'],\n                  evals_result=evals_results,num_boost_round=num_boost_round,early_stopping_rounds=early_stopping_rounds,\n                  verbose_eval=10,feval=feval)\n    n_estimators=bst1.best_iteration\n    print('Model Report')\n    print('n_estimators:',n_estimators)\n    print(metrics+':',evals_results['valid'][metrics][n_estimators-1])\n    return bst1","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"bf3a1936-bf47-473b-8a6c-61ea3ad4f6f6","_uuid":"02ed3ad484391e602a1ccb3a6327c4b8d50c84e3","trusted":true,"collapsed":true},"cell_type":"code","source":"path='../input/'\ndtypes={\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint32'\n}\nprint('load train...')#40000000\ntrain_df = pd.read_csv(path+\"train.csv\",skiprows=range(1,149903891), nrows=40000000, dtype=dtypes, usecols=['ip','app','device','os', 'channel', 'click_time', 'is_attributed'])\nprint('load test...')\ntest_df = pd.read_csv(path+\"test.csv\", dtype=dtypes, usecols=['ip','app','device','os', 'channel', 'click_time', 'click_id'])","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9895c471-0c88-4555-b93f-cde41bdf4e56","collapsed":true,"_uuid":"e7c71bf6ef69a0a3e390d17238008f5a7948f775","trusted":false},"cell_type":"code","source":"import gc\nlen_train = len(train_df)\ntrain_df=train_df.append(test_df)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f11ea879-5f49-4720-85a2-51cb15bc0ad1","_uuid":"990c31873248ecf4e9edc6d91045566550eb96a5","trusted":false,"collapsed":true},"cell_type":"code","source":"del test_df\ngc.collect()\nprint('data prep...')\ntrain_df['hour'] = pd.to_datetime(train_df.click_time).dt.hour.astype('uint8')\ntrain_df['day'] = pd.to_datetime(train_df.click_time).dt.day.astype('uint8')\n\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4341cf72-1b99-42a6-8947-5e3ab5ceda72","_uuid":"699d149f5b28813255729c82f0eadeefa99424be","trusted":false,"collapsed":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6c265ff8-979e-44c7-af68-a91924bf2a82","_uuid":"0fd3342a322ec412c3dd414519677131959d444f","trusted":false,"collapsed":true},"cell_type":"code","source":"#组合特征 1\nprint('group by...')\ngp = train_df[['ip','day','hour','channel']].groupby(by=['ip','day','hour'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'qty'})","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"05e83c20-986b-4502-ae98-4776ec336205","_uuid":"0e9cee3bbac6724901e019e4bdeff4bdef3d0793","trusted":false,"collapsed":true},"cell_type":"code","source":"gp.tail()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"559f40a0-d6d2-49da-a024-e7ba92c5bc58","_uuid":"67c0c32e9154467b6837e232f98aa7888e1479b2","trusted":false,"collapsed":true},"cell_type":"code","source":"print('merge...')\ntrain_df = train_df.merge(gp, on=['ip','day','hour'], how='left')\n\nprint(\"vars and data type: \")\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"8a1ff50e-197e-40eb-a17c-1955466ab466","_uuid":"99e6552bee8387eddcc512ec7326bd2be55e892d","trusted":false,"collapsed":true},"cell_type":"code","source":"train_df.tail()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d16533be15dbd1045e80cc953dc75e6a2dece971","_cell_guid":"cb9ea3c6-73ec-44a6-9d9f-7c6dd421fc3d","trusted":false,"collapsed":true},"cell_type":"code","source":"#组合特征 2\nprint('grouping by ip-app combination...')\ngp = train_df[['ip', 'app', 'channel']].groupby(by=['ip', 'app'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'ip_app_count'})\ntrain_df = train_df.merge(gp, on=['ip','app'], how='left')\ndel gp\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1d465995e47345a3a74d34e2292b64d1f0699f4","_cell_guid":"a0e9b555-2a8d-433c-b8bb-d8e64c40e218","trusted":false,"collapsed":true},"cell_type":"code","source":"#组合特征 3\nprint('grouping by ip-app-os combination...')\ngp = train_df[['ip','app', 'os', 'channel']].groupby(by=['ip', 'app', 'os'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'ip_app_os_count'})\ntrain_df = train_df.merge(gp, on=['ip','app', 'os'], how='left')\ndel gp\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"44e146656116375de7e64e3fbb365ea8b42c5888","_cell_guid":"4598496c-d5c9-42c3-b6e6-8f73b39e5cce","trusted":false,"collapsed":true},"cell_type":"code","source":"# 4\nprint('grouping by : ip_day_chl_var_hour')\ngp = train_df[['ip','day','hour','channel']].groupby(by=['ip','day','channel'])[['hour']].var().reset_index().rename(index=str, columns={'hour': 'ip_tchan_count'})\ntrain_df = train_df.merge(gp, on=['ip','day','channel'], how='left')\ndel gp\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"323a9dd93a98b870e394cc26650b751d28d0dab8","_cell_guid":"3f052ad5-6f63-4b40-bb37-50982d320088","trusted":false,"collapsed":true},"cell_type":"code","source":"# 5\nprint('grouping by : ip_app_os_var_hour')\ngp = train_df[['ip','app', 'os', 'hour']].groupby(by=['ip', 'app', 'os'])[['hour']].var().reset_index().rename(index=str, columns={'hour': 'ip_app_os_var'})\ntrain_df = train_df.merge(gp, on=['ip','app', 'os'], how='left')\ndel gp\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"58ead1bd59ea7ed0e2a92bddc07404970c7a2dbd","_cell_guid":"cfa17bf8-8325-4dc6-ab2a-fab351f64512","trusted":false,"collapsed":true},"cell_type":"code","source":"# 6\nprint('grouping by : ip_app_channel_var_day')\ngp = train_df[['ip','app', 'channel', 'day']].groupby(by=['ip', 'app', 'channel'])[['day']].var().reset_index().rename(index=str, columns={'day': 'ip_app_channel_var_day'})\ntrain_df = train_df.merge(gp, on=['ip','app', 'channel'], how='left')\ndel gp\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3dee7aae6e7bf50a77c76f8bcc319db884a27e0a","_cell_guid":"eea79eae-d821-4e1e-beac-b8aba3ee2413","trusted":false,"collapsed":true},"cell_type":"code","source":"train_df['qty'] = train_df['qty'].astype('uint16')\ntrain_df['ip_app_count'] = train_df['ip_app_count'].astype('uint16')\ntrain_df['ip_app_os_count'] = train_df['ip_app_os_count'].astype('uint16')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"71bf5751-c7a1-431b-8499-ddfd8b4dd892","_uuid":"3e56bf6dd3302ce90322b107b9d5400014f31ecd","trusted":false,"collapsed":true},"cell_type":"code","source":"test_df = train_df[len_train:]\nval_df = train_df[(len_train-3000000):len_train]\ntrain_df = train_df[:(len_train-3000000)]\n#train_df = train_df[:len_train]\nprint(\"train size: \", len(train_df))\nprint(\"valid size: \", len(val_df))\nprint(\"test size : \", len(test_df))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"460dc2e9-5803-4a3b-a90c-7f2effa2a890","_uuid":"11c90cb8f4fd7652c88c71e4dc850505d10b53aa","trusted":false,"collapsed":true},"cell_type":"code","source":"target = 'is_attributed'\npredictors = ['app','device','os', 'channel', 'hour', 'qty', \n              'ip_tchan_count', 'ip_app_count',\n              'ip_app_os_count', 'ip_app_os_var',\n              'ip_app_channel_var_day']\ncategorical = ['app','device','os', 'channel', 'hour']\n\n\nsub = pd.DataFrame()\nsub['click_id'] = test_df['click_id'].astype('int')\n\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"937e27c3-f9e9-428d-9481-050220491ad9","_uuid":"73999abff3815ebf530b5f40ce3f12fe26336246","trusted":false,"collapsed":true},"cell_type":"code","source":"print(\"Training...\")\nparams = {\n    'learning_rate': 0.1,\n    #'is_unbalance': 'true', # replaced with scale_pos_weight argument\n    'num_leaves': 1400,  # we should let it be smaller than 2^(max_depth)\n    'max_depth': 3,  # -1 means no limit\n    'min_child_samples': 200,#100  # Minimum number of data need in a child(min_data_in_leaf)\n    'max_bin': 100,  # Number of bucketed bin for feature values\n    'subsample': .7,  # Subsample ratio of the training instance.\n    'subsample_freq': 1,  # frequence of subsample, <=0 means no enable\n    'colsample_bytree': 0.7,  # Subsample ratio of columns when constructing each tree.\n    'min_child_weight': 0,  # Minimum sum of instance weight(hessian) needed in a child(leaf)\n    'scale_pos_weight':99 # because training data is extremely unbalanced \n}","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4506e197-0c72-477f-a904-c409f275ced4","_uuid":"5aabe618f4f624aa402ffa346dd09bf5971d0ce7","trusted":false,"collapsed":true},"cell_type":"code","source":"bst = lgb_modelfit_nocv(params, \n                        train_df, \n                        val_df, \n                        predictors, \n                        target, \n                        objective='binary', \n                        metrics='auc',\n                        early_stopping_rounds=50, \n                        verbose_eval=True, \n                        num_boost_round=400, #300\n                        categorical_features=categorical)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"cf0c3d13-f7c4-436a-8230-2e0302fa5d6c","collapsed":true,"_uuid":"294d1d4fc98dc63182b550cc2d53a4f3bca3e8e3","trusted":false},"cell_type":"code","source":"del train_df\ndel val_df\ngc.collect()\n\nprint(\"Predicting...\")\nsub['is_attributed'] = bst.predict(test_df[predictors])\nprint(\"writing...\")\nsub.to_csv('sub_lgb_balanced99.csv',index=False)\nprint(\"done...\")\nprint(sub.info())","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"90cc67e0-604b-48ce-b8ca-4dd06d4efad5","collapsed":true,"_uuid":"e11d5817ee6ac0b01dfcdfbeafb9a614c502b8cd","trusted":false},"cell_type":"code","source":"'''print(\"Predicting...\")\nsub['is_attributed'] = bst.predict(test_df[predictors])\nprint(\"writing...\")\nsub.to_csv('lgb_second.csv',index=False)\nprint(\"done...\")\nprint(sub.info())'''","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}