{"nbformat":4,"metadata":{"language_info":{"version":"3.6.5","nbconvert_exporter":"python","file_extension":".py","pygments_lexer":"ipython3","mimetype":"text/x-python","codemirror_mode":{"version":3,"name":"ipython"},"name":"python"},"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"}},"nbformat_minor":1,"cells":[{"cell_type":"code","metadata":{"_cell_guid":"88bef13f-e965-4ee5-ae53-f565018d3b0c","_uuid":"40db6b8510efb188da83fe268b04c0d5c915088d","collapsed":true},"source":"# Some predefined values that will be handy later on\n# The debug value will let to run either CV or directly work on test data\nDEBUG = True\n# You can choose to run the code in Google cloud where you can have more Memory flexibility\nWHERE = 'kaggle'\nFILENO = 4\nNCHUNK = 4000000\n# skipping offset rows\nOFFSET = 75000000\n\nMISSING32 = 999999999\nMISSING8 = 255\nPUBLIC_CUTOFF = 4032690","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"bcc9c39a-b576-45d3-907d-b22e5a01f735","_uuid":"443a5bad0c46b440b9a8fafcace21fbf4822a882","collapsed":true},"source":"# Path setup for the Kaggle and Google clouds\nif WHERE=='kaggle':\n    inpath = '../input/talkingdata-adtracking-fraud-detection/'\n    pickle_path ='../input/training-and-validation-data-pickle/'\n    suffix = ''\n    outpath = ''\n    savepath = ''\n    oofpath = ''\n    cores = 4\nelif WHERE=='gcloud':\n    inpath = '../.kaggle/competitions/talkingdata-adtracking-fraud-detection/'\n    pickle_path = '../data/'\n    suffix = '.zip'\n    outpath = '../sub/'\n    oofpath = '../oof/'\n    savepath = '../data/'\n    cores = 7","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true},"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport pandas as pd\nimport time\nimport numpy as np\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import Imputer\nimport lightgbm as lgb\nfrom sklearn.ensemble import RandomForestClassifier\nimport xgboost as xgb\n#from xgboost import plot_importance\nfrom sklearn.linear_model import LogisticRegression\nimport gc\nimport os\nimport matplotlib.pyplot as plt\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"b03dd3c8-a71f-4995-9909-150510eb1020","_uuid":"7138064cb875eb0a9837d654687673d58079f8b2","collapsed":true},"source":"RANDOM_SEED = 1\nimport random\nrandom.seed(RANDOM_SEED)\nnp.random.seed(RANDOM_SEED)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"7216bcd8-81d1-4555-babd-3824438ef16b","_uuid":"b0065ca71e1ca73f400b0b82b66f8e314e03214a","collapsed":true},"source":"# Debug and CV mode clarification\ndebug = DEBUG\nif debug:\n    print('*** Debug mode: this is a test run for debugging purposes and CV ***')\n    \nnchunk=NCHUNK\nif debug:\n    frm=0\n    nchunk=100000\n    val_size=10000\nto=frm+nchunk\nfileno = FILENO","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true},"source":"# Load the sample data: 100000 samples\nprint('Loading training and test data...')\ntrain_df = pd.read_csv('../input/train.csv', skiprows= range(1, 40000000), nrows = 100000, parse_dates=['click_time'])\n# To keep code simple for the debug and CV purposes choose first 100000 samples from train data\nif debug: \n    test_df = pd.read_csv('../input/train.csv', parse_dates=['click_time'], skiprows= range(1, 41000000), nrows = 100000)\nelse:\n    test_df = pd.read_csv('../input/test.csv', parse_dates=['click_time'], nrows = 100000)\n    ","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"eb4d8ac8-9ad6-4b57-97aa-b1374314b68c","_uuid":"61a1931452799e63b4b073c1e6401281c617a306","collapsed":true},"source":"# Merge train and test data to apply all feature engineering at once and in the same space\nlen_train = len(train_df)\n#test_df['is_attributed'] = MISSING8\n#test_df['is_attributed'] = test_df.is_attributed.astype('uint8')\ntrain_df=train_df.append(test_df)\n\ndel test_df\ngc.collect()","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"b02e171e-87f2-49c5-8da2-64ed59d7fbe2","_uuid":"446fdacec823ab018af93c5cb2253c3604d64297","collapsed":true},"source":"# convert the click time into proper day and hour \nprint('Converting click time into days and hours...')\ntrain_df['hour'] = pd.to_datetime(train_df.click_time).dt.hour.astype('uint8')\ntrain_df['day'] = pd.to_datetime(train_df.click_time).dt.day.astype('uint8')","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"785d0952-8b4c-4830-a877-5a20e90b2d4e","_uuid":"370c8dc79cd50d5fe3574940bd92c2d93f52a2eb","collapsed":true},"source":"# Feature engineering : count by aggregated attributes\ndef do_count( df, group_cols, agg_name, agg_type='uint32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Aggregating by \", group_cols , '...' )\n    gp = df[group_cols][group_cols].groupby(group_cols).size().rename(agg_name).to_frame().reset_index()\n    df = df.merge(gp, on=group_cols, how='left')\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"93a50754-9d13-4b24-aeac-a81466a28669","_uuid":"4ce9562295587ca57463c542ef9ef1c49a1c44f2","collapsed":true},"source":"# Lets first apply the do_count in these aggregation before jumping other methods\ntrain_df = do_count( train_df, ['ip', 'day', 'hour'], 'ip_tcount', show_max=True ); gc.collect()\ntrain_df = do_count( train_df, ['ip', 'app'], 'ip_app_count', show_max=True ); gc.collect()\ntrain_df = do_count( train_df, ['ip', 'app', 'os'], 'ip_app_os_count', 'uint16', show_max=True ); gc.collect()","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"09cb0f50-e8bb-409e-90cc-bd1486c5af3b","_uuid":"bbdce1af42716aa1e6fbed6d34c5af9424ae1c3f","collapsed":true},"source":"train_df.head(5)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"5461df26-557a-44c7-accf-8ffacfde983a","_uuid":"cad677c40be2aa273cab388f483015f55ce2b16f","collapsed":true},"source":"# Feature engineering: count the unique values of a attribute within different aggregated attributes\n# Ex: count # unique apps if [ip, day, hour] aggregated together\ndef do_countuniq( df, group_cols, counted, agg_name, agg_type='uint32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Counting unqiue \", counted, \" by \", group_cols , '...' )\n    gp = df[group_cols+[counted]].groupby(group_cols)[counted].nunique().reset_index().rename(columns={counted:agg_name})\n    df = df.merge(gp, on=group_cols, how='left')\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"2033c610-aa23-460d-99c9-9639f68817cb","_uuid":"4279983fc4a59a7fb237f82c2d339b2642a6d25d","collapsed":true},"source":"# Apply do_countuniq omn these peers of attribute and aggregated attributes\n# as you observe IP plays important role since it is the most distinguishing feature of the device\n# Of course beside click time that we already separated and replace by day and hour\ntrain_df = do_countuniq( train_df, ['ip'], 'channel', 'X0', 'uint8', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['ip', 'day'], 'hour', 'X2', 'uint8', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['ip'], 'app', 'X3', 'uint8', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['ip', 'app'], 'os', 'X4', 'uint8', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['ip'], 'device', 'X5', 'uint16', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['app'], 'channel', 'X6', show_max=True ); gc.collect()\ntrain_df = do_countuniq( train_df, ['ip', 'device', 'os'], 'app', 'X8', show_max=True ); gc.collect()\ntrain_df.head(10)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"f909824a-b55b-4eea-8a13-458ba9f44b38","_uuid":"8b953ba94a5254d10dad7931e43fdc0c819a1661","collapsed":true},"source":"# Feature engineering: Number  counted item in the aggregated group + counted attribute\n# from 0 to the length of that group - 1.\ndef do_cumcount( df, group_cols, counted, agg_name, agg_type='uint32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Cumulative count by \", group_cols , '...' )\n    gp = df[group_cols+[counted]].groupby(group_cols)[counted].cumcount()\n    df[agg_name]=gp.values\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"dc34d235-8401-454a-8295-fd015dc490fb","_uuid":"961cf3bb3e68d24e335adbc868d09d74ee450e14","collapsed":true},"source":"# Apply do_cumcount on these peers of attribute and aggregated attributes\ntrain_df = do_cumcount( train_df, ['ip', 'device', 'os'], 'app', 'X1', show_max=True ); gc.collect()\ntrain_df = do_cumcount( train_df, ['ip'], 'os', 'X7', show_max=True ); gc.collect()\ntrain_df.head(5)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"943cb0b0-b973-482a-a1fb-b82a86d23f8d","_uuid":"f0413621f8072a6a80859bd1a91238ceb9170041","collapsed":true},"source":"# Feature engineering: Calculating the mean of counted attributed within aggregated group \ndef do_mean( df, group_cols, counted, agg_name, agg_type='float32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Calculating mean of \", counted, \" by \", group_cols , '...' )\n    gp = df[group_cols+[counted]].groupby(group_cols)[counted].mean().reset_index().rename(columns={counted:agg_name})\n    df = df.merge(gp, on=group_cols, how='left')\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"b87a58dc-d675-4630-bc3a-4c68f2ae729f","_uuid":"00cd838fcc51022e16ebf8abe3df4c296d409607","collapsed":true},"source":"# Apply mean on hour within ['ip', 'app', 'channel'] group\ntrain_df = do_mean( train_df, ['ip', 'app', 'channel'], 'hour', 'ip_app_channel_mean_hour', show_max=True ); gc.collect()\ntrain_df.head(5)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"cb8faadf-1fc5-434d-87d7-c4da95a9bb6b","_uuid":"cc264db84c3e3133323137046b7fbf0558cb9ded","collapsed":true},"source":"# Feature engineering: Calculating the variance of counted attributed within aggregated group \ndef do_var( df, group_cols, counted, agg_name, agg_type='float32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Calculating variance of \", counted, \" by \", group_cols , '...' )\n    gp = df[group_cols+[counted]].groupby(group_cols)[counted].var().reset_index().rename(columns={counted:agg_name})\n    df = df.merge(gp, on=group_cols, how='left')\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"5b80f0be-4416-4db6-9292-9b0e3d64afd0","_uuid":"4511f4a6381d424fc92b0c8e594821b984bd0ea1","collapsed":true},"source":"# Apply var of hour and day by diff groups\ntrain_df = do_var( train_df, ['ip', 'day', 'channel'], 'hour', 'ip_tchan_var', show_max=True ); gc.collect()\ntrain_df = do_var( train_df, ['ip', 'app', 'os'], 'hour', 'ip_app_os_var', show_max=True ); gc.collect()\ntrain_df = do_var( train_df, ['ip', 'app', 'channel'], 'day', 'ip_app_channel_var_day', show_max=True ); gc.collect()\ntrain_df.head(5)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"5e079d92-5043-4faa-a7ca-5d4d076a218e","_uuid":"9d91b5bafe4e2a0637eca1f6291050e4b1d8bb6c","collapsed":true},"source":"# Feature engineering: Calculating the kurtosis of counted attributed within aggregated group \ndef do_kurt( df, group_cols, counted, agg_name, agg_type='float32', show_max=False, show_agg=True ):\n    if show_agg:\n        print( \"Calculating kurtosis of \", counted, \" by \", group_cols , '...' )\n    gp = df[group_cols+[counted]].groupby(group_cols)[counted].apply(pd.Series.kurt).reset_index().rename(columns={counted:agg_name})\n    df = df.merge(gp, on=group_cols, how='left')\n    del gp\n    if show_max:\n        print( agg_name + \" max value = \", df[agg_name].max() )\n    df[agg_name] = df[agg_name].astype(agg_type)\n    gc.collect()\n    return( df )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"7fe1734c-f70e-4c1d-9c06-4de7720f902b","_uuid":"5d75a3316d93e0ff56cadfb0f7bd8c441b8acd9f","collapsed":true},"source":"# Apply var of hour and day by diff groups\ntrain_df = do_kurt( train_df, ['ip', 'day', 'channel'], 'hour', 'ip_tchan_kurt', show_max=True ); gc.collect()\ntrain_df = do_kurt( train_df, ['ip', 'app', 'os'], 'hour', 'ip_app_os_kurt', show_max=True ); gc.collect()\ntrain_df = do_kurt( train_df, ['ip', 'app', 'channel'], 'day', 'ip_app_channel_kurt_day', show_max=True ); gc.collect()\ntrain_df.head(5)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"5af69fbd-ea5a-437e-ac2f-adfe3f3e6d40","_uuid":"7fd57c4946e35295ea6371926d1c0523f9a5629f","collapsed":true},"source":"# Calculating the next click time of the group from previous click time\nprint('Computing nextClick...')\npredictors=[]\nnew_feature = 'nextClick'\nD=2**26\ntrain_df['category'] = (train_df['ip'].astype(str) + \"_\" + train_df['app'].astype(str) + \"_\" + train_df['device'].astype(str) \\\n        + \"_\" + train_df['os'].astype(str)).apply(hash) % D\nclick_buffer= np.full(D, 3000000000, dtype=np.uint32)\ntrain_df['epochtime']= train_df['click_time'].astype(np.int64) // 10 ** 9\nnext_clicks= []\nfor category, t in zip(reversed(train_df['category'].values), reversed(train_df['epochtime'].values)):\n    next_clicks.append(click_buffer[category]-t)\n    click_buffer[category]= t\ndel(click_buffer)\nQQ= list(reversed(next_clicks))\n# Dropping unncessary attributes that we wont use anymore\ntrain_df.drop(['epochtime','category','click_time'], axis=1, inplace=True)\ntrain_df[new_feature] = pd.Series(QQ).astype('float32')\npredictors.append(new_feature)\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"29419704-889e-4367-849f-c07a55a507b0","_uuid":"4132eed75d740e3173e6ad8ee09477629ed8c559","collapsed":true},"source":"# The previous click time from next_click by shifting the series by 1\nprint('Computing previous Click...')\ntrain_df[new_feature+'_shift'] = train_df[new_feature].shift(+1).values\npredictors.append(new_feature+'_shift')\ndel QQ, next_clicks\ngc.collect()\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"019bed3d-5f39-459f-bbff-469b26cd7217","_uuid":"87ccd54ce80929e70fea6f12b59acef4b8a99431","collapsed":true},"source":"# Correcting data types for the new features\nprint(\"variables and data types: \")\ntrain_df.info()\ntrain_df['ip_tcount'] = train_df['ip_tcount'].astype('uint16')\ntrain_df['ip_app_count'] = train_df['ip_app_count'].astype('uint16')\ntrain_df['ip_app_os_count'] = train_df['ip_app_os_count'].astype('uint16')\n\n# Labeling target value and features clearly: numerical ones (predictors) & categorical ones\ntarget = 'is_attributed'\npredictors.extend(['app','device','os', 'channel', 'hour', 'day',\n              'ip_tcount', 'ip_tchan_var', 'ip_app_count',\n              'ip_app_os_count', 'ip_app_os_var',\n              'ip_app_channel_var_day','ip_app_channel_mean_hour',\n              'X0', 'X1', 'X2', 'X3', 'X4', 'X5', 'X6', 'X7', 'X8', 'ip_tchan_kurt', 'ip_app_os_kurt', 'ip_app_channel_kurt_day'])\ncategorical = ['app', 'device', 'os', 'channel', 'hour', 'day']\nprint('predictors',predictors)","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"e68b3e98-1bd5-403b-8119-00f34de3d4e5","_uuid":"5f44b741e4f3d2f9ae123641a43d0b7cf38ee408","collapsed":true},"source":"# Splitting test and training back since we are done with feature engineering\ntest_df = train_df[len_train:]\ntrain_df = train_df[:len_train]\n\nprint(\"train size: \", len(train_df))\nprint(\"test size : \", len(test_df))\n#test_df.to_pickle('test.pkl.gz')\n#del test_df\n#gc.collect()\n","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"ecd1dbc2-826e-46a0-a259-3bfeb3c715d9","_uuid":"f5ccd8e78aef3de23c0cb678333f010ececbb7fd","collapsed":true},"source":"# Create the imputer to replace missing values with the mean\n# You can change the function from mean to median or other to see better performace\nimp = Imputer(missing_values='NaN', strategy='mean', axis=0)\nimp = imp.fit(train_df[predictors])\n\n# Impute our data, then train\ntrain_imp = imp.transform(train_df[predictors])\ntest_imp  = imp.transform(test_df[predictors])","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"32d63ebd-f6c9-42b3-9054-0ce6b2e95b39","_uuid":"5355236bfde16db69e8482100883fe79498c9872","collapsed":true},"source":"# Random Forest parameters & Testing\nstart_time = time.time()\nprint(\"RF training...\")\nrf = RandomForestClassifier(bootstrap=True, class_weight=None, criterion='gini',\n            max_depth=9, max_features='auto', max_leaf_nodes=None,\n            min_impurity_decrease=0.0, min_impurity_split=None,\n            min_samples_leaf=1, min_samples_split=2,\n            min_weight_fraction_leaf=0.0, n_estimators=10, n_jobs=1,\n            oob_score=False, random_state=0, verbose=0, warm_start=False)\nrf.fit(train_imp, train_df[target])\nprint(\"Predicting...\")\ny_pred = rf.predict_proba(test_imp)\nstop_time = time.time()\n# converting predictions into desired competetion output formst\ndef convert_prediction(raw_preds):\n    preds = 1 - raw_preds[:, 0]\n    return preds\ny_pred = convert_prediction(y_pred)\nprint(\"RF elapsed time: \", stop_time - start_time)\nprint(  \"\\nRandom Forest test AUC score:    \", \n            roc_auc_score( test_df[target], y_pred )  )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"360a0846-16b4-4825-aaf8-8295cd92c5d3","scrolled":true,"_uuid":"5cbe8e616dbdf16a80b46ed4c278f7acca73e2ee","collapsed":true},"source":"# Logistic regression params\nstart_time = time.time()\nprint(\"Logistic Regression training...\")\nlg = LogisticRegression(C=1.0, class_weight=None, dual=False, fit_intercept=True,\n          intercept_scaling=1, max_iter=100, multi_class='ovr', n_jobs=1,\n          penalty='l2', random_state=None, solver='lbfgs', tol=0.0001,\n          verbose=0, warm_start=False)\nlg.fit(train_imp, train_df[target])\nprint(\" LR Predicting...\")\ny_pred = lg.predict_proba(test_imp)\nstop_time = time.time()\n# converting predictions into desired competetion output formst\ndef convert_prediction(raw_preds):\n    preds = 1 - raw_preds[:, 1]\n    return preds\ny_pred = convert_prediction(y_pred)\nprint(\" LR elapsed time: \", stop_time - start_time)\nprint(  \"\\n Logistic Regression test AUC score:    \", \n            roc_auc_score( test_df[target], y_pred )  )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"747fb66b-e883-4945-b2a3-04673cb0653c","_uuid":"ad094019d6df2f1e7248e7999207d702519c3a94","collapsed":true},"source":"\n# Set the params(this params from Pranav kernel) for xgboost model\nstart_time = time.time()\nprint(\"XGBoost training...\")\nxgb_params = {'eta': 0.3,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':9,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'random_state': 99, \n          'silent': True}\n          \ndtrain = xgb.DMatrix(train_df[predictors].values, train_df[target].values)\ndtest = xgb.DMatrix(test_df[predictors].values, label=test_df[target].values)\n\nwatchlist = [(dtrain, 'train'), (dtest, 'test')]\nmodel_xgb = xgb.train(xgb_params, dtrain, 200, watchlist, maximize=True, early_stopping_rounds = 30, verbose_eval=10)\n# Preticting\nprint(\" XGBoost Predicting...\")\ny_pred = model_xgb.predict(dtest, ntree_limit=model_xgb.best_ntree_limit)\nstop_time = time.time()\nprint(\" XGBoost elapsed time: \", stop_time - start_time)\nprint(  \"\\n  XGBoost test AUC score:    \", \n            roc_auc_score( test_df[target], y_pred )  )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"419c99cd-c687-4ecc-8b79-b531878d765a","_uuid":"3d76226250716042c708c198975eac577b6d882c","collapsed":true},"source":"# Setting the parameters of Light GBM model\nprint(\" LightGBM Training...\")\nstart_time = time.time()\n\nobjective='binary' \nmetrics='auc'\nearly_stopping_rounds=30 \nverbose_eval=True \nnum_boost_round=1000\n# Let the model know the categorical values\ncategorical_features=categorical\nlgb_params = {\n    'boosting_type': 'gbdt',\n    'objective': objective,\n    'metric':metrics,\n    'learning_rate': 0.10,\n    #'is_unbalance': 'true', # replaced with scale_pos_weight argument\n    'num_leaves': 7,  # 2^max_depth - 1\n    'max_depth': 3,  # -1 means no limit\n    'min_child_samples': 100,  # Minimum number of data need in a child(min_data_in_leaf)\n    'max_bin': 100,  # Number of bucketed bin for feature values\n    'subsample': 0.7,  # Subsample ratio of the training instance.\n    'subsample_freq': 1,  # frequence of subsample, <=0 means no enable\n    'colsample_bytree': 0.9,  # Subsample ratio of columns when constructing each tree.\n    'min_child_weight': 0,  # Minimum sum of instance weight(hessian) needed in a child(leaf)\n    'scale_pos_weight':200, # because training data is extremely unbalanced \n    'subsample_for_bin': 200000,  # Number of samples for constructing bin\n    'min_split_gain': 0,  # lambda_l1, lambda_l2 and min_gain_to_split to regularization\n    'reg_alpha': 0,  # L1 regularization term on weights\n    'reg_lambda': 5,  # L2 regularization term on weights\n    'nthread': cores,\n    'verbose': 0,\n    'metric':metrics\n}\n\n# lgb compatible training data\nxgtrain = lgb.Dataset(train_df[predictors].values, label=train_df[target].values,\n                      feature_name=predictors,\n                      categorical_feature=categorical)\nxgtest = lgb.Dataset(test_df[predictors].values, label=test_df[target].values,\n                      feature_name=predictors,\n                      categorical_feature=categorical)\nevals_results = {}                     \nprint( lgb_params )\nbst = lgb.train(lgb_params, \n                 xgtrain, \n                 valid_sets=[xgtrain, xgtest], \n                 valid_names=['train','test'], \n                 evals_result=evals_results, \n                 num_boost_round=num_boost_round,\n                 early_stopping_rounds=early_stopping_rounds,\n                 verbose_eval=10, \n                 feval=None)\n\n\n# Preticting\nprint(\" LIghtGBM Predicting...\")\ny_pred = bst.predict(test_df[predictors],num_iteration=bst.best_iteration)\nstop_time = time.time()\nprint(\" LGBM elapsed time: \", stop_time - start_time)\nprint(  \"\\n  LGBM test AUC score:    \", \n            roc_auc_score( test_df[target], y_pred )  )","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"64fa099f-6b75-44a6-82f9-72d9a16b8498","_uuid":"74c7c832a7035a3842778b689de4b1e7fa7d3f4b","collapsed":true},"source":"","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"56a90b6d-e928-4520-a90e-7b8a9b409964","_uuid":"4a50a8bfae0a689b86b1dad3dbd1ad1a5be6a1a6","collapsed":true},"source":"","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"b3d11a04-0015-4847-ad35-578c02919839","_uuid":"ce8b0b4dc7b6237fef13ab8aab121214c5546736","collapsed":true},"source":"","outputs":[],"execution_count":null},{"cell_type":"code","metadata":{"_cell_guid":"5d10f358-e0b6-4b19-b411-d69db7a626c5","_uuid":"b7da1909849767614e4fb41e80189a588bb6791a","collapsed":true},"source":"","outputs":[],"execution_count":null}]}