# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load in 

import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
import sys
# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

import os
import gc
import time
import numpy as np
import pandas as pd
from sklearn.cross_validation import train_test_split
import xgboost as xgb
from xgboost import plot_importance
import matplotlib.pyplot as plt
from contextlib import contextmanager
from memory_profiler import profile
from memory_profiler import memory_usage
from sklearn.utils import shuffle
from collections import defaultdict
from multiprocessing import Pool
from sklearn.model_selection import train_test_split as model_tts
import lightgbm as lgb
from datetime import datetime
# import matplotlib.pyplot as plt
# plt.switch_backend('agg')
print(os.listdir("../input"))
@contextmanager
def timer(name):
    t0=time.time()
    yield
    print(f'[{name}] done in {time.time()-t0:.0f} s')

def timeFeatures(df,drop_datetime=True):  
    df['datetime']=pd.to_datetime(df['click_time'])  
    df['dow']=df['datetime'].dt.dayofweek            
    df['doy']=df['datetime'].dt.dayofyear            
    if(drop_datetime):                            
        df.drop(['click_time','datetime'],axis=1,inplace=True)   
    return df
    
start_time = time.time()

train_columns = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']
test_columns  = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'click_id']

dtypes = {
        'ip'            : 'uint32',
        'app'           : 'uint16',
        'device'        : 'uint16',
        'os'            : 'uint16',
        'channel'       : 'uint16',
        'is_attributed' : 'uint8',
        'click_id'      : 'uint32'
        }
# Any results you write to the current directory are saved as output.
#train 10 xgboost model
models_predictions = defaultdict(list)
num_split=5
is_valid=False
path='../input/talkingdata-adtracking-fraud-detection/'
params = {'eta': 0.3,
             'tree_method': "hist",
             'grow_policy': "lossguide",
             'max_leaves': 1400,  
            'max_depth': 0, 
            'subsample': 0.9, 
              'colsample_bytree': 0.7, 
              'colsample_bylevel':0.7,
              'min_child_weight':0,
              'alpha':4,
              'objective': 'binary:logistic', 
              'scale_pos_weight':9,
              'eval_metric': 'auc', 
              'nthread':8,
              'random_state': 99, 
              'silent': True}
split_data_path='./'
######################################################################################################
############################  吸收别人的lgb解法，据说可以达到0.9692   ################################
######################################################################################################
from sklearn.model_selection import GridSearchCV
start = datetime.now()
VALIDATE = False
RANDOM_STATE = 50
VALID_SIZE = 0.90
MAX_ROUNDS = 1000
EARLY_STOP = 50
OPT_ROUNDS = 650
skiprows = range(1,134903891)
nrows = 50000000
output_filename = os.path.basename(__file__).split('.')[0]+'.csv'
print(output_filename)
pp_file='preposed_dataset/1_train.csv'

most_freq_hours_in_test_data = [4, 5, 9, 10, 13, 14]
least_freq_hours_in_test_data = [6, 11, 15]

target = 'is_attributed'
predictors = ['app','device','os', 'channel', 'hour', 'nip_day_test_hh', 'nip_day_hh', 'nip_hh_os', 'nip_hh_app', 'nip_hh_dev','next_click']
categorical = ['app', 'device', 'os', 'channel', 'hour']

def feature1(X_train, fea=None):
    ATTRIBUTION_CATEGORIES = [        
        # V1 Features #
        ###############
        ['ip'], ['app'], ['device'], ['os'], ['channel'],
        
        # V2 Features #
        ###############
        ['app', 'channel'],
        ['app', 'os'],
        ['app', 'device'],
    ]

    # Find frequency of is_attributed for each unique value in column
    freqs = {}
    for cols in ATTRIBUTION_CATEGORIES:
        
        # New feature name
        new_feature = '_'.join(cols)+'_confRate'   
        if fea is not None and new_feature not in fea:
            fea.append(new_feature) 
        
        # Perform the groupby
        group_object = X_train.groupby(cols)
        
        # Group sizes    
        group_sizes = group_object.size()
        log_group = 100000 # 1000 views -> 60% confidence, 100 views -> 40% confidence 
        
        # Aggregation function
        def rate_calculation(x):
            """Calculate the attributed rate. Scale by confidence"""
            rate = x.sum() / float(x.count())
            conf = np.min([1, np.log(x.count()) / log_group])
            return rate * conf
        
        # Perform the merge
        X_train = X_train.merge(
            group_object['is_attributed']. \
                apply(rate_calculation). \
                reset_index(). \
                rename( 
                    index=str,
                    columns={'is_attributed': new_feature}
                )[cols + [new_feature]],
            on=cols, how='left'
        )
    return X_train,predictors

def prep_data( df ):
    df['click_time']= pd.to_datetime(df['click_time'])
    with timer('add next click name'):
        D= 2**26
        df['category'] = (df['ip'].astype(str) + "_" + df['app'].astype(str) + "_" + df['device'].astype(str) \
             + "_" + df['os'].astype(str) + "_" + df['channel'].astype(str)).apply(hash) % D
        click_buffer= np.full(D, 3000000000, dtype=np.uint32)
        df['epochtime']= df['click_time'].astype(np.int64) // 10 ** 9
        next_clicks= []
        for category, time in zip(reversed(df['category'].values), reversed(df['epochtime'].values)):
            next_clicks.append(click_buffer[category]-time)
            click_buffer[category]= time
        del(click_buffer)
        df['next_click']= list(reversed(next_clicks))

    df['hour'] = df.click_time.dt.hour.astype('uint8')
    df['day'] = df.click_time.dt.day.astype('uint8')
    df.drop(['click_time'], axis=1, inplace=True)
    gc.collect()
    
    #little trick
    df['in_test_hh'] = (   3    
                         - 2*df['hour'].isin(  most_freq_hours_in_test_data ) 
                         - 1*df['hour'].isin( least_freq_hours_in_test_data ) ).astype('uint8')
    gp = df[['ip', 'day', 'in_test_hh', 'channel']].groupby(by=['ip', 'day', 'in_test_hh'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'nip_day_test_hh'})
    df = df.merge(gp, on=['ip','day','in_test_hh'], how='left')
    df.drop(['in_test_hh'], axis=1, inplace=True)
    df['nip_day_test_hh'] = df['nip_day_test_hh'].astype('uint32')
    del gp
    gc.collect()

    gp = df[['ip', 'day', 'hour', 'channel']].groupby(by=['ip', 'day', 'hour'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'nip_day_hh'})
    df = df.merge(gp, on=['ip','day','hour'], how='left')
    df['nip_day_hh'] = df['nip_day_hh'].astype('uint16')
    del gp
    gc.collect()
    
    gp = df[['ip', 'os', 'hour', 'channel']].groupby(by=['ip', 'os', 'hour'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'nip_hh_os'})
    df = df.merge(gp, on=['ip','os','hour'], how='left')
    df['nip_hh_os'] = df['nip_hh_os'].astype('uint16')
    del gp
    gc.collect()

    gp = df[['ip', 'app', 'hour', 'channel']].groupby(by=['ip', 'app',  'hour'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'nip_hh_app'})
    df = df.merge(gp, on=['ip','app','hour'], how='left')
    df['nip_hh_app'] = df['nip_hh_app'].astype('uint16')
    del gp
    gc.collect()

    gp = df[['ip', 'device', 'hour', 'channel']].groupby(by=['ip', 'device', 'hour'])[['channel']].count().reset_index().rename(index=str, columns={'channel': 'nip_hh_dev'})
    df = df.merge(gp, on=['ip','device','hour'], how='left')
    df['nip_hh_dev'] = df['nip_hh_dev'].astype('uint32')
    del gp
    gc.collect()

    # df.drop( ['ip','day'], axis=1, inplace=True )
    gc.collect()
    return df

if os.path.exists(path+pp_file):
    train_df=pd.read_csv(path+pp_file)
    print(f'train shape:{train_df.shape}')
    print(f'train.columns:{train_df.columns}')

else:    
    pass
    # train_df = pd.read_csv(path+"train.csv", skiprows=skiprows, nrows=nrows,dtype=dtypes, usecols=train_columns)
    # len_train = len(train_df)
    # gc.collect()

    # with timer('prep tain_df'):    
    #     train_df = prep_data(train_df)
    #     train_df,predictors= feature1(train_df,predictors)
    #     print(f'train columns:{train_df.columns}')
    #     print(f'predictors:{predictors}')
    # # with timer('save train df'):
    # #     train_df.to_csv(path+'preposed_dataset/1_train.csv')
    # gc.collect()

params = {
          'boosting_type': 'gbdt',
          'objective': 'binary',
          'metric':'auc',
          'learning_rate': 0.1,
          'num_leaves': 9,  # we should let it be smaller than 2^(max_depth)
          'max_depth': 5,  # -1 means no limit
          'min_child_samples': 100,  # Minimum number of data need in a child(min_data_in_leaf)
          'max_bin': 100,  # Number of bucketed bin for feature values
          'subsample': 0.9,  # Subsample ratio of the training instance.
          'subsample_freq': 1,  # frequence of subsample, <=0 means no enable
          'colsample_bytree': 0.7,  # Subsample ratio of columns when constructing each tree.
          'min_child_weight': 0,  # Minimum sum of instance weight(hessian) needed in a child(leaf)
          'min_split_gain': 0,  # lambda_l1, lambda_l2 and min_gain_to_split to regularization
          'nthread': 8,
          'verbose': 0,
          'scale_pos_weight':200, # because training data is extremely unbalanced 
          'num_threads':32
         }

if VALIDATE:
    if os.path.exists(path+'lightgbm_res/1_with_valid.txt'):
        model=lgb.Booster(model_file=path+'lightgbm_res/1_with_valid.txt')
    else:
        train_df, val_df = model_tts(train_df, test_size=VALID_SIZE, random_state=RANDOM_STATE, shuffle=True )
        print(f'用来训练的feature:{predictors}')
        dtrain = lgb.Dataset(train_df[predictors].values, 
                             label=train_df[target].values,
                             feature_name=predictors,
                             categorical_feature=categorical)
        del train_df
        gc.collect()

        dvalid = lgb.Dataset(val_df[predictors].values,
                             label=val_df[target].values,
                             feature_name=predictors,
                             categorical_feature=categorical)
        del val_df
        gc.collect()

        evals_results = {}
        with timer('train with valid'):
            model = lgb.train(params, 
                          dtrain, 
                          valid_sets=[dtrain, dvalid], 
                          valid_names=['train','valid'], 
                          evals_result=evals_results, 
                          num_boost_round=MAX_ROUNDS,
                          early_stopping_rounds=EARLY_STOP,
                          verbose_eval=50, 
                          feval=None)
        model.save_model('1_with_valid.txt')
        del dvalid,dtrain

else:
    if os.path.exists(path+'lightgbm_res/1_without_valid.txt'):
         model=lgb.Booster(model_file=path+'lightgbm_res/1_without_valid.txt')
    else:
        gc.collect()
        print(f'用来训练的feature:{predictors}')
        stepsize=10000000
        for epoch in range(OPT_ROUNDS):
            for batch in range(int(nrows/stepsize)):
                train_df = pd.read_csv(path+"train.csv", skiprows=skiprows+batch*stepsize, nrows=stepsize,dtype=dtypes, usecols=train_columns)
                # len_train = len(train_df)
                gc.collect()
            
                with timer(f'prep tain_df {epoch}:{batch}'):    
                    train_df = prep_data(train_df)
                    train_df,predictors= feature1(train_df,predictors)
                    print(f'train columns:{train_df.columns}')
                    print(f'predictors:{predictors}')
                # with timer('save train df'):
                #     train_df.to_csv(path+'preposed_dataset/1_train.csv')
                gc.collect()
                dtrain = lgb.Dataset(train_df[predictors].values, label=train_df[target].values,
                              feature_name=predictors,
                              categorical_feature=categorical
                              )
                del train_df
                gc.collect()
        
                evals_results = {}
                with timer('trian without valid'):
                    model = lgb.train(params, 
                                  dtrain, 
                                  valid_sets=[dtrain], 
                                  valid_names=['train'], 
                                  evals_result=evals_results, 
                                  num_boost_round=1,
                                  verbose_eval=1,
                                  feval=None)
                del dtrain
                
        # dtrain = lgb.Dataset(train_df[predictors].values, label=train_df[target].values,
        #                       feature_name=predictors,
        #                       categorical_feature=categorical
        #                       )
        # del train_df
        # gc.collect()

        # evals_results = {}
        # with timer('trian without valid'):
        #     model = lgb.train(params, 
        #                   dtrain, 
        #                   valid_sets=[dtrain], 
        #                   valid_names=['train'], 
        #                   evals_result=evals_results, 
        #                   num_boost_round=OPT_ROUNDS,
        #                   verbose_eval=50,
        #                   feval=None)
        model.save_model('1_without_valid.txt')
        # bst = lgb.Booster(model_file='mode.txt')
        # del dtrain
gc.collect()

if os.path.exists(path+'preposed_dataset/1_test.csv'):
    test_df=pd.read_csv(path+'preposed_dataset/1_test.csv')
    print(f'test_df columns:{test_df.columns}')
else:
    test_cols = ['ip','app','device','os', 'channel', 'click_time', 'click_id']
    with timer('read test'):
        test_df = pd.read_csv(path+"test.csv", dtype=dtypes, usecols=test_cols)
    with timer('prep test'):
        test_df = prep_data(test_df)
        test_df,_ =feature1(test_df)
        print(f'test_df columns:{test_df.columns}')
    # with timer('save test df'):
    #     test_df.to_csv(path+'preposed_dataset/1_test.csv')
    gc.collect()

sub = pd.DataFrame()
sub['click_id'] = test_df['click_id']
sub['is_attributed'] = model.predict(test_df[predictors])
sub.to_csv(output_filename, index=False, float_format='%.9f')
ax = lgb.plot_importance(model, max_num_features=30)
plt.gcf().savefig('1.png')
print('=='*35)
print('============================ Final Report ============================')
print('=='*35)
print(datetime.now(), '\n')
print('{:^17} : {:}'.format('train time', datetime.now()-start))
print('{:^17} : {:}'.format('output file', output_filename))
print('{:^17} : {:.5f}'.format('train auc', model.best_score['train']['auc']))
if VALIDATE:
    print('{:^17} : {:.5f}\n'.format('valid auc', model.best_score['valid']['auc']))
    print('{:^17} : {:}\n{:^17} : {}\n{:^17} : {}'.format('VALIDATE', VALIDATE, 'VALID_SIZE', VALID_SIZE, 'RANDOM_STATE', RANDOM_STATE))
print('{:^17} : {:}\n{:^17} : {}\n{:^17} : {}\n'.format('MAX_ROUNDS', MAX_ROUNDS, 'EARLY_STOP', EARLY_STOP, 'OPT_ROUNDS', model.best_iteration))
print('{:^17} : {:}\n{:^17} : {}\n'.format('skiprows', skiprows, 'nrows', nrows))
print('{:^17} : {:}\n{:^17} : {}\n'.format('variables', predictors, 'categorical', categorical))
print('{:^17} : {:}\n'.format('model params', params))
print('=='*35)
