# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load in 

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

import os
import gc
import numpy as np
import pandas as pd
import random
path = "../input/"

train_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']
test_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time']


dtypes = {
        'ip'            : 'uint32',
        'app'           : 'uint16',
        'device'        : 'uint16',
        'os'            : 'uint16',
        'channel'       : 'uint16',
        'is_attributed' : 'uint8',
        }
        
dtypes_test = {
        'ip'            : 'uint32',
        'app'           : 'uint16',
        'device'        : 'uint16',
        'os'            : 'uint16',
        'channel'       : 'uint16'
        }

lines = 184903891 # total number of rows in the complete dataset
skiplines = np.random.choice(np.arange(1, lines), size=lines-1-50000000, replace=False)
print("reading data...")
#sort the list
skiplines=np.sort(skiplines)
train = pd.read_csv(path+'train.csv', skiprows=skiplines, dtype=dtypes, usecols=train_cols, header=0)
train_attributed = pd.DataFrame()
chunksize = 10 ** 6
#in each chunk, filter for values that have 'is_attributed'==1, and merge these values into one dataframe
for chunk in pd.read_csv(path+'train.csv', chunksize=chunksize, dtype=dtypes):
    filtered = (chunk[(np.where(chunk['is_attributed']==1, True, False))])
    train_attributed = pd.concat([train_attributed, filtered], ignore_index=True)
    
train = pd.concat([train[train.is_attributed ==0], train_attributed], ignore_index=True)
test = pd.read_csv(path+'test.csv', dtype=dtypes_test, usecols=test_cols, header=0,)
len_train = len(train)
train=train.append(test)

del test
gc.collect()
del skiplines
gc.collect()
del train_attributed
gc.collect()
del filtered
gc.collect()

print("done reading data...")

train['click_time'] = pd.to_datetime(train['click_time'])
#train['period']= pd.cut(train.click_time.dt.hour,[-1,6,12,18,24],labels=['Night','Morning','Afternoon','Evening'])
train['hour']= train.click_time.dt.hour
train['day'] = train.click_time.dt.day.astype('uint8')

train.drop('click_time', axis=1,inplace=True)
train.drop('attributed_time', axis=1,inplace=True)


channel_count = train.groupby(['ip','day','hour'])['channel'].count().reset_index()
channel_count.columns = ['ip','day','hour','channel_count']
train = pd.merge(train, channel_count, how='left', on=['ip','day','hour'])
train['channel_count'].fillna(0, inplace=True)
del channel_count
gc.collect()

channel_count = train.groupby(['ip','app'])['channel'].count().reset_index()
channel_count.columns = ['ip','app','ip_app']
train = pd.merge(train, channel_count, how='left', on=['ip','app'])
train['ip_app'].fillna(0, inplace=True)
del channel_count
gc.collect()

channel_count = train.groupby(['ip','app','os'])['channel'].count().reset_index()
channel_count.columns = ['ip','app','os','ip_app_os']
train = pd.merge(train, channel_count, how='left', on=['ip','app','os'])
train['ip_app_os'].fillna(0, inplace=True)
del channel_count
gc.collect()
    
    
    
from sklearn.model_selection import train_test_split

### Split the train and test ###
test = train[len_train:]
train = train[:len_train]


print("splitting the data...")
X_train, X_test, y_train, y_test, = train_test_split(train.drop(['is_attributed','ip'], axis=1), 
                                                     train['is_attributed'], 
                                                     test_size = .1, 
                                                     random_state=123)

x_train, x_val, y_train, y_val = train_test_split(X_train, y_train,
                                                  test_size = .1,
                                                  random_state=12)

X_submission_id = pd.read_csv(path+'test.csv', dtype='int', usecols=['click_id'])
X_submission = test.drop(['is_attributed'], axis=1) 
X_submission.drop(['ip'], axis=1, inplace=True)


del test
gc.collect()

del train
gc.collect()

##### SMOTE #####
#from imblearn.over_sampling import SMOTE
#sm = SMOTE(random_state=123)
#x_train_res, y_train_res = sm.fit_sample(x_train, y_train)

print("start training the model...")

##### XGBOOST #####
import xgboost as xgb

params = {'eta': 0.05,
          'tree_method': "hist",
          'grow_policy': "lossguide",
          'max_leaves': 1400,  
          'max_depth': 0, 
          'subsample': 0.8, 
          'colsample_bytree': 0.7, 
          'colsample_bylevel':0.7,
          'min_child_weight':0,
          'alpha':4,
          'objective': 'binary:logistic', 
          'scale_pos_weight':9,
          'eval_metric': 'auc', 
          'nthread':8,
          'silent': True}

dtrain=xgb.DMatrix(x_train, y_train)
del x_train, y_train
gc.collect()

dvalid = xgb.DMatrix(x_val, y_val)
del x_val, y_val
gc.collect()

watchlist = [(dtrain, 'train'), (dvalid, 'valid')]
classifier = xgb.train(params, dtrain, 200, watchlist, early_stopping_rounds = 20, verbose_eval=5)
del watchlist, dtrain, dvalid
gc.collect()




#generate prediction for the test set
dtest=xgb.DMatrix(X_submission)
del X_submission
gc.collect()

X_submission_id['is_attributed'] = classifier.predict(dtest, ntree_limit=classifier.best_ntree_limit)
del dtest
gc.collect()

X_submission_id.to_csv('submission9.csv', float_format='%.8f', index = False)