import numpy as np # linear algebra
import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
import lightgbm as lgb
import gc
import os
import time

os.environ['OMP_NUM_THREADS'] = '4'  # Number of threads on the Kaggle server

def freq_hours(row):
    if row['hour'] in [4, 5, 9, 10, 13, 14]:
        return 1 #most frequent hours in test data
    elif row['hour'] in [6, 11, 15]:
        return 2 #least frequent hours in test data
    return 3 #none of them
    
def freq_os(row):
    if row['os'] in [13,17,18,22,10,8,6,25,20,15,9,16,14,3,37,41,12,1,35,607,23,27,28,32,53,11,47,30,26,31,49]:
        return 1
    if row['os'] in [36, 2, 40, 4, 42, 34, 46, 0, 43, 58, 48, 24, 7, 44, 65, 56, 5, 38, 70, 79, 66, 39, 21, 29, 64, 55, 52, 73, 97, 77, 62, 50]:
        return 2
    if row['os'] in [63, 98, 57, 90, 96, 76, 100, 85, 60, 74, 109, 83, 80, 59, 102, 67, 84, 106, 92, 112, 69, 81, 178, 87, 71, 68, 72, 118, 78, 133, 132, 140]:
        return 3
    if row['os'] in [164, 137, 113, 114, 86, 89, 101, 183, 145, 155, 265, 141, 107, 152, 105, 119, 95, 111, 117, 135, 124, 142, 245, 115, 153, 149, 110, 198, 174, 217, 88, 45]:
        return 4
    if row['os'] in [342, 193, 134, 156, 175, 214, 223, 125, 162, 438, 159, 172, 161, 188, 184, 147, 148, 284, 127, 171, 329, 260, 123, 177, 196, 128, 138, 93, 603, 326, 243, 104]:
        return 5
    if row['os'] in [213, 616, 541, 404, 647, 414, 54, 251, 169, 160, 129, 99, 75, 248, 712, 602, 610, 514, 126, 94, 702, 336, 219, 255, 617, 707, 197, 150, 465, 277, 692, 252]:
        return 6
    if row['os'] in [234, 512, 715, 620, 143, 645, 690, 302, 215, 236, 552, 120, 619, 300, 254, 61, 91, 209]:
        return 7
    return 8
    
zero_ips =  [83230, 17357, 35810, 45745, 161007]

def features(df):
    print('Making features')
    
    print('1 - datetime')
    df['datetime'] = pd.to_datetime(df['click_time'])
    df['dow'] = df['datetime'].dt.dayofweek.astype('uint8')
    df['month'] = df['datetime'].dt.month.astype('uint8')
    df['day'] = df["datetime"].dt.day.astype('uint8')
    df['hour'] = df["datetime"].dt.hour.astype('uint8')
    df['minute'] = df["datetime"].dt.minute.astype('uint8')
    df['second'] = df["datetime"].dt.second.astype('uint8')
    df['freq_hour'] = df.apply(lambda row: freq_hours(row) , axis=1)
    df['is_am'] = df.apply(lambda row: row['hour'] <= 12, axis=1)
    df['on_weekend'] = df.apply(lambda row: row['dow'] >= 5, axis=1)
    df.drop(['click_time', 'datetime'], axis=1, inplace=True)
    gc.collect()
    
    print('2 - Number of clicks for ip')
    ip_clicks = df[['ip','channel']].groupby(by=['ip'])[['channel']]\
        .count().reset_index().rename(columns={'channel': 'n_ip_clicks'})
    df = df.merge(ip_clicks, on=['ip'], how='left')
    del ip_clicks
    gc.collect()
    
    print('3 - Number of channels for ip within hour')
    n_chans = df[['ip','day','hour','channel']].groupby(by=['ip','day',
              'hour'])[['channel']].count().reset_index().rename(columns={'channel': 'n_channels'})
    df = df.merge(n_chans, on=['ip','day','hour'], how='left')
    del n_chans
    gc.collect()

    print('4 - Number of channels for ip and app')
    n_chans = df[['ip','app', 'channel']].groupby(by=['ip',
              'app'])[['channel']].count().reset_index().rename(columns={'channel': 'ip_app_count'})
    df = df.merge(n_chans, on=['ip','app'], how='left')
    del n_chans
    gc.collect()

    print('5 - Number of channels for ip, app and os')
    n_chans = df[['ip','app', 'os', 'channel']].groupby(by=['ip', 'app',
              'os'])[['channel']].count().reset_index().rename(columns={'channel': 'ip_app_os_count'})
    df = df.merge(n_chans, on=['ip','app', 'os'], how='left')
    del n_chans
    gc.collect()
    
    print('6 - IPs with 0 in is_atributed')
    df['ip_zero'] = df.apply(lambda row: row['ip'] in zero_ips, axis=1)
    
    print('7 - frequent is_attributed OS')
    df['freq_os'] = df.apply(lambda row: freq_os(row) , axis=1)

    print('Fixing types')
    df.info()
    for feat in ['n_channels', 'ip_app_count', 'ip_app_os_count', 'n_ip_clicks']:
        df[feat] = df[feat].astype('uint16')
        
    df.info()
    
    return df

train_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']
test_cols  = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'click_id']
col_types = {
        'ip'            : 'uint32',
        'app'           : 'uint16',
        'device'        : 'uint16',
        'os'            : 'uint16',
        'channel'       : 'uint16',
        'is_attributed' : 'uint8',
        'click_id'      : 'uint32'
        }

#n_rows_to_skip = 110000000
rows_to_read = 3000000 #memory limits :(
print('Loading training data')
train_raw = pd.read_csv('../input/train.csv', usecols = train_cols, dtype=col_types, nrows = rows_to_read)
print('Training data loaded')
gc.collect()

print('Processing training data')
train = features(train_raw)
y = train['is_attributed']
train.drop(['is_attributed'], axis=1, inplace=True)
print('Training data ready')
gc.collect()

print('Making train matrix')
dtrain = lgb.Dataset(train.values, label=y.values)
del train, y
gc.collect()
print('Train matrix ready')

print('Making model') 
params = {
    'boosting_type': 'gbdt',  # I think dart would be better, but takes too long to run
    # 'drop_rate': 0.09,  # Rate at which to drop trees
    'objective': 'binary',
    'metric': 'auc',
    'learning_rate': 0.1,
    'num_leaves': 11,  # Was 255: Reduced to control overfitting
    'max_depth': -1,  # Was 8: LightGBM splits leaf-wise, so control depth via num_leaves
    'min_child_samples': 100,
    'max_bin': 100,
    'subsample': 0.9,  # Was 0.7
    'subsample_freq': 1,
    'colsample_bytree': 0.7,
    'min_child_weight': 0,
    'subsample_for_bin': 200000,
    'min_split_gain': 0,
    'reg_alpha': 0,
    'reg_lambda': 0,
    'nthread': 4,
    'verbose': 0,
    'scale_pos_weight': 99.76  # Closer to ratio of positives in train set
}
model = lgb.train(params=params, train_set=dtrain, num_boost_round=300)
del dtrain
gc.collect()
print('Trained model')
print('Feature importances:', list(model.feature_importance()))

print('Loading testing data') 
test_raw = pd.read_csv('../input/test.csv', usecols = test_cols, dtype=col_types)
print('Testing data loaded')
gc.collect()

print('Procesing testing data')
test = features(test_raw)
del test_raw
gc.collect()
print('Testing data processed')

output = pd.DataFrame()
output['click_id'] = test['click_id']
test.drop(['click_id'], axis=1, inplace=True)
gc.collect()
print('Test data ready')

print('Making prediction')
output['is_attributed'] = model.predict(test, num_iteration=model.best_iteration)
gc.collect()

print('Saving data')
output.to_csv('answer-' + str(time.time()) + '.csv', float_format='%.8f', index=False)
print('Saved data')
# Any results you write to the current directory are saved as output.