import numpy as np
import pandas as pd
import datetime
from sklearn.cross_validation import KFold
from sklearn.cross_validation import train_test_split
import time
from sklearn import preprocessing
from xgboost import XGBRegressor
from sklearn.ensemble import RandomForestRegressor, ExtraTreesRegressor, GradientBoostingRegressor
from sklearn.grid_search import GridSearchCV
from sklearn.cross_validation import ShuffleSplit
from sklearn.metrics import make_scorer, mean_squared_error
from sklearn.linear_model import Ridge, LassoCV, LassoLarsCV, ElasticNet, RANSACRegressor
from sklearn.kernel_ridge import KernelRidge
from sklearn.neighbors import KNeighborsRegressor
from sklearn.svm import SVR
from scipy.stats import skew

if __name__=='__main__':

    def create_submission(prediction, score):
        now = datetime.datetime.now()
        sub_file = 'submission_' + str(score) + '_' + str(now.strftime("%Y-%m-%d-%H-%M")) + '.csv'
        # sub_file = 'prediction_training.csv'
        print ('Creating submission: ', sub_file)
        pd.DataFrame({'Id': test['Id'].values, 'SalePrice': prediction}).to_csv(sub_file, index=False)
    
    
    # train need to be test when do test prediction
    def data_preprocess(train, test):
        outliers = [1349, 968, 462, 1324, 812, 218, 1328, 666, 688, 431, 335, 523, 137, 3, 970, 632, 495, 30, 1432, 1453, 628,
                313, 669, 1181, 365, 1119, 142, 607, 197, 271, 66, 613, 185, 17, 1170, 1211, 238, 1292, 874, 691, 457, 1182,
                574, 1298, 1383, 1186, 150, 190, 463, 112]
        train.drop(train.index[outliers], inplace=True)
        all_data = pd.concat((train.loc[:, 'MSSubClass':'SaleCondition'],
                              test.loc[:, 'MSSubClass':'SaleCondition']))
    
        to_delete = ['Alley', 'FireplaceQu', 'PoolQC', 'Fence', 'MiscFeature']
        all_data = all_data.drop(to_delete, axis=1)
    
        train["SalePrice"] = np.log1p(train["SalePrice"])
        # log transform skewed numeric features
        numeric_feats = all_data.dtypes[all_data.dtypes != "object"].index
        skewed_feats = train[numeric_feats].apply(lambda x: skew(x.dropna()))  # compute skewness
        skewed_feats = skewed_feats[skewed_feats > 0.75]
        skewed_feats = skewed_feats.index
        all_data[skewed_feats] = np.log1p(all_data[skewed_feats])
        all_data = pd.get_dummies(all_data)
        all_data = all_data.fillna(all_data.mean())
        X_train = all_data[:train.shape[0]]
        X_test = all_data[train.shape[0]:]
        y = train.SalePrice
    
        return X_train, X_test, y
    
    
    def mean_squared_error_(ground_truth, predictions):
        return mean_squared_error(ground_truth, predictions) ** 0.5
    
    
    RMSE = make_scorer(mean_squared_error_, greater_is_better=False)
    
    
    class ensemble(object):
        def __init__(self, n_folds, stacker, base_models):
            self.n_folds = n_folds
            self.stacker = stacker
            self.base_models = base_models
    
        def fit_predict(self, train, test, ytr):
            X = train.values
            y = ytr.values
            T = test.values
            folds = list(KFold(len(y), n_folds=self.n_folds, shuffle=True, random_state=0))
            S_train = np.zeros((X.shape[0], len(self.base_models)))
            S_test = np.zeros((T.shape[0], len(self.base_models)))  # X need to be T when do test prediction
            for i, reg in enumerate(base_models):
                print ("Fitting the base model...")
                S_test_i = np.zeros((T.shape[0], len(folds)))  # X need to be T when do test prediction
                for j, (train_idx, test_idx) in enumerate(folds):
                    X_train = X[train_idx]
                    y_train = y[train_idx]
                    X_holdout = X[test_idx]
                    reg.fit(X_train, y_train)
                    y_pred = reg.predict(X_holdout)[:]
                    S_train[test_idx, i] = y_pred
                    S_test_i[:, j] = reg.predict(T)[:]
                # S_test_i[:,j] = reg.predict(X)[:]
                S_test[:, i] = S_test_i.mean(1)
    
            print ("Stacking base models...")
            param_grid = {
                'alpha': [1e-3, 5e-3, 1e-2, 5e-2, 1e-1, 0.2, 0.3, 0.4, 0.5, 0.8, 1e0, 3, 5, 7, 1e1, 2e1, 5e1],
            }
            grid = GridSearchCV(estimator=self.stacker, param_grid=param_grid, n_jobs=-1, cv=5, scoring=RMSE)
            grid.fit(S_train, y)
            try:
                print('Param grid:')
                print(param_grid)
                print('Best Params:')
                print(grid.best_params_)
                print('Best CV Score:')
                print(-grid.best_score_)
                print('Best estimator:')
                print(grid.best_estimator_)
                print(message)
            except:
                pass
    
            y_pred = grid.predict(S_test)[:]
            return y_pred, -grid.best_score_
    
    
    train = pd.read_csv("../input/train.csv")  # read train data
    test = pd.read_csv("../input/test.csv")  # read test data
    
    base_models = [
        RandomForestRegressor(
            n_jobs=-1, random_state=0,
            n_estimators=500, max_features=18, max_depth=11
        ),
        ExtraTreesRegressor(
            n_jobs=-1, random_state=0,
            n_estimators=500, max_features=20
        ),
        GradientBoostingRegressor(
            random_state=0,
            n_estimators=500, max_features=10, max_depth=6,
            learning_rate=0.05, subsample=0.8
        ),
        XGBRegressor(
            seed=0,
            n_estimators=500, max_depth=7,
            learning_rate=0.05, subsample=0.8, colsample_bytree=0.75
        ),
        RANSACRegressor(random_state=0,loss='squared_loss', max_trials=500)
    ]
    
    ensem = ensemble(
        n_folds=4,
        stacker=Ridge(),
        base_models=base_models
    )
    
    X_train, X_test, y_train = data_preprocess(train, test)
    y_pred, score = ensem.fit_predict(X_train, X_test, y_train)
    
    create_submission(np.expm1(y_pred), score)
