# This Python 3 environment comes with many helpful analytics libraries installed
# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python
# For example, here's several helpful packages to load in 

#coding:utf-8


import pandas as pd
import matplotlib.pyplot as plt
import numpy as np 
from sklearn.linear_model import Ridge

from sklearn.ensemble import RandomForestRegressor

 
def preprocess():
  
#   train_df = pd.read_csv('/home/jian/DATA_SETS/kaggle/house_price/train.csv', index_col=0)
#   test_df = pd.read_csv('/home/jian/DATA_SETS/kaggle/house_price/test.csv', index_col=0)
  train_df = pd.read_csv('../input/train.csv', index_col=0)
  test_df = pd.read_csv('../input/test.csv', index_col=0)
  #print(train_df.head(5))
  
  
  def show_price():
    prices = pd.DataFrame({"price":train_df['SalePrice'], "log(price+1)":np.log1p(train_df['SalePrice'])})
    prices.hist()
    plt.show()
  
  
  y_train = np.log1p(train_df.pop('SalePrice'))
  all_df = pd.concat((train_df, test_df), axis=0)
  #print(all_df.shape)
  
  
  # MSSubClass
  
  print(all_df['MSSubClass'].dtypes)
  
  all_df['MSSubClass'] = all_df['MSSubClass'].astype(str)
  print(all_df['MSSubClass'].value_counts())
  
  def show_Category():
    all_df['MSSubClass'].value_counts().plot(kind='bar')
    plt.show()
    
  print(pd.get_dummies(all_df['MSSubClass'], prefix='MSSubClass').head(5))
  
  # 对 category 数据 进行 One-hot
  all_dummy_df = pd.get_dummies(all_df)
  
  
  #numerical 数值型数据
  
  #  填充Missing Data
  all_dummy_df.isnull().sum().sort_values(ascending=False).head(10)
  mean_cols = all_dummy_df.mean()
  mean_cols.head(10)
  all_dummy_df = all_dummy_df.fillna(mean_cols)
  all_dummy_df.isnull().sum().sum()
  
  # 标准化
  numeric_cols = all_df.columns[all_df.dtypes != 'object']
  numeric_col_means = all_dummy_df.loc[:, numeric_cols].mean()
  numeric_col_std = all_dummy_df.loc[:, numeric_cols].std()
  all_dummy_df.loc[:, numeric_cols] = (all_dummy_df.loc[:, numeric_cols] - numeric_col_means) / numeric_col_std

  dummy_train_df = all_dummy_df.loc[train_df.index]
  dummy_test_df = all_dummy_df.loc[test_df.index]

  return dummy_train_df, dummy_test_df, y_train, test_df


def train():
  dummy_train_df, dummy_test_df, y_train, test_df= preprocess()
  X_train = dummy_train_df.values
  X_test = dummy_test_df.values
  ridge = Ridge(alpha=15)
  rf = RandomForestRegressor(n_estimators=500, max_features=.3)
  
  ridge.fit(X_train, y_train)
  rf.fit(X_train, y_train)
  
  y_ridge = np.expm1(ridge.predict(X_test))
  y_rf = np.expm1(rf.predict(X_test))
  
  y_final = (y_ridge + y_rf) / 2
  
  submission_df = pd.DataFrame(data= {'Id' : test_df.index, 'SalePrice': y_final}).to_csv('2017-03-19.csv', index =False)   

if __name__ == '__main__':
    train()