{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "8e8da892-22e7-9c47-1ddc-281fbef3236a"
      },
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d52b4c81-a8cb-9282-76b2-18e2bc8a3344"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "from sklearn.decomposition import PCA\n",
        "from sklearn.preprocessing import Imputer\n",
        "from sklearn.model_selection import KFold\n",
        "from sklearn import linear_model\n",
        "from sklearn.metrics import make_scorer\n",
        "from sklearn.ensemble import BaggingRegressor\n",
        "from sklearn.ensemble import RandomForestRegressor\n",
        "from sklearn import svm\n",
        "from sklearn.metrics import r2_score\n",
        "from sklearn.ensemble import AdaBoostRegressor\n",
        "from sklearn.model_selection import cross_val_score\n",
        "from sklearn.tree import DecisionTreeRegressor\n",
        "from sklearn.model_selection import GridSearchCV\n",
        "import matplotlib.pyplot as plt\n",
        "import tflearn\n",
        "import tensorflow as tf\n",
        "import seaborn\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fdfc434d-3a10-58da-9efd-54fec9c9a679"
      },
      "outputs": [],
      "source": [
        "train = pd.read_csv('../input/train.csv')\n",
        "#train.head()\n",
        "test = pd.read_csv('../input/test.csv')\n",
        "y = train.SalePrice\n",
        "data = pd.concat([train,test],ignore_index=True)\n",
        "data = data.drop(\"SalePrice\",1)\n",
        "print (data.shape)\n",
        "#print (train.shape)\n",
        "#print (test.shape)\n",
        "#train.head()\n",
        "#y = train.iloc[:,-1][:,np.newaxis]\n",
        "#y.head()\n",
        "#X = train.iloc[:,:-1]\n",
        "#print (X)\n",
        "#y.head()\n",
        "#print (X.shape)\n",
        "#print (y.shape)\n",
        "#X.isnull().sum()\n",
        "nans = pd.isnull(data).sum()\n",
        "nans = nans[nans>0]\n",
        "print (nans)\n",
        "#print (nans.columns)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1b689475-213f-9285-e119-30c74774964e"
      },
      "outputs": [],
      "source": [
        "'''from sklearn.decomposition import PCA\n",
        "from sklearn.model_selection import train_test_split\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=0)\n",
        "X_train_reduced = PCA(n_components=2).fit_transform(X_train)\n",
        "fig = plt.figure(1,figsize=(8,6))\n",
        "ax = Axes3D(fig, elev=-150, azim=110)\n",
        "ax.scatter(X_train_reduced[:,0],X_train_reduced[:,1],y_train[:],c='r',cmap=plt.cm.Paired)\n",
        "ax.set_title(\"First two PCA directions\")\n",
        "ax.set_xlabel(\"1st eigenvector\")\n",
        "ax.w_xaxis.set_ticklabels([])\n",
        "ax.set_ylabel(\"2nd eigenvector\")\n",
        "ax.w_yaxis.set_ticklabels([])\n",
        "ax.set_zlabel(\"the price\")\n",
        "ax.w_zaxis.set_ticklabels([])\n",
        "plt.show()\n",
        "'''"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1685c07b-d242-edef-55ff-ff2fd8896ad3"
      },
      "outputs": [],
      "source": [
        "data = data.drop('Id',1)\n",
        "data = data.drop('Alley',1)\n",
        "data = data.drop('Fence',1)\n",
        "data = data.drop('FireplaceQu',1)\n",
        "data = data.drop('MiscFeature',1)\n",
        "data = data.drop('PoolQC',1)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c00730b3-0345-07b8-00ec-ce0d6a6bf3db"
      },
      "outputs": [],
      "source": [
        "data.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b90a2c8f-01dc-f5d1-3efe-83b35588ddae"
      },
      "outputs": [],
      "source": [
        "data.dtypes.value_counts()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "28f47330-17fa-a787-3f84-dddce72c22c7"
      },
      "outputs": [],
      "source": [
        "all_columns = data.columns.values\n",
        "non_categorical = [\"LotFrontage\", \"LotArea\", \"MasVnrArea\", \"BsmtFinSF1\", \n",
        "                   \"BsmtFinSF2\", \"BsmtUnfSF\", \"TotalBsmtSF\", \"1stFlrSF\", \n",
        "                   \"2ndFlrSF\", \"LowQualFinSF\", \"GrLivArea\", \"GarageArea\", \n",
        "                   \"WoodDeckSF\", \"OpenPorchSF\", \"EnclosedPorch\", \"3SsnPorch\", \n",
        "                   \"ScreenPorch\",\"PoolArea\", \"MiscVal\"]\n",
        "categorical = [x for x in all_columns if x not in non_categorical]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1ad725e8-8f5d-ee3d-316e-ebb902450ce4"
      },
      "outputs": [],
      "source": [
        "data = pd.get_dummies(data)\n",
        "imp = Imputer(missing_values='NaN',strategy='most_frequent',axis=0)\n",
        "data = imp.fit_transform(data)\n",
        "# log transformation I do not understand\n",
        "data = np.log(data)\n",
        "labels = np.log(y)\n",
        "data[data==-np.inf] = 0"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3e874400-c660-9ff0-dbe8-4e4f72cf424c"
      },
      "outputs": [],
      "source": [
        "pca = PCA(whiten=True)\n",
        "pca.fit(data)\n",
        "variance = pd.DataFrame(pca.explained_variance_ratio_)\n",
        "print (np.cumsum(pca.explained_variance_ratio_))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0a54acb8-7d0c-b744-afa3-073b6fbdb775"
      },
      "outputs": [],
      "source": [
        "pca = PCA(n_components=36,whiten=True)\n",
        "pca = pca.fit(data)\n",
        "pca_data = pca.transform(data)\n",
        "pca_data.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b033b21a-651b-387a-a1aa-44741a2340af"
      },
      "outputs": [],
      "source": [
        "pca_train = pca_data[:1460,:]\n",
        "pca_test = pca_data[1460:,:]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "150a65a8-2617-6d3c-5a4b-273820c65aa1"
      },
      "outputs": [],
      "source": [
        "def lets_try(train,labels):\n",
        "    results={}\n",
        "    def test_model(clf):\n",
        "        \n",
        "        cv = KFold(n_splits=5,shuffle=True,random_state=45)\n",
        "        r2 = make_scorer(r2_score)\n",
        "        r2_val_score = cross_val_score(clf, train, labels, cv=cv,scoring=r2)\n",
        "        scores=[r2_val_score.mean()]\n",
        "        return scores\n",
        "\n",
        "    clf = linear_model.LinearRegression()\n",
        "    results[\"Linear\"]=test_model(clf)\n",
        "    \n",
        "    clf = linear_model.Ridge()\n",
        "    results[\"Ridge\"]=test_model(clf)\n",
        "    \n",
        "    clf = linear_model.BayesianRidge()\n",
        "    results[\"Bayesian Ridge\"]=test_model(clf)\n",
        "    \n",
        "    clf = linear_model.HuberRegressor()\n",
        "    results[\"Hubber\"]=test_model(clf)\n",
        "    \n",
        "    clf = linear_model.Lasso(alpha=1e-4)\n",
        "    results[\"Lasso\"]=test_model(clf)\n",
        "    \n",
        "    clf = BaggingRegressor()\n",
        "    results[\"Bagging\"]=test_model(clf)\n",
        "    \n",
        "    clf = RandomForestRegressor()\n",
        "    results[\"RandomForest\"]=test_model(clf)\n",
        "    \n",
        "    clf = AdaBoostRegressor()\n",
        "    results[\"AdaBoost\"]=test_model(clf)\n",
        "    \n",
        "    clf = svm.SVR()\n",
        "    results[\"SVM RBF\"]=test_model(clf)\n",
        "    \n",
        "    clf = svm.SVR(kernel=\"linear\")\n",
        "    results[\"SVM Linear\"]=test_model(clf)\n",
        "    \n",
        "    results = pd.DataFrame.from_dict(results,orient='index')\n",
        "    results.columns=[\"R Square Score\"] \n",
        "    results=results.sort(columns=[\"R Square Score\"],ascending=False)\n",
        "    results.plot(kind=\"bar\",title=\"Model Scores\")\n",
        "    axes = plt.gca()\n",
        "    axes.set_ylim([0.5,1])\n",
        "    return results\n",
        "\n",
        "lets_try(pca_train,labels)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}