{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bc614712-05ab-e0ad-c1cb-a9b437bbd083"
      },
      "outputs": [],
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7f8dc6fc-61cc-8cb3-64be-549094befac6"
      },
      "outputs": [],
      "source": [
        "train_data<-read.csv(file=\"../input/train.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5e4d8341-8717-4598-b18c-bafd32be5515"
      },
      "outputs": [],
      "source": [
        "library(dplyr)\n",
        "library(Hmisc)\n",
        "library(corrplot)\n",
        "library(mice)\n",
        "library(missForest)\n",
        "library(VIM)\n",
        "library(ClustOfVar)\n",
        "library(caret)\n",
        "library(gbm)\n",
        "library(dummies)\n",
        "library(moments)\n",
        "library(e1071)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "450bcf80-8781-ae60-b917-5fb60a5045f0"
      },
      "outputs": [],
      "source": [
        "test_data<- read.csv(file=\"../input/test.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "da6c069d-aa8e-0a8c-915d-5bb2e85b64ce"
      },
      "outputs": [],
      "source": [
        "## Taking a look at the test dataset and head of train datasets traget varibale(Sales Price)\n",
        "str(test_data)\n",
        "head(train_data$SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e2b88ebe-5820-7cb7-9232-eeaea900894a"
      },
      "outputs": [],
      "source": [
        "## Log transforming the salesprice variable for standard distribution.\n",
        "train_data$log_salesprice <- log(train_data$SalePrice +1)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fcb30f6e-744c-bd9b-5712-f1641a14062d"
      },
      "outputs": [],
      "source": [
        "#checking out the dimention and describtion of training data\n",
        "dim(train_data)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "2c067d0f-e1f8-042e-d36d-d8df986b57d6"
      },
      "source": [
        "Correlation of the numeric variables in train dataset"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0bd71201-f142-a947-6b0f-adc06411c710"
      },
      "outputs": [],
      "source": [
        "## Finding the correlation of the numeric variables in train dataset.\n",
        "numeric_values <- select_if(train_data,is.numeric) # Only the columns with numeric values are selected \n",
        "#Dimension of numeric_values\n",
        "dim(numeric_values)\n",
        "## Omitting ID and obervations with NA\n",
        "corplot.names <- colnames(numeric_values[2:dim(numeric_values)[2]])\n",
        "cordata <- na.omit(train_data[,(names(train_data) %in% corplot.names)])\n",
        "## Ploting correlation of the numberic variables using corrplot.mixed plot\n",
        "corrplot.mixed(cor(cordata), lower = \"square\", upper =\"circle\",\n",
        "                  tl= \"lt\",diag =\"l\",bg=\"blue\" )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fdd93d75-d452-723b-ed9e-6744f126faf0"
      },
      "outputs": [],
      "source": [
        "#selecting tables that contain N/A values in them\n",
        "#First we need to combine both training and test data together"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "4038c6d0-79b7-49b9-6dce-81eab46493aa"
      },
      "source": [
        "**Dealing with NA in variables**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8f83cf9c-a3a6-9449-ca5b-afc82c0591d9"
      },
      "outputs": [],
      "source": [
        "## First lets combine both test and training dataset\n",
        "dim(train_data)\n",
        "dim(test_data)\n",
        "##Need to remove the salesprice and logsalesprice from train data\n",
        "full_data <- rbind(train_data[,-c(dim(train_data)[2]-1,dim(train_data)[2])],test_data)\n",
        "dim(full_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f293bfe6-0987-85bc-42b2-0d579d9bbab8"
      },
      "outputs": [],
      "source": [
        "##Finding variables(Columns) with NA\n",
        "var_NA <- colnames(full_data)[colSums(is.na(full_data))>0]\n",
        "var_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0935bd2c-a758-ba1c-72df-e31eb3271409"
      },
      "outputs": [],
      "source": [
        "##Analysing further shows NA in certain features like fence is not actual missing \n",
        "# observations. NA in fence means that the house doesnot have fence. So lets remove NA from those\n",
        "##features and mark it as WT"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "448beb65-51e7-a889-25e2-eb21ee625671"
      },
      "outputs": [],
      "source": [
        "##Function to calculate the percentage of data missing in a feature.\n",
        "percentage_NA <- function(x){\n",
        "    (sum(is.na(x))/length(x))*100\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "943d7c01-8d43-c9d3-a40c-1a68c85b4216"
      },
      "outputs": [],
      "source": [
        "##Applying the function \"percentage_NA\" on our features in var_NA.\n",
        "perc<- apply(full_data[var_NA],2,percentage_NA)\n",
        "perc<- perc[order(-perc)]\n",
        "##Ploting to visualize the percentage of missing data.\n",
        "par(las=2)\n",
        "barplot(perc,main=\"percentage of NA\", horiz=TRUE,cex.names=0.5  )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "61af9984-94ce-a6e2-8653-7d838412c81b"
      },
      "outputs": [],
      "source": [
        "## Usually a safe maximum threshold is 5% of missing data in total for a large dataset.But in this case\n",
        "## We have features that have huge number of missing data. Lets have a look at few.\n",
        "describe(full_data$Fence)\n",
        "describe(full_data$Alley)\n",
        "describe(full_data$PoolQC)\n",
        "describe(full_data$MiscFeature)\n",
        "describe(full_data$LotFrontage)\n",
        "describe(full_data$GarageYrBlt)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "04e1529f-4599-9253-2f68-f47869032d54"
      },
      "outputs": [],
      "source": [
        "##It can be observed that PoolQC has 99% of NA and MiscFeature has 96% NA and so on\n",
        "perc"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "610a0433-891f-1d5f-d2de-033f5b08af86"
      },
      "outputs": [],
      "source": [
        "## Converting \"BsmtFullBath\",\"BsmtHalfBath\" NA into 0\n",
        "full_data$BsmtFullBath[is.na(full_data$BsmtFullBath)] <- 0\n",
        "full_data$BsmtHalfBath[is.na(full_data$BsmtHalfBath)] <- 0\n",
        "\n",
        "##Now the following features NA means they do not have that particular amenity, hence lets replace\n",
        "## NA with wt(Without)\n",
        "var_None <- c(\"Alley\",\"BsmtQual\",\"BsmtCond\",\"BsmtExposure\",\"BsmtFinType1\",\"BsmtFinType2\",\n",
        "            \"FireplaceQu\",\"GarageType\",\"GarageFinish\",\"GarageQual\",\"GarageCond\",\"PoolQC\",\n",
        "            \"Fence\",\"MiscFeature\")\n",
        "\n",
        "fill_None <- function(data,feature){\n",
        "    levels(data[,feature]) <- c(levels(data[,feature]), \"None\")\n",
        "    data[,feature][is.na(data[,feature])] <- \"None\"\n",
        "    return(data[,feature])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6963005b-4e29-d497-e7ba-3efbd29ca12e"
      },
      "outputs": [],
      "source": [
        "comp_data <- full_data\n",
        "for(i in 1:length(var_None)){\n",
        "    comp_data[,var_None][i] <- fill_None(comp_data,var_None[i])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dae4fe17-44eb-2e85-44fe-f1731e3cc8dd"
      },
      "outputs": [],
      "source": [
        "##After replacing NA with None for respective features. Lets sort the remaining features that has NA.\n",
        "rem_NA<-colnames(comp_data[apply(comp_data,2,percentage_NA)>0])\n",
        "rem_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6461d2b2-8a0d-b95f-56f2-c7a70fe42dc7"
      },
      "outputs": [],
      "source": [
        "## Lets inpute missing data using missForest Package.\n",
        "## Take look how many observations are missing in rem_NA\n",
        "apply(comp_data[rem_NA],2,describe)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f79a67dd-d746-bf0f-9af1-8b467640b511"
      },
      "outputs": [],
      "source": [
        "imp.arg <- missForest(comp_data[rem_NA],verbose= TRUE, maxiter=3,ntree=20)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ec553108-e495-8b0c-3663-d0591e414389"
      },
      "outputs": [],
      "source": [
        "## Normalized Root Mean Squared error and proposition of falized classification looks satisfying.\n",
        "## Random forest has imputed plausible data.\n",
        "imp.arg$OOBerror"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24e214be-1c0b-5a49-f66c-7b926a6e7b2b"
      },
      "outputs": [],
      "source": [
        "## comp_data before imputation\n",
        "plot1 <- aggr(comp_data, col = c(\"Navy Blue\",\"Red\"),\n",
        "             numbers = TRUE, sortVars = TRUE,\n",
        "             labels = names(comp_data), cex.axis =0.7,\n",
        "             gap=2, ylab = c(\"Missing data Before Imputation\", \"Pattern\"))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "317f5ae1-a653-1bf6-5cb1-9f82077fb3b6"
      },
      "outputs": [],
      "source": [
        "##Total no. of NA in comp_data before imputation\n",
        "sum(is.na(comp_data))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "60956698-16e0-7276-118f-4fe79d4ba69a"
      },
      "outputs": [],
      "source": [
        "imp_data <- imp.arg$ximp\n",
        "## Replacing missing observations with imputed values\n",
        "for(i in 1:length(rem_NA)){\n",
        "    comp_data[,rem_NA][i] <- imp_data[i]\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "92a9fd38-7652-08a1-090f-0e9ea25c4a58"
      },
      "outputs": [],
      "source": [
        "##After replacing it with imputed values\n",
        "sum(is.na(comp_data))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fea567dd-972c-39ac-b140-f61eb7203fda"
      },
      "outputs": [],
      "source": [
        "## Lets plot using VIM\n",
        "plot2 <- aggr(comp_data, col =c(\"Navy blue\", \"Red\"),\n",
        "             numbers= TRUE, sortVars = TRUE,\n",
        "             labels = names(comp_data),cex.axis = 0.7,\n",
        "             gap=2, ylab= c(\"Missing data AFTER imputation\", \"Pattern\"))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "acd8b9f1-30d7-5f0c-4983-54d301c58955"
      },
      "outputs": [],
      "source": [
        "new_data <- comp_data"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24995765-30c6-8451-ab07-1c4ba494ab8f"
      },
      "outputs": [],
      "source": [
        " drop <- c(\"Street\",\"Utilities\",\"PoolQC\")\n",
        "new_data <- new_data[!names(new_data) %in% drop]\n",
        "\n",
        "levels(new_data$Condition1) = c(levels(new_data$Condition1), \"Pos\", \"RRe\", \"RRn\")\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"PosA\", \"PosN\")] = \"Pos\"\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"RRNe\", \"RRAe\")] = \"RRe\"\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"RRNn\", \"RRAn\")] = \"RRn\"\n",
        "new_data$Condition1 <- as.factor(new_data$Condition1)\n",
        "\n",
        "levels(new_data$Condition2) = c(levels(new_data$Condition2), \"Pos\", \"RR\")\n",
        "new_data$Condition2[new_data$Condition2 %in% c(\"PosA\", \"PosN\")] <- \"Pos\"\n",
        "new_data$Condition2[new_data$Condition2 %in% c(\"RRNe\", \"RRAe\",\"RRNn\", \"RRAn\",\"RRe\",\"RRn\")] <- \"RR\"\n",
        "new_data$Condition2 <- as.factor(new_data$Condition2)\n",
        "\n",
        "levels(new_data$RoofMatl) = c(levels(new_data$RoofMatl), \"Others\")\n",
        "new_data$RoofMatl[new_data$RoofMatl %in% c(\"ClyTile\",\"Membran\",\"Metal\",\"Roll\",\"WdShake\", \"WdShngl\")] <- \"Others\"\n",
        "new_data$RoofMatl <- as.factor(new_data$RoofMatl)\n",
        "\n",
        "levels(new_data$HouseStyle) <- c(levels(new_data$HouseStyle), \"2.5Com\")\n",
        "new_data$HouseStyle[new_data$HouseStyle %in% c(\"2.5Unf\",\"2.5Fin\")] <- \"2.5Com\"\n",
        "new_data$HouseStyle <- as.factor(new_data$HouseStyle)\n",
        "\n",
        "levels(new_data$Exterior1st) <- c(levels(new_data$Exterior1st), \"Others\")\n",
        "new_data$Exterior1st[new_data$Exterior1st %in% c(\"AsphShn\", \"BrkComm\", \"CBlock\", \n",
        "                                                 \"ImStucc\",  \"Stone\",  \"Stucco\" )] <- \"Others\"\n",
        "new_data$Exterior1st <- as.factor(new_data$Exterior1st)\n",
        "\n",
        "levels(new_data$Exterior2nd) <- c(levels(new_data$Exterior2nd), \"Others\")\n",
        "new_data$Exterior2nd[new_data$Exterior2nd %in% c(\"AsphShn\", \"Brk Cmn\", \"CBlock\", \n",
        "                                                 \"ImStucc\",  \"Stone\",  \"Stucco\",\"Other\" )] = \"Others\"\n",
        "new_data$Exterior2nd <- as.factor(new_data$Exterior2nd)\n",
        "\n",
        "levels(new_data$Electrical) <- c(levels(new_data$Electrical), \"Others\")\n",
        "new_data$Electrical[new_data$Electrical %in% c(\"FuseP\", \"Mix\", \"SBrkr\")] <- \"Others\"\n",
        "new_data$Electrical <- as.factor(new_data$Electrical)\n",
        "\n",
        "levels(new_data$Heating) <- c(levels(new_data$Heating), \"Others\")\n",
        "new_data$Heating[new_data$Heating %in% c(\"Floor\", \"Grav\",  \"OthW\",  \"Wall\")] <- \"Others\"\n",
        "new_data$Heating <- as.factor(new_data$Heating)\n",
        "\n",
        "new_data[new_data$GarageYrBlt == 2207,\"GarageYrBlt\"] <- 2007\n",
        "\n",
        "new_data$GarageQual[new_data$GarageQual %in% c(\"Ex\")] <- \"Gd\"\n",
        "new_data$GarageQual <- as.factor(new_data$GarageQual)\n",
        "\n",
        "new_data$MiscFeature[new_data$MiscFeature %in%\"TenC\"] <- \"Othr\"\n",
        "new_data$MiscFeature <- as.factor(new_data$MiscFeature)\n",
        "\n",
        "\n",
        "levels(new_data$RoofStyle) <- c(levels(new_data$RoofStyle), \"MansardOrShed\")\n",
        "new_data$RoofStyle[new_data$RoofStyle %in% c(\"Mansard\",\"Shed\")] <- \"MansardOrShed\"\n",
        "new_data$RoofStyle <- as.factor(new_data$RoofStyle)\n",
        "\n",
        "new_data$ExterCond <- as.integer(new_data$ExterCond)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "28d176df-fe46-57d4-d6d5-4370b7b50186"
      },
      "outputs": [],
      "source": [
        "only_numeric <- new_data %>% select_if(is.numeric)\n",
        "skewd_data <- sapply(only_numeric, skewness )\n",
        "skewd<-skewd_data[skewd_data > 0.75]\n",
        "dropp <- c(\"KitchenAbvGr\",\"BsmtHalfBath\")\n",
        "skewd <- skewd[!names(skewd) %in% dropp]\n",
        "\n",
        "name_skew <-names(skewd)\n",
        "new_data[name_skew] <- sapply(new_data[name_skew] , function(x) log(x+1) )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f9afe550-2e74-5fe2-928b-dc4634a27747"
      },
      "outputs": [],
      "source": [
        "new_data$HeatingQC <- as.numeric(new_data$HeatingQC)\n",
        "new_data$BsmtCond <- as.numeric(new_data$BsmtCond)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e643d43f-22c8-0037-e8e8-9854a3725aae"
      },
      "outputs": [],
      "source": [
        "##Seperating train and test original\n",
        "new_train <- new_data[1:1460,-1 ]\n",
        "new_test <- new_data[1461:2919,-1]\n",
        "\n",
        "# Adding Log(saleprice +1) to new_train\n",
        "SalePrice <- train_data$log_salesprice\n",
        "new_train <- cbind(new_train,SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2612ff7e-10d1-acc0-4057-f302268f695f"
      },
      "outputs": [],
      "source": [
        "# Splitting train data into train and validation \n",
        "set.seed(1234)\n",
        "sample_length <- floor(nrow(new_train) * 0.70)\n",
        "index_train <- sample(seq_len(nrow(new_train)), size = sample_length)\n",
        "sample_train <- new_train[index_train,]\n",
        "validation <- new_train[-index_train,]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1bd2242c-979d-f337-5f75-a1b6e79a3260"
      },
      "outputs": [],
      "source": [
        "sample_train_with_sales <- sample_train\n",
        "validation_with_sales <- validation\n",
        "my_sample_data <- rbind(sample_train,validation)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "362e3615-c262-de72-af63-5b2dc52a2e15"
      },
      "outputs": [],
      "source": [
        "new_sample_data <- dummy.data.frame(my_sample_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8e19bbf0-4bbc-ebf0-c112-3d7ac958b09f"
      },
      "outputs": [],
      "source": [
        "sample_pca_train <- new_sample_data[1:nrow(sample_train),]\n",
        "sample_pca_test <- new_sample_data[-(1:nrow(sample_train)),] ##validation"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5d8b7bdc-befd-e003-fa95-2abf5c3c28da"
      },
      "outputs": [],
      "source": [
        "dropp <-names(sample_pca_train[colSums(sample_pca_train) == 0])\n",
        "sample_pca_train <- sample_pca_train[!names(sample_pca_train) %in% dropp]\n",
        "sample_pca_test <- sample_pca_test[!names(sample_pca_test) %in% dropp]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "516c5569-7c30-e441-be75-4d473c8ee3ff"
      },
      "outputs": [],
      "source": [
        "sample_prin_comp <-  prcomp(sample_pca_train, scale. =T)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5a627d3e-534d-1a67-6d6e-913581db9c9c"
      },
      "outputs": [],
      "source": [
        "biplot(sample_prin_comp, scale = 0)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "af91bd81-a700-0172-ec4e-80e9d56fba84"
      },
      "outputs": [],
      "source": [
        "sample_sdev <- sample_prin_comp$sdev\n",
        "sample_var <- sample_sdev ^2\n",
        "sample_var[1:10]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "eba24d88-f207-a962-2034-435a818609e9"
      },
      "outputs": [],
      "source": [
        "prop_var <- sample_var/ sum(sample_var)\n",
        "prop_var[1:10]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "45393290-5378-5853-2c62-21693a84f77b"
      },
      "outputs": [],
      "source": [
        "plot(cumsum(prop_var), xlab = \"Principal Component\",\n",
        "             ylab = \"Proportion of Variance Explained\",\n",
        "             type = \"b\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "276fcebf-9268-89ba-5b6e-e60f32896896"
      },
      "outputs": [],
      "source": [
        "sample_train_data <- data.frame(saleprice = sample_train$SalePrice,sample_prin_comp$x )\n",
        "sample_train_data <- sample_train_data[,1:191]\n",
        "\n",
        "sample_test_data <- predict(sample_prin_comp,sample_pca_test)\n",
        "sample_test_data <- as.data.frame(sample_test_data[,1:190])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a4d005a6-9e21-c937-cd62-d96c1884e5cf"
      },
      "outputs": [],
      "source": [
        "  sample.lm <- lm(saleprice~.,sample_train_data)\n",
        "lm_sample <- predict(sample.lm, newdata = sample_test_data )\n",
        "\n",
        "final_prediction <- exp(lm_sample) -1\n",
        "\n",
        "original_prediction <- exp(validation$SalePrice) -1\n",
        "sqrt(mean(final_prediction - original_prediction)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "15e99f4e-69c9-ae1e-e5a6-1eae97109883"
      },
      "outputs": [],
      "source": [
        "str(final_prediction)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ab3edc55-4bc2-9e15-d268-bba9accee637"
      },
      "outputs": [],
      "source": [
        "str(original_prediction)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c8b159d7-68a2-00d3-7950-a6e01617bd0c"
      },
      "outputs": [],
      "source": [
        " mytraincontrol <- trainControl(method =\"repeatedCv\", number= 10, repeats = 4 )\n",
        "sample_model <- train(saleprice ~ ., sample_train_data, trControl = mytraincontrol, method = \"glmnet\",\n",
        "                    tuneGrid = expand.grid(.alpha = seq(0,1, by =0.05),\n",
        "                                          .lambda = seq(0,0.07, by =0.01)))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b572b08f-87e3-cdbf-6691-c33f97e30245"
      },
      "outputs": [],
      "source": [
        "fit_sample <- predict(sample_model, newdata = sample_test_data )\n",
        "\n",
        "final_prediction <- exp(fit_sample) -1\n",
        "\n",
        "original_prediction <- exp(validation$SalePrice) -1\n",
        "sqrt(mean(final_prediction - original_prediction)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "98e51671-7c7f-d298-0d43-4a76b4aa7a3b"
      },
      "outputs": [],
      "source": [
        "my_data <- dummy.data.frame(new_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f72eabc6-0aad-0b3c-715b-03bdff48e8ad"
      },
      "outputs": [],
      "source": [
        "pca.train <- my_data[1:nrow(train_data),]\n",
        "pca.test <- my_data[-(1:nrow(train_data)),]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "22e2a0d3-aaee-6fb2-7c72-ee4d87c38e8c"
      },
      "outputs": [],
      "source": [
        "prin_comp <- prcomp(pca.train, scale. =T)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8d1e16e7-c2c1-4369-ba48-b56bc20d8fde"
      },
      "outputs": [],
      "source": [
        "biplot(prin_comp, scale= 0)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7f038f5c-1d9f-25cb-95c2-1b3b830048e7"
      },
      "outputs": [],
      "source": [
        "pr_sdev <- prin_comp$sdev\n",
        "pr_var <- pr_sdev ^2\n",
        "pr_var[1:10]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "718a0921-6aaf-6111-a629-17b6c75ae617"
      },
      "outputs": [],
      "source": [
        "prop_var <- pr_var/ sum(pr_var)\n",
        "prop_var[1:10]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "878467a9-2edf-d668-8bd9-515e72aae460"
      },
      "outputs": [],
      "source": [
        "plot(cumsum(prop_var), xlab = \"Principal Component\",\n",
        "             ylab = \"Proportion of Variance Explained\",\n",
        "             type = \"b\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9b90d026-05a8-4245-d773-abcfb2b734ac"
      },
      "outputs": [],
      "source": [
        "final_train_data <- data.frame(saleprice = new_train$SalePrice,prin_comp$x )\n",
        "final_train_data <- final_train_data[,1:191]\n",
        "\n",
        "final_test_data <- predict(prin_comp,pca.test)\n",
        "final_test_data <- as.data.frame(final_test_data[,1:190])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "25d9c17e-aa61-dd14-7484-22f73899af4c"
      },
      "outputs": [],
      "source": [
        "final.lm <- lm(saleprice~.,final_train_data)\n",
        "lm_final <- predict(final.lm, newdata = final_test_data )\n",
        "\n",
        "final_prediction <- exp(lm_final) -1"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "82b50c08-7370-c0b1-6e1d-e78e3e5ba2ec"
      },
      "outputs": [],
      "source": [
        "final_prediction<-unname(final_prediction)\n",
        "\n",
        "submit <- data.frame(Id=test_data$Id,SalePrice= final_prediction)\n",
        "write.csv(submit,file= \"LmVignesh.csv\", row.names = F)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c1e6a1ee-7c46-8f22-eff1-f263eccc608f"
      },
      "outputs": [],
      "source": [
        "lm_final1 <- predict(sample.lm, newdata = final_test_data )\n",
        "\n",
        "final_prediction1 <- exp(lm_final1) -1\n",
        "\n",
        "final_prediction1<-unname(final_prediction1)\n",
        "\n",
        "submit1 <- data.frame(Id=test_data$Id,SalePrice= final_prediction1)\n",
        "write.csv(submit,file= \"1LmVignesh.csv\", row.names = F)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "202dc0fa-7c23-83b8-50dc-557a8ed0b632"
      },
      "outputs": [],
      "source": [
        "lm_sample <- predict(final.lm, newdata = sample_test_data )\n",
        "\n",
        "final_prediction1 <- exp(lm_sample) -1\n",
        "\n",
        "original_prediction <- exp(validation$SalePrice) -1\n",
        "sqrt(mean(final_prediction1 - original_prediction)^2)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}