{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bc614712-05ab-e0ad-c1cb-a9b437bbd083"
      },
      "outputs": [],
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7f8dc6fc-61cc-8cb3-64be-549094befac6"
      },
      "outputs": [],
      "source": [
        "train_data<-read.csv(file=\"../input/train.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5e4d8341-8717-4598-b18c-bafd32be5515"
      },
      "outputs": [],
      "source": [
        "library(dplyr)\n",
        "library(Hmisc)\n",
        "library(corrplot)\n",
        "library(mice)\n",
        "library(missForest)\n",
        "library(VIM)\n",
        "library(ClustOfVar)\n",
        "library(caret)\n",
        "library(gbm)\n",
        "library(dummies)\n",
        "library(moments)\n",
        "library(e1071)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "450bcf80-8781-ae60-b917-5fb60a5045f0"
      },
      "outputs": [],
      "source": [
        "test_data<- read.csv(file=\"../input/test.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "da6c069d-aa8e-0a8c-915d-5bb2e85b64ce"
      },
      "outputs": [],
      "source": [
        "## Taking a look at the test dataset and head of train datasets traget varibale(Sales Price)\n",
        "str(test_data)\n",
        "head(train_data$SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e2b88ebe-5820-7cb7-9232-eeaea900894a"
      },
      "outputs": [],
      "source": [
        "## Log transforming the salesprice variable for standard distribution.\n",
        "train_data$log_salesprice <- log(train_data$SalePrice +1)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fcb30f6e-744c-bd9b-5712-f1641a14062d"
      },
      "outputs": [],
      "source": [
        "#checking out the dimention and describtion of training data\n",
        "dim(train_data)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "2c067d0f-e1f8-042e-d36d-d8df986b57d6"
      },
      "source": [
        "Correlation of the numeric variables in train dataset"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0bd71201-f142-a947-6b0f-adc06411c710"
      },
      "outputs": [],
      "source": [
        "## Finding the correlation of the numeric variables in train dataset.\n",
        "numeric_values <- select_if(train_data,is.numeric) # Only the columns with numeric values are selected \n",
        "#Dimension of numeric_values\n",
        "dim(numeric_values)\n",
        "## Omitting ID and obervations with NA\n",
        "corplot.names <- colnames(numeric_values[2:dim(numeric_values)[2]])\n",
        "cordata <- na.omit(train_data[,(names(train_data) %in% corplot.names)])\n",
        "## Ploting correlation of the numberic variables using corrplot.mixed plot\n",
        "corrplot.mixed(cor(cordata), lower = \"square\", upper =\"circle\",\n",
        "                  tl= \"lt\",diag =\"l\",bg=\"blue\" )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fdd93d75-d452-723b-ed9e-6744f126faf0"
      },
      "outputs": [],
      "source": [
        "#selecting tables that contain N/A values in them\n",
        "#First we need to combine both training and test data together"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "4038c6d0-79b7-49b9-6dce-81eab46493aa"
      },
      "source": [
        "**Dealing with NA in variables**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8f83cf9c-a3a6-9449-ca5b-afc82c0591d9"
      },
      "outputs": [],
      "source": [
        "## First lets combine both test and training dataset\n",
        "dim(train_data)\n",
        "dim(test_data)\n",
        "##Need to remove the salesprice and logsalesprice from train data\n",
        "full_data <- rbind(train_data[,-c(dim(train_data)[2]-1,dim(train_data)[2])],test_data)\n",
        "dim(full_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f293bfe6-0987-85bc-42b2-0d579d9bbab8"
      },
      "outputs": [],
      "source": [
        "##Finding variables(Columns) with NA\n",
        "var_NA <- colnames(full_data)[colSums(is.na(full_data))>0]\n",
        "var_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0935bd2c-a758-ba1c-72df-e31eb3271409"
      },
      "outputs": [],
      "source": [
        "##Analysing further shows NA in certain features like fence is not actual missing \n",
        "# observations. NA in fence means that the house doesnot have fence. So lets remove NA from those\n",
        "##features and mark it as WT"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "448beb65-51e7-a889-25e2-eb21ee625671"
      },
      "outputs": [],
      "source": [
        "##Function to calculate the percentage of data missing in a feature.\n",
        "percentage_NA <- function(x){\n",
        "    (sum(is.na(x))/length(x))*100\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "943d7c01-8d43-c9d3-a40c-1a68c85b4216"
      },
      "outputs": [],
      "source": [
        "##Applying the function \"percentage_NA\" on our features in var_NA.\n",
        "perc<- apply(full_data[var_NA],2,percentage_NA)\n",
        "perc<- perc[order(-perc)]\n",
        "##Ploting to visualize the percentage of missing data.\n",
        "par(las=2)\n",
        "barplot(perc,main=\"percentage of NA\", horiz=TRUE,cex.names=0.5  )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "61af9984-94ce-a6e2-8653-7d838412c81b"
      },
      "outputs": [],
      "source": [
        "## Usually a safe maximum threshold is 5% of missing data in total for a large dataset.But in this case\n",
        "## We have features that have huge number of missing data. Lets have a look at few.\n",
        "describe(full_data$Fence)\n",
        "describe(full_data$Alley)\n",
        "describe(full_data$PoolQC)\n",
        "describe(full_data$MiscFeature)\n",
        "describe(full_data$LotFrontage)\n",
        "describe(full_data$GarageYrBlt)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "04e1529f-4599-9253-2f68-f47869032d54"
      },
      "outputs": [],
      "source": [
        "##It can be observed that PoolQC has 99% of NA and MiscFeature has 96% NA and so on\n",
        "perc"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "610a0433-891f-1d5f-d2de-033f5b08af86"
      },
      "outputs": [],
      "source": [
        "## Converting \"BsmtFullBath\",\"BsmtHalfBath\" NA into 0\n",
        "full_data$BsmtFullBath[is.na(full_data$BsmtFullBath)] <- 0\n",
        "full_data$BsmtHalfBath[is.na(full_data$BsmtHalfBath)] <- 0\n",
        "\n",
        "##Now the following features NA means they do not have that particular amenity, hence lets replace\n",
        "## NA with wt(Without)\n",
        "var_None <- c(\"Alley\",\"BsmtQual\",\"BsmtCond\",\"BsmtExposure\",\"BsmtFinType1\",\"BsmtFinType2\",\n",
        "            \"FireplaceQu\",\"GarageType\",\"GarageFinish\",\"GarageQual\",\"GarageCond\",\"PoolQC\",\n",
        "            \"Fence\",\"MiscFeature\")\n",
        "\n",
        "fill_None <- function(data,feature){\n",
        "    levels(data[,feature]) <- c(levels(data[,feature]), \"None\")\n",
        "    data[,feature][is.na(data[,feature])] <- \"None\"\n",
        "    return(data[,feature])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6963005b-4e29-d497-e7ba-3efbd29ca12e"
      },
      "outputs": [],
      "source": [
        "comp_data <- full_data\n",
        "for(i in 1:length(var_None)){\n",
        "    comp_data[,var_None][i] <- fill_None(comp_data,var_None[i])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dae4fe17-44eb-2e85-44fe-f1731e3cc8dd"
      },
      "outputs": [],
      "source": [
        "##After replacing NA with None for respective features. Lets sort the remaining features that has NA.\n",
        "rem_NA<-colnames(comp_data[apply(comp_data,2,percentage_NA)>0])\n",
        "rem_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6461d2b2-8a0d-b95f-56f2-c7a70fe42dc7"
      },
      "outputs": [],
      "source": [
        "## Lets inpute missing data using missForest Package.\n",
        "## Take look how many observations are missing in rem_NA\n",
        "apply(comp_data[rem_NA],2,describe)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f79a67dd-d746-bf0f-9af1-8b467640b511"
      },
      "outputs": [],
      "source": [
        "imp.arg <- missForest(comp_data[rem_NA],verbose= TRUE, maxiter=3,ntree=20)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ec553108-e495-8b0c-3663-d0591e414389"
      },
      "outputs": [],
      "source": [
        "## Normalized Root Mean Squared error and proposition of falized classification looks satisfying.\n",
        "## Random forest has imputed plausible data.\n",
        "imp.arg$OOBerror"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24e214be-1c0b-5a49-f66c-7b926a6e7b2b"
      },
      "outputs": [],
      "source": [
        "## comp_data before imputation\n",
        "plot1 <- aggr(comp_data, col = c(\"Navy Blue\",\"Red\"),\n",
        "             numbers = TRUE, sortVars = TRUE,\n",
        "             labels = names(comp_data), cex.axis =0.7,\n",
        "             gap=2, ylab = c(\"Missing data Before Imputation\", \"Pattern\"))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "317f5ae1-a653-1bf6-5cb1-9f82077fb3b6"
      },
      "outputs": [],
      "source": [
        "##Total no. of NA in comp_data before imputation\n",
        "sum(is.na(comp_data))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "60956698-16e0-7276-118f-4fe79d4ba69a"
      },
      "outputs": [],
      "source": [
        "imp_data <- imp.arg$ximp\n",
        "## Replacing missing observations with imputed values\n",
        "for(i in 1:length(rem_NA)){\n",
        "    comp_data[,rem_NA][i] <- imp_data[i]\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "92a9fd38-7652-08a1-090f-0e9ea25c4a58"
      },
      "outputs": [],
      "source": [
        "##After replacing it with imputed values\n",
        "sum(is.na(comp_data))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fea567dd-972c-39ac-b140-f61eb7203fda"
      },
      "outputs": [],
      "source": [
        "## Lets plot using VIM\n",
        "plot2 <- aggr(comp_data, col =c(\"Navy blue\", \"Red\"),\n",
        "             numbers= TRUE, sortVars = TRUE,\n",
        "             labels = names(comp_data),cex.axis = 0.7,\n",
        "             gap=2, ylab= c(\"Missing data AFTER imputation\", \"Pattern\"))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "25247e52-e078-6d7a-1e0f-6ffb6c52a348"
      },
      "outputs": [],
      "source": [
        "## Seperating variables as numeric and categorical and storing them with diff names\n",
        "train_comp_quanti <- select_if(comp_data,is.numeric)[1:1460,-1]\n",
        "test_comp_quanti <- select_if(comp_data,is.numeric)[1461:2919,-1]\n",
        "train_comp_catg <- select_if(comp_data,is.factor)[1:1460,]\n",
        "test_comp_catg <- select_if(comp_data,is.factor)[1461:2919,]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e8dc315d-4a7b-327c-563c-19257fb8bfd7"
      },
      "outputs": [],
      "source": [
        "## Dividing train into into test and train dataset for calculating error.\n",
        "set.seed(123)\n",
        "samp_leng <- nrow(train_data)*0.70\n",
        "train_index <- sample(seq_len(nrow(train_data)),size= samp_leng)\n",
        "samp_train_quanti <- train_comp_quanti[train_index,]\n",
        "samp_test_quanti <- train_comp_quanti[-train_index,]\n",
        "samp_train_catg <- train_comp_catg[train_index,]\n",
        "samp_train_catg <- subset( samp_train_catg, select= -c(Utilities))\n",
        "samp_test_catg <- train_comp_catg[-train_index,]\n",
        "samp_test_catg <- subset( samp_test_catg , select = -c(Utilities))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "acd8b9f1-30d7-5f0c-4983-54d301c58955"
      },
      "outputs": [],
      "source": [
        "new_data <- comp_data"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "24995765-30c6-8451-ab07-1c4ba494ab8f"
      },
      "outputs": [],
      "source": [
        " drop <- c(\"Street\",\"Utilities\",\"PoolQC\")\n",
        "new_data <- new_data[!names(new_data) %in% drop]\n",
        "\n",
        "levels(new_data$Condition1) = c(levels(new_data$Condition1), \"Pos\", \"RRe\", \"RRn\")\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"PosA\", \"PosN\")] = \"Pos\"\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"RRNe\", \"RRAe\")] = \"RRe\"\n",
        "new_data$Condition1[new_data$Condition1 %in% c(\"RRNn\", \"RRAn\")] = \"RRn\"\n",
        "new_data$Condition1 <- as.factor(new_data$Condition1)\n",
        "\n",
        "levels(new_data$Condition2) = c(levels(new_data$Condition2), \"Pos\", \"RR\")\n",
        "new_data$Condition2[new_data$Condition2 %in% c(\"PosA\", \"PosN\")] <- \"Pos\"\n",
        "new_data$Condition2[new_data$Condition2 %in% c(\"RRNe\", \"RRAe\",\"RRNn\", \"RRAn\",\"RRe\",\"RRn\")] <- \"RR\"\n",
        "new_data$Condition2 <- as.factor(new_data$Condition2)\n",
        "\n",
        "levels(new_data$RoofMatl) = c(levels(new_data$RoofMatl), \"Others\")\n",
        "new_data$RoofMatl[new_data$RoofMatl %in% c(\"ClyTile\",\"Membran\",\"Metal\",\"Roll\",\"WdShake\", \"WdShngl\")] <- \"Others\"\n",
        "new_data$RoofMatl <- as.factor(new_data$RoofMatl)\n",
        "\n",
        "levels(new_data$HouseStyle) <- c(levels(new_data$HouseStyle), \"2.5Com\")\n",
        "new_data$HouseStyle[new_data$HouseStyle %in% c(\"2.5Unf\",\"2.5Fin\")] <- \"2.5Com\"\n",
        "new_data$HouseStyle <- as.factor(new_data$HouseStyle)\n",
        "\n",
        "levels(new_data$Exterior1st) <- c(levels(new_data$Exterior1st), \"Others\")\n",
        "new_data$Exterior1st[new_data$Exterior1st %in% c(\"AsphShn\", \"BrkComm\", \"CBlock\", \n",
        "                                                 \"ImStucc\",  \"Stone\",  \"Stucco\" )] <- \"Others\"\n",
        "new_data$Exterior1st <- as.factor(new_data$Exterior1st)\n",
        "\n",
        "levels(new_data$Exterior2nd) <- c(levels(new_data$Exterior2nd), \"Others\")\n",
        "new_data$Exterior2nd[new_data$Exterior2nd %in% c(\"AsphShn\", \"Brk Cmn\", \"CBlock\", \n",
        "                                                 \"ImStucc\",  \"Stone\",  \"Stucco\",\"Other\" )] = \"Others\"\n",
        "new_data$Exterior2nd <- as.factor(new_data$Exterior2nd)\n",
        "\n",
        "levels(new_data$Electrical) <- c(levels(new_data$Electrical), \"Others\")\n",
        "new_data$Electrical[new_data$Electrical %in% c(\"FuseP\", \"Mix\", \"SBrkr\")] <- \"Others\"\n",
        "new_data$Electrical <- as.factor(new_data$Electrical)\n",
        "\n",
        "levels(new_data$Heating) <- c(levels(new_data$Heating), \"Others\")\n",
        "new_data$Heating[new_data$Heating %in% c(\"Floor\", \"Grav\",  \"OthW\",  \"Wall\")] <- \"Others\"\n",
        "new_data$Heating <- as.factor(new_data$Heating)\n",
        "\n",
        "new_data[new_data$GarageYrBlt == 2207,\"GarageYrBlt\"] <- 2007\n",
        "\n",
        "new_data$GarageQual[new_data$GarageQual %in% c(\"Ex\")] <- \"Gd\"\n",
        "new_data$GarageQual <- as.factor(new_data$GarageQual)\n",
        "\n",
        "new_data$MiscFeature[new_data$MiscFeature %in%\"TenC\"] <- \"Othr\"\n",
        "new_data$MiscFeature <- as.factor(new_data$MiscFeature)\n",
        "\n",
        "\n",
        "levels(new_data$RoofStyle) <- c(levels(new_data$RoofStyle), \"MansardOrShed\")\n",
        "new_data$RoofStyle[new_data$RoofStyle %in% c(\"Mansard\",\"Shed\")] <- \"MansardOrShed\"\n",
        "new_data$RoofStyle <- as.factor(new_data$RoofStyle)\n",
        "\n",
        "new_data$ExterCond <- as.integer(new_data$ExterCond)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "28d176df-fe46-57d4-d6d5-4370b7b50186"
      },
      "outputs": [],
      "source": [
        "only_numeric <- new_data %>% select_if(is.numeric)\n",
        "skewd_data <- sapply(only_numeric, skewness )\n",
        "skewd<-skewd_data[skewd_data > 0.75]\n",
        "dropp <- c(\"KitchenAbvGr\",\"BsmtHalfBath\")\n",
        "skewd <- skewd[!names(skewd) %in% dropp]\n",
        "\n",
        "name_skew <-names(skewd)\n",
        "new_data[name_skew] <- sapply(new_data[name_skew] , function(x) log(x+1) )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f9afe550-2e74-5fe2-928b-dc4634a27747"
      },
      "outputs": [],
      "source": [
        "new_data$HeatingQC <- as.numeric(new_data$HeatingQC)\n",
        "new_data$BsmtCond <- as.numeric(new_data$BsmtCond)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e643d43f-22c8-0037-e8e8-9854a3725aae"
      },
      "outputs": [],
      "source": [
        "##Seperating train and test original\n",
        "new_train <- new_data[1:1460,-1 ]\n",
        "new_test <- new_data[1461:2919,-1]\n",
        "\n",
        "# Adding Log(saleprice +1) to new_train\n",
        "SalePrice <- train_data$log_salesprice\n",
        "new_train <- cbind(new_train,SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2612ff7e-10d1-acc0-4057-f302268f695f"
      },
      "outputs": [],
      "source": [
        "# Splitting train data into train and validation \n",
        "set.seed(1234)\n",
        "sample_length <- floor(nrow(new_train) * 0.70)\n",
        "index_train <- sample(seq_len(nrow(new_train)), size = sample_length)\n",
        "sample_train <- new_train[index_train,]\n",
        "validation <- new_train[-index_train,]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "eece3fa7-921d-22e9-9465-cf55cb1beff8"
      },
      "outputs": [],
      "source": [
        "att1 <- sample_train[sample_train$MiscFeature== \"Gar2\",][1,]\n",
        "validation <- rbind(validation,att1)\n",
        "\n",
        "pp <- sample_train[sample_train$MiscFeature ==\"Gar2\",][2,]\n",
        "sample_train <- sample_train[!sample_train$MiscFeature ==\"Gar2\",]\n",
        "sample_train <-rbind(sample_train,pp)\n",
        "\n",
        "att2 <- sample_train[sample_train$SaleType== \"Oth\",][1,]\n",
        "validation <- rbind(validation,att2)\n",
        "\n",
        "pp1 <- sample_train[sample_train$SaleType ==\"Oth\",][c(2,3),]\n",
        "sample_train <- sample_train[!sample_train$SaleType ==\"Oth\",]\n",
        "sample_train <-rbind(sample_train,pp1)\n",
        "\n",
        "sample_train$Functional[sample_train$Functional %in% \"Sev\"] <- \"Typ\"\n",
        "sample_train$Functional <- as.factor(sample_train$Functional)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3369f5d2-dbd9-aa54-6cac-01850d23fc83"
      },
      "outputs": [],
      "source": [
        "sampTrainSale <- sample_train\n",
        "valiSale <- validation\n",
        "\n",
        "sample_train <- sample_train[,-77]\n",
        "validation <- validation[,-77]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a35a2b86-32a9-ba95-88da-a1d637e47041"
      },
      "outputs": [],
      "source": [
        "# Divide sample_train into catgorical and numeric for clustering of variables using kmeans\n",
        "train_cat <- sample_train %>% select_if(is.factor)\n",
        "train_num <- sample_train %>% select_if(is.numeric)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c19fa5a8-721b-108d-f1a2-49410820804b"
      },
      "outputs": [],
      "source": [
        "# Building kmeans\n",
        "sample_km <- kmeansvar(train_num, train_cat, init= 9, nstart= 10)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "18cfec8f-32ab-8fed-723d-5a69f1334ce5"
      },
      "outputs": [],
      "source": [
        "# The gain in cohesion looks good.\n",
        "sample_km$E"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8aed3a9b-f4f9-5785-b33f-f8a170f70bd7"
      },
      "outputs": [],
      "source": [
        "sample_km$var"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a461ded4-a260-a3b1-b0dd-1dec77f8c50f"
      },
      "outputs": [],
      "source": [
        "test_cat <- validation %>% select_if(is.factor)\n",
        "test_num <- validation %>% select_if(is.numeric)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "30b5ace9-651a-1185-284d-0607e28fc9d3"
      },
      "outputs": [],
      "source": [
        "## Predicting the cluster matrix(PCA) for the validation set\n",
        "mat_vali<- predict(sample_km, test_num, test_cat)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "52bea671-025d-01e1-cd9b-4cf8ab37482b"
      },
      "outputs": [],
      "source": [
        "## Adding log of sale price to the training matrix (Score of sample_km Cluster)\n",
        "mat_train<- sample_km$scores\n",
        "mat_train <- cbind(mat_train, SalePrice = sampTrainSale[,77])\n",
        "mat_train <- data.frame(mat_train)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2734d37c-54e3-daa3-ba48-671da347dff8"
      },
      "outputs": [],
      "source": [
        "normal_linear <- lm(SalePrice ~. , mat_train)\n",
        "mat_vali <- data.frame(mat_vali)\n",
        "samp_actual <- valiSale$SalePrice\n",
        "samp_prediction1 <- predict(normal_linear, mat_vali)\n",
        "sqrt(mean(samp_prediction1-samp_actual)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "daf9ffa0-29e6-44f4-dd46-dfb83157f2af"
      },
      "outputs": [],
      "source": [
        "ori_samp_actual <- exp(samp_actual) -1\n",
        "samp_pred_actual <- exp(samp_prediction1) -1\n",
        "sqrt(mean(samp_pred_actual - ori_samp_actual)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "50860796-5c17-3364-efb3-5f4ea4ff9415"
      },
      "outputs": [],
      "source": [
        "mytraincontrol <- trainControl(method =\"repeatedCv\", number= 10, repeats = 4 )\n",
        "sample_model1 <- train(SalePrice ~ ., mat_train, trControl = mytraincontrol, method = \"glmnet\",\n",
        "                    tuneGrid = expand.grid(.alpha = seq(0,1, by =0.05),\n",
        "                                          .lambda = seq(0,0.07, by =0.01)))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fd8d0f09-e998-7bce-bba4-2ca32f3ab289"
      },
      "outputs": [],
      "source": [
        "samp_prediction2 <- predict(sample_model1,mat_vali)\n",
        "samp_pred_actual <- exp(samp_prediction2) -1\n",
        "sqrt(mean(samp_pred_actual - ori_samp_actual)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "48646883-fb90-e4e2-6916-12bac7f79f27"
      },
      "outputs": [],
      "source": [
        "samp_svm <- svm(SalePrice~., data = mat_train, scale = F , center=F, kernel = 'linear', \n",
        "               shrinking = T, cross = 15, cost = 1, epislon = 0.5)\n",
        "pred_svm <- predict(samp_svm,mat_vali)\n",
        "samp_pred_actual <- exp(pred_svm) -1\n",
        "sqrt(mean(samp_pred_actual - ori_samp_actual)^2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a5b9e705-df83-588b-3ffa-c049b4d82817"
      },
      "outputs": [],
      "source": [
        "##Taking in new_train data\n",
        "final_train <- new_train[,-77] ## Removing Sales price\n",
        "final_test <- new_test\n",
        "\n",
        "levels(final_train$Condition2) <- c(levels(final_train$Condition2), \"RR&Pos\")\n",
        "final_train$Condition2[final_train$Condition2 %in% c(\"Pos\",\"RR\")] <- \"RR&Pos\"\n",
        "final_train$Condition2 <- as.factor(final_train$Condition2)\n",
        "\n",
        "levels(final_test$Condition2) <- c(levels(final_test$Condition2), \"RR&Pos\")\n",
        "final_test$Condition2[final_test$Condition2 %in% c(\"Pos\")] <- \"RR&Pos\"\n",
        "final_test$Condition2 <- as.factor(final_test$Condition2)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "eb6ff6e4-d29d-51dd-c17e-899ab5392597"
      },
      "outputs": [],
      "source": [
        "## Separting the categorical and numerical variables\n",
        "final_cat <- final_train %>% select_if(is.factor)\n",
        "final_num <- final_train %>% select_if(is.numeric)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "33ee6a1b-d4cc-4d2b-37aa-95c48e53f851"
      },
      "outputs": [],
      "source": [
        "## Clusting variables with kmeans and deriving PCA\n",
        "final_km <- kmeansvar(final_num,final_cat, init= 9, nstart= 15)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c2c0327f-981a-ec62-9bde-d5ddcb91b57b"
      },
      "outputs": [],
      "source": [
        "## The Gain in cohesion looks goods.\n",
        "final_km$E"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d6806900-b213-9f18-a2bd-5643c7fd2c29"
      },
      "outputs": [],
      "source": [
        "final_test_cat <- final_test %>% select_if(is.factor)\n",
        "final_test_num <- final_test %>% select_if(is.numeric)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "75468d0c-3c4a-53d3-22e4-9e97a8bd2c4f"
      },
      "outputs": [],
      "source": [
        "final_test_matrix <- predict(final_km, final_test_num, final_test_cat)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "068014b5-51ec-a9fc-adcd-06dfc94ca80d"
      },
      "outputs": [],
      "source": [
        "## Final dataset for prediction\n",
        "final_train_mat <- final_km$scores\n",
        "final_train_mat <- cbind(final_train_mat,SalePrice = new_train$SalePrice)\n",
        "final_train_mat <- data.frame(final_train_mat)\n",
        "final_test_mat <- data.frame(final_test_matrix)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1d720fe7-e47b-c1a0-69f4-cf9c7559ef77"
      },
      "outputs": [],
      "source": [
        "final.svm.fit <- svm(SalePrice~., data = final_train_mat, scale = F , center=F, kernel = 'linear', \n",
        "               shrinking = T, cross = 15, cost = 1, epislon = 0.5)\n",
        "pred_svm <- predict(final.svm.fit, newdata = final_test_mat )\n",
        "\n",
        "final_prediction <- exp(pred_svm) -1 "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6ec71c6e-204b-f832-59a0-30eb8285e5f9"
      },
      "outputs": [],
      "source": [
        "final_prediction <- unname(final_prediction)\n",
        "submit <- data.frame(Id=test_data$Id,SalePrice= final_prediction)\n",
        "write.csv(submit,file= \"VigneshSVM.csv\", row.names = F)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "362e3615-c262-de72-af63-5b2dc52a2e15"
      },
      "outputs": [],
      "source": [
        "mytraincontrol <- trainControl(method =\"repeatedCv\", number= 10, repeats = 4 )\n",
        "sample_model <- train(SalePrice ~ ., final_train_mat, trControl = mytraincontrol, method = \"glmnet\",\n",
        "                    tuneGrid = expand.grid(.alpha = seq(0,1, by =0.05),\n",
        "                                          .lambda = seq(0,0.07, by =0.01)))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d4c39c34-ca58-21a8-cced-7a949b3ba32d"
      },
      "outputs": [],
      "source": [
        "pred_lm <- predict(sample_model, newdata = final_test_mat )\n",
        "pred_lm <- unname(pred_lm)\n",
        "pred_lm <- exp(pred_lm)-1\n",
        "submit2 <- data.frame(Id=test_data$Id,SalePrice= pred_lm)\n",
        "write.csv(submit,file= \"Vigneshlm.csv\", row.names = F)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1733baae-05c6-e396-77f3-157b8efa82c3"
      },
      "outputs": [],
      "source": [
        "submit2"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8dee18f4-87e3-31e5-ab95-70e715b58a54"
      },
      "outputs": [],
      "source": [
        "submit"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4bbddc91-f745-2b5f-99d9-bf43c403d13f"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}