{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bc614712-05ab-e0ad-c1cb-a9b437bbd083"
      },
      "outputs": [],
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7f8dc6fc-61cc-8cb3-64be-549094befac6"
      },
      "outputs": [],
      "source": [
        "train_data<-read.csv(file=\"../input/train.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5e4d8341-8717-4598-b18c-bafd32be5515"
      },
      "outputs": [],
      "source": [
        "library(dplyr)\n",
        "library(Hmisc)\n",
        "library(corrplot)\n",
        "library(mice)\n",
        "library(missForest)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "450bcf80-8781-ae60-b917-5fb60a5045f0"
      },
      "outputs": [],
      "source": [
        "test_data<- read.csv(file=\"../input/test.csv\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "da6c069d-aa8e-0a8c-915d-5bb2e85b64ce"
      },
      "outputs": [],
      "source": [
        "## Taking a look at the test dataset and head of train datasets traget varibale(Sales Price)\n",
        "str(test_data)\n",
        "head(train_data$SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e2b88ebe-5820-7cb7-9232-eeaea900894a"
      },
      "outputs": [],
      "source": [
        "## Log transforming the salesprice variable for standard distribution.\n",
        "train_data$log_salesprice <- log(train_data$SalePrice)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fcb30f6e-744c-bd9b-5712-f1641a14062d"
      },
      "outputs": [],
      "source": [
        "#checking out the dimention and describtion of training data\n",
        "dim(train_data)\n"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "2c067d0f-e1f8-042e-d36d-d8df986b57d6"
      },
      "source": [
        "Correlation of the numeric variables in train dataset"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0bd71201-f142-a947-6b0f-adc06411c710"
      },
      "outputs": [],
      "source": [
        "## Finding the correlation of the numeric variables in train dataset.\n",
        "numeric_values <- select_if(train_data,is.numeric) # Only the columns with numeric values are selected \n",
        "#Dimension of numeric_values\n",
        "dim(numeric_values)\n",
        "## Omitting ID and obervations with NA\n",
        "corplot.names <- colnames(numeric_values[2:dim(numeric_values)[2]])\n",
        "cordata <- na.omit(train_data[,(names(train_data) %in% corplot.names)])\n",
        "## Ploting correlation of the numberic variables using corrplot.mixed plot\n",
        "corrplot.mixed(cor(cordata), lower = \"square\", upper =\"circle\",\n",
        "                  tl= \"lt\",diag =\"l\",bg=\"blue\" )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fdd93d75-d452-723b-ed9e-6744f126faf0"
      },
      "outputs": [],
      "source": [
        "#selecting tables that contain N/A values in them\n",
        "#First we need to combine both training and test data together"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "4038c6d0-79b7-49b9-6dce-81eab46493aa"
      },
      "source": [
        "**Dealing with NA in variables**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8f83cf9c-a3a6-9449-ca5b-afc82c0591d9"
      },
      "outputs": [],
      "source": [
        "## First lets combine both test and training dataset\n",
        "dim(train_data)\n",
        "dim(test_data)\n",
        "##Need to remove the salesprice and logsalesprice from train data\n",
        "full_data <- rbind(train_data[,-c(dim(train_data)[2]-1,dim(train_data)[2])],test_data)\n",
        "dim(full_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f293bfe6-0987-85bc-42b2-0d579d9bbab8"
      },
      "outputs": [],
      "source": [
        "##Finding variables(Columns) with NA\n",
        "var_NA <- colnames(full_data)[colSums(is.na(full_data))>0]\n",
        "var_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0935bd2c-a758-ba1c-72df-e31eb3271409"
      },
      "outputs": [],
      "source": [
        "##Analysing further shows NA in certain features like fence is not actual missing \n",
        "# observations. NA in fence means that the house doesnot have fence. So lets remove NA from those\n",
        "##features and mark it as WT\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "448beb65-51e7-a889-25e2-eb21ee625671"
      },
      "outputs": [],
      "source": [
        "##Function to calculate the percentage of data missing in a feature.\n",
        "percentage_NA <- function(x){\n",
        "    (sum(is.na(x))/length(x))*100\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "943d7c01-8d43-c9d3-a40c-1a68c85b4216"
      },
      "outputs": [],
      "source": [
        "##Applying the function \"percentage_NA\" on our features in var_NA.\n",
        "perc<- apply(full_data[var_NA],2,percentage_NA)\n",
        "perc<- perc[order(-perc)]\n",
        "##Ploting to visualize the percentage of missing data.\n",
        "par(las=2)\n",
        "barplot(perc,main=\"percentage of NA\", horiz=TRUE,cex.names=0.5  )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "61af9984-94ce-a6e2-8653-7d838412c81b"
      },
      "outputs": [],
      "source": [
        "## Usually a safe maximum threshold is 5% of missing data in total for a large dataset.But in this case\n",
        "## We have features that have huge number of missing data. Lets have a look at few.\n",
        "describe(full_data$Fence)\n",
        "describe(full_data$Alley)\n",
        "describe(full_data$PoolQC)\n",
        "describe(full_data$MiscFeature)\n",
        "describe(full_data$LotFrontage)\n",
        "describe(full_data$GarageYrBlt)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "04e1529f-4599-9253-2f68-f47869032d54"
      },
      "outputs": [],
      "source": [
        "##It can be observed that PoolQC has 99% of NA and MiscFeature has 96% NA and so on\n",
        "perc"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "610a0433-891f-1d5f-d2de-033f5b08af86"
      },
      "outputs": [],
      "source": [
        "##Now the following features NA means they do not have that particular amenity, hence lets replace\n",
        "## NA with wt(Without)\n",
        "var_None <- c(\"Alley\",\"BsmtQual\",\"BsmtCond\",\"BsmtExposure\",\"BsmtFinType1\",\"BsmtFinType2\",\n",
        "            \"FireplaceQu\",\"GarageType\",\"GarageFinish\",\"GarageQual\",\"GarageCond\",\"PoolQC\",\n",
        "            \"Fence\",\"MiscFeature\")\n",
        "\n",
        "fill_None <- function(data,feature){\n",
        "    levels(data[,feature]) <- c(levels(data[,feature]), \"None\")\n",
        "    data[,feature][is.na(data[,feature])] <- \"None\"\n",
        "    return(data[,feature])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6963005b-4e29-d497-e7ba-3efbd29ca12e"
      },
      "outputs": [],
      "source": [
        "comp_data <- full_data\n",
        "for(i in 1:length(var_None)){\n",
        "    comp_data[,var_None][i] <- fill_None(comp_data,var_None[i])\n",
        "}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dae4fe17-44eb-2e85-44fe-f1731e3cc8dd"
      },
      "outputs": [],
      "source": [
        "##After replacing NA with None for respective features. Lets sort the remaining features that has NA.\n",
        "rem_NA<-colnames(comp_data[apply(comp_data,2,percentage_NA)>0])\n",
        "rem_NA"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6461d2b2-8a0d-b95f-56f2-c7a70fe42dc7"
      },
      "outputs": [],
      "source": [
        "## Lets inpute missing data using missForest Package.\n",
        "## Take look how many observations are missing in rem_NA\n",
        "apply(comp_data[rem_NA],2,describe)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a1cf4e42-93d3-e2ed-ecd6-96bd42c4d0a6"
      },
      "outputs": [],
      "source": [
        "data.imp<- missForest(comp_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d93f844a-c073-21b3-f43a-f6cde189467a"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}