{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "74010f41-eca3-87e2-e73c-3676bda225c5"
      },
      "outputs": [],
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "library(corrplot)\n",
        "library(reshape2)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "51de109e-8c14-4c1b-8e1a-0e9f410bf16e"
      },
      "outputs": [],
      "source": [
        "train <- read.csv(\"../input/train.csv\", stringsAsFactors = T)\n",
        "test <- read.csv(\"../input/test.csv\", stringsAsFactors = T)\n",
        "\n",
        "#str(train)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "47fdd0a1-24d2-6342-316e-b0245ac33092"
      },
      "outputs": [],
      "source": [
        "cormat <- cor(train[, as.vector(lapply(train, is.numeric) == T)])\n",
        "cormat[is.na(cormat)] <- 0\n",
        "head(cormat)\n",
        "corrplot(cormat, method = \"circle\", order = \"hclust\", rect.col = \"black\", rect.lwd = 5, outline = T)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "030a0b09-f7a4-2c7c-c959-dd5927dd10f2"
      },
      "outputs": [],
      "source": [
        "lm <- lm(SalePrice ~ OverallQual + GarageArea + GrLivArea, data = train)\n",
        "summary(lm)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c657cdaa-42a3-630b-0696-43914f0acef3"
      },
      "outputs": [],
      "source": [
        "submission <- data.frame(Id = integer(nrow(test)), SalePrice = numeric(nrow(test)))\n",
        "str(submission)\n",
        "\n",
        "\n",
        "prediction <- predict(lm, newdata = test[ , c(\"OverallQual\", \"GarageArea\", \"GrLivArea\")])\n",
        "submission$Id <- test[, \"Id\"]\n",
        "submission$SalePrice <- prediction\n",
        "submission[is.na(submission)] <- 0\n",
        "#head(submission)\n",
        "str(submission)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1dd5237f-c00d-06f3-0fe0-14691bcc7bbb"
      },
      "outputs": [],
      "source": [
        "write.csv(submission, \"20170323_v1.csv\", row.names = F)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}