{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "1cc4d7ef-35f8-cc51-8bdb-a8a317555aa1"
      },
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2deddcd9-dbf1-1d73-484a-857a095ce038"
      },
      "outputs": [],
      "source": [
        "system(\"ls ../input\")\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "971125d5-9289-b900-24ea-82e4ea49d344"
      },
      "outputs": [],
      "source": [
        "library(xgboost)\n",
        "library(Matrix)\n",
        "\n",
        "set.seed(1234)\n",
        "\n",
        "\n",
        "train <- read.csv(\"../input/train.csv\")\n",
        "test  <- read.csv(\"../input/test.csv\")\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "25acfcd4-b57c-2eff-663b-32a30c7d91bd"
      },
      "outputs": [],
      "source": [
        "\n",
        "##### Removing IDs\n",
        "train$ID <- NULL\n",
        "test.id <- test$ID\n",
        "test$ID <- NULL\n",
        "\n",
        "##### Extracting TARGET\n",
        "train.y <- train$TARGET\n",
        "train$TARGET <- NULL\n",
        "\n",
        "##### 0 count per line\n",
        "count0 <- function(x) {\n",
        "  return( sum(x == 0) )\n",
        "}\n",
        "train$n0 <- apply(train, 1, FUN=count0)\n",
        "test$n0 <- apply(test, 1, FUN=count0)\n",
        "\n",
        "##### Removing constant features\n",
        "cat(\"\\n## Removing the constants features.\\n\")\n",
        "for (f in names(train)) {\n",
        "  if (length(unique(train[[f]])) == 1) {\n",
        "    cat(f, \"is constant in train. We delete it.\\n\")\n",
        "    train[[f]] <- NULL\n",
        "    test[[f]] <- NULL\n",
        "  }\n",
        "}\n",
        "\n",
        "##### Removing identical features\n",
        "features_pair <- combn(names(train), 2, simplify = F)\n",
        "toRemove <- c()\n",
        "for(pair in features_pair) {\n",
        "  f1 <- pair[1]\n",
        "  f2 <- pair[2]\n",
        "  \n",
        "  if (!(f1 %in% toRemove) & !(f2 %in% toRemove)) {\n",
        "    if (all(train[[f1]] == train[[f2]])) {\n",
        "      cat(f1, \"and\", f2, \"are equals.\\n\")\n",
        "      toRemove <- c(toRemove, f2)\n",
        "    }\n",
        "  }\n",
        "}\n",
        "\n",
        "feature.names <- setdiff(names(train), toRemove)\n",
        "\n",
        "train <- train[, feature.names]\n",
        "test <- test[, feature.names]\n",
        "\n",
        "train$TARGET <- train.y"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "54ad8a2e-5f9f-704f-63a2-b5be7b664008"
      },
      "outputs": [],
      "source": [
        "train$TARGET<-factor(train$TARGET)\n",
        "library(randomForest)\n",
        "rf_model<-randomForest(TARGET~.,data=train,importance=T,proximity=F,ntree=100,cutoff=c(0.8,0.2))\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8fdc7bf4-d205-a305-cc72-9c8e08393332"
      },
      "outputs": [],
      "source": [
        "predict_test<-predict(rf_model,newdata=test,type=\"prob\")\n",
        "result<-as.data.frame(cbind(ID=test.id,TARGET=predict_test[,2]))\n",
        "write.csv(result,file=\"submission.csv\",row.names = F)\n"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}