{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ccfc18fe-6fa6-acb7-f347-aaf1f33041cf"
      },
      "outputs": [],
      "source": [
        "# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages\n",
        "# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "library(ggplot2) # Data visualization\n",
        "library(readr) # CSV file I/O, e.g. the read_csv function\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "system(\"ls ../input\")\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "166769b8-dd04-6b75-6999-a01bd364ea05"
      },
      "outputs": [],
      "source": [
        "# Loading required packages\n",
        "library(caret)\n",
        "library(e1071)\n",
        "train=read.csv(\"../input/train.csv\")\n",
        "test=read.csv(\"../input/test.csv\")\n",
        "dim(train);dim(test)\n",
        "#summary(train[,1:10])\n",
        "test$label=1                                # Adding label column in the test dataset\n",
        "# Combining data\n",
        "comb_data=rbind(train,test)\n",
        "comb_data$label=as.factor(comb_data$label)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "7fe6f083-dca0-51e1-6329-c3227878527b"
      },
      "source": [
        "**Data Preprocessing**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6fda7036-2bba-4f0d-64c4-32b6887e084c"
      },
      "outputs": [],
      "source": [
        "# All pixels having 0 variance are removed\n",
        "store=c(1)\n",
        "for(i in 2:785){\n",
        "  if (var(comb_data[,i])==0){store=append(store,i)}\n",
        "}\n",
        "comb_data=comb_data[,-store]"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "e49b5f49-e835-31f3-cd7b-21ebfad4a3c1"
      },
      "source": [
        "**Dimensionality reduction by PCA (Principal Component Analysis)**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "47e3e24c-7935-ee79-e1d5-213cadec0b0c"
      },
      "outputs": [],
      "source": [
        "pca.train=comb_data[1:nrow(train),]\n",
        "pca.test=comb_data[42001:nrow(comb_data),]\n",
        "prin_comp=prcomp(pca.train)\n",
        "std_dev = prin_comp$sdev\n",
        "pr_var = std_dev^2\n",
        "prop_varex = pr_var/sum(pr_var)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "e3c00bf6-a601-fd67-d98d-4922353f1d59"
      },
      "source": [
        "**Visualizing Principal Components**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d7a0ed51-599e-6d2d-61c7-ff81410bfd0d"
      },
      "outputs": [],
      "source": [
        "#plot(prop_varex, xlab = \"Principal Component\",ylab = \"Proportion of Variance Explained\")\n",
        "#plot(cumsum(prop_varex), xlab = \"Principal Component\",ylab = \"Cumulative Proportion of Variance Explained\",type = \"b\")\n",
        "#pca.train=pca.train/255\n",
        "#pca.test=pca.test/255"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "2afe02b7-ab17-284c-07fa-1bfac87cfbc1"
      },
      "source": [
        "**Obtaining training and test data with variables as principal components**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f8e2ffc9-50e8-0ec0-76ec-fda7d50f110e"
      },
      "outputs": [],
      "source": [
        "train.data=data.frame(label=train$label,predict(prin_comp,pca.train))\n",
        "test.data=data.frame(predict(prin_comp,pca.test))\n",
        "\n",
        "rownames(test.data)=NULL\n",
        "head(test.data)\n",
        "train.data$label=as.factor(train.data$label)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "ef87dcc4-e9eb-1fbc-9f6d-d3b5971fe522"
      },
      "source": [
        "**Optimizing number of components to be selected**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "47d14b6c-65a3-0195-1660-2d9b55011485"
      },
      "outputs": [],
      "source": [
        "#SVM = function(x){\n",
        "#    obj=svm(label~.,data=train.data[,1:x])\n",
        "#    return (predict(obj,newdata=train.data[1:100,2:x]))\n",
        "#}\n",
        "#no_of_components=c(10,20,50,70)\n",
        "#tested=data.frame(labels=train.data$label[1:100])\n",
        "#for (i in no_of_components){\n",
        "#    mylabels=SVM(i)\n",
        "#    tested=cbind(tested,mylabels)\n",
        "#    paste(i,sum(diag(table(tested$labels,tested$mylabels))))\n",
        "#}"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "6fe7bc01-a73c-6f7b-c605-2c1061834958"
      },
      "source": [
        "**Applying SVM on the chosen components**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c289d3e3-a9ec-8c02-c41a-6847476d28b8"
      },
      "outputs": [],
      "source": [
        "#svm_tune=tune(svm,label~.,data=train.data[sample(1:42000,1000),1:51],kernel=\"radial\", ranges=list(cost=10^(1:2), gamma=2^(-8:-3)))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7aa971a4-017b-c1c1-f975-8a0f1fa4bf77"
      },
      "outputs": [],
      "source": [
        "#str(svm_tune)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c185192b-5360-45eb-978b-c99debc7813f"
      },
      "outputs": [],
      "source": [
        "# optimal cost=10,gamma=0.03125\n",
        "obj=svm(label~.,data=train.data[,1:71],cost=10,gamma=0.03125)       ## 70 principal components \n",
        "subm = data.frame(ImageId=1:nrow(test),Label=predict(obj,newdata=test.data[,1:70]))\n",
        "write.csv(subm,\"mysub2.csv\",row.names=F)\n",
        "#write.csv(subm,file=gzfile(\"mysub2.csv.gz\"),row.names=F)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6114d963-a7a2-321a-47ac-57a69cbd2253"
      },
      "outputs": [],
      "source": [
        "#write.csv(subm,\"mysub3.csv\",row.names=F)\n",
        "#write.csv(subm,file=gzfile(\"https://www.kaggle.com/vikrantthakur14/digit-recognizer/fork-of-digit-recognizer-pca-svm/output/mysub3.csv.gz\"),row.names=F)\n",
        "head(subm)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "R",
      "language": "R",
      "name": "ir"
    },
    "language_info": {
      "codemirror_mode": "r",
      "file_extension": ".r",
      "mimetype": "text/x-r-source",
      "name": "R",
      "pygments_lexer": "r",
      "version": "3.3.3"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}