{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "17f9c881-5dd8-b9e6-9583-6af61f43df94"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np\n",
        "import pandas as pd\n",
        "from sklearn.decomposition import PCA,RandomizedPCA\n",
        "from sklearn.model_selection import train_test_split\n",
        "from sklearn import svm\n",
        "import matplotlib.pyplot as plt\n"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "e8a72291-8952-3308-e204-5016f29bbf43"
      },
      "source": [
        "**Load Data**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f74ed763-63ca-1a8a-f29d-4b4643f02120"
      },
      "outputs": [],
      "source": [
        "labeled_images = pd.read_csv(\"../input/train.csv\")\n",
        "images = labeled_images.iloc[:,1:]\n",
        "labels = labeled_images.iloc[:,:1]"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "6d516a2e-a9e9-dac2-8ac8-775fd2ec42d7"
      },
      "source": [
        "**PCA**\n",
        "> Use PCA to reduce the hight-dimesion, and then slice to the train sets and test sets\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f81e04f3-b48d-8f7d-f37c-c368b75dc149"
      },
      "outputs": [],
      "source": [
        "pca = PCA(n_components=64, whiten=True).fit(images)\n",
        "new_images = pca.transform(images)\n",
        "train_images,test_images, train_labels, test_labels = train_test_split(new_images,labels, train_size = 0.8)\n",
        "\n"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "ace6ce79-b15e-d4ef-8597-2dabe0b1b051"
      },
      "source": [
        "**SVM**\n",
        "> use SVM to classify the images"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a85cbc83-6fe8-4bf9-fb91-49aa233e0b99"
      },
      "outputs": [],
      "source": [
        "clf = svm.SVC()\n",
        "clf.fit(train_images, train_labels.values.ravel())\n",
        "clf.score(test_images, test_labels)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "9d5619ff-2fa7-1918-b9ff-e5b1cbbc9f2e"
      },
      "source": [
        "**Predication**\n",
        "> reduce the test data dimension by PCA,  and Then using traind svm to predicate the labels"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6d3dfe03-990d-06d8-c796-9aec8f7f8cda"
      },
      "outputs": [],
      "source": [
        "test_images = pd.read_csv(\"../input/test.csv\")\n",
        "new_test_images = pca.transform(test_images)\n",
        "results = clf.predict(new_test_images)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "c853d11e-5d3a-3d92-4ab3-e60761a05891"
      },
      "source": [
        "**Saving Result**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1846abd1-1aa3-238f-2506-764c2878d432"
      },
      "outputs": [],
      "source": [
        "df = pd.DataFrame(results)\n",
        "df.index.name='ImageId'\n",
        "df.index+=1\n",
        "df.columns=['Label']\n",
        "df.to_csv('results.csv', header=True, index_label=\"ImageId\")"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}