{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3c9c1fc0-d4e2-24d9-c2e9-1a9ef0e74a08"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d5f43349-868f-7f48-3c5b-ce044a87626e"
      },
      "outputs": [],
      "source": [
        "from sklearn.metrics import f1_score\n",
        "from sklearn.cross_validation import train_test_split\n",
        "from sklearn.grid_search import GridSearchCV\n",
        "from sklearn import svm\n",
        "from time import time"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2d9a3be6-59d9-92a2-229d-f23d8b79e695"
      },
      "outputs": [],
      "source": [
        "def train_predict(clf, X_train, y_train, X_test, y_test):\n",
        "    ''' Train and predict using a classifer based on F1 score. '''\n",
        "    \n",
        "    # Indicate the classifier and the training set size\n",
        "    print (\"Training a {} using a training set size of {}. . .\".format(clf.__class__.__name__, len(X_train)))\n",
        "    \n",
        "    # Train the classifier\n",
        "    train_classifier(clf, X_train, y_train)\n",
        "    \n",
        "    # Print the results of prediction for both training and testing\n",
        "    train_score = predict_labels(clf, X_train, y_train)\n",
        "    test_score = predict_labels(clf, X_train, y_train)\n",
        "    print (\"F1 score for training set: {:.4f}.\".format(predict_labels(clf, X_train, y_train)))\n",
        "\n",
        "    test_score = predict_labels(clf, X_test, y_test)\n",
        "    print (\"F1 score for test set: {:.4f}.\".format(test_score))\n",
        "    return test_score"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7c008c06-c344-497d-5abf-1c23ad989629"
      },
      "outputs": [],
      "source": [
        "def train_classifier(clf, X_train, y_train):\n",
        "    ''' Fits a classifier to the training data. '''\n",
        "    \n",
        "    # Start the clock, train the classifier, then stop the clock\n",
        "    start = time()\n",
        "    clf.fit(X_train, y_train)\n",
        "    end = time()\n",
        "    \n",
        "    # Print the results\n",
        "    diff = end-start\n",
        "    print (\"Trained model in {:.4f} seconds\".format(diff))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "03b16b1b-2146-48ad-c579-f91e06c283a9"
      },
      "outputs": [],
      "source": [
        "def predict_labels(clf, features, target):\n",
        "    ''' Makes predictions using a fit classifier based on F1 score. '''\n",
        "    \n",
        "    # Start the clock, make predictions, then stop the clock\n",
        "    start = time()\n",
        "    y_pred = clf.predict(features)\n",
        "    end = time()\n",
        "    \n",
        "    # Print and return results\n",
        "    diff = end-start\n",
        "    print (\"Made predictions in {:.4f} seconds.\".format(diff))\n",
        "    return f1_score(target.values, y_pred, pos_label=1, average='weighted')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f7339bb3-8f89-6c21-7e1c-cc911c38ac7c"
      },
      "outputs": [],
      "source": [
        "def add_pred_cols(clf, df):\n",
        "    feature_cols, target_col, X_all, y_all = create_analysis_vars(df)\n",
        "    y_pred = clf.predict(df[feature_cols])\n",
        "    pred = [int(y) for y in y_pred]\n",
        "    return pred"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8dbf56d8-2a8d-531b-90ef-a881794aa2f2"
      },
      "outputs": [],
      "source": [
        "def create_analysis_vars(df):\n",
        "    feature_cols = list(df.columns[1:])\n",
        "    target_col = df.columns[0]\n",
        "    X_all = df[feature_cols]\n",
        "    y_all = df[target_col]\n",
        "    return (feature_cols, target_col, X_all, y_all)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "73b17b50-08a4-c205-e789-8178ce447384"
      },
      "outputs": [],
      "source": [
        "data = pd.read_csv(r'../input/train.csv')\n",
        "#test = pd.read_csv(r'../input/test.csv')\n",
        "\n",
        "clf = svm.SVC(random_state=0)\n",
        "\n",
        "feature_cols, target_col, X_all, y_all = create_analysis_vars(data)\n",
        "#normalise\n",
        "X_all = X_all.asty\n",
        "X_all /= 255\n",
        "X_train, X_test, y_train, y_test = train_test_split(X_all, y_all)\n",
        "score = train_predict(clf, X_train, y_train, X_test, y_test)\n",
        "print(\"Score: {:.4f}\".format(score))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a96b3d59-af15-7686-9247-280c3bfe6eee"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6e3a8960-25e6-f0b8-fc1f-963520346988"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7828e6c7-8be1-d0b3-414a-0c8921271e4f"
      },
      "outputs": [],
      "source": [
        "\n",
        "\n",
        "\n"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}