{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "print(\"hello\")"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "from math import log\nfrom collections import defaultdict\n\nrestricted = []\nfor i in range(23):\n    if i not in [0,11,12]:\n        restricted.append(i)\n\nfile = open(\"../input/train.csv\", \"r\")\ncluster = defaultdict(int)\nfeature = [defaultdict(lambda:defaultdict(int)) for i in range(23)]\ncount = 0\nfor line in file:\n    if count % 2000000 == 0:\n        print(count)\n    raw = line.strip().split(\",\")\n    c = int(raw[-1])\n    cluster[c] += 1\n    for i in restricted:\n        if raw[i] != \"\":\n            feature[i][raw[i]][c] += 1\n    count += 1"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def gainratio(D, infoCluster):\n    info_a = 0\n    split_info = 0\n    d = 0\n    for i, Di in D.items():\n        di = 0\n        for j, dij in Di.items():\n            info_a -= dij*log(dij)\n            di += dij\n        info_a += di*log(di)\n        split_info -= di*log(di)\n        d += di\n    split_info += d*log(d)\n    return (infoCluster - info_a) / split_info\n\nd, infoCluster = 0, 0\nfor j, cj in cluster.items():\n    infoCluster -= cj*log(cj)\n    d += cj\ninfoCluster += d*log(d)\nfile = open(\"../input/train.csv\", \"r\")\nattr = file.readline().strip().split(\",\")\nfile.close()\nratio = []\nfor i in restricted:\n    ratio.append( (gainratio(feature[i], infoCluster), attr[i]) )"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "ratio.sort(reverse=True)\nratio"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}