{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# Load 1000th rows of train.csv\ntrain = pd.read_csv('../input/train.csv',nrows=1000)\n# Look at the head of this sample\ntrain.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#Get columns from train\ncolumns = train.columns\n# Analyze features size (if categorical or not)\nfor col in columns:\n    print(col)\n    print(len(set(train[col])))\n    print(79*'*')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# \n#from pandas.tools.plotting import scatter_matrix\n#scatter_matrix(train, alpha=0.2, figsize=(20, 20), diagonal='kde')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# Correlation each feature with hotel_cluster:\nfor col in columns:\n    if (train[col].dtype)!='object':\n        print(col)\n        print(train[col].corr(train['hotel_cluster']))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# correlation between features:\ncol_used = []\nfor col in columns:\n    if col not in col_used:\n        col_used.append(col_used)\n        for col2 in columns:\n            if (train[col].dtype)!='object' and train[col2].dtype!='object' and col2 !=col:\n                #col_used.append(col2)\n                if np.abs(train[col].corr(train[col2]))>0.5:\n                    print('correlation :\\t'+col+' \\tand '+col2+' : \\t'+str(train[col].corr(train[col2])))"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}