{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\ndist = pd.read_csv(\"../input/destinations.csv\")\n\ndestinations = pd.read_csv(\"../input/destinations.csv\")\n\ntrain = pd.read_csv('../input/train.csv',\n                            usecols=[\"date_time\", \"user_location_country\", \"user_location_region\", \"user_location_city\",\n                                    \"user_id\", \"is_booking\", \"orig_destination_distance\",\n                                     \"hotel_cluster\", \"srch_ci\", \"srch_co\", \"srch_destination_id\", \n                                     \"hotel_continent\", \"hotel_country\", \"hotel_market\"],\n                            dtype={'date_time':np.str_, 'user_location_country':np.int8, \n                                   'user_location_region':np.int8, 'user_location_city':np.int8, \n                                   'user_id':np.int32, 'is_booking':np.int8,\n                                   \"orig_destination_distance\":np.float64,\n                                   \"hotel_cluster\":np.int8,\n                                   'srch_ci':np.str_, 'srch_co':np.str_,\n                                   \"srch_destination_id\":np.int32,\n                                   \"hotel_continent\":np.int8,\n                                   \"hotel_country\":np.int8,\n                                   \"hotel_market\":np.int8}                        \n                           )\n                           \ntest = pd.read_csv('../input/test.csv',\n                           usecols=[\"id\", \"date_time\", \"user_location_country\", \"user_location_region\", \"user_location_city\",\n                                \"user_id\", \"orig_destination_distance\",\n                                   \"srch_ci\", \"srch_co\", \"srch_destination_id\",\n                                   \"hotel_continent\", \"hotel_country\", \"hotel_market\"],\n                            dtype={'id':np.int32, 'date_time':np.str_, 'user_location_country':np.int8, \n                            'user_location_region':np.int8, 'user_location_city':np.int8, \n                            'user_id':np.int32, \n                            \"orig_destination_distance\":np.float64, 'srch_ci':np.str_, 'srch_co':np.str_,\n                                   \"srch_destination_id\":np.int32,\n                                   \"hotel_continent\":np.int8,\n                                   \"hotel_country\":np.int8,\n                                   \"hotel_market\":np.int8})\t\ntrain.shape\ntest.shape\ntrain.head(5)\ntest.head(5)\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# assesing feature importance for random forest\nfrom sklearn.ensemble import RandomForestClassifier\nfeat_labels = df_hotel.columns[1:]\nforest = RandomForestClassifier(\n    n_estimators=10000,\n    random_state=0\n)\nforest.fit(x_train, y_train)\nimportances=forest.feature_importances_\nindices = np.argsort(importances)[::-1]\nfor f in range(x_train.shape[1]):\n    print(\"%2d) %-*s %f\" % (f + 1, 30, feat_labels[f],importances[indices[f]]))    \n    \n    \n"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}