{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\n#https://www.dataquest.io/blog/python-vs-r/"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import random\nfilename = \"../input/train.csv\"\nsamples = 100000\n\n# Read some samples - TAKES A LOT TO COMPUTE\n#n = sum(1 for line in open(filename)) - 1 #excludes header\n#print(n)\nn = 37670294\n#skip = sorted(random.sample(range(1,n+1),n-samples)) #the 0-indexed header will not be included in the skip list\n\n#train_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], skiprows=skip)\n#train_data.info()\n\ntrain_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], nrows=samples)\ntrain_data.info()\n\nprint('---------------')\nprint('We use %.1f%% of a data' % (samples/n))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train_data.head(5)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import seaborn as sns\nimport matplotlib.pyplot as plt\nsns.countplot(y='hotel_continent', data=train_data)\nsns.plt.title('Continent destination')\nplt.show() #get rid of <matplotlib.text.Text at 0x7f ..."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "f, ax = plt.subplots(figsize=(15, 25))\nsns.countplot(y='hotel_country', data=train_data)\nplt.title('Search of hotels per country')\nplt.show()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "sns.pairplot(train_data[['hotel_country', 'user_location_country']], size=6)\nplt.show()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train_data.loc[:,['is_mobile', 'is_package', 'orig_destination_distance', 'srch_adults_cnt', \n                  'srch_children_cnt', 'srch_rm_cnt', 'is_booking', 'cnt']].describe()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#https://www.dataquest.io/blog/python-data-science/\ntrain_data.groupby('hotel_continent').count().sort_values(by='hotel_country')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "corr = train_data.corr()\n\n# Generate a mask for the upper triangle\nmask = np.zeros_like(corr, dtype=np.bool)\nmask[np.triu_indices_from(mask)] = True\n\n# Set up the matplotlib figure\nf, ax = plt.subplots(figsize=(12, 15))\n\n# Draw the heatmap with the mask and correct aspect ratio\nsns.heatmap(corr, mask=mask, vmax=.3, square=True, linewidths=.5, cbar_kws={\"shrink\": .5}, ax=ax)\nplt.show()"
 },
 {
  "cell_type": "markdown",
  "metadata": {},
  "source": "Machine Learning\n---"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#Divide data into training and test\nfrom sklearn.model_selection import train_test_split\ntrain, test, y_train, y_test = train_test_split(\n    train_data[['hotel_continent', 'user_location_country', 'hotel_country']], \n    train_data['hotel_cluster'], \n    test_size=0.33, \n    random_state=13)\n#display samle data with label\ntrain.head().join(y_train.head())"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#Fit a model\nfrom sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(n_estimators=100, min_samples_leaf=10)\nmodel.fit(train, y_train)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#mean squared error\nfrom sklearn.metrics import mean_squared_error\nimport math\n\npredictions = model.predict(test)\nmean_squared_error(predictions, y_test)"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}