{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# Get first 10000 rows and print some info about columns\ntrain = pd.read_csv(\"../input/train.csv\", parse_dates=['srch_ci', 'srch_co'], nrows=20000)\ntrain.info()\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nsns.set_style(\"whitegrid\")"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# distribution of the total number of people per cluster\nsrc_total_cnt = train.srch_adults_cnt + train.srch_children_cnt\ntrain['src_total_cnt'] = src_total_cnt\nax = sns.kdeplot(train['hotel_cluster'], train['src_total_cnt'], cmap=\"Purples_d\")\nlim = ax.set(ylim=(0.5, 4.5))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# plot all columns countplots\nimport numpy as np\nrows = train.columns.size//2 - 1\nfig, axes = plt.subplots(nrows=rows, ncols=2, figsize=(12,36))\nfig.tight_layout()\ni = 0\nj = 0\nfor col in train.columns:\n    if j >= 2:\n        j = 0\n        i += 1\n    # avoid to plot by date    \n    if train[col].dtype == np.int64:\n        sns.countplot(x=col, data=train, ax=axes[i][j])\n        j += 1"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# putting the two above together\nwith sns.color_palette(\"ocean\"):\n    sns.countplot(x='is_booking', hue='is_mobile', data=train)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# putting the two above together\nwith sns.color_palette(\"Set1\"):\n    sns.countplot(x='cnt', hue='is_booking', data=train)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import numpy as np\n# get number of booked nights as difference between check in and check out\nis_same = (train['hotel_country'] - train['user_location_country'])*10000000000000000\n#is_same = (is_same / np.timedelta64(1, 'D')).astype(float) # convert to float to avoid NA problems\ntrain['is_same'] = is_same\nwith sns.color_palette(\"husl\"):\n    plt.figure(figsize=(11, 9))\n    ax = sns.violinplot(x='hotel_continent', y='is_same', data=train)\n#lim = ax.set(ylim=(0, 15))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#import numpy as np\n# get number of booked nights as difference between check in and check out\nplt.figure(figsize=(19, 9))\nax = sns.violinplot(x='hotel_cluster', y='is_same', data=train)\nlim = ax.set(xlim=(-1, 50))"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}