{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\n#print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n%matplotlib inline\nimport numpy as np\nimport pandas as pd \n#print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\nimport datetime\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train = pd.read_csv(\"../input/train.csv\", parse_dates=['date_time'], nrows=10000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1 = pd.read_csv(\"../input/train.csv\", parse_dates=['date_time'], nrows=1000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1['total'] = train1['srch_adults_cnt']+train1['srch_children_cnt']"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import seaborn as sns\nimport matplotlib.pyplot as plt"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "sns.distplot(train['user_location_country'], label=\"User country\")\nsns.distplot(train['hotel_country'], label=\"Hotel country\")\nplt.legend()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1[\"srch_ci\"] = pd.to_datetime(train1[\"srch_ci\"], format='%Y-%m-%d', errors=\"coerce\")\ntrain1[\"srch_co\"] = pd.to_datetime(train1[\"srch_co\"], format='%Y-%m-%d', errors=\"coerce\")\ntrain1[\"stay_span\"] = (train1[\"srch_co\"] - train1[\"srch_ci\"]).astype('timedelta64[D]')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "xmin = train1['total'].min()\nxmax = train1['total'].max()\nymin = train1['hotel_country'].min()\nymax = train1['hotel_country'].max()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1.plot.hexbin(x='total', y='hotel_country', C='stay_span', gridsize=30,xscale = 'linear', yscale = 'linear')"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}