{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom matplotlib import pyplot as plt\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "\ndest = pd.read_csv('../input/destinations.csv',\n                    #dtype={'is_booking':bool,'srch_destination_id':np.int32, 'hotel_cluster':np.int32},\n                    #usecols=['srch_destination_id','is_booking','hotel_cluster'],\n                    #chunksize=1000000\n                   )"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "dest"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "\ntrain = pd.read_csv('../input/train.csv',\n                    #dtype={'is_booking':bool,'srch_destination_id':np.int32, 'hotel_cluster':np.int32},\n                    #usecols=['srch_destination_id','is_booking','hotel_cluster'],\n                    chunksize=1000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "aggs = []\nprint('-'*38)\nfor chunk in train:\n    agg = chunk.groupby(['srch_adults_cnt'])['is_booking'].agg(['sum','count'])\n    agg.reset_index(inplace=True)\n    aggs.append(agg)\n    print('.',end='')\nprint('')\naggs = pd.concat(aggs, axis=0)\naggs.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train = pd.read_csv('../input/train.csv'\n                    ,nrows=100000\n                    #,chunksize=100000\n                   )"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "from matplotlib import pyplot as plt"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "plt.scatter(aggs['sum'][1:100],aggs['count'][1:100])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "fit = np.polyfit(aggs['sum'][1:100],aggs['count'][1:100],1)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "fit"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "plt.plot(x,y, 'yo', x, fit_fn(x), '--k')\nplt.xlim(0, 5)\nplt.ylim(0, 12)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test = pd.read_csv('../input/test.csv',\n                    nrows = 100000\n                    #dtype={'is_booking':bool,'srch_destination_id':np.int32, 'hotel_cluster':np.int32},\n                    #usecols=['srch_destination_id','is_booking','hotel_cluster'],\n                    #chunksize=1000000)\n                  )"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test.describe()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train.describe()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train.groupby('srch_adults_cnt')"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "plt.scatter(aggs['srch_adults_cnt'],aggs['sum']/aggs['count'])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "plt.scatter(aggs['srch_adults_cnt'],aggs['sum']/aggs['count'])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\noutput_notebook()\n\nfrom bokeh.plotting import figure, output_notebook, show, vplot, ColumnDataSource\nfrom bokeh.charts import TimeSeries\nfrom bokeh.models import HoverTool, CrosshairTool\nfrom bokeh.palettes import brewer\nimport gc\nimport dask.dataframe as dd"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train  =  pd.read_csv('../input/train.csv', usecols = ('date_time', 'srch_destination_type_id','is_booking'), \n                      parse_dates = ['date_time'])"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train['dow'] = train.date_time.dt.weekday\ntrain['year'] = train.date_time.dt.year\ntrain['month'] = train.date_time.dt.month\ntrain['day'] = train.date_time.dt.day"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train_agg = train.groupby(['dow','year','month','day', 'srch_destination_type_id']).agg(['sum', 'count'] )\ntrain_agg.columns = ('bookings', 'total')\ntrain_agg.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "del(train)\n\ngc.collect()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "pv_agg = train_agg.reset_index()\npv_agg['dt'] = pd.to_datetime( pv_agg.year*10000 + pv_agg.month*100 + pv_agg.day\n                                  , format='%Y%m%d')\npv_agg = pv_agg.pivot(index = 'dt', columns = 'srch_destination_type_id', values = 'bookings')\npv_agg.columns = [str(i) for i in pv_agg.columns]\npv_agg['dt'] = pv_agg.index\npv_agg['dow'] = pv_agg.dt.dt.weekday\npv_agg.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def make_hc_plot(df, start, stop):\n#    hover = HoverTool(\n#        tooltips=[\n#            (\"Date\", \"@day\"),\n#            (\"Day of week\", \"@dow\"),\n#            (\"srch_destination_type_id\", \"@srch_destination_type_id\"),\n#            (\"bookings\", \"@bookings\"),\n#        ]\n#    )\n    \n    #colors = brewer['RdYlBu'][stop-start]\n    colors = ['red', 'darkmagenta', 'green', 'darkorange', 'blue']\n#    ch = CrosshairTool(dimensions = ['height'], line_color='red')\n    p = figure(x_axis_type = 'datetime',plot_width=800, plot_height=400, tools=[hover, ch, 'pan,wheel_zoom,save,box_zoom,reset,resize'])\n    p.title = 'Expedia bookings for subset of srch_destination_type_id {} to {}'.format(start, stop)\n    p.xaxis.axis_label = 'Date'\n    p.yaxis.axis_label = 'Number of bookings'\n    \n    for i in range(start, stop):\n        src  = ColumnDataSource({'day': df.dt.dt.strftime('%Y-%m-%d'), \n                                 'dow': df.dow.tolist(),\n                                 'bookings': df[str(i)],\n                                 'srch_destination_type_id': [i]*df.shape[0]})\n        \n        p.line((df['dt']), df[str(i)], color=colors[i-start], legend = str(i), source = src)\n\n    return p"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "\ntslines = []\n\nfor i in range(0,9,3):\n    tsline = make_hc_plot(pv_agg, i, i+3)\n    tslines.append(tsline)\n    \nshow(vplot(*tslines))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}