{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\n#print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\nfrom collections import defaultdict\nfrom datetime import datetime"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "usecols=['srch_destination_id','is_booking','hotel_cluster','srch_adults_cnt','srch_children_cnt','srch_rm_cnt','srch_destination_type_id']"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "li_cols = ['srch_destination_id','srch_adults_cnt','srch_children_cnt','srch_rm_cnt','srch_destination_type_id']"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train = pd.read_csv('../input/train.csv', parse_dates=['date_time'], nrows=100000)\ntest = pd.read_csv('../input/test.csv', parse_dates=['date_time'], nrows=1000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotel_cluster_count = train.groupby(['srch_destination_id','is_booking','hotel_cluster','srch_adults_cnt','srch_children_cnt','srch_rm_cnt','srch_destination_type_id']).is_booking.agg(['sum','count']).reset_index()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotel_cluster_count_test = test[li_cols]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotel_cluster_count_test.describe()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotel_cluster_count['is_booking'] = 0.8456 * hotel_cluster_count['sum'] + (1 - 0.8456) * hotel_cluster_count['count']"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "hotel_cluster_count.head(30)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def popular_hotels(gp):\n    p = gp.values\n    # order hotel_clusters by score then reverse it and take the 5 first\n    clusters = p[:, 0][p[:,1].argsort()[::-1]][:5].astype(np.int8)\n    return np.array_str(clusters)[1:-1]# remove square brackets"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "dest_top_five = hotel_cluster_count.groupby(['srch_destination_id', 'srch_adults_cnt', 'srch_children_cnt','srch_rm_cnt','srch_destination_type_id'])['hotel_cluster', 'is_booking'].apply(popular_hotels).reset_index()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "dest_top_five = pd.DataFrame(dest_top_five).rename(columns={0:'hotel_cluster'})"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "dest_top_five.head(10)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "merge2 = hotel_cluster_count_test.merge(dest_top_five, how = 'left', on = ('srch_destination_id','srch_adults_cnt','srch_children_cnt','srch_rm_cnt','srch_destination_type_id') , suffixes=('_train_top', '_test_merge'))"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "merge2.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "most_pop_all = hotel_cluster_count.groupby('hotel_cluster')['is_booking'].sum().nlargest(5).index\nmost_pop_all = np.array_str(most_pop_all)[1:-1]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "merge2['hotel_cluster'].fillna(most_pop_all,inplace=True)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "merge2[['srch_destination_id','hotel_cluster']].head()"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}