{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "import numpy as np\nimport pandas as pd \nfrom subprocess import check_output\n#print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\nimport datetime\nimport time"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train1 = pd.read_csv('../input/train.csv', dtype={'is_booking':bool,'srch_destination_id':np.int32, 'hotel_cluster':np.int32, 'srch_children_cnt':np.int32,'srch_adults_cnt':np.int32,'srch_destination_type_id':np.int32, 'hotel_cluster':np.int32,'orig_destination_distance':np.float64},\n                    usecols=['date_time','srch_ci','srch_co','srch_destination_id','is_booking','srch_children_cnt','srch_adults_cnt','srch_destination_type_id','hotel_cluster','orig_destination_distance'], chunksize=1000000)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "train = [ ]\nfor train2 in train1:\n    train2[\"srch_ci\"] = pd.to_datetime(train2[\"srch_ci\"], format='%Y-%m-%d', errors=\"coerce\")\n    train2[\"srch_co\"] = pd.to_datetime(train2[\"srch_co\"], format='%Y-%m-%d', errors=\"coerce\")\n    train2[\"stay_span\"] = (train2[\"srch_co\"] - train2[\"srch_ci\"]).astype('timedelta64[D]')\n    #train2 = train2.drop('srch_co', axis=1)\n    train2[\"date_time\"] = pd.to_datetime(train2[\"date_time\"], format='%Y-%m-%d', errors=\"coerce\")\n    train2[\"search_span\"] = (train2[\"srch_ci\"] - train2[\"date_time\"]).astype('timedelta64[D]')\n    #train2 = train2.drop('srch_ci', axis=1)\n    train2['year'] = train2['date_time'].dt.year\n    train2['month'] = train2['date_time'].dt.month\n    train2['day_of_week'] = train2['date_time'].dt.dayofweek\n    train2['hour'] = train2['date_time'].dt.hour\n    #train2 = train2.drop('date_time', axis=1)\n    train2.ix[(train2['hour'] >= 10) & (train2['hour'] < 18), 'hour'] = 1\n    train2.ix[(train2['hour'] >= 18) & (train2['hour'] < 22), 'hour'] = 2\n    train2.ix[(train2['hour'] >= 22) & (train2['hour'] == 24), 'hour'] = 3\n    train2.ix[(train2['hour'] >= 1) & (train2['hour'] < 10), 'hour'] = 3\n    #train2['Individuals'] = train2['srch_adults_cnt']+train2['srch_children_cnt']\n    #train2 = train2.drop('srch_adults_cnt', axis=1)\n    #train2 = train2.drop('srch_children_cnt', axis=1)\n    #train2 = train2.drop('search_span', axis=1)\n    #train2 = train2.drop('user_location_city', axis=1)\n    #train2 = train2.drop('hotel_country', axis=1)\n    train2 = train2[['orig_destination_distance','srch_destination_id','srch_destination_type_id','is_booking','hotel_cluster','stay_span','search_span','month','day_of_week','hour', 'srch_adults_cnt', 'srch_children_cnt']]\n    agg = train2.groupby(['srch_destination_id','srch_destination_type_id','hotel_cluster','day_of_week','hour', 'srch_adults_cnt', 'srch_children_cnt'], sort = False)['is_booking'].agg(['sum','count'])\n    agg.reset_index(inplace=True)\n    train.append(agg)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "# test = pd.read_csv('../input/test.csv', dtype={'srch_destination_id':np.int32}, usecols=['srch_destination_id'])\ntrain = pd.concat(train, axis=0)\ntrain.head()\nCLICK_WEIGHT = 0.1"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "agg = train.groupby(['srch_destination_id','srch_destination_type_id','hotel_cluster', 'day_of_week','hour','srch_adults_cnt', 'srch_children_cnt']).sum().reset_index()\n#agg.head()\nagg['count'] -= agg['sum']\nagg = agg.rename(columns={'sum':'bookings','count':'clicks'})\nagg['relevance'] = agg['bookings'] + CLICK_WEIGHT * agg['clicks']\n#agg.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "def most_popular(group, n_max=5):\n    relevance = group['relevance'].values\n    hotel_cluster = group['hotel_cluster'].values\n    most_popular = hotel_cluster[np.argsort(relevance)[::-1]][:n_max]\n    return np.array_str(most_popular)[1:-1]"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "most_pop = agg.groupby(['srch_destination_id','srch_destination_type_id', 'day_of_week','hour', 'srch_adults_cnt', 'srch_children_cnt']).apply(most_popular)\nmost_pop = pd.DataFrame(most_pop).rename(columns={0:'hotel_cluster'})\n#most_pop.head()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1 = pd.read_csv('../input/test.csv')\ntest1[\"date_time\"] = pd.to_datetime(test1[\"date_time\"], format='%Y-%m-%d', errors=\"coerce\")\ntest1['day_of_week'] = test1['date_time'].dt.dayofweek\ntest1['hour'] = test1['date_time'].dt.hour\ntest1 = test1[['srch_destination_id','srch_destination_type_id', 'day_of_week','hour', 'srch_adults_cnt', 'srch_children_cnt']]\n#test1.info()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1 = test1.merge(most_pop, how='left',left_on=['srch_destination_id','srch_destination_type_id', 'day_of_week','hour', 'srch_adults_cnt', 'srch_children_cnt'],right_index=True)"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#test.head()\ntest1.hotel_cluster.isnull().sum()"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "most_pop_all = agg.groupby('hotel_cluster')['relevance'].sum().nlargest(5).index\nmost_pop_all = np.array_str(most_pop_all)[1:-1]\nmost_pop_all"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "test1.hotel_cluster.fillna(most_pop_all,inplace=True)\ntest1.hotel_cluster.to_csv('predicted_with_pandas.csv',header=True, index_label='id')"
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}