{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d8a46760-f063-60e5-8068-c5c3445235f8"
      },
      "outputs": [],
      "source": [
        "import datetime\n",
        "import pandas as pd\n",
        "import numpy as np\n",
        "import matplotlib.pyplot as plt\n",
        "%matplotlib inline\n",
        "\n",
        "from sklearn.ensemble import RandomForestClassifier\n",
        "from sklearn.cross_validation import train_test_split\n",
        "import ml_metrics as metrics"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "87b9a102-fafb-1049-2b8c-c21049933732"
      },
      "outputs": [],
      "source": [
        "dtype={'is_booking':bool,\n",
        "        'srch_ci' : np.str_,\n",
        "        'srch_co' : np.str_,\n",
        "        'srch_adults_cnt' : np.int32,\n",
        "        'srch_children_cnt' : np.int32,\n",
        "        'srch_rm_cnt' : np.int32,\n",
        "        'srch_destination_id':np.str_,\n",
        "        'user_location_country' : np.str_,\n",
        "        'user_location_region' : np.str_,\n",
        "        'user_location_city' : np.str_,\n",
        "        'hotel_cluster' : np.str_,\n",
        "        'orig_destination_distance':np.float64,\n",
        "        'date_time':np.str_,\n",
        "        'hotel_market':np.str_}\n",
        "# feature selection\n",
        "# downsample the data: 60% of the 2014 booking data\n",
        "# originally have 30million training data, 3million test data, but only ~20 features,\n",
        "# so we can down sample the data\n",
        "\n",
        "#Specifying dtypes helps reduce memory requirements for reading in csv file later."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "acaf8378-0525-2995-f1b5-67b630dd5680"
      },
      "outputs": [],
      "source": [
        "df0 = pd.read_csv('../input/train.csv',dtype=dtype, usecols=dtype, parse_dates=['date_time'] ,sep=',',nrows=2000000)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1b7f560c-2540-5740-fe21-c1ba5e23a168"
      },
      "outputs": [],
      "source": [
        "df0.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "978a84ee-f3f4-4b10-01fe-89de858096fe"
      },
      "outputs": [],
      "source": [
        "\n",
        "# take data from 2014 as sampling 50%\n",
        "df0['year']=df0['date_time'].dt.year\n",
        "train = df0.query('is_booking==True & year==2014').sample(frac=0.6)\n",
        "train.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e971ac47-4738-495e-d344-f09747311966"
      },
      "outputs": [],
      "source": [
        "\n",
        "train.tail()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a4117555-4723-6f8e-3123-81ffa6e4fe3d"
      },
      "outputs": [],
      "source": [
        "train.isnull().sum(axis=0)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8eaf35b8-913d-efea-3b43-82bb034f6393"
      },
      "outputs": [],
      "source": [
        "#datetime features\n",
        "train['srch_ci']=pd.to_datetime(train['srch_ci'],infer_datetime_format = True,errors='coerce')\n",
        "train['srch_co']=pd.to_datetime(train['srch_co'],infer_datetime_format = True,errors='coerce')\n",
        "\n",
        "train['month']= train['date_time'].dt.month\n",
        "train['plan_time'] = ((train['srch_ci']-train['date_time'])/np.timedelta64(1,'D')).astype(float)\n",
        "train['hotel_nights']=((train['srch_co']-train['srch_ci'])/np.timedelta64(1,'D')).astype(float)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cfe374c4-6971-7271-32fc-08502540c1c3"
      },
      "outputs": [],
      "source": [
        "train.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cbfa1df9-920d-f506-f9a7-a27f871e7b54"
      },
      "outputs": [],
      "source": [
        "#fill Missing Values\n",
        "\n",
        "#fill orig_destination_distance with mean of the whole or mean of the same orig_destination pair\n",
        "m=train.orig_destination_distance.mean()\n",
        "train['orig_destination_distance']=train.orig_destination_distance.fillna(m)\n",
        "\n",
        "#fill missing dates with -1\n",
        "train.fillna(-1,inplace=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "89836f16-cfbf-50c0-caee-a373e0265708"
      },
      "outputs": [],
      "source": [
        "# Since we extract the plan_time from srch_ci and date_time, we drop date_time and srch_ci\n",
        "# we extract how many nights of stay, so we drop srch_co\n",
        "lst_drop=['date_time','srch_ci','srch_co']\n",
        "train.drop(lst_drop,axis=1,inplace=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1e358629-e522-7dad-5f05-2b9670fdc553"
      },
      "outputs": [],
      "source": [
        "train.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "33b770cf-5195-cf37-cdbf-38ff3477e2f4"
      },
      "outputs": [],
      "source": [
        "y=train['hotel_cluster']\n",
        "X=train.drop(['hotel_cluster','is_booking','year'],axis=1) # in training dataset, have clicking and booking event"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0101e086-f223-5795-db20-6444a56324ea"
      },
      "outputs": [],
      "source": [
        "y.shape,X.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "af62650b-13b4-034b-d7f2-6d44dc225334"
      },
      "outputs": [],
      "source": [
        "y.nunique()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "459a6f1c-d671-21df-2cbc-991098d41370"
      },
      "outputs": [],
      "source": [
        "X_train, X_test, y_train, y_test = train_test_split(X,y, test_size=0.33)\n",
        "rf_tree = RandomForestClassifier(n_estimators=31,max_depth=10,random_state=123)\n",
        "rf_tree.fit(X_train,y_train)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cd0195bc-03d4-35f5-2d1d-3812f58ab1d9"
      },
      "outputs": [],
      "source": [
        "importance = rf_tree.feature_importances_\n",
        "indices=np.argsort(importance)[::-1][:10]\n",
        "importance[indices]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ccef79eb-22e9-62f9-6fd4-b2b9fcd4e4a7"
      },
      "outputs": [],
      "source": [
        "plt.barh(range(10), importance[indices],color='r')\n",
        "plt.yticks(range(10),X_train.columns[indices])\n",
        "plt.xlabel('Feature Importance')\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4cf522cf-070d-3c3a-e60f-892be6148b1d"
      },
      "outputs": [],
      "source": [
        "rf_tree.classes_"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9374136c-756a-69ca-e43d-332edd3db17f"
      },
      "outputs": [],
      "source": [
        "dict_cluster = {}\n",
        "for (k,v) in enumerate(rf_tree.classes_):\n",
        "    dict_cluster[k] = v"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3caab8e1-bf23-81f3-f87b-2841d0201aa1"
      },
      "outputs": [],
      "source": [
        "y_pred=rf_tree.predict_proba(X_test)\n",
        "#take largest 5 probablities' indexes\n",
        "a=y_pred.argsort(axis=1)[:,-5:]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a01ed0f1-098d-1bb9-54c4-73ba6d3a6979"
      },
      "outputs": [],
      "source": [
        "y_pred"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d728b9f6-58c9-c2f0-af26-e4c6fc3257ef"
      },
      "outputs": [],
      "source": [
        "a"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b42c8270-9c3a-0bfb-39ca-2a45b18a33bf"
      },
      "outputs": [],
      "source": [
        "\n",
        "#take the corresonding cluster of the 5 top indices\n",
        "b = []\n",
        "for i in a.flatten():\n",
        "    b.append(dict_cluster.get(i))\n",
        "cluster_pred = np.array(b).reshape(a.shape)\n",
        "cluster_pred"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f9b323e0-e707-7b0e-978a-a0582b9ac97b"
      },
      "outputs": [],
      "source": [
        "print(\"score:\",metrics.mapk(y_test,cluster_pred,k=5))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "eebfbe9f-c165-5713-8b45-2550446504ab"
      },
      "outputs": [],
      "source": [
        "metrics.mapk?"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4c9226e5-f023-d100-ab79-2e785737e278"
      },
      "outputs": [],
      "source": [
        "y_test.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4d635c1a-3c68-60f6-9711-0efe4d017138"
      },
      "outputs": [],
      "source": [
        "#import and process test data\n",
        "dtype1={'srch_ci' : np.str_,\n",
        "        'srch_co' : np.str_,\n",
        "        'srch_adults_cnt' : np.int32,\n",
        "        'srch_children_cnt' : np.int32,\n",
        "        'srch_rm_cnt' : np.int32,\n",
        "        'srch_destination_id':np.str_,\n",
        "        'user_location_country' : np.str_,\n",
        "        'user_location_region' : np.str_,\n",
        "        'user_location_city' : np.str_,\n",
        "        'orig_destination_distance':np.float64,\n",
        "        'date_time':np.str_,\n",
        "        'hotel_market':np.str_}"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f423a62d-005b-b78a-16f4-98aa9ae4e2d6"
      },
      "outputs": [],
      "source": [
        "# feature engineering on test data\n",
        "test = pd.read_csv('../input/test.csv',dtype=dtype1,usecols=dtype1,parse_dates=['date_time'] ,sep=',')\n",
        "test['srch_ci']=pd.to_datetime(test['srch_ci'],infer_datetime_format = True,errors='coerce')\n",
        "test['srch_co']=pd.to_datetime(test['srch_co'],infer_datetime_format = True,errors='coerce')\n",
        "\n",
        "test['month']=test['date_time'].dt.month\n",
        "test['plan_time'] = ((test['srch_ci']-test['date_time'])/np.timedelta64(1,'D')).astype(float)\n",
        "test['hotel_nights']=((test['srch_co']-test['srch_ci'])/np.timedelta64(1,'D')).astype(float)\n",
        "\n",
        "n=test.orig_destination_distance.mean()\n",
        "test['orig_destination_distance']=test.orig_destination_distance.fillna(m)\n",
        "test.fillna(-1,inplace=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "42b86dfa-d423-eb31-3f13-1eebbce21624"
      },
      "outputs": [],
      "source": [
        "test1=test.sample(frac=0.1) # random sampled 5% of the test data"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "53cdda79-47a9-d662-9d9e-3502d0ce4af8"
      },
      "outputs": [],
      "source": [
        "test1.shape, train.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3f84fca8-455a-1d1c-008d-200ed5f018a0"
      },
      "outputs": [],
      "source": [
        "lst_drop=['date_time','srch_ci','srch_co']\n",
        "test1.drop(lst_drop,axis=1, inplace=True)\n",
        "target=train['hotel_cluster']\n",
        "train1=train.drop(['hotel_cluster','is_booking','year'],axis=1)\n",
        "train1.shape, test1.shape"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7d47d663-942e-fff1-2949-461f04eba51d"
      },
      "outputs": [],
      "source": [
        "#on All training sample\n",
        "rf_all = RandomForestClassifier(n_estimators=31,max_depth=10,random_state=123)\n",
        "rf_all.fit(train1,target)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f0510b1b-9f44-f569-72f6-1b8db55f96be"
      },
      "outputs": [],
      "source": [
        "importance = rf_all.feature_importances_\n",
        "indices=np.argsort(importance)[::-1][:10]\n",
        "importance[indices]\n",
        "\n",
        "plt.barh(range(10), importance[indices],color='r')\n",
        "plt.yticks(range(10),train1.columns[indices])\n",
        "plt.xlabel('Feature Importance')\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "18f8f337-f447-54d7-00f3-0cde4dde415b"
      },
      "outputs": [],
      "source": [
        "y_pred=rf_all.predict_proba(test1) # predict on test dataset\n",
        "y_pred"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "25c80be7-87ec-5842-5e0c-c24436a10b18"
      },
      "outputs": [],
      "source": [
        "#take largest 5 probablities' indexes\n",
        "a=y_pred.argsort(axis=1)[:,-5:]"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ca65374c-ddbc-5b82-3865-4ba25815bd05"
      },
      "outputs": [],
      "source": [
        "a"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c17e3d51-b05f-b1c8-bf06-ad33373cb62f"
      },
      "outputs": [],
      "source": [
        "dict_cluster = {}\n",
        "for (k,v) in enumerate(rf_tree.classes_):\n",
        "    dict_cluster[k] = v\n",
        "b = []\n",
        "for i in a.flatten():\n",
        "    b.append(dict_cluster.get(i))\n",
        "predict_class=np.array(b).reshape(a.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f163f938-1e2c-eec7-7cd1-788ec5443c02"
      },
      "outputs": [],
      "source": [
        "predict_class"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f3642c74-4c18-e808-39b2-0e87abc0dd3a"
      },
      "outputs": [],
      "source": [
        "predict_class=map(lambda x: ' '.join(map(str,x)), predict_class)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a75a140f-bffa-092e-6594-b63e8812253f"
      },
      "outputs": [],
      "source": [
        "submission = pd.DataFrame()\n",
        "submission['hotel_cluster'] = predict_class\n",
        "submission.to_csv('rf01expedia.csv', index=False)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9bdab1c8-1aeb-96c7-dd8e-346612643d55"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d3b0944a-880c-285b-648c-6bdd41054a8e"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0625df38-512b-4e80-93c1-e46a7d2c12d2"
      },
      "outputs": [],
      "source": [
        "# IMPORTANT! - Another Method for Hotel Cluster Prediction\n",
        "# Expedia Hotel Cluster Predictions\n",
        "# Link: https://www.kaggle.com/omarelgabry/expedia-hotel-recommendations/expedia-hotel-cluster-predictions"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}