{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e01e5bc7-20f6-4303-bb68-e706e523f8ce"},"outputs":[],"source":"# Imports\n\n# pandas\nimport pandas as pd\nfrom pandas import Series,DataFrame\n\n# numpy, matplotlib, seaborn\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('whitegrid')\n%matplotlib inline\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nimport xgboost as xgb"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"720a108b-b593-441e-818f-521b3279f4a7"},"outputs":[],"source":"# get expedia & test csv files as a DataFrame\nexpedia_df = pd.read_csv('../input/train.csv', nrows=10000)\ntest_df    = pd.read_csv('../input/test.csv')\n\n# preview the data\nexpedia_df.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f2e6817a-736e-4e49-9486-c6ea37b31fd5"},"outputs":[],"source":"expedia_df.info()\nprint(\"----------------------------\")\ntest_df.info()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e8d76f8b-341e-439f-a00c-eea025328eb6"},"outputs":[],"source":"# drop unnecessary columns, these columns won't be useful in analysis and prediction\nexpedia_df = expedia_df.drop(['date_time','site_name', 'user_location_region', 'user_location_city', 'orig_destination_distance', \n                              'user_id', 'srch_co', 'srch_adults_cnt', 'srch_children_cnt', 'srch_rm_cnt', 'cnt'], axis=1)\ntest_df    = test_df.drop(['date_time','site_name', 'user_location_region', 'user_location_city', 'orig_destination_distance', \n                              'user_id', 'srch_co', 'srch_adults_cnt', 'srch_children_cnt', 'srch_rm_cnt'], axis=1)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d60bdb20-c0ad-4cf4-b616-b8cc2dfc7e04"},"outputs":[],"source":"# Plot \n\nfig, (axis1,axis2) = plt.subplots(2,1,figsize=(15,10))\n\nbookings_df = expedia_df[expedia_df[\"is_booking\"] == 1]\n\n# What are the most countries the customer travel from?\nsns.countplot('user_location_country',data=bookings_df.sort_values(by=['user_location_country']),ax=axis1,palette=\"Set3\")\n\n# What are the most countries the customer travel to?\nsns.countplot('hotel_country',data=bookings_df.sort_values(by=['hotel_country']),ax=axis2,palette=\"Set3\")\n\n# Combine both plots\n# fig, (axis1) = plt.subplots(1,1,figsize=(15,5))\n\n# sns.distplot(bookings_df[\"hotel_country\"], kde=False, rug=False, bins=25, ax=axis1)\n# sns.distplot(bookings_df[\"user_location_country\"], kde=False, rug=False, bins=25, ax=axis1)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"dc90d157-ee83-4877-a57a-4d063b97673a"},"outputs":[],"source":"# Where do most of the customers from a country travel?\nuser_country_id = 66\n\nfig, (axis1) = plt.subplots(1,1,figsize=(15,10))\n\ncountry_customers = expedia_df[expedia_df[\"user_location_country\"] == user_country_id]\ncountry_customers[\"hotel_country\"].value_counts().plot(kind='bar',colormap=\"Set3\",figsize=(15,5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"28a40ae4-822c-4898-b473-3177c1e14eae"},"outputs":[],"source":"# Plot frequency for each hotel_clusters\n\nexpedia_df[\"hotel_cluster\"].value_counts().plot(kind='bar',colormap=\"Set3\",figsize=(15,5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"ba4bea84-f087-42d9-8219-bb563070fa87"},"outputs":[],"source":"# What are the most frequent hotel clusters booked by customers from a country?\nuser_country_id = 66\n\nfig, (axis1) = plt.subplots(1,1,figsize=(15,10))\n\ncustomer_clusters = expedia_df[expedia_df[\"user_location_country\"] == user_country_id][\"hotel_cluster\"]\ncustomer_clusters.value_counts().plot(kind='bar',colormap=\"Set3\",figsize=(15,5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"615ad416-3a16-482a-9617-bb84497eff68"},"outputs":[],"source":"# What are the most frequent hotel clusters in a country?\ncountry_id = 50\n\nfig, (axis1) = plt.subplots(1,1,figsize=(15,10))\n\ncountry_clusters = expedia_df[expedia_df[\"hotel_country\"] == country_id][\"hotel_cluster\"]\ncountry_clusters.value_counts().plot(kind='bar',colormap=\"Set3\",figsize=(15,5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"92ae349b-56ed-4581-a703-46a48039bc0a"},"outputs":[],"source":"# Plot post_continent & hotel_continent\n\nfig, ((axis1,axis2),(axis3,axis4)) = plt.subplots(2,2,figsize=(15,10))\n\n# Plot frequency for each posa_continent\nsns.countplot('posa_continent', data=expedia_df,order=[0,1,2,3,4],palette=\"Set3\",ax=axis1)\n\n# Plot frequency for each posa_continent decomposed by hotel_continent\nsns.countplot('posa_continent', hue='hotel_continent',data=expedia_df,order=[0,1,2,3,4],palette=\"Set3\",ax=axis2)\n\n# Plot frequency for each hotel_continent\nsns.countplot('hotel_continent', data=expedia_df,order=[0,2,3,4,5,6],palette=\"Set3\",ax=axis3)\n\n# Plot frequency for each hotel_continent decomposed by posa_continent\nsns.countplot('hotel_continent', hue='posa_continent', data=expedia_df, order=[0,2,3,4,5,6],palette=\"Set3\",ax=axis4)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f46eeb80-e1b2-47ed-825b-ce832b3e8df8"},"outputs":[],"source":"# Plot frequency of is_mobile & is_package\n\nfig, (axis1,axis2) = plt.subplots(1,2,figsize=(15,3))\n\n# What's the frequency of bookings through mobile?\nsns.countplot(x='is_mobile',data=bookings_df, order=[0,1], palette=\"Set3\", ax=axis1)\n\n# What's the frequency of bookings with package?\nsns.countplot(x='is_package',data=bookings_df, order=[0,1], palette=\"Set3\", ax=axis2)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"95801995-c418-4637-b846-2df63d0bd8e5"},"outputs":[],"source":"# What's the most impactful channel?\n\nfig, (axis1) = plt.subplots(1,1,figsize=(15,3))\n\nsns.countplot(x='channel', order=list(range(0,10)), data=expedia_df, palette=\"Set3\")"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0fde1b19-2d05-4181-9f11-1cd3f57d94b3"},"outputs":[],"source":"# Convert srch_ci to Year, Month, and Week\n\nexpedia_df['Year']   = expedia_df['srch_ci'].apply(lambda x: int(str(x)[:4]) if x == x else np.nan)\nexpedia_df['Month']  = expedia_df['srch_ci'].apply(lambda x: int(str(x)[5:7]) if x == x else np.nan)\nexpedia_df['Week']   = expedia_df['srch_ci'].apply(lambda x: int(str(x)[8:10]) if x == x else np.nan)\n\nfig, (axis1,axis2,axis3) = plt.subplots(1,3,sharex=True,figsize=(15,5))\n\n# Plot How many bookings in each month\nsns.countplot('Month',data=expedia_df[expedia_df[\"is_booking\"] == 1],order=list(range(1,13)),palette=\"Set3\",ax=axis1)\n\n# Plot The percentage of bookings of each month(sum of month bookings / count of bookings(=1 OR =0) of a month)\n# sns.factorplot('Month',\"is_booking\",data=expedia_df, order=list(range(1,13)), palette=\"Set3\",ax=axis2)\nsns.barplot('Month',\"is_booking\",data=expedia_df, order=list(range(1,13)), palette=\"Set3\",ax=axis2)\n\n# Plot The percentage of bookings of each month compared to all bookings(sum of month bookings / count of bookings(=1) of all months)\nmonth_sum = expedia_df[['Month', 'is_booking']].groupby(['Month'],as_index=False).sum()\nmonth_sum['is_booking'] = month_sum['is_booking'] / len(expedia_df[expedia_df['is_booking'] == 1])\n\nsns.barplot(x='Month', y='is_booking', order=list(range(1,13)), data=month_sum,ax=axis3) "},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"89e1e6d5-24e7-488c-8a2b-3eb1bed55ad9"},"outputs":[],"source":"# Convert srch_ci column to Date(Y-M)\nexpedia_df['Date']  = expedia_df['srch_ci'].apply(lambda x: (str(x)[:7]) if x == x else np.nan)\n\n# Plot number of bookings over Date\ndate_bookings  = expedia_df.groupby('Date')[\"is_booking\"].sum()\nax1 = date_bookings.plot(legend=True,marker='o',title=\"Total Bookings\", figsize=(15,5)) \nax1.set_xticks(range(len(date_bookings)))\nxlabels = ax1.set_xticklabels(date_bookings.index.tolist(), rotation=90)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a6fde302-5347-4245-ad90-86e957c31be9"},"outputs":[],"source":"# .... continue with plot Date column\n\n# Plot important values(min,max,quartiles) for number of bookings over Date\nfig, (axis1) = plt.subplots(1,1,figsize=(15,3))\n\nax2 = sns.boxplot([date_bookings.values], whis=np.inf,ax=axis1)\nax2.set_title('Important Values')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"4863e563-34c6-4da0-92db-038abbb58c5e"},"outputs":[],"source":"# Correlation between hotel_country in number of bookings through 2013, 2014, & 2015\n\nhotel_country_piv       = pd.pivot_table(expedia_df,values='is_booking', index='Date', columns=['hotel_country'],aggfunc='sum')\nhotel_country_piv       = hotel_country_piv.fillna(0)\nhotel_country_piv.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a38e89f6-866f-4f49-80e8-4780ed98e70f"},"outputs":[],"source":"# .... continue Correlation\n\n# Plot correlation between range of hotel_country\ncountry_ids = [1,5,7,8,47,50,182,185]\n\nfig, (axis1) = plt.subplots(1,1,figsize=(15,5))\n\n# using summation of booking values for each hotel_country \nsns.heatmap(hotel_country_piv[country_ids].corr(),annot=True,linewidths=2,cmap=\"YlGnBu\")"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"74f35b38-f60f-4db2-aa10-79e0fa6d0465"},"outputs":[],"source":"# .... continue Correlation\n\n# Reformat the heatmap so similar hotel_country are next to each other\nsns.clustermap(hotel_country_piv[country_ids].corr(), cmap=\"YlGnBu\")"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8d9f95b8-3d5b-4ca5-8dfd-68026f34e77d"},"outputs":[],"source":"# Define training and testing sets\n\ntrain_df = pd.read_csv('../input/train.csv', usecols=['is_booking', 'srch_destination_id', 'hotel_cluster'])\ntest_df  = test_df[['id', 'srch_destination_id']]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"95fe05f0-b176-4622-8658-1e7de2a17381"},"outputs":[],"source":"# Group by srch_destination_id & hotel_cluster\n# Then for each destination id & hotel cluster, compute summation of bookings, and count number of clicks(no-booking)\n\ntrain_df = train_df.groupby(['srch_destination_id','hotel_cluster'])['is_booking'].agg(['sum','count'])\ntrain_df['count'] = train_df['count'] - train_df['sum']\ntrain_df.rename(columns={'sum': 'sum_bookings', 'count': 'clicks'}, inplace=True)\n\n# For each destination id & hotel cluster, \n# the relevance will be the number of bookings made + number of clicks(no-bookings) * 0.1\n# meaning for every 10 clicks, they will be counted as 1 booking\n\ntrain_df['relevance'] = train_df['sum_bookings'] + (train_df['clicks'] * 0.1)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"aee9759f-cc5d-4642-b7b9-8223d0a4b992"},"outputs":[],"source":"# For each srch_destination_id group, get top 5 hotel clusters with max relevance\n\ndef get_top_clusters(group):\n    indexes      = group.relevance.nlargest(5).index\n    top_clusters = group.hotel_cluster[indexes].values\n    if(len(top_clusters) < 5):\n        top_clusters = (list(top_clusters) + list(ferq_clusters.index))[:5]\n    return np.array_str(np.array(top_clusters))[1:-1]\n\ntrain_df      = train_df.reset_index()\nferq_clusters = train_df['hotel_cluster'].value_counts()[:5]\ntop_clusters  = train_df.groupby(['srch_destination_id']).apply(get_top_clusters)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d8263368-14b9-4c86-9edf-c033e92eb617"},"outputs":[],"source":"# Create top_clusters_df\n\ntop_clusters_df = pd.DataFrame(top_clusters).rename(columns={0: 'hotel_cluster'})\ntop_clusters_df.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"fabfc60c-df3e-42a8-b277-057e919a086b"},"outputs":[],"source":"# Merge test_df with top_clusters_df\n\n# For every destination id in test_df, merge it with the corresponding id in top_clusters_df \ntest_df = pd.merge(test_df, top_clusters_df, how='left',left_on='srch_destination_id', right_index=True)\n\n# Fill NaN values with most frequent clusters\ntest_df.hotel_cluster.fillna(np.array_str(ferq_clusters.index)[1:-1],inplace=True)\n\nY_pred = test_df[\"hotel_cluster\"]"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b2f4ccdf-8c04-4cc4-8a46-6524781679bf"},"outputs":[],"source":"# Create submission\n\nsubmission = pd.DataFrame()\nsubmission[\"id\"]            = test_df[\"id\"]\nsubmission[\"hotel_cluster\"] = Y_pred\n\nsubmission.to_csv('expedia.csv', index=False)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8e2b0e5d-68c7-499c-9745-b8f34fc1bf87"},"outputs":[],"source":"# IMPORTANT! - Another Method for Hotel Cluster Prediction\n# Expedia Hotel Cluster Predictions\n# Link: https://www.kaggle.com/omarelgabry/expedia-hotel-recommendations/expedia-hotel-cluster-predictions"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}