{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2466b9c5ce772f58f15dfc6da013453812f47117"},"cell_type":"code","source":"import matplotlib as mpl\n\nmpl.rcParams['figure.figsize'] = [15, 7]\nmpl.rcParams['figure.dpi'] = 80\nmpl.rcParams['savefig.dpi'] = 100\n\nmpl.rcParams['font.size'] = 14\nmpl.rcParams['legend.fontsize'] = 'large'\nmpl.rcParams['figure.titlesize'] = 'medium'","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"!head -n 5 ../input/train.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d4a404a41654806a3049e53e3ffba95124c5394a"},"cell_type":"code","source":"!wc -l ../input/train.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85f81a30ea5adcd4311bb1eda134e579c3cbf97f"},"cell_type":"code","source":"data = pd.read_csv(\"../input/train.csv\", nrows=100000)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de71260d360730ff42407349dd145900caa394ae"},"cell_type":"markdown","source":"# Goals:\n- Hotel_clusters are dependent on which categorical variable?\n    - Graph the top 4\n- Hotel_clusters are dependent on which numerical variable?\n- How correlated are the user's country to the hotel's country? Continent?\n\n# Models to check out\n- The closest 5 hotels in the latent space clicked by each person should be recommended at the top 5 (Content-based filtering)\n- Create a pairwise ranking matrix factorization model of user to hotel cluster (Collaborative filtering)\n- Factorization machine of the dependent categoricals and numericals (Hybrid)"},{"metadata":{"trusted":true,"_uuid":"7d0372db34c549842df9207fddac25ce9475e2c1"},"cell_type":"code","source":"CATEGORICALS = [\"site_name\", \"posa_continent\", \"user_location_country\", \"user_location_region\", \"user_location_city\", \"is_mobile\", \"is_package\", \"channel\", \n               \"srch_destination_type_id\", \"hotel_continent\", \"hotel_country\", \"hotel_market\", \"srch_destination_id\"]\nNUMERICALS = [\"orig_destination_distance\", \"srch_adults_cnt\", \"srch_children_cnt\", \"srch_rm_cnt\", \"cnt\"]\nUSER_ID = \"user_id\"\nIS_BOOKING = \"is_booking\"\nHOTEL_CLUSTER = \"hotel_cluster\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"415d3669347e19198c9034df7ad65009ad6b4b74"},"cell_type":"code","source":"ax = data[HOTEL_CLUSTER].value_counts().plot.bar(color='dodgerblue')\nax.set_xticklabels([])\nplt.title(\"Hotel clusters' clicks - Looks like the beginnings of a power-law distribution\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3bd24d530723dd094b1acd8e6153bdec56412ba"},"cell_type":"code","source":"data_cat_dummies = pd.get_dummies(pd.get_dummies(data[CATEGORICALS].astype('category')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b98bde04ca7acd377ee6ea89d709ec6cfa4b8bb"},"cell_type":"code","source":"def name_scores(featurecoef, col_names, label=\"Score\", sort=False):\n    df_feature_importance = pd.DataFrame([dict(zip(col_names, featurecoef))]).T.reset_index()\n    df_feature_importance.columns = [\"Feature\", label]\n    if sort:\n        return df_feature_importance.sort_values(ascending=False, by=label)\n    return df_feature_importance\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c32c60eccf59295ef7477b7719b4db4cb33f13bf"},"cell_type":"code","source":"from sklearn.feature_selection import chi2\n\nsample_n = 10000\ndata_cat_dummies_sample = data_cat_dummies.sample(sample_n)\nchi2_scores = chi2(data_cat_dummies_sample, data[HOTEL_CLUSTER].loc[data_cat_dummies_sample.index])\ndf_chi2_scores = name_scores(chi2_scores[0], data_cat_dummies_sample.columns)\ndf_chi2_scores = df_chi2_scores.sort_values(by=\"Score\", ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70e3d9da46ea3473403a7667e18643b8fbf28e9c"},"cell_type":"code","source":"# get the top 100 features and graph the categorical\nn = 100\ntop_n_features = df_chi2_scores[:n][\"Feature\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4ce36b271fb612411f3c1dae2ba67b4bb8b3357"},"cell_type":"code","source":"top_n_features.apply(lambda s : ' '.join(s.split(\"_\")[:len(s.split(\"_\"))-1])).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ed95c10676abe6f97f692cc8a815a59a644c6d4"},"cell_type":"markdown","source":"* The graph below gets the features whose dummy variables contain the most entries in the top 100 by chi2 scores.\n* Let's graph hotel market, country and user's city with respect to the hotel clusters. \n* We'll dedicate a section for srch_destination_id, but that seems very relevant.\n* For ease of readability, we'll only get the top 5 hotel_clusters, their market, country and user location."},{"metadata":{"_uuid":"039daf4d4f969339572978a81fe9c47ff8165e71"},"cell_type":"markdown","source":"## Some basic findings\n- Looks like user city ~17 and ~90 is very active all throughout the different hotel clusters\n- Looks like hotel_market ~59 and 62 are related to many hotel_clusters\n- Looks like hotel_country ~8 is related to many hotel_clusters\n"},{"metadata":{"trusted":true,"_uuid":"999090d2dfeca88af5132ffb876887f358a43543"},"cell_type":"code","source":"def create_matrix(data, group_column, val_column):\n    grouped_data = data.groupby(group_column)[val_column].value_counts()\n    grouped_data = grouped_data.groupby(level=0).nlargest(3)\n    grouped_data.index = grouped_data.index.droplevel(0)\n    # transfrom to a square matrix\n    grouped_data_unstacked = grouped_data.unstack()\n    return grouped_data_unstacked.fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be0b6d7e5f49f266f3fbbdd9493aaafb82ddce3f"},"cell_type":"code","source":"hotel_user_city_vc_matrix = create_matrix(data, \"hotel_cluster\", \"user_location_city\")\nhotel_market_vc_matrix = create_matrix(data, \"hotel_cluster\", \"hotel_market\")\nhotel_country_vc_matrix = create_matrix(data, \"hotel_cluster\", \"hotel_country\")\nhotel_continent_vc_matrix = create_matrix(data, \"hotel_cluster\", \"hotel_continent\")\n\n# hotel user city\nfig, axes = plt.subplots(2, 2, figsize=(15, 15))\n\naxes[0][0].set_title(\"Looks like user city ~17 and ~90 is very active\\n all throughout the different hotel clusters\")\naxes[0][0].imshow(hotel_user_city_vc_matrix, cmap='gray')\n\naxes[0][1].set_title(\"Looks like hotel_market ~59 and 62 are\\nrelated to many hotel_clusters\")\naxes[0][1].imshow(hotel_market_vc_matrix, cmap='gray')\n\naxes[1][0].set_title(\"Looks like hotel_country ~8 is\\nrelated to many hotel_clusters (France?)\")\naxes[1][0].imshow(hotel_country_vc_matrix, cmap='gray');\n\naxes[1][1].set_title(\"I think hotel_continent 1 is Europe!\")\naxes[1][1].imshow(hotel_continent_vc_matrix, cmap='gray');","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22705e54dcc8856480ffbc761b86befe8cbbf02e"},"cell_type":"markdown","source":"# Numericals\n- Since orig_destination_distance has null values, let's see its distribution then decide on the imputation method."},{"metadata":{"trusted":true,"_uuid":"5f827b10b41893570e5b7adeede38c883384fb94"},"cell_type":"code","source":"sns.distplot(data[\"orig_destination_distance\"].dropna())\n\nplt.title(\"Majority of destination places are close to the origin. \\nThus, let's just use the median for imputation.\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c869aee25184ee0f8a829030ae753e27723cdefe"},"cell_type":"code","source":"sample_n = 10000\ndata_numericals_sample = data[NUMERICALS].sample(sample_n)\nchi2_scores = chi2(data_numericals_sample.fillna(data_numericals_sample.median()), data[HOTEL_CLUSTER].loc[data_numericals_sample.index])\ndf_chi2_scores = name_scores(chi2_scores[0], data_numericals_sample.columns)\n\ndf_chi2_pvalues = name_scores(chi2_scores[1], data_numericals_sample.columns)\ndf_chi2_pvalues.columns = [\"Feature\", \"PValue\"]\n\ndf_chi2_scores.merge(df_chi2_pvalues).sort_values(by=\"Score\", ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2648c1a9a523b64117c239bb4a6a5e1cf549270c"},"cell_type":"markdown","source":"### Hypotheses:\n- Distance is important to travelers. Hotel clusters seem to be dependent on the distance.\n- Certain hotel clusters seem to be children friendly\n- Certain hotel_clusters seem also to rake in \"bandwagoners\" (thru cnt variable)"},{"metadata":{"trusted":true,"_uuid":"ee8486c8d1a37f1f7ffde680f1e767aee0a25859"},"cell_type":"code","source":"sns.boxplot(data=data, x=HOTEL_CLUSTER, y=\"orig_destination_distance\")\n\nplt.title(\"There's a lot of outlier distances for each hotel_cluster\");","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eea92b008d6485aece30ac36f19cac12a1add555"},"cell_type":"code","source":"data_cluster = data[[HOTEL_CLUSTER]].copy()\ndata_cluster[\"cluster_num\"] = pd.cut(data_cluster[HOTEL_CLUSTER], bins=4, labels=range(4))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"124958a01bf46a091cab1e12ba0ed406f3e124a7"},"cell_type":"code","source":"clusters_1 = data[data[HOTEL_CLUSTER].isin(data_cluster.loc[data_cluster[\"cluster_num\"] == 0, HOTEL_CLUSTER])]\nclusters_1_unstacked = clusters_1.groupby(HOTEL_CLUSTER)[\"srch_children_cnt\"].value_counts().unstack().fillna(0)\n\nclusters_1_unstacked= clusters_1_unstacked.drop(0, axis=1).sort_values(by=2,)\nax = clusters_1_unstacked.plot.barh(stacked=True)\nax.legend(loc='upper right')\n\nplt.title(\"Some hotel_clusters are more perceived to be family friendly\", );","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb2e3c88a68ce690a257a4a002634d9d64c7d09f"},"cell_type":"markdown","source":"# What is the relationship of srch_destination_id to hotel_clusters?"},{"metadata":{"trusted":true,"_uuid":"8f37159ce4c688a2d5867394ed7c15ee8052805a"},"cell_type":"code","source":"print(\"Search destination id's nunique: \", data[\"srch_destination_id\"].nunique())\nprint(\"Search destination types nunique: \", data[\"srch_destination_type_id\"].nunique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e29067ab96c69301f71850ff2017833bbc99353d"},"cell_type":"markdown","source":"## For each hotel cluster, how is srch type related?"},{"metadata":{"trusted":true,"_uuid":"08b9e61abd25e7d49215f63c18cfc11ab9c0b275"},"cell_type":"code","source":"hotel_search_type_matrix = create_matrix(data, HOTEL_CLUSTER, \"srch_destination_type_id\")\nprint(\"Looks like the search types tend around 1 and 6.\")\ndisplay(hotel_search_type_matrix[:10])\nplt.imshow(hotel_search_type_matrix, cmap='gray');","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c0bad4902c75a835df6495a313ca8e9b324e5749"},"cell_type":"markdown","source":"## For each hotel cluster, how is search_id related. How many unique search ids are there for each hotel cluster?\n\n- Seems like there is a non-trivial amount of search id of just 1. These may be the \"weird\" searches that corresponded to some hotel clusters.\n- On the other hand, there's a lot of search ids per hotel cluster. Seems we can work with this for our content-based filtering algorithm. We can spend time tweaking embeddings for each hotel clsuter. Or we can just average them out."},{"metadata":{"trusted":true,"_uuid":"eddde81be9bd43a2a82b2c2d42034c527f4e762e"},"cell_type":"code","source":"hotel_to_search_n = data.groupby(HOTEL_CLUSTER)[\"srch_destination_id\"].nunique()\nax = sns.distplot(hotel_to_search_n, bins=50, kde=False)\nax.set_xlabel(\"Number of unique search IDs per hotel cluster\")\n\ndisplay(data.groupby(HOTEL_CLUSTER)[\"srch_destination_id\"].nunique().describe().to_frame(\n    \"Stats of the number of unique search IDs per hotel cluster\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83206672f41ee0ff3d7b3b88594ab4718bc1e7c7"},"cell_type":"markdown","source":"# Latent space variables look like a 10-20 'expressed' variables out of a hundred fifty"},{"metadata":{"trusted":true,"_uuid":"6d80d596299fc52c95a4e10e9557bdf8210951a9"},"cell_type":"code","source":"destinations = pd.read_csv(\"../input/destinations.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1fb161020f43acd0f327942ee8ba2dbeb3ea797b"},"cell_type":"code","source":"def create_latent_search_img(index, ax):\n    # to make the image 10x15, we create a 150th feature with the mean of the array\n    img = np.array(destinations.loc[index].values[1:].tolist() + [destinations.loc[index].values[1:].mean()])\n    img = img.reshape((15,10))\n    sns.heatmap(img, cmap='gray', ax=ax)\n    ax.set_xticklabels([])\n    ax.set_yticklabels([])\n    return ax","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f738f77609285352f73729ae5f8d27eba73033b2"},"cell_type":"code","source":"ax1 = create_latent_search_img(0, plt.subplot(2, 3, 1))\nax2 = create_latent_search_img(1, plt.subplot(2, 3, 2))\nax3 = create_latent_search_img(2, plt.subplot(2, 3, 3))\nax4 = create_latent_search_img(3, plt.subplot(2, 3, 4))\nax5 = create_latent_search_img(3, plt.subplot(2, 3, 5))\nax6 = create_latent_search_img(3, plt.subplot(2, 3, 6))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d0345fc277a292b61a3dadbe722bc5c8a2a558f"},"cell_type":"markdown","source":"# Let's average out the search ids per hotel cluster\n- Keep in mind the minimum number of unique search ids per hotel_cluster is 5, so we should have unique \"feature images\" per hotel_cluster.\n- When we search, we may click or not click on a hotel cluster. We take into account that act of clicking, which means the more search ids associated to a hotel_cluster, the more expressed the variables of those ids should be.\n   - We can either make this a sum or weighted mean aggregation operation. This can be for exploration in the modeling stage.\n   - For now, we'll do a sum."},{"metadata":{"trusted":true,"_uuid":"8973a6f41a3c8584200a4b172484f9106053ae5c"},"cell_type":"code","source":"def get_latent_search_hotel_array(hotel_cluster_index):\n    values = data.loc[data[HOTEL_CLUSTER] == hotel_cluster_index, \"srch_destination_id\"].to_frame().merge(destinations)\n    values = values.drop(\"srch_destination_id\", axis=1)\n    values = values.sum()\n    return values\n\ndef create_latent_search_hotel_image(hotel_cluster_index, ax):\n    img = get_latent_search_hotel_array(hotel_cluster_index)\n    img = np.array(img.tolist() + [img.mean()])\n    img = img.reshape((15,10))\n    \n    ax.imshow(img, cmap='gray')\n    ax.set_title(\"Cluster \" + str(hotel_cluster_index))\n    ax.set_xticklabels([])\n    ax.set_yticklabels([])\n    return ax","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33348d4f0da29272fbf3304feb47717e6403b695"},"cell_type":"code","source":"hotel_arrays = []\nfor i in range(data[HOTEL_CLUSTER].nunique()):\n    hotel_arrays.append(get_latent_search_hotel_array(i))\nhotel_arrays = np.array(hotel_arrays)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57cc873cee42266a021c872876a6dc009ee8f26a"},"cell_type":"code","source":"from scipy.cluster import hierarchy\nimport matplotlib.pyplot as plt\n\nZ = hierarchy.linkage(hotel_arrays, 'single')\nplt.figure(figsize=(20, 8))\ndn = hierarchy.dendrogram(Z)\n\n# We change the fontsize of minor ticks label \nplt.tick_params(axis='both', which='major', labelsize=10)\nplt.tick_params(axis='both', which='minor', labelsize=10)\nplt.xticks(rotation=0)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb6b63fc48d4ce2dc3b0b58b7a8d0c9b42a553e6"},"cell_type":"code","source":"fig = plt.figure(figsize=(20, 20))\n\nfor i in range(data[HOTEL_CLUSTER].nunique()):\n    create_latent_search_hotel_image(i, fig.add_subplot(10, 10, i+1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fa81a5336add48a061236614292f72f7e1018f8f"},"cell_type":"markdown","source":"# Analysis:\n- There's some minute differences across clusters although there are the same latent variables that are always expressed.\n- From the dendrogram, some clusters are very close while those on the left seem very unique.\n\n# Stats for Collaborative Filtering\n- Looks like we can do CF - Matrix Factorization here. Non-zero percentage is 28%, very trivial for CF.\n- We should use dimensionality lower than 100 as to not overfit."},{"metadata":{"trusted":true,"_uuid":"6f2c9fac0dd500b6226357485c1cec66fa5f67cf"},"cell_type":"code","source":"user_id_col = \"user_id\"\nitem_id_col = HOTEL_CLUSTER\nratings = data[[user_id_col, item_id_col]]\n\nnum_users = ratings[user_id_col].nunique()\nnum_items = ratings[item_id_col].nunique()\npossible_combinations = num_users * num_items\nnnz = len(ratings)\nnnz_percent = nnz / possible_combinations\n\nprint(\"Num Users:\", num_users)\nprint(\"Num Items:\", num_items)\nprint(\"Sparsity:\", nnz_percent)\nprint(\"Not very sparse. CF will work wonders here.\")\n\n# average number of hotel_clusters per user\nhotel_per_user = ratings.groupby(user_id_col)[item_id_col].nunique()\n\nfig = plt.figure(figsize=(10, 8))\nax = fig.add_subplot(211)\nsns.distplot(hotel_per_user, kde=False, ax=ax)\nax.set_title(\"Mean number of clusters per user: {:.2f}\".format(hotel_per_user.mean()))\n\n# # average number of users per hotel\nuser_per_hotel = ratings.groupby(item_id_col)[user_id_col].nunique()\nax = fig.add_subplot(212)\nsns.distplot(user_per_hotel, kde=False, ax=ax)\nax.set_title(\"Mean number of users per cluster: {:.2f}\".format(user_per_hotel.mean()))\n\nfig.tight_layout()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}