{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"9d5e5c27-3638-4245-8f23-5b1d708a1e8b"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n%matplotlib inline\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"583eb9d3-7ed0-4051-9a90-1c1717ff7000"},"outputs":[],"source":"# Get first 10000 rows and print some info about columns\ntrain = pd.read_csv(\"../input/train.csv\", parse_dates=['srch_ci', 'srch_co'], nrows=10000)\ntrain.info()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3df7cd3b-e57b-4881-a3fb-d98766f5011e"},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# preferred continent destinations\nsns.countplot(x='hotel_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f0392a6c-02c0-44d7-abca-d45af98abb6a"},"outputs":[],"source":"# most of people booking are from continent 3 I guess is one of the rich continent?\nsns.countplot(x='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"073b34be-6e70-451b-b071-4052d4e92492"},"outputs":[],"source":"# putting the two above together\nsns.countplot(x='hotel_continent', hue='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"4f059fc3-0575-4f73-8972-6e1c0634ade9"},"outputs":[],"source":"# how many people by continent are booking from mobile\nsns.countplot(x='posa_continent', hue='is_mobile', data = train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"949ea2e5-a987-4f23-8ce6-b133fcfd7910"},"outputs":[],"source":"# Difference between user and destination country\nsns.distplot(train['user_location_country'], label=\"User country\")\nsns.distplot(train['hotel_country'], label=\"Hotel country\")\nplt.legend()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e113bb33-56ef-4c1a-83c5-04c6b5d784f4"},"outputs":[],"source":"import numpy as np\n# get number of booked nights as difference between check in and check out\nhotel_nights = train['srch_co'] - train['srch_ci'] \nhotel_nights = (hotel_nights / np.timedelta64(1, 'D')).astype(float) # convert to float to avoid NA problems\ntrain['hotel_nights'] = hotel_nights\nplt.figure(figsize=(11, 9))\nax = sns.boxplot(x='hotel_continent', y='hotel_nights', data=train)\nlim = ax.set(ylim=(0, 15))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"3780b9a5-618b-49a5-9f87-741d1aee9306"},"outputs":[],"source":"plt.figure(figsize=(11, 9))\nsns.countplot(x=\"hotel_nights\", data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"bcdb85c7-7743-4385-a871-aa592f00977c"},"outputs":[],"source":"# distribution of the total number of people per cluster\nsrc_total_cnt = train.srch_adults_cnt + train.srch_children_cnt\ntrain['src_total_cnt'] = src_total_cnt\nax = sns.kdeplot(train['hotel_cluster'], train['src_total_cnt'], cmap=\"Purples_d\")\nlim = ax.set(ylim=(0.5, 4.5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"4814ecbd-24b7-4f09-9e22-cc04720f629d"},"outputs":[],"source":"# plot all columns countplots\nimport numpy as np\nrows = train.columns.size//3 - 1\nfig, axes = plt.subplots(nrows=rows, ncols=3, figsize=(12,18))\nfig.tight_layout()\ni = 0\nj = 0\nfor col in train.columns:\n    if j >= 3:\n        j = 0\n        i += 1\n    # avoid to plot by date    \n    if train[col].dtype == np.int64:\n        sns.countplot(x=col, data=train, ax=axes[i][j])\n        j += 1"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}