{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"cd4b2ae9-a645-4df1-8fca-eeb7efb4343e"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n%matplotlib inline\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"7637f38d-88cc-4c1a-9f77-1af8858cc5d6"},"outputs":[],"source":"# Get first 10000 rows and print some info about columns\ntrain = pd.read_csv(\"../input/train.csv\", parse_dates=['srch_ci', 'srch_co'], nrows=10000)\ntrain.info()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6bf40dba-5a8a-45d3-baad-bb981e2fdf19"},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# preferred continent destinations\nsns.countplot(x='hotel_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"ea35dbe3-87aa-4584-9a32-37a1a44a41d3"},"outputs":[],"source":"# most of people booking are from continent 3 I guess is one of the rich continent?\nsns.countplot(x='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"61943bf2-5823-4761-8ea8-fac7053e305f"},"outputs":[],"source":"# putting the two above together\nsns.countplot(x='hotel_continent', hue='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"9dc6e1e4-9bcd-4841-9215-1dd1e61c2e80"},"outputs":[],"source":"# how many people by continent are booking from mobile\nsns.countplot(x='posa_continent', hue='is_mobile', data = train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"171033c5-c001-4766-a2cd-11126398050b"},"outputs":[],"source":"# Difference between user and destination country\nsns.distplot(train['user_location_country'], label=\"User country\")\nsns.distplot(train['hotel_country'], label=\"Hotel country\")\nplt.legend()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"029a8ac8-8987-47e8-b526-d721b48316fc"},"outputs":[],"source":"import numpy as np\n# get number of booked nights as difference between check in and check out\nhotel_nights = train['srch_co'] - train['srch_ci'] \nhotel_nights = (hotel_nights / np.timedelta64(1, 'D')).astype(float) # convert to float to avoid NA problems\ntrain['hotel_nights'] = hotel_nights\nplt.figure(figsize=(11, 9))\nax = sns.boxplot(x='hotel_continent', y='hotel_nights', data=train)\nlim = ax.set(ylim=(0, 15))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8d39f20f-4a9c-4b77-9121-24d2c65b136b"},"outputs":[],"source":"plt.figure(figsize=(11, 9))\nsns.countplot(x=\"hotel_nights\", data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"11a5dc04-e25f-488f-a92e-1307ee849218"},"outputs":[],"source":"# distribution of the total number of people per cluster\nsrc_total_cnt = train.srch_adults_cnt + train.srch_children_cnt\ntrain['src_total_cnt'] = src_total_cnt\nax = sns.kdeplot(train['hotel_cluster'], train['src_total_cnt'], cmap=\"Purples_d\")\nlim = ax.set(ylim=(0.5, 4.5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"9d2d1f83-3882-41ae-bbda-68adc98c5177"},"outputs":[],"source":"# plot all columns countplots\nimport numpy as np\nrows = train.columns.size//3 - 1\nfig, axes = plt.subplots(nrows=rows, ncols=3, figsize=(12,18))\nfig.tight_layout()\ni = 0\nj = 0\nfor col in train.columns:\n    if j >= 3:\n        j = 0\n        i += 1\n    # avoid to plot by date    \n    if train[col].dtype == np.int64:\n        sns.countplot(x=col, data=train, ax=axes[i][j])\n        j += 1"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}