{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f71231cb-6094-492c-ab6d-6c88ac5fa7cb"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n%matplotlib inline\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"5c0fa4f6-fc69-4924-bc91-d7b7cd25014f"},"outputs":[],"source":"# Get first 10000 rows and print some info about columns\ntrain = pd.read_csv(\"../input/train.csv\", parse_dates=['srch_ci', 'srch_co'], nrows=10000)\ntrain.info()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"867468e6-c445-42b6-9a80-c60a23ca882b"},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# preferred continent destinations\nsns.countplot(x='hotel_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f21bc50e-88f8-4b94-9fdf-2baa3dba51f1"},"outputs":[],"source":"# most of people booking are from continent 3 I guess is one of the rich continent?\nsns.countplot(x='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"67b567c6-c726-433f-b6b1-b2eedd8d0b6c"},"outputs":[],"source":"# putting the two above together\nsns.countplot(x='hotel_continent', hue='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"6d0f8108-aed3-4a2f-8f5b-731c3dba34c7"},"outputs":[],"source":"# how many people by continent are booking from mobile\nsns.countplot(x='posa_continent', hue='is_mobile', data = train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"b619ea0b-bb56-4ffd-bfd5-90c870ea18be"},"outputs":[],"source":"# Difference between user and destination country\nsns.distplot(train['user_location_country'], label=\"User country\")\nsns.distplot(train['hotel_country'], label=\"Hotel country\")\nplt.legend()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"dd39a1a7-e4d9-440a-995e-13a6159e17d3"},"outputs":[],"source":"import numpy as np\n# get number of booked nights as difference between check in and check out\nhotel_nights = train['srch_co'] - train['srch_ci'] \nhotel_nights = (hotel_nights / np.timedelta64(1, 'D')).astype(float) # convert to float to avoid NA problems\ntrain['hotel_nights'] = hotel_nights\nplt.figure(figsize=(11, 9))\nax = sns.boxplot(x='hotel_continent', y='hotel_nights', data=train)\nlim = ax.set(ylim=(0, 15))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"8384ee2f-329a-4fcc-ac62-3717291dbc69"},"outputs":[],"source":"plt.figure(figsize=(11, 9))\nsns.countplot(x=\"hotel_nights\", data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"2dbb4087-564c-45ab-b118-db7aa636255c"},"outputs":[],"source":"# distribution of the total number of people per cluster\nsrc_total_cnt = train.srch_adults_cnt + train.srch_children_cnt\ntrain['src_total_cnt'] = src_total_cnt\nax = sns.kdeplot(train['hotel_cluster'], train['src_total_cnt'], cmap=\"Purples_d\")\nlim = ax.set(ylim=(0.5, 4.5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"388da0f5-125f-439f-a2a0-6b6f1babed9a"},"outputs":[],"source":"# plot all columns countplots\nimport numpy as np\nrows = train.columns.size//3 - 1\nfig, axes = plt.subplots(nrows=rows, ncols=3, figsize=(12,18))\nfig.tight_layout()\ni = 0\nj = 0\nfor col in train.columns:\n    if j >= 3:\n        j = 0\n        i += 1\n    # avoid to plot by date    \n    if train[col].dtype == np.int64:\n        sns.countplot(x=col, data=train, ax=axes[i][j])\n        j += 1"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}