{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f6304670-9b70-4e1c-8117-23575262881c"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n%matplotlib inline\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"21a05c04-7218-4672-9a18-fa4a81f3a223"},"outputs":[],"source":"# Get first 10000 rows and print some info about columns\ntrain = pd.read_csv(\"../input/train.csv\", parse_dates=['srch_ci', 'srch_co'], nrows=10000)\ntrain.info()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"00c3b669-63fd-407d-b96d-19c3a58ec6af"},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n# preferred continent destinations\nsns.countplot(x='hotel_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"48893890-e947-401b-b120-3e6b17f71e9d"},"outputs":[],"source":"# most of people booking are from continent 3 I guess is one of the rich continent?\nsns.countplot(x='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f7f7156f-6fe4-448c-a2df-9cb0b3889151"},"outputs":[],"source":"# putting the two above together\nsns.countplot(x='hotel_continent', hue='posa_continent', data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"69315f85-3762-4ca6-a4f5-e33ecdf640c0"},"outputs":[],"source":"# how many people by continent are booking from mobile\nsns.countplot(x='posa_continent', hue='is_mobile', data = train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"0556ffb7-e5c4-4636-9f22-78ede5da67c3"},"outputs":[],"source":"# Difference between user and destination country\nsns.distplot(train['user_location_country'], label=\"User country\")\nsns.distplot(train['hotel_country'], label=\"Hotel country\")\nplt.legend()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"353fa3e6-123c-4e1a-96c8-4baf0dd21b58"},"outputs":[],"source":"import numpy as np\n# get number of booked nights as difference between check in and check out\nhotel_nights = train['srch_co'] - train['srch_ci'] \nhotel_nights = (hotel_nights / np.timedelta64(1, 'D')).astype(float) # convert to float to avoid NA problems\ntrain['hotel_nights'] = hotel_nights\nplt.figure(figsize=(11, 9))\nax = sns.boxplot(x='hotel_continent', y='hotel_nights', data=train)\nlim = ax.set(ylim=(0, 15))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"cdffa6cf-8231-4709-b5d1-5d00b384101a"},"outputs":[],"source":"plt.figure(figsize=(11, 9))\nsns.countplot(x=\"hotel_nights\", data=train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"178ff611-e95b-4483-a9f8-0f9928e8c0de"},"outputs":[],"source":"# distribution of the total number of people per cluster\nsrc_total_cnt = train.srch_adults_cnt + train.srch_children_cnt\ntrain['src_total_cnt'] = src_total_cnt\nax = sns.kdeplot(train['hotel_cluster'], train['src_total_cnt'], cmap=\"Purples_d\")\nlim = ax.set(ylim=(0.5, 4.5))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"d4c0200c-b2db-46fa-bf03-825bdad2dc59"},"outputs":[],"source":"# plot all columns countplots\nimport numpy as np\nrows = train.columns.size//3 - 1\nfig, axes = plt.subplots(nrows=rows, ncols=3, figsize=(12,18))\nfig.tight_layout()\ni = 0\nj = 0\nfor col in train.columns:\n    if j >= 3:\n        j = 0\n        i += 1\n    # avoid to plot by date    \n    if train[col].dtype == np.int64:\n        sns.countplot(x=col, data=train, ax=axes[i][j])\n        j += 1"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.5.2"}},"nbformat":4,"nbformat_minor":0}