{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\n#https://www.dataquest.io/blog/python-vs-r/"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"import random\nfilename = \"../input/train.csv\"\nsamples = 100000\n\n# Read some samples - TAKES A LOT TO COMPUTE\n#n = sum(1 for line in open(filename)) - 1 #excludes header\n#print(n)\nn = 37670294\n#skip = sorted(random.sample(range(1,n+1),n-samples)) #the 0-indexed header will not be included in the skip list\n\n#train_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], skiprows=skip)\n#train_data.info()\n\ntrain_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], nrows=samples)\ntrain_data.info()\n\nprint('---------------')\nprint('We use %.1f%% of a data' % (samples/n))"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"train_data.head(5)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nsns.countplot(y='hotel_continent', data=train_data)\nsns.plt.title('Continent destination')\nplt.show() #get rid of <matplotlib.text.Text at 0x7f ..."},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"f, ax = plt.subplots(figsize=(15, 25))\nsns.countplot(y='hotel_country', data=train_data)\nplt.title('Search of hotels per country')\nplt.show()"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"sns.pairplot(train_data[['hotel_country', 'user_location_country']], size=6)\nplt.show()"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"train_data.loc[:,['is_mobile', 'is_package', 'orig_destination_distance', 'srch_adults_cnt', \n                  'srch_children_cnt', 'srch_rm_cnt', 'is_booking', 'cnt']].describe()"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"#https://www.dataquest.io/blog/python-data-science/\ntrain_data.groupby('hotel_continent').count().sort_values(by='hotel_country')"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"corr = train_data.corr()\n\n# Generate a mask for the upper triangle\nmask = np.zeros_like(corr, dtype=np.bool)\nmask[np.triu_indices_from(mask)] = True\n\n# Set up the matplotlib figure\nf, ax = plt.subplots(figsize=(12, 15))\n\n# Draw the heatmap with the mask and correct aspect ratio\nsns.heatmap(corr, mask=mask, vmax=.3, square=True, linewidths=.5, cbar_kws={\"shrink\": .5}, ax=ax)\nplt.show()"},{"cell_type":"markdown","metadata":{},"source":"Machine Learning\n---"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"#Divide data into training and test\nfrom sklearn.model_selection import train_test_split\ntrain, test, y_train, y_test = train_test_split(\n    train_data[['hotel_continent', 'user_location_country', 'hotel_country']], \n    train_data['hotel_cluster'], \n    test_size=0.33, \n    random_state=13)\n#display samle data with label\ntrain.head().join(y_train.head())"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"#Fit a model\nfrom sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(n_estimators=100, min_samples_leaf=10)\nmodel.fit(train, y_train)"},{"cell_type":"code","execution_count":null,"metadata":{"collapsed":false},"outputs":[],"source":"#mean squared error\nfrom sklearn.metrics import mean_squared_error\nimport math\n\npredictions = model.predict(test)\nmean_squared_error(predictions, y_test)"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":0}