{"cells":[{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"dab1d745-d81e-4e14-8d2c-34f944492cc7"},"outputs":[],"source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.\n\n#https://www.dataquest.io/blog/python-vs-r/"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"bcd24fa3-2928-4253-bf0b-36757e86dfde"},"outputs":[],"source":"import random\nfilename = \"../input/train.csv\"\nsamples = 100000\n\n# Read some samples - TAKES A LOT TO COMPUTE\n#n = sum(1 for line in open(filename)) - 1 #excludes header\n#print(n)\nn = 37670294\n#skip = sorted(random.sample(range(1,n+1),n-samples)) #the 0-indexed header will not be included in the skip list\n\n#train_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], skiprows=skip)\n#train_data.info()\n\ntrain_data = pd.read_csv(filename, parse_dates=['date_time', 'srch_ci', 'srch_co'], nrows=samples)\ntrain_data.info()\n\nprint('---------------')\nprint('We use %.1f%% of a data' % (samples/n))"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"95e6907a-a07c-49c5-8cc1-819a06515301"},"outputs":[],"source":"train_data.head(5)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"a71f9e54-2269-47bc-9159-444a1a87e556"},"outputs":[],"source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nsns.countplot(y='hotel_continent', data=train_data)\nsns.plt.title('Continent destination')\nplt.show() #get rid of <matplotlib.text.Text at 0x7f ..."},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"56274ace-ee22-4f9a-9161-d9af9e0450f1"},"outputs":[],"source":"f, ax = plt.subplots(figsize=(15, 25))\nsns.countplot(y='hotel_country', data=train_data)\nplt.title('Search of hotels per country')\nplt.show()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"443f65b2-7966-4a73-82ae-46662557deb2"},"outputs":[],"source":"sns.pairplot(train_data[['hotel_country', 'user_location_country']], size=6)\nplt.show()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"f26c7529-35a7-43e1-a58f-2d0179526b34"},"outputs":[],"source":"train_data.loc[:,['is_mobile', 'is_package', 'orig_destination_distance', 'srch_adults_cnt', \n                  'srch_children_cnt', 'srch_rm_cnt', 'is_booking', 'cnt']].describe()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"14b44f9b-6144-40b6-98df-551e75f3bf48"},"outputs":[],"source":"#https://www.dataquest.io/blog/python-data-science/\ntrain_data.groupby('hotel_continent').count().sort_values(by='hotel_country')"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"e94d8c63-b972-4390-818c-80e170d065ab"},"outputs":[],"source":"corr = train_data.corr()\n\n# Generate a mask for the upper triangle\nmask = np.zeros_like(corr, dtype=np.bool)\nmask[np.triu_indices_from(mask)] = True\n\n# Set up the matplotlib figure\nf, ax = plt.subplots(figsize=(12, 15))\n\n# Draw the heatmap with the mask and correct aspect ratio\nsns.heatmap(corr, mask=mask, vmax=.3, square=True, linewidths=.5, cbar_kws={\"shrink\": .5}, ax=ax)\nplt.show()"},{"cell_type":"markdown","metadata":{"_cell_guid":"ce33b935-03f4-44b7-b1f9-e06441fc839c"},"source":"Machine Learning\n---"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"470c63fc-0f41-415e-afc1-a52f84eb5500"},"outputs":[],"source":"#Divide data into training and test\nfrom sklearn.model_selection import train_test_split\ntrain, test, y_train, y_test = train_test_split(\n    train_data[['hotel_continent', 'user_location_country', 'hotel_country']], \n    train_data['hotel_cluster'], \n    test_size=0.33, \n    random_state=13)\n#display samle data with label\ntrain.head().join(y_train.head())"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"004e7e66-60fd-816c-4b21-0b65744739d7"},"outputs":[],"source":"y_train.head()"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"922b2374-0cc7-4e67-82b3-f3f1f6d955c6"},"outputs":[],"source":"#Fit a model\nfrom sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(n_estimators=100, min_samples_leaf=10)\nmodel.fit(train, y_train)"},{"cell_type":"code","execution_count":null,"metadata":{"_cell_guid":"49e4ffc5-8b40-4bde-98d8-94ff9a9f8949"},"outputs":[],"source":"#mean squared error\nfrom sklearn.metrics import mean_squared_error\nimport math\n\npredictions = model.predict(test)\nmean_squared_error(predictions, y_test)"}],"metadata":{"_change_revision":0,"_is_fork":false,"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.0"}},"nbformat":4,"nbformat_minor":0}