{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import pandas as pd\ntrain = pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv', usecols = ['user_id'])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"users = train['user_id'].unique()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"users.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nselect_users = np.random.choice(users, 100000, replace=False)\nselect_users.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"select_users","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = pd.DataFrame(select_users)\ny.columns = ['id']\ny","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\npath = '/kaggle/input/expedia-hotel-recommendations/train.csv'\niter_csv = pd.read_csv(path, iterator=True, chunksize=1000)\nselect = pd.concat([chunk.loc[chunk.user_id.isin(y['id'])] for chunk in iter_csv])\nselect.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = select","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**EDA**\nStart visualization\n\n1. timeline\n1. click vs booking\n1. Destination country overlay Destimation booking"},{"metadata":{"trusted":true},"cell_type":"code","source":"df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['date'] = pd.to_datetime(df['date_time'])\ndf['date'] = df['date'].dt.date\ndf['month-year'] = pd.to_datetime(df['date']).dt.to_period('M')\ndf['hours'] = pd.to_datetime(df['date_time'])\ndf['hours'] = df['hours'].dt.hour\ndf = df.sort_values(by = 'month-year')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfig = plt.figure(figsize=(30,20))\nfig.subplots_adjust(hspace=0.4, wspace=0.4)\n\nax = fig.add_subplot(2, 2, 1)\nsns.countplot(df['posa_continent'], ax=ax)\n\nax = fig.add_subplot(2, 2, 2)\nsns.countplot(df['month-year'], ax=ax)\nplt.xticks(rotation=90)\n\nax = fig.add_subplot(2, 2, 3)\nsns.countplot(df['hotel_cluster'], ax=ax)\nplt.xticks(rotation=90)\n\nax = fig.add_subplot(2, 2, 4)\nsns.countplot(df['is_booking'], ax=ax)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.barplot(df['hotel_cluster'], hue =df['is_booking'])\nplt.xticks(rotation=90)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(x=df['hotel_cluster'], hue=df['is_booking'])\nplt.xticks(rotation=90)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(df['is_booking'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(df['month-year'],hue=df['is_booking'])\nplt.xticks(rotation=90)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nfig = plt.figure(figsize=(30,20))\nfig.subplots_adjust(hspace=0.4, wspace=0.4)\nax = fig.add_subplot(2, 2, 1)\nsns.countplot(df['hours'], ax=ax)\nax = fig.add_subplot(2, 2, 2)\nsns.countplot(df['is_package'], ax=ax)\nplt.xticks(rotation=90)\nax = fig.add_subplot(2, 2, 3)\nsns.countplot(df['channel'], ax=ax)\nplt.xticks(rotation=90)\nax = fig.add_subplot(2, 2, 4)\nsns.distplot(df['srch_adults_cnt'], ax=ax)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.FacetGrid(df, hue=\"is_booking\", size=6) \\\n   .map(plt.hist, \"hotel_cluster\") \\\n   .add_legend()\nplt.title('book or click')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['day'] = pd.to_datetime(df['date']).dt.day\ndf['month'] = pd.to_datetime(df['date']).dt.month\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head().transpose()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1 = df[['site_name', 'posa_continent', 'user_location_country',\n       'user_location_region', 'user_location_city',\n       'user_id', 'is_mobile', 'is_package',\n       'channel', 'srch_destination_id', 'srch_destination_type_id',\n       'is_booking', 'cnt', 'hotel_continent', 'hotel_country', 'hotel_market',\n       'hotel_cluster', 'day', 'month', 'hours']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = pd.DataFrame(df_1.groupby(['user_location_country']).size())\nx.transpose()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1['is_clicking'] = df_1['is_booking']\ndf_1.is_clicking[df_1.is_clicking == 0] = 2\ndf_1.is_clicking[df_1.is_clicking == 1] = 0\ndf_1.is_clicking[df_1.is_clicking == 2] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1[['is_booking', 'is_clicking']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_2 = df_1.groupby(['user_id', 'hotel_cluster', 'site_name', 'posa_continent', 'user_location_country','channel', 'hotel_continent', 'hotel_country', 'hotel_market']).sum()[['is_booking', 'is_clicking', 'is_mobile', 'is_package', 'cnt']].reset_index()\ndf_2.head\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_2.tail(20).transpose()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_2.head(20).transpose()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1.user_location_country[df_1.user_location_country == \"NaN\"] = 1000001\ndf_1.user_location_region[df_1.user_location_region == \"NaN\"] = 1000001\ndf_1.user_location_city[df_1.user_location_city == \"NaN\"] = 1000001\ndf_1.srch_destination_id[df_1.srch_destination_id == \"NaN\"] = 1000001\ndf_1.srch_destination_type_id[df_1.srch_destination_type_id == \"NaN\"] = 1000001\ndf_1.user_location_country[df_1.user_location_country == \"NaN\"] = 1000001\ndf_1.hotel_continent[df_1.hotel_continent == \"NaN\"] = 1000001\ndf_1.hotel_country[df_1.hotel_country == \"NaN\"] = 1000001\ndf_1.hotel_market[df_1.hotel_market == \"NaN\"] = 1000001","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"site_name = pd.get_dummies(df_1[\"site_name\"], prefix = 'site_name: ')\nposa_continent = pd.get_dummies(df_1[\"posa_continent\"], prefix = 'posa_continent: ')\nuser_location_country = pd.get_dummies(df_1[\"user_location_country\"], prefix = 'user_location_country: ')\nuser_location_region = pd.get_dummies(df_1[\"user_location_region\"], prefix = 'user_location_region: ')\n#user_location_city = pd.get_dummies(df_1[\"user_location_city\"], prefix = 'user_location_city: ')\n#srch_destination_id = pd.get_dummies(df_1[\"srch_destination_id\"], prefix = 'srch_destination_id: ')\nsrch_destination_type_id = pd.get_dummies(df_1[\"srch_destination_type_id\"], prefix = 'srch_destination_type_id: ')\nhotel_continent = pd.get_dummies(df_1[\"hotel_continent\"], prefix = 'hotel_continent: ')\nhotel_country = pd.get_dummies(df_1[\"hotel_country\"], prefix = 'hotel_country: ')\n#hotel_market = pd.get_dummies(df_1[\"hotel_market\"], prefix = 'hotel_market: ')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_2 = pd.concat([df_1.drop(['site_name', 'posa_continent',\"user_location_country\",\"user_location_region\",\"srch_destination_type_id\", \"hotel_continent\", \"hotel_country\"], axis = 1), \n                  site_name, posa_continent, user_location_country, user_location_region, srch_destination_type_id,\n                 hotel_continent, hotel_country], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_2.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn import datasets \nfrom sklearn.metrics import confusion_matrix \nfrom sklearn.model_selection import train_test_split \n\nX = df_2.drop(['hotel_cluster'] , axis=1) \ny = df_2['hotel_cluster'] \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, train_size=0.8) \n  \n# training a Naive Bayes classifier \nfrom sklearn.naive_bayes import GaussianNB \ngnb = GaussianNB().fit(X_train, y_train) \ngnb_predictions = gnb.predict(X_test) \n  \n# accuracy on X_test \naccuracy = gnb.score(X_test, y_test) \nprint (accuracy) \n  \n# creating a confusion matrix \ncm = confusion_matrix(y_test, gnb_predictions) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}