{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":5056,"databundleVersionId":868325,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport warnings\n\n# Suppress all warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-08T17:44:00.598456Z","iopub.execute_input":"2024-09-08T17:44:00.598887Z","iopub.status.idle":"2024-09-08T17:44:00.607650Z","shell.execute_reply.started":"2024-09-08T17:44:00.598848Z","shell.execute_reply":"2024-09-08T17:44:00.606243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only using a sample_df of size 10000","metadata":{}},{"cell_type":"code","source":"sample_df=pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv',nrows=10000)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:13.435101Z","iopub.execute_input":"2024-09-08T17:44:13.435637Z","iopub.status.idle":"2024-09-08T17:44:13.470574Z","shell.execute_reply.started":"2024-09-08T17:44:13.435594Z","shell.execute_reply":"2024-09-08T17:44:13.469774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Basic EDA","metadata":{}},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:13.472165Z","iopub.execute_input":"2024-09-08T17:44:13.472487Z","iopub.status.idle":"2024-09-08T17:44:13.492147Z","shell.execute_reply.started":"2024-09-08T17:44:13.472453Z","shell.execute_reply":"2024-09-08T17:44:13.491208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nfig.set_size_inches(25, 8)\nsns.countplot(x='hotel_cluster',data=sample_df, ax=ax)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:13.493266Z","iopub.execute_input":"2024-09-08T17:44:13.493575Z","iopub.status.idle":"2024-09-08T17:44:14.548660Z","shell.execute_reply.started":"2024-09-08T17:44:13.493538Z","shell.execute_reply":"2024-09-08T17:44:14.547696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"data seems to be little skewed","metadata":{}},{"cell_type":"code","source":"def stacked_barcount_with_hue(count,hue):\n    # Create a cross-tabulation to calculate the counts of `hotel_cluster` stacked by `hotel_continent`\n    dummy=pd.DataFrame\n    dummy=sample_df['hotel_continent'].astype('category')\n\n    cross_tab = pd.crosstab(sample_df[count], sample_df[hue])\n\n    # Create the plot\n    fig, ax = plt.subplots(figsize=(25, 8))\n\n    # Plot the stacked bar chart\n    bottom = None  # Start with no offset for the first stack\n    for i, continent in enumerate(cross_tab.columns):\n        ax.bar(cross_tab.index, cross_tab[continent], label=continent, bottom=bottom)\n        # Update the bottom to stack bars\n        if bottom is None:\n            bottom = cross_tab[continent].values\n        else:\n            bottom += cross_tab[continent].values\n\n    # Add a title and labels if needed\n    ax.set_title('Count of Hotel Clusters Stacked by'+hue)\n    ax.set_xlabel(count)\n    ax.set_ylabel('Count')\n\n    # Adjust the legend for better visualization\n    ax.legend(title=hue, loc='upper right')\n\n    # Show the plot\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:17.956381Z","iopub.execute_input":"2024-09-08T17:44:17.956807Z","iopub.status.idle":"2024-09-08T17:44:17.964752Z","shell.execute_reply.started":"2024-09-08T17:44:17.956767Z","shell.execute_reply":"2024-09-08T17:44:17.963772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='hotel_continent',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','hotel_continent')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:17.966551Z","iopub.execute_input":"2024-09-08T17:44:17.966991Z","iopub.status.idle":"2024-09-08T17:44:19.332895Z","shell.execute_reply.started":"2024-09-08T17:44:17.966959Z","shell.execute_reply":"2024-09-08T17:44:19.331955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='posa_continent',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','posa_continent')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:19.334143Z","iopub.execute_input":"2024-09-08T17:44:19.334452Z","iopub.status.idle":"2024-09-08T17:44:20.569018Z","shell.execute_reply.started":"2024-09-08T17:44:19.334419Z","shell.execute_reply":"2024-09-08T17:44:20.568061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='is_package',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','is_package')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:20.570420Z","iopub.execute_input":"2024-09-08T17:44:20.571392Z","iopub.status.idle":"2024-09-08T17:44:21.363785Z","shell.execute_reply.started":"2024-09-08T17:44:20.571354Z","shell.execute_reply":"2024-09-08T17:44:21.362761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='is_booking',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','is_booking')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:21.366440Z","iopub.execute_input":"2024-09-08T17:44:21.366788Z","iopub.status.idle":"2024-09-08T17:44:22.721862Z","shell.execute_reply.started":"2024-09-08T17:44:21.366754Z","shell.execute_reply":"2024-09-08T17:44:22.720864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='is_mobile',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','is_mobile')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:22.723172Z","iopub.execute_input":"2024-09-08T17:44:22.723487Z","iopub.status.idle":"2024-09-08T17:44:23.470577Z","shell.execute_reply.started":"2024-09-08T17:44:22.723455Z","shell.execute_reply":"2024-09-08T17:44:23.469494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='is_package',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','is_package')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:23.472515Z","iopub.execute_input":"2024-09-08T17:44:23.472977Z","iopub.status.idle":"2024-09-08T17:44:24.143131Z","shell.execute_reply.started":"2024-09-08T17:44:23.472928Z","shell.execute_reply":"2024-09-08T17:44:24.142142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='srch_adults_cnt',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','srch_adults_cnt')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:24.144562Z","iopub.execute_input":"2024-09-08T17:44:24.144908Z","iopub.status.idle":"2024-09-08T17:44:25.938513Z","shell.execute_reply.started":"2024-09-08T17:44:24.144874Z","shell.execute_reply":"2024-09-08T17:44:25.937588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='srch_children_cnt',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','srch_children_cnt')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:25.939947Z","iopub.execute_input":"2024-09-08T17:44:25.940732Z","iopub.status.idle":"2024-09-08T17:44:27.776995Z","shell.execute_reply.started":"2024-09-08T17:44:25.940665Z","shell.execute_reply":"2024-09-08T17:44:27.776049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='srch_rm_cnt',data=sample_df)\nstacked_barcount_with_hue('hotel_cluster','srch_rm_cnt')","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:27.778461Z","iopub.execute_input":"2024-09-08T17:44:27.778897Z","iopub.status.idle":"2024-09-08T17:44:29.454859Z","shell.execute_reply.started":"2024-09-08T17:44:27.778852Z","shell.execute_reply":"2024-09-08T17:44:29.453965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We can see that there is no discernable pattern in the data and the hotel cluster predicted","metadata":{}},{"cell_type":"markdown","source":"# Basic Feature Engineering and Data Cleanup\n","metadata":{}},{"cell_type":"markdown","source":"creating a new column for the month of srch_ci(check in month) and srch_month(when the search was made) ","metadata":{}},{"cell_type":"code","source":"sample_df['srch_ci']=pd.to_datetime(sample_df['srch_ci'])\nsample_df['srch_co']=pd.to_datetime(sample_df['srch_co'])\nsample_df['date_time']=pd.to_datetime(sample_df['date_time'])\n\nsample_df['ci_month']=sample_df['srch_ci'].apply(lambda x:x.month)\nsample_df['srch_month']=sample_df['date_time'].apply(lambda x:x.month)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:29.455983Z","iopub.execute_input":"2024-09-08T17:44:29.456265Z","iopub.status.idle":"2024-09-08T17:44:29.509976Z","shell.execute_reply.started":"2024-09-08T17:44:29.456233Z","shell.execute_reply":"2024-09-08T17:44:29.509114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='srch_month',data=sample_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:29.511377Z","iopub.execute_input":"2024-09-08T17:44:29.511817Z","iopub.status.idle":"2024-09-08T17:44:29.799357Z","shell.execute_reply.started":"2024-09-08T17:44:29.511773Z","shell.execute_reply":"2024-09-08T17:44:29.798441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='ci_month',data=sample_df)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:29.802909Z","iopub.execute_input":"2024-09-08T17:44:29.803226Z","iopub.status.idle":"2024-09-08T17:44:30.088342Z","shell.execute_reply.started":"2024-09-08T17:44:29.803192Z","shell.execute_reply":"2024-09-08T17:44:30.087367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"creating a new column of days_spent","metadata":{}},{"cell_type":"code","source":"sample_df['days_spent']=(sample_df['srch_co']-sample_df['srch_ci'])\nsample_df['days_spent']=sample_df['days_spent'].dt.days\nsample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.089639Z","iopub.execute_input":"2024-09-08T17:44:30.090011Z","iopub.status.idle":"2024-09-08T17:44:30.113798Z","shell.execute_reply.started":"2024-09-08T17:44:30.089977Z","shell.execute_reply":"2024-09-08T17:44:30.112824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.114923Z","iopub.execute_input":"2024-09-08T17:44:30.115250Z","iopub.status.idle":"2024-09-08T17:44:30.127650Z","shell.execute_reply.started":"2024-09-08T17:44:30.115213Z","shell.execute_reply":"2024-09-08T17:44:30.126727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"since there very few null values of days_spent, srch_ci,etc ; we remove the rows for those values","metadata":{}},{"cell_type":"code","source":"sample_df=sample_df.dropna(subset='srch_ci')\nsample_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.128700Z","iopub.execute_input":"2024-09-08T17:44:30.129046Z","iopub.status.idle":"2024-09-08T17:44:30.146161Z","shell.execute_reply.started":"2024-09-08T17:44:30.129014Z","shell.execute_reply":"2024-09-08T17:44:30.145397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for orig_destination_destination we replace the null value with mean of other non-null values.","metadata":{}},{"cell_type":"code","source":"dist_mean=sample_df['orig_destination_distance'].mean()\nsample_df['orig_destination_distance']=sample_df['orig_destination_distance'].fillna(dist_mean)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.147085Z","iopub.execute_input":"2024-09-08T17:44:30.147344Z","iopub.status.idle":"2024-09-08T17:44:30.152902Z","shell.execute_reply.started":"2024-09-08T17:44:30.147315Z","shell.execute_reply":"2024-09-08T17:44:30.151934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.154036Z","iopub.execute_input":"2024-09-08T17:44:30.154328Z","iopub.status.idle":"2024-09-08T17:44:30.168437Z","shell.execute_reply.started":"2024-09-08T17:44:30.154297Z","shell.execute_reply":"2024-09-08T17:44:30.167425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalising int values to improve model's accuracy","metadata":{}},{"cell_type":"code","source":"ds_std=sample_df['days_spent'].std()\nds_mean=sample_df['days_spent'].mean()\nsample_df['days_spent']=(sample_df['days_spent']-ds_mean)/ds_std\n\nprint(sample_df['days_spent'].mean())\nprint(sample_df['days_spent'].std())","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.169455Z","iopub.execute_input":"2024-09-08T17:44:30.169937Z","iopub.status.idle":"2024-09-08T17:44:30.177630Z","shell.execute_reply.started":"2024-09-08T17:44:30.169896Z","shell.execute_reply":"2024-09-08T17:44:30.176665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ad_mean=sample_df['srch_adults_cnt'].mean()\nad_std=sample_df['srch_adults_cnt'].std()\nch_mean=sample_df['srch_children_cnt'].mean()\nch_std=sample_df['srch_children_cnt'].std()\n\n\ncnt_mean=sample_df['cnt'].mean()\ncnt_std=sample_df['cnt'].std()\n\nroom_mean=sample_df['srch_rm_cnt'].mean()\nroom_std=sample_df['srch_rm_cnt'].std()\n\nsample_df['srch_adults_cnt']=(sample_df['srch_adults_cnt']-ad_mean)/ad_std\nsample_df['srch_children_cnt']=(sample_df['srch_children_cnt']-ch_mean)/ch_std\nsample_df['cnt']=(sample_df['cnt']-cnt_mean)/cnt_std\nsample_df['srch_rm_cnt']=(sample_df['srch_rm_cnt']-room_mean)/room_std\n\n\n\nprint(sample_df['srch_adults_cnt'].mean())\nprint(sample_df['srch_children_cnt'].std())\nprint(sample_df['srch_adults_cnt'].mean())\nprint(sample_df['srch_children_cnt'].std())\n\nprint(sample_df['cnt'].mean())\nprint(sample_df['cnt'].std())\nprint(sample_df['srch_rm_cnt'].mean())\nprint(sample_df['srch_rm_cnt'].std())\n\nprint(sample_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.178914Z","iopub.execute_input":"2024-09-08T17:44:30.179924Z","iopub.status.idle":"2024-09-08T17:44:30.195366Z","shell.execute_reply.started":"2024-09-08T17:44:30.179872Z","shell.execute_reply":"2024-09-08T17:44:30.194399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"droping site name and user id because that does not have any real relation with the choosing of the hotel_cluster. In case of site name that is already in being factored in location id of the srch","metadata":{}},{"cell_type":"markdown","source":"# Little bit more EDA","metadata":{}},{"cell_type":"markdown","source":"ploting a heat map to see if there is a significant correlation bewtween any two columns","metadata":{}},{"cell_type":"code","source":"corr = sample_df.corr()\nplt.figure(figsize=(20, 20))  # Adjust the figure size as needed\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\", linewidths=0.5)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:30.196665Z","iopub.execute_input":"2024-09-08T17:44:30.197297Z","iopub.status.idle":"2024-09-08T17:44:32.704172Z","shell.execute_reply.started":"2024-09-08T17:44:30.197255Z","shell.execute_reply":"2024-09-08T17:44:32.703188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:32.705634Z","iopub.execute_input":"2024-09-08T17:44:32.706054Z","iopub.status.idle":"2024-09-08T17:44:32.732774Z","shell.execute_reply.started":"2024-09-08T17:44:32.706011Z","shell.execute_reply":"2024-09-08T17:44:32.731796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_count = sample_df['hotel_country'].nunique()\nprint(unique_count)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:32.733914Z","iopub.execute_input":"2024-09-08T17:44:32.734259Z","iopub.status.idle":"2024-09-08T17:44:32.739846Z","shell.execute_reply.started":"2024-09-08T17:44:32.734226Z","shell.execute_reply":"2024-09-08T17:44:32.738847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df_copy=sample_df","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:32.741072Z","iopub.execute_input":"2024-09-08T17:44:32.741433Z","iopub.status.idle":"2024-09-08T17:44:32.750108Z","shell.execute_reply.started":"2024-09-08T17:44:32.741390Z","shell.execute_reply":"2024-09-08T17:44:32.749353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One Hot Encoding the Categorical Values","metadata":{}},{"cell_type":"code","source":"categorical_columns=['hotel_continent', 'hotel_country', 'hotel_market','user_location_country','user_location_region', 'user_location_city','posa_continent','ci_month','srch_month','channel']\nalternative_df=sample_df.drop(columns=categorical_columns)\nsample_df=pd.get_dummies(sample_df,columns=categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:44:49.595533Z","iopub.execute_input":"2024-09-08T16:44:49.595949Z","iopub.status.idle":"2024-09-08T16:44:49.630527Z","shell.execute_reply.started":"2024-09-08T16:44:49.595899Z","shell.execute_reply":"2024-09-08T16:44:49.629660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_drops=['date_time','srch_ci','srch_co','user_id']\nsample_df=sample_df.drop(columns=column_drops)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:44:32.751133Z","iopub.execute_input":"2024-09-08T17:44:32.751493Z","iopub.status.idle":"2024-09-08T17:44:32.769302Z","shell.execute_reply.started":"2024-09-08T17:44:32.751450Z","shell.execute_reply":"2024-09-08T17:44:32.768498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Applying Multi-Class Classification Models","metadata":{}},{"cell_type":"markdown","source":"importing necessary library","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import AdaBoostClassifier, RandomForestClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom xgboost import XGBClassifier \n","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:42:26.865135Z","iopub.execute_input":"2024-09-08T16:42:26.865583Z","iopub.status.idle":"2024-09-08T16:42:27.682966Z","shell.execute_reply.started":"2024-09-08T16:42:26.865541Z","shell.execute_reply":"2024-09-08T16:42:27.681631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using common Multi-Class Classification Models(and using some with advanced algorithms)","metadata":{}},{"cell_type":"code","source":"# 1. Load Dataset\ndata = sample_df  # You can replace this with your own dataset\nX = sample_df.drop(columns='hotel_cluster')  # Features\ny = sample_df['hotel_cluster']  # Original target (not used for now)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nclassifiers = {\n        \"Decision Tree\": DecisionTreeClassifier(),\n        \"Naive Bayes\": GaussianNB(),\n        \"Multinomial Logistic Regression\": LogisticRegression(multi_class='multinomial', solver='lbfgs', max_iter=200),\n        \"AdaBoost with Decision Trees\": AdaBoostClassifier(estimator=DecisionTreeClassifier(),n_estimators=100, random_state=42),\n        \"XGBoost Classifier\": XGBClassifier(use_label_encoder=False, eval_metric='mlogloss', random_state=42),\n        \"K-Nearest Neighbors\": KNeighborsClassifier(n_neighbors=5),\n        \"Random Forest\": RandomForestClassifier(n_estimators=100, random_state=42)\n    }\n\n    # Train and evaluate each classifier\nfor name, model in classifiers.items():\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_test)\n    accuracy = accuracy_score(y_test, y_pred)\n    print(f\"{name} Accuracy: {accuracy:.4f}\")\n    print(\"\\n\" + \"-\" * 50 + \"\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:42:29.085368Z","iopub.execute_input":"2024-09-08T16:42:29.085843Z","iopub.status.idle":"2024-09-08T16:42:48.194960Z","shell.execute_reply.started":"2024-09-08T16:42:29.085799Z","shell.execute_reply":"2024-09-08T16:42:48.192864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Multi-class Classification Algorithms with K-means Clustering of the Hotel Clusters","metadata":{}},{"cell_type":"markdown","source":"The Hotels are divided into 100 clusters and the clusters are categorical in nature. But having 100 different classes would greatly reduce the accuracy of the model.","metadata":{}},{"cell_type":"markdown","source":"So we clusters the hotel clusters further to improve model's accuracy further. We use K-means clustering to keep the variance of hotel_cluster relatively same","metadata":{}},{"cell_type":"code","source":"# 1. Load Dataset\ndata = sample_df  # You can replace this with your own dataset\nX = sample_df.drop(columns='hotel_cluster')  # Features\ny = sample_df['hotel_cluster']  # Original target (not used for now)\n\n# 2. Define Function to Perform Clustering and Classification with Multiple Algorithms\ndef cluster_and_classify(X, y, k):\n    # Apply K-Means clustering\n    y = np.array(y).reshape(-1, 1)\n    kmeans = KMeans(n_clusters=k, init='k-means++', random_state=42)\n    cluster_labels = kmeans.fit_predict(y)\n\n    # Split the data into training and test sets using cluster labels as the target\n    X_train, X_test, y_train, y_test = train_test_split(X, cluster_labels, test_size=0.2, random_state=42)\n\n    # Dictionary to store classifiers\n    classifiers = {\n        \"Decision Tree\": DecisionTreeClassifier(),\n        \"Naive Bayes\": GaussianNB(),\n        \"Multinomial Logistic Regression\": LogisticRegression(multi_class='multinomial', solver='lbfgs', max_iter=200),\n        \"AdaBoost with Decision Trees\": AdaBoostClassifier(estimator=DecisionTreeClassifier(), n_estimators=100, random_state=42),\n        \"XGBoost Classifier\": XGBClassifier(use_label_encoder=False, eval_metric='mlogloss', random_state=42),\n        \"K-Nearest Neighbors\": KNeighborsClassifier(n_neighbors=5),\n        \"Random Forest\": RandomForestClassifier(n_estimators=100, random_state=42)\n    }\n\n    # Train and evaluate each classifier\n    print(f\"Results for k={k}:\")\n    for name, model in classifiers.items():\n        model.fit(X_train, y_train)\n        y_pred = model.predict(X_test)\n        accuracy = accuracy_score(y_test, y_pred)\n        print(f\"{name} Accuracy: {accuracy:.4f}\")\n    print(\"\\n\" + \"-\" * 50 + \"\\n\")\n\n# 4. Apply Clustering and Classification for Different k Values\nfor k in [5, 10, 20, 50]:\n    cluster_and_classify(X, y, k)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:42:48.196432Z","iopub.status.idle":"2024-09-08T16:42:48.198850Z","shell.execute_reply.started":"2024-09-08T16:42:48.198456Z","shell.execute_reply":"2024-09-08T16:42:48.198496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making Cluster Prediction using Deep Nueral Network\n","metadata":{}},{"cell_type":"code","source":"# Remove the 'hotel_cluster' column and store it in a variable\nhotel_cluster_column = sample_df.pop('hotel_cluster')\n\n# Add the 'hotel_cluster' column back as the last column\nsample_df['hotel_cluster'] = hotel_cluster_column","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:16.418661Z","iopub.execute_input":"2024-09-08T17:45:16.419064Z","iopub.status.idle":"2024-09-08T17:45:16.424993Z","shell.execute_reply.started":"2024-09-08T17:45:16.419027Z","shell.execute_reply":"2024-09-08T17:45:16.424053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.columns.size","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:17.373347Z","iopub.execute_input":"2024-09-08T17:45:17.374304Z","iopub.status.idle":"2024-09-08T17:45:17.380784Z","shell.execute_reply.started":"2024-09-08T17:45:17.374249Z","shell.execute_reply":"2024-09-08T17:45:17.379888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"importing necessary libraries","metadata":{}},{"cell_type":"code","source":"pip install scikeras","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:46:52.454724Z","iopub.execute_input":"2024-09-08T16:46:52.455131Z","iopub.status.idle":"2024-09-08T16:47:10.879613Z","shell.execute_reply.started":"2024-09-08T16:46:52.455094Z","shell.execute_reply":"2024-09-08T16:47:10.878457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install keras ","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:49:52.958494Z","iopub.execute_input":"2024-09-08T16:49:52.959354Z","iopub.status.idle":"2024-09-08T16:50:05.768526Z","shell.execute_reply.started":"2024-09-08T16:49:52.959305Z","shell.execute_reply":"2024-09-08T16:50:05.767364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install np_utils","metadata":{"execution":{"iopub.status.busy":"2024-09-08T16:50:29.389512Z","iopub.execute_input":"2024-09-08T16:50:29.390372Z","iopub.status.idle":"2024-09-08T16:50:44.878032Z","shell.execute_reply.started":"2024-09-08T16:50:29.390331Z","shell.execute_reply":"2024-09-08T16:50:44.876768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom scikeras.wrappers import KerasClassifier\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.pipeline import Pipeline","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:23.613089Z","iopub.execute_input":"2024-09-08T17:45:23.613474Z","iopub.status.idle":"2024-09-08T17:45:23.619191Z","shell.execute_reply.started":"2024-09-08T17:45:23.613440Z","shell.execute_reply":"2024-09-08T17:45:23.618214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load dataset\ndataset = sample_df.values\nX = dataset[:,0:22].astype(float)\nY = dataset[:,22]","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:24.078319Z","iopub.execute_input":"2024-09-08T17:45:24.078726Z","iopub.status.idle":"2024-09-08T17:45:24.185957Z","shell.execute_reply.started":"2024-09-08T17:45:24.078677Z","shell.execute_reply":"2024-09-08T17:45:24.185151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode class values as integers\nencoder = LabelEncoder()\nencoder.fit(Y)\nencoded_Y = encoder.transform(Y)\n# convert integers to dummy variables (i.e. one hot encoded)\ndummy_y = to_categorical(encoded_Y)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:25.953157Z","iopub.execute_input":"2024-09-08T17:45:25.953545Z","iopub.status.idle":"2024-09-08T17:45:25.961742Z","shell.execute_reply.started":"2024-09-08T17:45:25.953510Z","shell.execute_reply":"2024-09-08T17:45:25.960731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_y","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:45:28.613664Z","iopub.execute_input":"2024-09-08T17:45:28.614538Z","iopub.status.idle":"2024-09-08T17:45:28.621024Z","shell.execute_reply.started":"2024-09-08T17:45:28.614497Z","shell.execute_reply":"2024-09-08T17:45:28.620033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define baseline model\ndef baseline_model():\n\t# create model\n\tmodel = Sequential()\n\tmodel.add(Dense(4, input_dim=22, activation='relu'))\n\tmodel.add(Dense(100, activation='softmax'))\n\t# Compile model\n\tmodel.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\n\treturn model","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:46:26.263726Z","iopub.execute_input":"2024-09-08T17:46:26.264140Z","iopub.status.idle":"2024-09-08T17:46:26.270124Z","shell.execute_reply.started":"2024-09-08T17:46:26.264104Z","shell.execute_reply":"2024-09-08T17:46:26.269001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"estimator = KerasClassifier(build_fn=baseline_model, epochs=200, batch_size=5, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:46:26.683632Z","iopub.execute_input":"2024-09-08T17:46:26.684044Z","iopub.status.idle":"2024-09-08T17:46:26.688604Z","shell.execute_reply.started":"2024-09-08T17:46:26.684009Z","shell.execute_reply":"2024-09-08T17:46:26.687682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = KFold(n_splits=10, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:46:27.453059Z","iopub.execute_input":"2024-09-08T17:46:27.453825Z","iopub.status.idle":"2024-09-08T17:46:27.458053Z","shell.execute_reply.started":"2024-09-08T17:46:27.453785Z","shell.execute_reply":"2024-09-08T17:46:27.457074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = cross_val_score(estimator, X, dummy_y, cv=kfold)\nprint(\"Baseline: %.2f%% (%.2f%%)\" % (results.mean()*100, results.std()*100))","metadata":{"execution":{"iopub.status.busy":"2024-09-08T17:46:27.828670Z","iopub.execute_input":"2024-09-08T17:46:27.829417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}