{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-11T02:38:52.78207Z","iopub.execute_input":"2021-12-11T02:38:52.782356Z","iopub.status.idle":"2021-12-11T02:38:52.79476Z","shell.execute_reply.started":"2021-12-11T02:38:52.782313Z","shell.execute_reply":"2021-12-11T02:38:52.793731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 37 Million Entries for training data, I will use 10k as a sample so it will not take as long to process models","metadata":{}},{"cell_type":"code","source":"sample_df=pd.read_csv('/kaggle/input/expedia-hotel-recommendations/train.csv',nrows=10000)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:52.79777Z","iopub.execute_input":"2021-12-11T02:38:52.798421Z","iopub.status.idle":"2021-12-11T02:38:52.850761Z","shell.execute_reply.started":"2021-12-11T02:38:52.798371Z","shell.execute_reply":"2021-12-11T02:38:52.850144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:52.851713Z","iopub.execute_input":"2021-12-11T02:38:52.852225Z","iopub.status.idle":"2021-12-11T02:38:52.871896Z","shell.execute_reply.started":"2021-12-11T02:38:52.852194Z","shell.execute_reply":"2021-12-11T02:38:52.870945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nsns.countplot(x='hotel_continent', data=sample_df)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:52.873273Z","iopub.execute_input":"2021-12-11T02:38:52.873565Z","iopub.status.idle":"2021-12-11T02:38:53.236774Z","shell.execute_reply.started":"2021-12-11T02:38:52.873532Z","shell.execute_reply":"2021-12-11T02:38:53.236211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax=plt.subplots()\nfig.set_size_inches(20,15)\nsns.heatmap(sample_df.corr(),cmap='coolwarm',ax=ax,annot=True,linewidths=2)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:53.238567Z","iopub.execute_input":"2021-12-11T02:38:53.239286Z","iopub.status.idle":"2021-12-11T02:38:56.218331Z","shell.execute_reply.started":"2021-12-11T02:38:53.239233Z","shell.execute_reply":"2021-12-11T02:38:56.2174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nfig.set_size_inches(13, 8)\nsns.countplot(x='hotel_cluster',data=sample_df, ax=ax)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:56.219596Z","iopub.execute_input":"2021-12-11T02:38:56.219826Z","iopub.status.idle":"2021-12-11T02:38:57.71468Z","shell.execute_reply.started":"2021-12-11T02:38:56.219799Z","shell.execute_reply":"2021-12-11T02:38:57.713735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nfig.set_size_inches(13, 8)\nsns.countplot(x='is_package',data=sample_df, order=[0,1], ax=ax)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:57.715888Z","iopub.execute_input":"2021-12-11T02:38:57.716104Z","iopub.status.idle":"2021-12-11T02:38:57.940529Z","shell.execute_reply.started":"2021-12-11T02:38:57.716078Z","shell.execute_reply":"2021-12-11T02:38:57.939702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Converting Dates into month attribute, this is because the month has the most seasonal attribute","metadata":{}},{"cell_type":"code","source":"sample_df['srch_ci']=pd.to_datetime(sample_df['srch_ci'])\nsample_df['srch_co']=pd.to_datetime(sample_df['srch_co'])\nsample_df['date_time']=pd.to_datetime(sample_df['date_time'])\n\nsample_df['check_in_month']=sample_df['srch_ci'].apply(lambda x:x.month)\nsample_df['date_time_month']=sample_df['date_time'].apply(lambda x:x.month)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:57.94193Z","iopub.execute_input":"2021-12-11T02:38:57.94214Z","iopub.status.idle":"2021-12-11T02:38:58.075534Z","shell.execute_reply.started":"2021-12-11T02:38:57.942113Z","shell.execute_reply":"2021-12-11T02:38:58.074552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting month distribution\nfig, ax = plt.subplots()\nfig.set_size_inches(13, 8)\nsns.countplot('date_time_month',data=sample_df[sample_df[\"is_booking\"] == 1],order=list(range(1,13)),ax=ax)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.076997Z","iopub.execute_input":"2021-12-11T02:38:58.077219Z","iopub.status.idle":"2021-12-11T02:38:58.357951Z","shell.execute_reply.started":"2021-12-11T02:38:58.077192Z","shell.execute_reply":"2021-12-11T02:38:58.357061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Adding days spent variable.","metadata":{}},{"cell_type":"code","source":"sample_df['time_delta']=(sample_df['srch_co']-sample_df['srch_ci'])\nsample_df['days_spent']=sample_df['time_delta'].dt.days\nsample_df=sample_df.drop(columns=['time_delta'])","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.359291Z","iopub.execute_input":"2021-12-11T02:38:58.359546Z","iopub.status.idle":"2021-12-11T02:38:58.369314Z","shell.execute_reply.started":"2021-12-11T02:38:58.359516Z","shell.execute_reply":"2021-12-11T02:38:58.368431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dealing with NA values","metadata":{}},{"cell_type":"code","source":"sample_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.3707Z","iopub.execute_input":"2021-12-11T02:38:58.370899Z","iopub.status.idle":"2021-12-11T02:38:58.38323Z","shell.execute_reply.started":"2021-12-11T02:38:58.370873Z","shell.execute_reply":"2021-12-11T02:38:58.382404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"orig_destination_distance is the big offender, there are NA values in srch_ci and srch_co but since there are only 7 I will remove those rows","metadata":{}},{"cell_type":"code","source":"sample_df=pd.DataFrame(sample_df)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.384689Z","iopub.execute_input":"2021-12-11T02:38:58.385066Z","iopub.status.idle":"2021-12-11T02:38:58.390663Z","shell.execute_reply.started":"2021-12-11T02:38:58.384984Z","shell.execute_reply":"2021-12-11T02:38:58.389552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.394031Z","iopub.execute_input":"2021-12-11T02:38:58.394241Z","iopub.status.idle":"2021-12-11T02:38:58.419871Z","shell.execute_reply.started":"2021-12-11T02:38:58.394215Z","shell.execute_reply":"2021-12-11T02:38:58.419063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This dropna call removes the 7 offending NA values.","metadata":{}},{"cell_type":"code","source":"sample_df=sample_df.dropna(subset=['srch_ci'])\nsample_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.421451Z","iopub.execute_input":"2021-12-11T02:38:58.421868Z","iopub.status.idle":"2021-12-11T02:38:58.439935Z","shell.execute_reply.started":"2021-12-11T02:38:58.421731Z","shell.execute_reply":"2021-12-11T02:38:58.439002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Then we must deal with orig_destination_distance. I will swap orig_destination_distance's mean value as the replacement.","metadata":{}},{"cell_type":"code","source":"dist_mean=sample_df['orig_destination_distance'].mean()\nsample_df['orig_destination_distance']=sample_df['orig_destination_distance'].fillna(dist_mean)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.441546Z","iopub.execute_input":"2021-12-11T02:38:58.442131Z","iopub.status.idle":"2021-12-11T02:38:58.447821Z","shell.execute_reply.started":"2021-12-11T02:38:58.442091Z","shell.execute_reply":"2021-12-11T02:38:58.446948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking that there are no more NaN values and indeed there are no longer any values.","metadata":{}},{"cell_type":"code","source":"sample_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.449281Z","iopub.execute_input":"2021-12-11T02:38:58.449741Z","iopub.status.idle":"2021-12-11T02:38:58.463198Z","shell.execute_reply.started":"2021-12-11T02:38:58.4497Z","shell.execute_reply":"2021-12-11T02:38:58.462279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We must now standardize the orig_destination_distance variable","metadata":{}},{"cell_type":"code","source":"odd_std=sample_df['orig_destination_distance'].std()\nodd_mean=sample_df['orig_destination_distance'].mean()\n\nsample_df['orig_destination_distance']=(sample_df['orig_destination_distance']-odd_mean)/odd_std","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.4648Z","iopub.execute_input":"2021-12-11T02:38:58.465109Z","iopub.status.idle":"2021-12-11T02:38:58.473176Z","shell.execute_reply.started":"2021-12-11T02:38:58.465067Z","shell.execute_reply":"2021-12-11T02:38:58.472405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking that mean=0 and std=1","metadata":{}},{"cell_type":"code","source":"print(sample_df['orig_destination_distance'].mean())\nprint(sample_df['orig_destination_distance'].std())","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.474536Z","iopub.execute_input":"2021-12-11T02:38:58.47499Z","iopub.status.idle":"2021-12-11T02:38:58.486952Z","shell.execute_reply.started":"2021-12-11T02:38:58.474948Z","shell.execute_reply":"2021-12-11T02:38:58.485899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"orig_destination_distance is now standardized","metadata":{}},{"cell_type":"markdown","source":"We must also standardize the days_spent varaible","metadata":{}},{"cell_type":"code","source":"ds_std=sample_df['days_spent'].std()\nds_mean=sample_df['days_spent'].mean()\nsample_df['days_spent']=(sample_df['days_spent']-ds_mean)/ds_std","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.488326Z","iopub.execute_input":"2021-12-11T02:38:58.489391Z","iopub.status.idle":"2021-12-11T02:38:58.499796Z","shell.execute_reply.started":"2021-12-11T02:38:58.489328Z","shell.execute_reply":"2021-12-11T02:38:58.499045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking that mean=0 and std=1","metadata":{}},{"cell_type":"code","source":"print(sample_df['days_spent'].mean())\nprint(sample_df['days_spent'].std())","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.500958Z","iopub.execute_input":"2021-12-11T02:38:58.501657Z","iopub.status.idle":"2021-12-11T02:38:58.511486Z","shell.execute_reply.started":"2021-12-11T02:38:58.50162Z","shell.execute_reply":"2021-12-11T02:38:58.510747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"days_spent is now standardized","metadata":{}},{"cell_type":"markdown","source":"Standardizing srch_adults_cnt,srch_children_cnt srch_rm_cnt and cnt","metadata":{}},{"cell_type":"code","source":"ad_mean=sample_df['srch_adults_cnt'].mean()\nad_std=sample_df['srch_adults_cnt'].std()\nch_mean=sample_df['srch_children_cnt'].mean()\nch_std=sample_df['srch_children_cnt'].std()\n\n\ncnt_mean=sample_df['cnt'].mean()\ncnt_std=sample_df['cnt'].std()\n\nroom_mean=sample_df['srch_rm_cnt'].mean()\nroom_std=sample_df['srch_rm_cnt'].std()\n\nsample_df['srch_adults_cnt']=(sample_df['srch_adults_cnt']-ad_mean)/ad_std\nsample_df['srch_children_cnt']=(sample_df['srch_children_cnt']-ch_mean)/ch_std\nsample_df['cnt']=(sample_df['cnt']-cnt_mean)/cnt_std\nsample_df['srch_rm_cnt']=(sample_df['srch_rm_cnt']-room_mean)/room_std\n\n\n\nprint(sample_df['srch_adults_cnt'].mean())\nprint(sample_df['srch_children_cnt'].std())\nprint(sample_df['srch_adults_cnt'].mean())\nprint(sample_df['srch_children_cnt'].std())\n\nprint(sample_df['cnt'].mean())\nprint(sample_df['cnt'].std())\nprint(sample_df['srch_rm_cnt'].mean())\nprint(sample_df['srch_rm_cnt'].std())\n\nprint(sample_df.columns)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.512611Z","iopub.execute_input":"2021-12-11T02:38:58.512828Z","iopub.status.idle":"2021-12-11T02:38:58.533671Z","shell.execute_reply.started":"2021-12-11T02:38:58.512803Z","shell.execute_reply":"2021-12-11T02:38:58.532829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are many variables that are categorical data but not ordinal. These are hotel_continent, hotel_country, hotel_market,user_location_country,user_location_region, user_location_city,site_name, posa_continent check_in_month,date_time_month and channel . I will use one hot encoding on all these variables.","metadata":{}},{"cell_type":"code","source":"categorical_columns=['hotel_continent', 'hotel_country', 'hotel_market','user_location_country','user_location_region', 'user_location_city','site_name','posa_continent','check_in_month','date_time_month','channel']\nalternative_df=sample_df.drop(columns=categorical_columns)\nsample_df=pd.get_dummies(sample_df,columns=categorical_columns)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.534809Z","iopub.execute_input":"2021-12-11T02:38:58.535031Z","iopub.status.idle":"2021-12-11T02:38:58.612219Z","shell.execute_reply.started":"2021-12-11T02:38:58.535003Z","shell.execute_reply":"2021-12-11T02:38:58.611352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Certian variables have been used or will not be useful for model processing. There are date_time, srch_ci, srch_co and user_id.","metadata":{}},{"cell_type":"code","source":"column_drops=['date_time','srch_ci','srch_co','user_id']\nsample_df=sample_df.drop(columns=column_drops)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.613459Z","iopub.execute_input":"2021-12-11T02:38:58.613704Z","iopub.status.idle":"2021-12-11T02:38:58.694996Z","shell.execute_reply.started":"2021-12-11T02:38:58.613672Z","shell.execute_reply":"2021-12-11T02:38:58.694241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Taking target variable and storing it as y","metadata":{}},{"cell_type":"code","source":"y=sample_df['hotel_cluster']","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.696325Z","iopub.execute_input":"2021-12-11T02:38:58.696592Z","iopub.status.idle":"2021-12-11T02:38:58.702269Z","shell.execute_reply.started":"2021-12-11T02:38:58.69656Z","shell.execute_reply":"2021-12-11T02:38:58.701274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Taking dataset without hotel cluster and storing as x","metadata":{}},{"cell_type":"code","source":"x=sample_df.drop(columns='hotel_cluster')","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.703843Z","iopub.execute_input":"2021-12-11T02:38:58.704748Z","iopub.status.idle":"2021-12-11T02:38:58.726Z","shell.execute_reply.started":"2021-12-11T02:38:58.704701Z","shell.execute_reply":"2021-12-11T02:38:58.725093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train_test_split set up","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test=train_test_split(x,y,test_size=.3,random_state=10)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.727242Z","iopub.execute_input":"2021-12-11T02:38:58.727502Z","iopub.status.idle":"2021-12-11T02:38:58.785906Z","shell.execute_reply.started":"2021-12-11T02:38:58.727471Z","shell.execute_reply":"2021-12-11T02:38:58.784948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model test 1: Random Forest","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrf_model=RandomForestClassifier()\nrf_model.fit(X_train,y_train)\ny_pred_rf=rf_model.predict(X_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:38:58.787432Z","iopub.execute_input":"2021-12-11T02:38:58.78775Z","iopub.status.idle":"2021-12-11T02:39:16.318354Z","shell.execute_reply.started":"2021-12-11T02:38:58.787715Z","shell.execute_reply":"2021-12-11T02:39:16.317435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_important=rf_model.feature_importances_\nindex_list=sorted(range(len(rf_model.feature_importances_)),key=lambda i: rf_model.feature_importances_[i])[-10:]\nimpFeatures=list(x.columns[index_list])\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:16.319771Z","iopub.execute_input":"2021-12-11T02:39:16.320075Z","iopub.status.idle":"2021-12-11T02:39:47.269085Z","shell.execute_reply.started":"2021-12-11T02:39:16.320037Z","shell.execute_reply":"2021-12-11T02:39:47.268259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in index_list:\n    print (round(rf_model.feature_importances_[i],3))\nprint(impFeatures)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:47.270595Z","iopub.execute_input":"2021-12-11T02:39:47.270939Z","iopub.status.idle":"2021-12-11T02:39:47.473016Z","shell.execute_reply.started":"2021-12-11T02:39:47.270896Z","shell.execute_reply":"2021-12-11T02:39:47.472063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking accuracy","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics\nfrom sklearn.metrics import mean_squared_error\nprint(metrics.accuracy_score(y_test,y_pred_rf))\nprint(metrics.mean_squared_error(y_test,y_pred_rf))","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:47.474226Z","iopub.execute_input":"2021-12-11T02:39:47.474894Z","iopub.status.idle":"2021-12-11T02:39:47.482369Z","shell.execute_reply.started":"2021-12-11T02:39:47.474855Z","shell.execute_reply":"2021-12-11T02:39:47.481281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model test 2: K means clustering","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nkn_model=KNeighborsClassifier()\nkn_model.fit(X_train,y_train)\ny_pred_kn=kn_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:47.483622Z","iopub.execute_input":"2021-12-11T02:39:47.484007Z","iopub.status.idle":"2021-12-11T02:39:51.332519Z","shell.execute_reply.started":"2021-12-11T02:39:47.483977Z","shell.execute_reply":"2021-12-11T02:39:51.331684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking accuracy","metadata":{}},{"cell_type":"code","source":"print(metrics.accuracy_score(y_test,y_pred_kn))\nprint(metrics.mean_squared_error(y_test,y_pred_kn))","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:51.333705Z","iopub.execute_input":"2021-12-11T02:39:51.333945Z","iopub.status.idle":"2021-12-11T02:39:51.340271Z","shell.execute_reply.started":"2021-12-11T02:39:51.333915Z","shell.execute_reply":"2021-12-11T02:39:51.339425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model test 3: Decision Tree","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nmodel_dt=DecisionTreeClassifier()\nmodel_dt.fit(X_train,y_train)\ny_pred_dt=model_dt.predict(X_test)\n\nprint(metrics.accuracy_score(y_test,y_pred_dt))","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:39:51.341857Z","iopub.execute_input":"2021-12-11T02:39:51.342161Z","iopub.status.idle":"2021-12-11T02:39:52.461956Z","shell.execute_reply.started":"2021-12-11T02:39:51.342113Z","shell.execute_reply":"2021-12-11T02:39:52.461077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model 4: MLPCClassifier","metadata":{}},{"cell_type":"code","source":"from sklearn.neural_network import MLPClassifier\nmodel_nn=MLPClassifier(solver='adam')\nmodel_nn.fit(X_train,y_train)\ny_pred_nn=model_nn.predict(X_test)\n\nprint(metrics.accuracy_score(y_test,y_pred_nn))\n","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:58:18.902607Z","iopub.execute_input":"2021-12-11T02:58:18.902911Z","iopub.status.idle":"2021-12-11T02:58:48.293085Z","shell.execute_reply.started":"2021-12-11T02:58:18.902879Z","shell.execute_reply":"2021-12-11T02:58:48.292147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.values","metadata":{"execution":{"iopub.status.busy":"2021-12-11T02:59:24.429037Z","iopub.execute_input":"2021-12-11T02:59:24.429298Z","iopub.status.idle":"2021-12-11T02:59:24.435423Z","shell.execute_reply.started":"2021-12-11T02:59:24.429269Z","shell.execute_reply":"2021-12-11T02:59:24.434607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking accuracy","metadata":{}}]}