{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#mporting required libraries\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn import datasets\nfrom sklearn import model_selection\nfrom sklearn import tree\nimport graphviz\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.tree import DecisionTreeRegressor \nfrom sklearn.ensemble import StackingClassifier\nfrom mlxtend.plotting import plot_confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"id":"VFo11ICDpp7i","execution":{"iopub.status.busy":"2022-09-27T13:00:11.277093Z","iopub.execute_input":"2022-09-27T13:00:11.277903Z","iopub.status.idle":"2022-09-27T13:00:13.283077Z","shell.execute_reply.started":"2022-09-27T13:00:11.277795Z","shell.execute_reply":"2022-09-27T13:00:13.280776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loading the data\ndf = pd.read_csv('../input/clean-wheat-seeds-dataset/seeds_dataset (2).csv')","metadata":{"id":"6vpEOF5BqknU","execution":{"iopub.status.busy":"2022-09-27T13:00:13.285850Z","iopub.execute_input":"2022-09-27T13:00:13.286391Z","iopub.status.idle":"2022-09-27T13:00:13.325108Z","shell.execute_reply.started":"2022-09-27T13:00:13.286336Z","shell.execute_reply":"2022-09-27T13:00:13.324259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking columns in our data\ndf.columns","metadata":{"id":"CJlC4qgsqwE9","outputId":"740fdbbb-8cf1-42d7-949b-da094d53dda9","execution":{"iopub.status.busy":"2022-09-01T06:58:18.967772Z","iopub.execute_input":"2022-09-01T06:58:18.968743Z","iopub.status.idle":"2022-09-01T06:58:18.977812Z","shell.execute_reply.started":"2022-09-01T06:58:18.968697Z","shell.execute_reply":"2022-09-01T06:58:18.977068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"1gKVQNekqzRN","outputId":"5251b75a-cfe5-4453-8d43-2d46c4afacf6","execution":{"iopub.status.busy":"2022-09-01T06:58:20.096145Z","iopub.execute_input":"2022-09-01T06:58:20.096546Z","iopub.status.idle":"2022-09-01T06:58:20.123733Z","shell.execute_reply.started":"2022-09-01T06:58:20.096513Z","shell.execute_reply":"2022-09-01T06:58:20.122445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"id":"bSFJGOzvq1Q2","outputId":"802b8722-0606-47d7-e885-bc90a5bd372a","execution":{"iopub.status.busy":"2022-09-01T06:58:20.635138Z","iopub.execute_input":"2022-09-01T06:58:20.635562Z","iopub.status.idle":"2022-09-01T06:58:20.656845Z","shell.execute_reply.started":"2022-09-01T06:58:20.635525Z","shell.execute_reply":"2022-09-01T06:58:20.656033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#taking info of our data\ndf.info()","metadata":{"id":"rcNjHUkOq2oF","outputId":"e82812bb-9e28-4611-ab92-43160d05188c","execution":{"iopub.status.busy":"2022-09-01T06:58:21.119126Z","iopub.execute_input":"2022-09-01T06:58:21.120024Z","iopub.status.idle":"2022-09-01T06:58:21.149265Z","shell.execute_reply.started":"2022-09-01T06:58:21.119972Z","shell.execute_reply":"2022-09-01T06:58:21.148124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#droping the unwaned columns\ndf.drop(['Unnamed: 8','Unnamed: 9'],axis=1,inplace=True)","metadata":{"id":"iJ0LQO6pq5kB","execution":{"iopub.status.busy":"2022-09-01T06:58:21.538595Z","iopub.execute_input":"2022-09-01T06:58:21.539400Z","iopub.status.idle":"2022-09-01T06:58:21.547289Z","shell.execute_reply.started":"2022-09-01T06:58:21.539358Z","shell.execute_reply":"2022-09-01T06:58:21.545619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"id":"VLyNKi0Oz5a_","outputId":"a6b033ac-addf-4e1b-871d-7c423bd21d39","execution":{"iopub.status.busy":"2022-09-01T06:58:21.951891Z","iopub.execute_input":"2022-09-01T06:58:21.953153Z","iopub.status.idle":"2022-09-01T06:58:21.976298Z","shell.execute_reply.started":"2022-09-01T06:58:21.953094Z","shell.execute_reply":"2022-09-01T06:58:21.974486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"id":"iKmeUuMi2K6k","outputId":"2015f089-7496-46bc-9060-3da6a35b39cb","execution":{"iopub.status.busy":"2022-09-01T06:58:22.353787Z","iopub.execute_input":"2022-09-01T06:58:22.354644Z","iopub.status.idle":"2022-09-01T06:58:22.363071Z","shell.execute_reply.started":"2022-09-01T06:58:22.354595Z","shell.execute_reply":"2022-09-01T06:58:22.362115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df)","metadata":{"id":"Ziy_ibWZ0diY","outputId":"13a5ed68-c234-4f84-d8a9-f132af043b01","execution":{"iopub.status.busy":"2022-09-01T06:58:35.760157Z","iopub.execute_input":"2022-09-01T06:58:35.760564Z","iopub.status.idle":"2022-09-01T06:58:48.780228Z","shell.execute_reply.started":"2022-09-01T06:58:35.760531Z","shell.execute_reply":"2022-09-01T06:58:48.779118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## boxplot to find the outliers\n\nplt.figure(figsize=(10,30),facecolor='white')\nplotnumber=1\nfor i in df:\n    ax=plt.subplot(8,1,plotnumber)\n    sns.boxplot(df[i],color='green')\n    plotnumber=plotnumber + 1 \nplt.show()","metadata":{"id":"VP1VKC9Y2-rr","outputId":"56d211cb-9f87-4aa5-9f37-8f2da528c41a","execution":{"iopub.status.busy":"2022-09-01T06:58:48.782107Z","iopub.execute_input":"2022-09-01T06:58:48.782671Z","iopub.status.idle":"2022-09-01T06:58:49.806314Z","shell.execute_reply.started":"2022-09-01T06:58:48.782635Z","shell.execute_reply":"2022-09-01T06:58:49.805075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we have outliers in compactness and in asymmetry coefficient","metadata":{"id":"za4PDW3o7QLY"}},{"cell_type":"code","source":"plt.figure(figsize=(15,15),facecolor='white')\n\nplotnum=1 #counter\n\nfor i in df:\n    if(plotnum<9):\n        a=plt.subplot(4,2,plotnum)#plotting 8 graph\n        sns.distplot(df[i],color='purple')#to know distribution\n    plotnum+=1#increment counter\nplt.tight_layout() \n\n## Distribution plot \n\n\"\"\"plt.figure(figsize=(25,140))\nplotnumber=1\nfor a in df:\n    ax=plt.subplot(20,2,plotnumber)\n    sns.distplot(x=df[a],color='purple')\n    #plt.xticks(rotation=70)\n    plotnumber+=1\nplt.show() \"\"\"","metadata":{"id":"ko1Rjt6v0ysp","outputId":"2a86c3fb-022e-454a-e8e8-498b0971fa68","execution":{"iopub.status.busy":"2022-09-01T06:58:49.808121Z","iopub.execute_input":"2022-09-01T06:58:49.808851Z","iopub.status.idle":"2022-09-01T06:58:52.006536Z","shell.execute_reply.started":"2022-09-01T06:58:49.808793Z","shell.execute_reply":"2022-09-01T06:58:52.004843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we have to make distribution normalise","metadata":{"id":"PF51tGS750ZP"}},{"cell_type":"markdown","source":"# scaling","metadata":{"id":"UTOFBd44kAAr"}},{"cell_type":"code","source":"# Build Machine Learning Model\n#Lets create feature matrix X  and y labels\nX = df.drop(('Class (1, 2, 3)'),axis=1)\ny = df['Class (1, 2, 3)']\n\nprint('X shape=', X.shape)\nprint('y shape=', y.shape)","metadata":{"id":"9nnNftjhj2d_","outputId":"6fcd1e95-8300-4475-958b-1ecb05fa4bee","execution":{"iopub.status.busy":"2022-09-01T06:58:52.009411Z","iopub.execute_input":"2022-09-01T06:58:52.009798Z","iopub.status.idle":"2022-09-01T06:58:52.018204Z","shell.execute_reply.started":"2022-09-01T06:58:52.009762Z","shell.execute_reply":"2022-09-01T06:58:52.016888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# copy the data\ndf_min_max_scaled = X.copy()\n  \n# apply normalization techniques\nfor column in df_min_max_scaled.columns:\n    df_min_max_scaled[column] = (df_min_max_scaled[column] - df_min_max_scaled[column].min()) / (df_min_max_scaled[column].max() - df_min_max_scaled[column].min())    \n  \n# view normalized data\nprint(df_min_max_scaled)\n\n\ndf2 = df_min_max_scaled","metadata":{"id":"d3PmxSgy5ag6","outputId":"78790370-6454-4bd6-9579-924b451be688","execution":{"iopub.status.busy":"2022-09-01T06:58:52.020115Z","iopub.execute_input":"2022-09-01T06:58:52.020491Z","iopub.status.idle":"2022-09-01T06:58:52.044676Z","shell.execute_reply.started":"2022-09-01T06:58:52.020459Z","shell.execute_reply":"2022-09-01T06:58:52.042678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15),facecolor='white')\n\nplotnum=1 #counter\n\nfor i in df_min_max_scaled:\n    if(plotnum<9):\n        a=plt.subplot(4,2,plotnum)#plotting 8 graph\n        sns.distplot(df_min_max_scaled[i],color='purple')#to know distribution\n    plotnum+=1#increment counter\nplt.tight_layout() ","metadata":{"id":"-YvrusYo7-r-","outputId":"1a4c4e47-cdf5-4b8f-b8d1-ef54d0728868","execution":{"iopub.status.busy":"2022-09-01T06:58:52.045714Z","iopub.execute_input":"2022-09-01T06:58:52.046095Z","iopub.status.idle":"2022-09-01T06:58:53.532554Z","shell.execute_reply.started":"2022-09-01T06:58:52.046061Z","shell.execute_reply":"2022-09-01T06:58:53.531309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# copy the data\ndf_z_scaled = df.copy()\n  \n# apply normalization techniques\nfor column in df_z_scaled.columns:\n    df_z_scaled[column] = (df_z_scaled[column] -\n                           df_z_scaled[column].mean()) / df_z_scaled[column].std()    \n  \n# view normalized data   \ndisplay(df_z_scaled)","metadata":{"id":"MkrF7MEA8OZy","outputId":"525276a2-3cf8-4444-c56a-8d38a2eaf076","execution":{"iopub.status.busy":"2022-09-01T06:58:53.534611Z","iopub.execute_input":"2022-09-01T06:58:53.535128Z","iopub.status.idle":"2022-09-01T06:58:53.561849Z","shell.execute_reply.started":"2022-09-01T06:58:53.535079Z","shell.execute_reply":"2022-09-01T06:58:53.560979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15),facecolor='white')\n\nplotnum=1 #counter\n\nfor i in df_z_scaled:\n    if(plotnum<9):\n        a=plt.subplot(4,2,plotnum)#plotting 8 graph\n        sns.distplot(df_z_scaled[i],color='purple')#to know distribution\n    plotnum+=1#increment counter\nplt.tight_layout() ","metadata":{"id":"sbnl3rsx8hMO","outputId":"cc6dd6a5-8a63-4cd3-b590-af5f25d3f821","execution":{"iopub.status.busy":"2022-09-01T06:58:53.563269Z","iopub.execute_input":"2022-09-01T06:58:53.563859Z","iopub.status.idle":"2022-09-01T06:58:55.163356Z","shell.execute_reply.started":"2022-09-01T06:58:53.563817Z","shell.execute_reply":"2022-09-01T06:58:55.162205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# copy the data\ndf_max_scaled = df.copy()\n  \n# apply normalization techniques\nfor column in df_max_scaled.columns:\n    df_max_scaled[column] = df_max_scaled[column]  / df_max_scaled[column].abs().max()\n      \n# view normalized data\ndisplay(df_max_scaled)","metadata":{"id":"aTDXX7nx8nNF","outputId":"53414347-5c0a-46f8-9033-9f8ac83cf697","execution":{"iopub.status.busy":"2022-09-01T06:58:55.164976Z","iopub.execute_input":"2022-09-01T06:58:55.166226Z","iopub.status.idle":"2022-09-01T06:58:55.193621Z","shell.execute_reply.started":"2022-09-01T06:58:55.166180Z","shell.execute_reply":"2022-09-01T06:58:55.192086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,15),facecolor='white')\n\nplotnum=1 #counter\n\nfor i in df_max_scaled:\n    if(plotnum<9):\n        a=plt.subplot(4,2,plotnum)#plotting 8 graph\n        sns.distplot(df_max_scaled[i],color='purple')#to know distribution\n    plotnum+=1#increment counter\nplt.tight_layout()","metadata":{"id":"FTtK8EAZ83lx","outputId":"c63acde6-a7f3-407e-d6a7-f682a610d984","execution":{"iopub.status.busy":"2022-09-01T06:58:55.198894Z","iopub.execute_input":"2022-09-01T06:58:55.199737Z","iopub.status.idle":"2022-09-01T06:58:57.060087Z","shell.execute_reply.started":"2022-09-01T06:58:55.199696Z","shell.execute_reply":"2022-09-01T06:58:57.058974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import skew\n\nnumerical_features = df.dtypes[df.dtypes != 'object'].index\n\n# checking the skewness in all the numerical features\nskewed_features = df[numerical_features].apply(lambda x: skew(x.dropna())).sort_values(ascending = False)\n\n# converting the features into a dataframe\nskewness = pd.DataFrame({'skew':skewed_features})\n\n# checking the head of skewness dataset\nskewness","metadata":{"id":"GQtK_H2e86_j","outputId":"011c6d41-ba5c-49a4-80a8-fb3bf097bee7","execution":{"iopub.status.busy":"2022-09-01T06:58:57.061815Z","iopub.execute_input":"2022-09-01T06:58:57.062507Z","iopub.status.idle":"2022-09-01T06:58:57.082652Z","shell.execute_reply.started":"2022-09-01T06:58:57.062463Z","shell.execute_reply":"2022-09-01T06:58:57.081353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Builinding model**","metadata":{"id":"YGwmpMoNjhIH"}},{"cell_type":"code","source":"# Build Machine Learning Model\n#Lets create feature matrix X  and y labels\nX = df2\ny = df['Class (1, 2, 3)']\n\nprint('X shape=', X.shape)\nprint('y shape=', y.shape)","metadata":{"id":"10H-lPJ7kb26","outputId":"9d544990-4d6a-4143-cb4e-87d0f79484f8","execution":{"iopub.status.busy":"2022-09-01T06:58:57.084193Z","iopub.execute_input":"2022-09-01T06:58:57.085183Z","iopub.status.idle":"2022-09-01T06:58:57.098561Z","shell.execute_reply.started":"2022-09-01T06:58:57.085138Z","shell.execute_reply":"2022-09-01T06:58:57.097593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#correlation plot\n#Thier no correlation among variables.\nplt.figure(figsize=(18,8))\ncorr = X.corr()\nmask = np.triu(np.ones_like(X.corr()))\nsns.heatmap(corr , cmap = 'YlGnBu' , annot = True,fmt='.2f',mask=mask);","metadata":{"id":"bMf2uNaAkjiS","outputId":"eb2f2dce-af00-4e9c-8210-a1bd91bb7ee9","execution":{"iopub.status.busy":"2022-09-01T06:58:57.099948Z","iopub.execute_input":"2022-09-01T06:58:57.101281Z","iopub.status.idle":"2022-09-01T06:58:57.549051Z","shell.execute_reply.started":"2022-09-01T06:58:57.101232Z","shell.execute_reply":"2022-09-01T06:58:57.547914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking vif","metadata":{"id":"Gn_qYAuWPzrM"}},{"cell_type":"code","source":"import statsmodels.api as sm\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\n#X = df[list(df.columns[:-1])]\n\nvif_info = pd.DataFrame()\nvif_info['VIF'] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\nvif_info['Column'] = X.columns\nvif_info.sort_values('VIF', ascending=False)","metadata":{"id":"9nnm_AHkDNh0","outputId":"6f307f39-b118-462b-fc21-6415c9ce04c5","execution":{"iopub.status.busy":"2022-09-01T06:58:57.550560Z","iopub.execute_input":"2022-09-01T06:58:57.550928Z","iopub.status.idle":"2022-09-01T06:58:58.343112Z","shell.execute_reply.started":"2022-09-01T06:58:57.550893Z","shell.execute_reply":"2022-09-01T06:58:58.341820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"very high vif value for Area we will drop the Area variable and again check for vif value","metadata":{"id":"wL2UeU4Jk4B-"}},{"cell_type":"code","source":"X1 = X.drop(('Area'),axis=1)\nvif_info = pd.DataFrame()\nvif_info['VIF'] = [variance_inflation_factor(X1.values, i) for i in range(X1.shape[1])]\nvif_info['Column'] = X1.columns\nvif_info.sort_values('VIF', ascending=False)","metadata":{"id":"RXx6vH7rk3uN","outputId":"c14a6367-38bc-4598-bf6f-03719703298d","execution":{"iopub.status.busy":"2022-09-01T06:58:58.345202Z","iopub.execute_input":"2022-09-01T06:58:58.345680Z","iopub.status.idle":"2022-09-01T06:58:58.368482Z","shell.execute_reply.started":"2022-09-01T06:58:58.345635Z","shell.execute_reply":"2022-09-01T06:58:58.367487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2 = X1.drop(('Perimeter'),axis=1)\nvif_info = pd.DataFrame()\nvif_info['VIF'] = [variance_inflation_factor(X2.values, i) for i in range(X2.shape[1])]\nvif_info['Column'] = X2.columns\nvif_info.sort_values('VIF', ascending=False)","metadata":{"id":"CBxAMr4tQC5a","outputId":"42c23e69-37fb-474b-c920-3babec729dd9","execution":{"iopub.status.busy":"2022-09-01T06:58:58.370359Z","iopub.execute_input":"2022-09-01T06:58:58.370793Z","iopub.status.idle":"2022-09-01T06:58:58.391057Z","shell.execute_reply.started":"2022-09-01T06:58:58.370751Z","shell.execute_reply":"2022-09-01T06:58:58.389897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#correlation plot\n#Thier no correlation among variables.\nplt.figure(figsize=(18,8))\ncorr = X2.corr()\nmask = np.triu(np.ones_like(X2.corr()))\nsns.heatmap(corr , cmap = 'YlGnBu' , annot = True,fmt='.2f',mask=mask);","metadata":{"id":"nEZfe7gypPrx","outputId":"6bac2365-35a8-403f-b5e9-077f5622db9f","execution":{"iopub.status.busy":"2022-09-01T06:58:58.392717Z","iopub.execute_input":"2022-09-01T06:58:58.393099Z","iopub.status.idle":"2022-09-01T06:58:58.993692Z","shell.execute_reply.started":"2022-09-01T06:58:58.393060Z","shell.execute_reply":"2022-09-01T06:58:58.992536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **split**","metadata":{"id":"OOJ8lNEVm_-X"}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train,X_test, y_train, y_test = train_test_split(X2, y, test_size= 0.2, random_state= 1)\nprint('X_train dimension= ', X_train.shape)\nprint('X_test dimension= ', X_test.shape)\nprint('y_train dimension= ', y_train.shape)\nprint('y_train dimension= ', y_test.shape)","metadata":{"id":"Gy-FSs8-m9Hw","outputId":"5095364d-1970-44b1-f59a-29a5b1cb360d","execution":{"iopub.status.busy":"2022-09-01T06:58:58.995777Z","iopub.execute_input":"2022-09-01T06:58:58.996262Z","iopub.status.idle":"2022-09-01T06:58:59.005644Z","shell.execute_reply.started":"2022-09-01T06:58:58.996216Z","shell.execute_reply":"2022-09-01T06:58:59.004491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# fitting model","metadata":{"id":"77BtIJeluUHT"}},{"cell_type":"code","source":"\"\"\"\nTo obtain a deterministic behaviour during fitting always set value for 'random_state' attribute\nAlso note that default value of criteria to split the data is 'gini'\n\"\"\"\ncls = tree.DecisionTreeClassifier(random_state= 1)\ncls.fit(X_train ,y_train)","metadata":{"id":"VKANkmnrhEu3","outputId":"c5a64f72-b16b-478f-ae50-d7c352ebce5d","execution":{"iopub.status.busy":"2022-09-01T06:58:59.007122Z","iopub.execute_input":"2022-09-01T06:58:59.007462Z","iopub.status.idle":"2022-09-01T06:58:59.032311Z","shell.execute_reply.started":"2022-09-01T06:58:59.007431Z","shell.execute_reply":"2022-09-01T06:58:59.031433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Model Score\n\n#Check the model score using test data\n\ncls.score(X_test, y_test)","metadata":{"id":"3esGI6rHrD5N","outputId":"19def4c4-be07-4910-b743-a81be511451a","execution":{"iopub.status.busy":"2022-09-01T06:58:59.033857Z","iopub.execute_input":"2022-09-01T06:58:59.034231Z","iopub.status.idle":"2022-09-01T06:58:59.043852Z","shell.execute_reply.started":"2022-09-01T06:58:59.034197Z","shell.execute_reply":"2022-09-01T06:58:59.043032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tree.plot_tree(cls)","metadata":{"id":"NN4xLmC2rnjH","outputId":"b106d2de-c2a7-4cf1-a461-2cb0c0a8fa91","execution":{"iopub.status.busy":"2022-09-01T06:58:59.045440Z","iopub.execute_input":"2022-09-01T06:58:59.046209Z","iopub.status.idle":"2022-09-01T06:59:00.957621Z","shell.execute_reply.started":"2022-09-01T06:58:59.046173Z","shell.execute_reply":"2022-09-01T06:59:00.956311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting Decision Tree Regression to the dataset\nfrom sklearn.tree import DecisionTreeRegressor\nregressor = DecisionTreeRegressor()\nregressor.fit(X_train, y_train)\ny_pred_train = regressor.predict(X_train)","metadata":{"id":"bdYxZza_rsXv","execution":{"iopub.status.busy":"2022-09-01T06:59:00.959168Z","iopub.execute_input":"2022-09-01T06:59:00.959547Z","iopub.status.idle":"2022-09-01T06:59:00.968728Z","shell.execute_reply.started":"2022-09-01T06:59:00.959515Z","shell.execute_reply":"2022-09-01T06:59:00.967604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#confusion matrix\nfrom sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_train, y_pred_train)\nprint(cm)","metadata":{"id":"Q5-lAc0usAhn","outputId":"01e1f035-3c19-4913-ff11-c47cf15a0e37","execution":{"iopub.status.busy":"2022-09-01T06:59:00.969995Z","iopub.execute_input":"2022-09-01T06:59:00.970509Z","iopub.status.idle":"2022-09-01T06:59:00.987280Z","shell.execute_reply.started":"2022-09-01T06:59:00.970474Z","shell.execute_reply":"2022-09-01T06:59:00.985978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#confusion matrix\nfrom sklearn.metrics import confusion_matrix\n\ny_pred_test = regressor.predict(X_test)\n\ncm = confusion_matrix(y_test, y_pred_test)\nprint(cm)","metadata":{"id":"ATCk1zOMut0V","outputId":"8c74fe5b-003f-4b65-f8f1-549272510aed","execution":{"iopub.status.busy":"2022-09-01T06:59:00.989364Z","iopub.execute_input":"2022-09-01T06:59:00.990204Z","iopub.status.idle":"2022-09-01T06:59:01.000901Z","shell.execute_reply.started":"2022-09-01T06:59:00.990158Z","shell.execute_reply":"2022-09-01T06:59:00.999614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfrom sklearn.metrics import classification_report\nprint (\"Accuracy : \",accuracy_score(y_test,y_pred_test)*100)\n\t\nprint(\"Report : \",classification_report(y_test, y_pred_test))","metadata":{"id":"xgvgA_9au7JL","outputId":"85ab8e88-ada5-4df4-abe8-30ad8e53abae","execution":{"iopub.status.busy":"2022-09-01T06:59:01.002612Z","iopub.execute_input":"2022-09-01T06:59:01.003290Z","iopub.status.idle":"2022-09-01T06:59:01.017454Z","shell.execute_reply.started":"2022-09-01T06:59:01.003244Z","shell.execute_reply":"2022-09-01T06:59:01.016091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Stacking","metadata":{"id":"Qg5IzMIuwOxg"}},{"cell_type":"code","source":"!apt-get -qq install -y libarchive-dev && pip install -U libarchive\nimport libarchive","metadata":{"id":"K2eYlOO6E0YU","outputId":"ff985631-95c0-488b-865f-11dc7d462ed7","execution":{"iopub.status.busy":"2022-09-01T06:59:01.023305Z","iopub.execute_input":"2022-09-01T06:59:01.023682Z","iopub.status.idle":"2022-09-01T06:59:22.656889Z","shell.execute_reply.started":"2022-09-01T06:59:01.023650Z","shell.execute_reply":"2022-09-01T06:59:22.655423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mlrose","metadata":{"id":"7yGmhSZ6FICp","outputId":"65598c2f-fd5f-4214-ae78-bfee52f0cd1f","execution":{"iopub.status.busy":"2022-09-01T06:59:22.658581Z","iopub.execute_input":"2022-09-01T06:59:22.658983Z","iopub.status.idle":"2022-09-01T06:59:34.258572Z","shell.execute_reply.started":"2022-09-01T06:59:22.658943Z","shell.execute_reply":"2022-09-01T06:59:34.257097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import six\nimport sys\nsys.modules['sklearn.externals.six'] = six\nimport mlrose","metadata":{"id":"EdjKymkwE-_P","execution":{"iopub.status.busy":"2022-09-01T06:59:34.261080Z","iopub.execute_input":"2022-09-01T06:59:34.261567Z","iopub.status.idle":"2022-09-01T06:59:34.275537Z","shell.execute_reply.started":"2022-09-01T06:59:34.261514Z","shell.execute_reply":"2022-09-01T06:59:34.274654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport pandas as pd\nfrom mlxtend.plotting import plot_confusion_matrix\nfrom mlxtend.classifier import StackingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score","metadata":{"id":"dJLuWuGmEjlp","execution":{"iopub.status.busy":"2022-09-01T06:59:34.277552Z","iopub.execute_input":"2022-09-01T06:59:34.277981Z","iopub.status.idle":"2022-09-01T06:59:34.303061Z","shell.execute_reply.started":"2022-09-01T06:59:34.277937Z","shell.execute_reply":"2022-09-01T06:59:34.302094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"training KNeighbors Classifier","metadata":{"id":"_dGfnj8OzATG"}},{"cell_type":"code","source":"KNC = KNeighborsClassifier()   # initialising KNeighbors Classifier\nNB = GaussianNB()              # initialising Naive Bayes","metadata":{"id":"U_E9pipNvCRT","execution":{"iopub.status.busy":"2022-09-01T06:59:34.304284Z","iopub.execute_input":"2022-09-01T06:59:34.305529Z","iopub.status.idle":"2022-09-01T06:59:34.310279Z","shell.execute_reply.started":"2022-09-01T06:59:34.305490Z","shell.execute_reply":"2022-09-01T06:59:34.309462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_kNeighborsClassifier = KNC.fit(X_train, y_train)   # fitting Training Set\npred_knc = model_kNeighborsClassifier.predict(X_test)   # Predicting on test dataset","metadata":{"id":"6rbGq1JIxMmn","execution":{"iopub.status.busy":"2022-09-01T06:59:34.311686Z","iopub.execute_input":"2022-09-01T06:59:34.311989Z","iopub.status.idle":"2022-09-01T06:59:34.330477Z","shell.execute_reply.started":"2022-09-01T06:59:34.311960Z","shell.execute_reply":"2022-09-01T06:59:34.329465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_knc = accuracy_score(y_test, pred_knc)  # evaluating accuracy score\nprint('accuracy score of KNeighbors Classifier is:', acc_knc * 100)","metadata":{"id":"TeWgmGljynUk","outputId":"ea55cb1e-a1dc-46d8-8ee0-76abe9b73fe5","execution":{"iopub.status.busy":"2022-09-01T06:59:34.332050Z","iopub.execute_input":"2022-09-01T06:59:34.332632Z","iopub.status.idle":"2022-09-01T06:59:34.345101Z","shell.execute_reply.started":"2022-09-01T06:59:34.332597Z","shell.execute_reply":"2022-09-01T06:59:34.343862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training Naive Bayes Classifier","metadata":{"id":"S7MgINnyyyyQ"}},{"cell_type":"code","source":"model_NaiveBayes = NB.fit(X_train, y_train)\npred_nb = model_NaiveBayes.predict(X_test)","metadata":{"id":"PORcpvm2ytXo","execution":{"iopub.status.busy":"2022-09-01T06:59:34.346820Z","iopub.execute_input":"2022-09-01T06:59:34.347861Z","iopub.status.idle":"2022-09-01T06:59:34.360354Z","shell.execute_reply.started":"2022-09-01T06:59:34.347814Z","shell.execute_reply":"2022-09-01T06:59:34.358920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluation of Naive Bayes Classifier ","metadata":{"id":"4eimwdJdGKPM"}},{"cell_type":"code","source":"acc_nb = accuracy_score(y_test, pred_nb)\nprint('Accuracy of Naive Bayes Classifier:', acc_nb * 100)","metadata":{"id":"bVU3JVTVz6CG","outputId":"26aaed42-6b44-403a-897e-25d242e303ee","execution":{"iopub.status.busy":"2022-09-01T06:59:34.361852Z","iopub.execute_input":"2022-09-01T06:59:34.362597Z","iopub.status.idle":"2022-09-01T06:59:34.370411Z","shell.execute_reply.started":"2022-09-01T06:59:34.362551Z","shell.execute_reply":"2022-09-01T06:59:34.369388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Implementing Stacking Classifier**","metadata":{"id":"gjDPDWY8GPe7"}},{"cell_type":"code","source":"lr = LogisticRegression()  # defining meta-classifier\nclf_stack = StackingClassifier(classifiers =[KNC, NB], meta_classifier = lr, use_probas = True, use_features_in_secondary = True)","metadata":{"id":"Uo3A6F_OFhb2","execution":{"iopub.status.busy":"2022-09-01T06:59:34.375126Z","iopub.execute_input":"2022-09-01T06:59:34.375774Z","iopub.status.idle":"2022-09-01T06:59:34.380780Z","shell.execute_reply.started":"2022-09-01T06:59:34.375737Z","shell.execute_reply":"2022-09-01T06:59:34.379725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training Stacking Classifier \n\n\nmodel_stack = clf_stack.fit(X_train, y_train)   # training of stacked model\npred_stack = model_stack.predict(X_test)       # predictions on test data using stacked model","metadata":{"id":"DVV0Skyaz-4l","execution":{"iopub.status.busy":"2022-09-01T06:59:34.382360Z","iopub.execute_input":"2022-09-01T06:59:34.382859Z","iopub.status.idle":"2022-09-01T06:59:34.418584Z","shell.execute_reply.started":"2022-09-01T06:59:34.382828Z","shell.execute_reply":"2022-09-01T06:59:34.417435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluating Stacking Classifier \n\nacc_stack = accuracy_score(y_test, pred_stack)  # evaluating accuracy\nprint('accuracy score of Stacked model:', acc_stack * 100)","metadata":{"id":"vNYgGlBM94AH","outputId":"65fbf860-29c4-4424-d163-a12a39b6552d","execution":{"iopub.status.busy":"2022-09-01T06:59:34.420082Z","iopub.execute_input":"2022-09-01T06:59:34.420730Z","iopub.status.idle":"2022-09-01T06:59:34.428260Z","shell.execute_reply.started":"2022-09-01T06:59:34.420683Z","shell.execute_reply":"2022-09-01T06:59:34.427087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Method 2**","metadata":{"id":"ucu3njSqGA5k"}},{"cell_type":"code","source":"pip install vecstack","metadata":{"id":"t6fcD7tm-C0w","outputId":"949eb0a2-b2fd-43dc-cc2a-2a8cba33d465","execution":{"iopub.status.busy":"2022-09-01T06:59:34.429560Z","iopub.execute_input":"2022-09-01T06:59:34.430329Z","iopub.status.idle":"2022-09-01T06:59:45.382774Z","shell.execute_reply.started":"2022-09-01T06:59:34.430294Z","shell.execute_reply":"2022-09-01T06:59:45.381123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom vecstack import stacking","metadata":{"id":"olfH0ExE9369","execution":{"iopub.status.busy":"2022-09-01T06:59:45.385101Z","iopub.execute_input":"2022-09-01T06:59:45.386546Z","iopub.status.idle":"2022-09-01T06:59:45.549616Z","shell.execute_reply.started":"2022-09-01T06:59:45.386490Z","shell.execute_reply":"2022-09-01T06:59:45.548527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = [\n    KNeighborsClassifier(n_neighbors=5,\n                        n_jobs=-1),\n        \n    RandomForestClassifier(random_state=0, n_jobs=-1, \n                           n_estimators=100, max_depth=3),\n        ]","metadata":{"id":"y4tWUNRI5dWl","execution":{"iopub.status.busy":"2022-09-01T07:01:32.996762Z","iopub.execute_input":"2022-09-01T07:01:32.997240Z","iopub.status.idle":"2022-09-01T07:01:33.003995Z","shell.execute_reply.started":"2022-09-01T07:01:32.997200Z","shell.execute_reply":"2022-09-01T07:01:33.002628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"S_train, S_test = stacking(models,X_train, y_train, X_test,regression=False,mode='oof_pred_bag',needs_proba=False,save_dir=None,metric=accuracy_score,\n                           n_folds=4,stratified=True,shuffle=True,random_state=0,verbose=2)","metadata":{"id":"Af6WQs7F9S92","outputId":"67239917-15da-4596-dc75-aecd25b2695f","execution":{"iopub.status.busy":"2022-09-01T07:01:34.331822Z","iopub.execute_input":"2022-09-01T07:01:34.332707Z","iopub.status.idle":"2022-09-01T07:01:36.987377Z","shell.execute_reply.started":"2022-09-01T07:01:34.332661Z","shell.execute_reply":"2022-09-01T07:01:36.986186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RandomForestClassifier(random_state=0, n_jobs=-1,\n                      n_estimators=100, max_depth=3)\n    \nmodel = model.fit(S_train, y_train)\ny_pred = model.predict(S_test)\nprint('Final prediction score: [%.8f]' % accuracy_score(y_test, y_pred))","metadata":{"id":"Dnju5zrY-Mig","outputId":"9382c5a8-2cdb-43b9-ff6e-1ceb5d11e58b","execution":{"iopub.status.busy":"2022-09-01T07:03:43.883695Z","iopub.execute_input":"2022-09-01T07:03:43.884547Z","iopub.status.idle":"2022-09-01T07:03:44.240850Z","shell.execute_reply.started":"2022-09-01T07:03:43.884486Z","shell.execute_reply":"2022-09-01T07:03:44.239434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Bagging**","metadata":{"id":"NjkXrnogd2no"}},{"cell_type":"code","source":"from sklearn import model_selection\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nimport pandas as pd\n\n# load the data\n\nX = X\nY =y\n\nseed = 8\nkfold = model_selection.KFold(n_splits = 3 , shuffle = True)\n\n# initialize the base classifier\nbase_cls = DecisionTreeClassifier()\n\n# no. of base classifier\nnum_trees = 500\n\n# bagging classifier\nmodel = BaggingClassifier(base_estimator = base_cls,\n\t\t\t\t\t\tn_estimators = num_trees,\n\t\t\t\t\t\trandom_state = seed)\n\nresults = model_selection.cross_val_score(model, X, Y, cv = kfold)\nprint(\"accuracy :\")\nprint(results)\n","metadata":{"id":"yGzNByWed6bb","outputId":"0a0b59b5-9514-42a5-da27-adb28ee2eb17","execution":{"iopub.status.busy":"2022-09-01T07:03:45.929676Z","iopub.execute_input":"2022-09-01T07:03:45.930097Z","iopub.status.idle":"2022-09-01T07:03:48.672364Z","shell.execute_reply.started":"2022-09-01T07:03:45.930061Z","shell.execute_reply":"2022-09-01T07:03:48.671019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import model_selection\nfrom sklearn.ensemble import BaggingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nimport pandas as pd\n  \n# load the data\nX = X\nY = y\n  \nseed = 8\nkfold = model_selection.KFold(n_splits = 3 , shuffle = True)\n  \n# initialize the base classifier\nbase_cls = DecisionTreeClassifier()\n  \n# no. of base classifier\nnum_trees = 500\n  \n# bagging classifier\nmodel = BaggingClassifier(base_estimator = base_cls,\n                          n_estimators = num_trees,\n                          random_state = seed)\n  \nresults = model_selection.cross_val_score(model, X, Y, cv = kfold)\nprint(\"accuracy :\")\nprint(results.mean())","metadata":{"id":"H7qL7EY3iFFT","outputId":"7e057c00-7dc8-4039-e3e7-df0c07064e3d","execution":{"iopub.status.busy":"2022-09-01T07:03:48.674221Z","iopub.execute_input":"2022-09-01T07:03:48.674654Z","iopub.status.idle":"2022-09-01T07:03:51.403973Z","shell.execute_reply.started":"2022-09-01T07:03:48.674617Z","shell.execute_reply":"2022-09-01T07:03:51.402675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}