{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T14:40:34.590889Z","iopub.execute_input":"2022-08-12T14:40:34.591462Z","iopub.status.idle":"2022-08-12T14:40:34.604367Z","shell.execute_reply.started":"2022-08-12T14:40:34.591422Z","shell.execute_reply":"2022-08-12T14:40:34.602419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"1\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">IMPORTING LIBRARIES</p>","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport plotly.graph_objects as go\nfrom datetime import timedelta\nimport warnings\nwarnings.filterwarnings('ignore')\n\npalette = 'Pastel1'\n%matplotlib inline\nsns.set_style('darkgrid')\nsns.set_context(rc={\"grid.linewidth\": 3.2})\ncolors = ['#1b1341', '#3a417a', '#3d57b1', '#abb2e1', '#d3d5f2']","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:38.014499Z","iopub.execute_input":"2022-08-12T14:40:38.015316Z","iopub.status.idle":"2022-08-12T14:40:38.608191Z","shell.execute_reply.started":"2022-08-12T14:40:38.015267Z","shell.execute_reply":"2022-08-12T14:40:38.606889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ndf_test=pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\")\nss=pd.read_csv(\"../input/tabular-playground-series-aug-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:41.028202Z","iopub.execute_input":"2022-08-12T14:40:41.028750Z","iopub.status.idle":"2022-08-12T14:40:41.333349Z","shell.execute_reply.started":"2022-08-12T14:40:41.028705Z","shell.execute_reply":"2022-08-12T14:40:41.331760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Exploring Data</p>","metadata":{}},{"cell_type":"code","source":"df.head(5).style.background_gradient(cmap='Purples')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:44.162535Z","iopub.execute_input":"2022-08-12T14:40:44.164034Z","iopub.status.idle":"2022-08-12T14:40:44.285904Z","shell.execute_reply.started":"2022-08-12T14:40:44.163956Z","shell.execute_reply":"2022-08-12T14:40:44.284405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(5).style.background_gradient(cmap='Purples')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:47.077470Z","iopub.execute_input":"2022-08-12T14:40:47.078421Z","iopub.status.idle":"2022-08-12T14:40:47.139679Z","shell.execute_reply.started":"2022-08-12T14:40:47.078369Z","shell.execute_reply":"2022-08-12T14:40:47.138454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()\nprint(\"train data shape is :\",df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:50.051883Z","iopub.execute_input":"2022-08-12T14:40:50.053411Z","iopub.status.idle":"2022-08-12T14:40:50.080782Z","shell.execute_reply.started":"2022-08-12T14:40:50.053331Z","shell.execute_reply":"2022-08-12T14:40:50.079049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().any().any(),df_test.isna().any().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:54.239251Z","iopub.execute_input":"2022-08-12T14:40:54.240743Z","iopub.status.idle":"2022-08-12T14:40:54.263452Z","shell.execute_reply.started":"2022-08-12T14:40:54.240657Z","shell.execute_reply":"2022-08-12T14:40:54.261730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().style.background_gradient(cmap='Purples')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:40:57.273138Z","iopub.execute_input":"2022-08-12T14:40:57.273908Z","iopub.status.idle":"2022-08-12T14:40:57.414090Z","shell.execute_reply.started":"2022-08-12T14:40:57.273866Z","shell.execute_reply":"2022-08-12T14:40:57.412924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"3\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Product Code</p>","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1,figsize=(11,10))\nsns.histplot(x= \"product_code\", data=df, shrink=.7, palette ='Purples',hue = \"failure\", ax= ax[0]);\nsns.histplot(x= \"product_code\", data=df_test, shrink=.7, color='#abb2e1',ax=ax[1]);\nax[0].set(title='train product code')\nax[1].set(title='test product code')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:01.440490Z","iopub.execute_input":"2022-08-12T14:41:01.441562Z","iopub.status.idle":"2022-08-12T14:41:01.921345Z","shell.execute_reply.started":"2022-08-12T14:41:01.441514Z","shell.execute_reply":"2022-08-12T14:41:01.920079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Loading Distribution</p>","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1,figsize=(11,10))\nsns.histplot(x= \"loading\", data=df,hue =\"failure\",palette ='Purples', kde=True, ax= ax[0]);\nsns.histplot(x= \"loading\", data=df_test,color='#abb2e1',kde=True,ax=ax[1])\nax[0].set(title='train loading distribution ')\nax[1].set(title='test loading distrinution')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:05.962209Z","iopub.execute_input":"2022-08-12T14:41:05.962772Z","iopub.status.idle":"2022-08-12T14:41:07.448525Z","shell.execute_reply.started":"2022-08-12T14:41:05.962728Z","shell.execute_reply":"2022-08-12T14:41:07.447202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Attribute</p>","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(15,12))\nsns.histplot(x= \"attribute_0\", shrink = .4,data=df, palette ='Purples',hue =\"failure\", ax= ax[0,0])\nsns.histplot(x= \"attribute_1\", shrink = .4,data=df, palette ='Purples',hue =\"failure\", ax= ax[0,1]);\nsns.histplot(x= \"attribute_0\", shrink=2,data=df_test,color='#abb2e1',ax=ax[1,0])\nsns.histplot(x= \"attribute_1\", shrink=2,data=df_test,color='#d3d5f2',ax=ax[1,1])\nax[0,0].set(title='train attribute_0 ')\nax[0,1].set(title='train attribute_1 ')\nax[1,0].set(title='test attribute_0')\nax[1,1].set(title='test attribute_1')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:14.826139Z","iopub.execute_input":"2022-08-12T14:41:14.826717Z","iopub.status.idle":"2022-08-12T14:41:15.663800Z","shell.execute_reply.started":"2022-08-12T14:41:14.826668Z","shell.execute_reply":"2022-08-12T14:41:15.662454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(15,12))\nsns.histplot(x= \"attribute_2\", shrink = 3.5,data=df, palette ='Purples',hue =\"failure\", ax= ax[0,0])\nsns.histplot(x= \"attribute_3\", shrink = 3.5,data=df, palette ='Purples',hue =\"failure\", ax= ax[0,1]);\nsns.histplot(x= \"attribute_2\", shrink=3.5,data=df_test,color='#abb2e1',ax=ax[1,0])\nsns.histplot(x= \"attribute_3\", shrink=3.5,data=df_test,color='#d3d5f2',ax=ax[1,1])\nax[0,0].set(title='train attribute_2 ')\nax[0,1].set(title='train attribute_3 ')\nax[1,0].set(title='test attribute_2')\nax[1,1].set(title='test attribute_3')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:20.329545Z","iopub.execute_input":"2022-08-12T14:41:20.330012Z","iopub.status.idle":"2022-08-12T14:41:21.730048Z","shell.execute_reply.started":"2022-08-12T14:41:20.329974Z","shell.execute_reply":"2022-08-12T14:41:21.728849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"6\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Target Distribution</p>","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(9,5))\nsns.countplot(x= \"failure\", data=df , palette=\"Purples_r\")\nplt.title(\"Target distribution\", pad= 15)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:25.335679Z","iopub.execute_input":"2022-08-12T14:41:25.336105Z","iopub.status.idle":"2022-08-12T14:41:25.564116Z","shell.execute_reply.started":"2022-08-12T14:41:25.336073Z","shell.execute_reply":"2022-08-12T14:41:25.562568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\ndf1 = df.select_dtypes(include=numerics)\nnewdf=df1.sample(1000)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:41:28.795497Z","iopub.execute_input":"2022-08-12T14:41:28.795963Z","iopub.status.idle":"2022-08-12T14:41:28.808116Z","shell.execute_reply.started":"2022-08-12T14:41:28.795919Z","shell.execute_reply":"2022-08-12T14:41:28.806812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows = 6, ncols = 4)    # axes is 2d array (3x3)\naxes = axes.flatten()         # Convert axes to 1d array of length 9\nfig.set_size_inches(30, 20)\n\nfor ax, col in zip(axes, newdf.columns):\n  sns.distplot(newdf[col], ax = ax, color=\"#abb2e1\")\n  ax.set_title(col)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:58:21.502097Z","iopub.execute_input":"2022-08-12T14:58:21.505055Z","iopub.status.idle":"2022-08-12T14:58:27.506703Z","shell.execute_reply.started":"2022-08-12T14:58:21.504973Z","shell.execute_reply":"2022-08-12T14:58:27.505433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"7\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Correlation</p>","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\npx.data.medals_wide(indexed=True)\nfig = px.imshow(newdf.corr(),color_continuous_scale=\"Ice_r\")\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:58:59.463752Z","iopub.execute_input":"2022-08-12T14:58:59.465210Z","iopub.status.idle":"2022-08-12T14:59:00.487634Z","shell.execute_reply.started":"2022-08-12T14:58:59.465151Z","shell.execute_reply":"2022-08-12T14:59:00.486410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"8\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Train Model</p>","metadata":{}},{"cell_type":"code","source":"numerics_col_list = list(df.select_dtypes(include=['int16', 'int32', 'int64', 'float16', 'float32', 'float64']).columns)\ncat_col_list = list(set(df.columns)-set(numerics_col_list))\ndf=df.dropna() # Remove missing values ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T14:59:26.236826Z","iopub.execute_input":"2022-08-12T14:59:26.238515Z","iopub.status.idle":"2022-08-12T14:59:26.268319Z","shell.execute_reply.started":"2022-08-12T14:59:26.238448Z","shell.execute_reply":"2022-08-12T14:59:26.266727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline \n# Pipeline can be used to chain multiple estimators into one. This is useful as  there is often a fixed sequence of steps  \n# in processing the data,for example feature selection, normalization and classification.\n\nfrom sklearn.impute import SimpleImputer\n# Replace missing values using a descriptive statistic\n# (e.g. mean, median, or most frequent) along each column, or using a constant value.\n\nfrom sklearn.preprocessing import OrdinalEncoder\n# Encode categorical features as an integer array.\n\nfrom sklearn.compose import ColumnTransformer\n# Applies transformers to columns of an array or pandas DataFrame.\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:00:52.743007Z","iopub.execute_input":"2022-08-12T15:00:52.743744Z","iopub.status.idle":"2022-08-12T15:00:52.872000Z","shell.execute_reply.started":"2022-08-12T15:00:52.743679Z","shell.execute_reply":"2022-08-12T15:00:52.870250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_trans = Pipeline(steps = [('imputer', SimpleImputer(strategy='mean'))])\ncat_trans = Pipeline(steps = [('imputer', SimpleImputer(strategy = 'most_frequent')),\n                                (('odi', (OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value = np.nan))))])\ntransformer = ColumnTransformer(transformers=[(\"num\", num_trans, numerics_col_list),\n                                           (\"cat\", cat_trans, cat_col_list)])\ndf = pd.DataFrame(transformer.fit_transform(df), columns = (numerics_col_list+cat_col_list))\ny=df['failure']\nX=df.drop(['failure'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:00:56.925815Z","iopub.execute_input":"2022-08-12T15:00:56.926392Z","iopub.status.idle":"2022-08-12T15:00:56.984960Z","shell.execute_reply.started":"2022-08-12T15:00:56.926343Z","shell.execute_reply":"2022-08-12T15:00:56.983297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_test, y_train, y_test = train_test_split( X, y, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:01:00.481794Z","iopub.execute_input":"2022-08-12T15:01:00.482353Z","iopub.status.idle":"2022-08-12T15:01:00.496725Z","shell.execute_reply.started":"2022-08-12T15:01:00.482308Z","shell.execute_reply":"2022-08-12T15:01:00.494869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"9\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Logistic Regression</p>","metadata":{}},{"cell_type":"markdown","source":"> *Logistic regression estimates the probability of an event occurring, such as voted or didn't vote, based on a given dataset of independent variables.*\n> *Since the outcome is a probability, the dependent variable is bounded between 0 and 1.*\n> * For example, for classifying an email, the algorithm will use the words in the email as features and based on that make a prediction whether the email is spam or not. ","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nlr=LogisticRegression(max_iter=1000)\nlr.fit(x_train, y_train)\n\nprint(\"score on test: \" + str(lr.score(x_test, y_test)))\nprint(\"score on train: \"+ str(lr.score(x_train, y_train)))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:04:51.310796Z","iopub.execute_input":"2022-08-12T15:04:51.311348Z","iopub.status.idle":"2022-08-12T15:04:51.462842Z","shell.execute_reply.started":"2022-08-12T15:04:51.311310Z","shell.execute_reply":"2022-08-12T15:04:51.460847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"10\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">K-Nearest Neighbour</p>","metadata":{}},{"cell_type":"markdown","source":"> * K-Nearest Neighbour is one of the simplest Machine Learning algorithms based on Supervised Learning technique.\n> * K-NN algorithm assumes the similarity between the new case/data and available cases and put the new case into the category that is most similar to the available categories.\n> * K-NN algorithm stores all the available data and classifies a new data point based on the similarity. This means when new data appears then it can be easily classified into a well suite category by using K- NN algorithm.\n> * K-NN algorithm can be used for Regression as well as for Classification but mostly it is used for the Classification problems.","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier(algorithm = 'brute', n_neighbors=2)\nknn.fit(x_train, y_train)\n\nprint(\"train shape: \" + str(x_train.shape))\nprint(\"score on test: \" + str(knn.score(x_test, y_test)))\nprint(\"score on train: \"+ str(knn.score(x_train, y_train)))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:04:59.985611Z","iopub.execute_input":"2022-08-12T15:04:59.987167Z","iopub.status.idle":"2022-08-12T15:05:02.259175Z","shell.execute_reply.started":"2022-08-12T15:04:59.987075Z","shell.execute_reply":"2022-08-12T15:05:02.257599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"11\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Decision Tree Classifier</p>","metadata":{}},{"cell_type":"markdown","source":"> * Decision Tree is a Supervised learning technique that can be used for both classification and Regression problems, but mostly it is preferred for solving Classification problems. It is a tree-structured classifier, where internal nodes represent the features of a dataset, branches represent the decision rules and each leaf node represents the outcome.\n> * In a Decision tree, there are two nodes, which are the Decision Node and Leaf Node. Decision nodes are used to make any decision and have multiple branches, whereas Leaf nodes are the output of those decisions and do not contain any further branches.\n> * A decision tree simply asks a question, and based on the answer (Yes/No), it further split the tree into subtrees.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nclf = DecisionTreeClassifier()\nclf.fit(x_train, y_train)\n\nprint(\"train shape: \" + str(x_train.shape))\nprint(\"score on test: \"  + str(clf.score(x_test, y_test)))\nprint(\"score on train: \" + str(clf.score(x_train, y_train)))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:05:08.386022Z","iopub.execute_input":"2022-08-12T15:05:08.387026Z","iopub.status.idle":"2022-08-12T15:05:08.789342Z","shell.execute_reply.started":"2022-08-12T15:05:08.386963Z","shell.execute_reply":"2022-08-12T15:05:08.787835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:05:12.110365Z","iopub.execute_input":"2022-08-12T15:05:12.110858Z","iopub.status.idle":"2022-08-12T15:05:12.122029Z","shell.execute_reply.started":"2022-08-12T15:05:12.110817Z","shell.execute_reply":"2022-08-12T15:05:12.120531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"12\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Confusion Matrix For Decision Tree</p>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nresult = confusion_matrix(y_test, y_pred)\nprint(\"Confusion Matrix:\")\nprint(result)\nresult1 = classification_report(y_test, y_pred)\nprint(\"Classification Report:\",)\nprint (result1)\nresult2 = accuracy_score(y_test,y_pred)\nprint(\"Accuracy:\",result2)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:05:16.979155Z","iopub.execute_input":"2022-08-12T15:05:16.980503Z","iopub.status.idle":"2022-08-12T15:05:17.007438Z","shell.execute_reply.started":"2022-08-12T15:05:16.980447Z","shell.execute_reply":"2022-08-12T15:05:17.005891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"13\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Random Forest</p>","metadata":{}},{"cell_type":"markdown","source":"> * Random forest is a supervised learning algorithm which is used for both classification as well as regression. But however, it is mainly used for classification problems. \n> * As we know that a forest is made up of trees and more trees means more robust forest. \n> * Similarly, random forest algorithm creates decision trees on data samples and then gets the prediction from each of them and finally selects the best solution by means of voting.\n> * It is an ensemble method which is better than a single decision tree because it reduces the over-fitting by averaging the result.","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nclassifier = RandomForestClassifier(n_estimators = 50)\nclassifier.fit(x_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:06:20.828822Z","iopub.execute_input":"2022-08-12T15:06:20.829301Z","iopub.status.idle":"2022-08-12T15:06:23.389865Z","shell.execute_reply.started":"2022-08-12T15:06:20.829265Z","shell.execute_reply":"2022-08-12T15:06:23.388514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = classifier.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:06:25.869818Z","iopub.execute_input":"2022-08-12T15:06:25.870244Z","iopub.status.idle":"2022-08-12T15:06:25.926212Z","shell.execute_reply.started":"2022-08-12T15:06:25.870212Z","shell.execute_reply":"2022-08-12T15:06:25.924932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"14\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">Confusion Matrix For Random Forest</p>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nresult = confusion_matrix(y_test, y_pred)\nprint(\"Confusion Matrix:\")\nprint(result)\nresult1 = classification_report(y_test, y_pred)\nprint(\"Classification Report:\",)\nprint (result1)\nresult2 = accuracy_score(y_test,y_pred)\nprint(\"Accuracy:\",result2)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T15:06:28.294951Z","iopub.execute_input":"2022-08-12T15:06:28.295379Z","iopub.status.idle":"2022-08-12T15:06:28.322548Z","shell.execute_reply.started":"2022-08-12T15:06:28.295345Z","shell.execute_reply":"2022-08-12T15:06:28.321180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"15\"></a>\n# <p style=\"background-color:#3a417a;font-family:newtimeroman;color:#FFF9ED;font-size:150%;text-align:center;border-radius:10px 10px;\">End</p>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}