{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Setup and Initialization**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T17:54:30.281189Z","iopub.execute_input":"2022-08-11T17:54:30.281574Z","iopub.status.idle":"2022-08-11T17:54:30.292183Z","shell.execute_reply.started":"2022-08-11T17:54:30.281545Z","shell.execute_reply":"2022-08-11T17:54:30.290958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Collecting the data**\nLoad the train and test dataset from csv file to pandas dataframe","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\n\n# Printing the first 5 rows of the train dataset.\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.294159Z","iopub.execute_input":"2022-08-11T17:54:30.294962Z","iopub.status.idle":"2022-08-11T17:54:30.328533Z","shell.execute_reply.started":"2022-08-11T17:54:30.294926Z","shell.execute_reply":"2022-08-11T17:54:30.327221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Analysis**","metadata":{}},{"cell_type":"code","source":"# Printing the Total rows and columns of training data\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.330907Z","iopub.execute_input":"2022-08-11T17:54:30.332107Z","iopub.status.idle":"2022-08-11T17:54:30.339112Z","shell.execute_reply.started":"2022-08-11T17:54:30.332032Z","shell.execute_reply":"2022-08-11T17:54:30.337729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting some information about the training data\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.340967Z","iopub.execute_input":"2022-08-11T17:54:30.341356Z","iopub.status.idle":"2022-08-11T17:54:30.362342Z","shell.execute_reply.started":"2022-08-11T17:54:30.341324Z","shell.execute_reply":"2022-08-11T17:54:30.361117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so, we can observe that there are some missing values in the table.\nfor example, Age column is missing for many rows.Out of 891 rows, the Age value is present only in 714 rows. ","metadata":{}},{"cell_type":"code","source":"# checking the nimber of the missing data in each column [for Training data]\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.365395Z","iopub.execute_input":"2022-08-11T17:54:30.366515Z","iopub.status.idle":"2022-08-11T17:54:30.378626Z","shell.execute_reply.started":"2022-08-11T17:54:30.366465Z","shell.execute_reply":"2022-08-11T17:54:30.377395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting some statistical measure about the data\ntrain_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.380209Z","iopub.execute_input":"2022-08-11T17:54:30.381195Z","iopub.status.idle":"2022-08-11T17:54:30.422877Z","shell.execute_reply.started":"2022-08-11T17:54:30.381135Z","shell.execute_reply":"2022-08-11T17:54:30.422026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Cleanup Data**","metadata":{}},{"cell_type":"code","source":"# deleting unnecessary columns from the dataframe\ntrain_data.drop(['Name','PassengerId','Ticket'], axis=1, inplace=True)\n\n# deleting corrupted ( many missing values) columns from the dataframe\ntrain_data.drop(['Cabin'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.424016Z","iopub.execute_input":"2022-08-11T17:54:30.424817Z","iopub.status.idle":"2022-08-11T17:54:30.432751Z","shell.execute_reply.started":"2022-08-11T17:54:30.424784Z","shell.execute_reply":"2022-08-11T17:54:30.431476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.434528Z","iopub.execute_input":"2022-08-11T17:54:30.434993Z","iopub.status.idle":"2022-08-11T17:54:30.454892Z","shell.execute_reply.started":"2022-08-11T17:54:30.434951Z","shell.execute_reply":"2022-08-11T17:54:30.454061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Chart information for Features**","metadata":{}},{"cell_type":"code","source":"# making a chart for survived\nimport seaborn as sns\nsns.countplot('Survived', data = train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.456226Z","iopub.execute_input":"2022-08-11T17:54:30.456549Z","iopub.status.idle":"2022-08-11T17:54:30.629402Z","shell.execute_reply.started":"2022-08-11T17:54:30.456518Z","shell.execute_reply":"2022-08-11T17:54:30.628144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"sns.countplot('Sex',hue='Survived', data= train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.632742Z","iopub.execute_input":"2022-08-11T17:54:30.633132Z","iopub.status.idle":"2022-08-11T17:54:30.834907Z","shell.execute_reply.started":"2022-08-11T17:54:30.633098Z","shell.execute_reply":"2022-08-11T17:54:30.833571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Chart shows that  Women more likely survivied than Men","metadata":{}},{"cell_type":"code","source":"sns.countplot('Pclass',hue='Survived', data= train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.836346Z","iopub.execute_input":"2022-08-11T17:54:30.836739Z","iopub.status.idle":"2022-08-11T17:54:30.994584Z","shell.execute_reply.started":"2022-08-11T17:54:30.836706Z","shell.execute_reply":"2022-08-11T17:54:30.993332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The Chart shows that the **1st class** more likely survivied than other classes\n* The Chart shows that the  **3rd class** more likely dead than other classes","metadata":{}},{"cell_type":"code","source":"sns.countplot('Embarked',hue='Survived', data= train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:30.996190Z","iopub.execute_input":"2022-08-11T17:54:30.997152Z","iopub.status.idle":"2022-08-11T17:54:31.159597Z","shell.execute_reply.started":"2022-08-11T17:54:30.997105Z","shell.execute_reply":"2022-08-11T17:54:31.158386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The Chart confirms a person aboarded from C slightly more likely survived\n* The Chart confirms a person aboarded from Q more likely dead\n* The Chart confirms a person aboarded from S more likely dead","metadata":{}},{"cell_type":"markdown","source":"# **Replacing the missing values**","metadata":{}},{"cell_type":"code","source":"# replacing the missing values in age column with mean value\n\ntrain_data['Age'].fillna(train_data['Age'].mean() , inplace=True )\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.161617Z","iopub.execute_input":"2022-08-11T17:54:31.162063Z","iopub.status.idle":"2022-08-11T17:54:31.176755Z","shell.execute_reply.started":"2022-08-11T17:54:31.162000Z","shell.execute_reply":"2022-08-11T17:54:31.175559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# printing the most repeated value and its frist index\nprint(train_data['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.178808Z","iopub.execute_input":"2022-08-11T17:54:31.179322Z","iopub.status.idle":"2022-08-11T17:54:31.188662Z","shell.execute_reply.started":"2022-08-11T17:54:31.179286Z","shell.execute_reply":"2022-08-11T17:54:31.187560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values with the most repeated value\ntrain_data['Embarked'].fillna(train_data['Embarked'].mode()[0] , inplace=True )","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.190316Z","iopub.execute_input":"2022-08-11T17:54:31.191225Z","iopub.status.idle":"2022-08-11T17:54:31.200195Z","shell.execute_reply.started":"2022-08-11T17:54:31.191180Z","shell.execute_reply":"2022-08-11T17:54:31.198787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.201597Z","iopub.execute_input":"2022-08-11T17:54:31.202290Z","iopub.status.idle":"2022-08-11T17:54:31.214139Z","shell.execute_reply.started":"2022-08-11T17:54:31.202256Z","shell.execute_reply":"2022-08-11T17:54:31.213100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we have no missing values in any column.","metadata":{}},{"cell_type":"markdown","source":"# **Replacing text to numerical**","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.215518Z","iopub.execute_input":"2022-08-11T17:54:31.216090Z","iopub.status.idle":"2022-08-11T17:54:31.234299Z","shell.execute_reply.started":"2022-08-11T17:54:31.216024Z","shell.execute_reply":"2022-08-11T17:54:31.233033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing male with 0 and female with 1\ntrain_data.replace({'Sex':{'male':0,'female':1}}, inplace=True)\n\n# replacing male with 0 and female with 1\ntrain_data.replace({'Embarked':{'S':0,'C':1,'Q':3}}, inplace=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.235973Z","iopub.execute_input":"2022-08-11T17:54:31.236473Z","iopub.status.idle":"2022-08-11T17:54:31.249657Z","shell.execute_reply.started":"2022-08-11T17:54:31.236429Z","shell.execute_reply":"2022-08-11T17:54:31.248630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.251174Z","iopub.execute_input":"2022-08-11T17:54:31.251604Z","iopub.status.idle":"2022-08-11T17:54:31.271701Z","shell.execute_reply.started":"2022-08-11T17:54:31.251563Z","shell.execute_reply":"2022-08-11T17:54:31.270640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preper the model**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\nX = train_data.drop('Survived', axis=1)\ny = train_data['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.273277Z","iopub.execute_input":"2022-08-11T17:54:31.273890Z","iopub.status.idle":"2022-08-11T17:54:31.281546Z","shell.execute_reply.started":"2022-08-11T17:54:31.273855Z","shell.execute_reply":"2022-08-11T17:54:31.280489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Cross Validation**","metadata":{}},{"cell_type":"code","source":"# K-fold : new concept that i leared in this project form this documentation:\n# https://scikit-learn.org/stable/modules/generated/sklearn.model_selection.KFold.html\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import cross_val_score\nk_fold = KFold(n_splits=10, shuffle=True, random_state=0)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.282930Z","iopub.execute_input":"2022-08-11T17:54:31.283921Z","iopub.status.idle":"2022-08-11T17:54:31.294931Z","shell.execute_reply.started":"2022-08-11T17:54:31.283873Z","shell.execute_reply":"2022-08-11T17:54:31.293816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Decision Tree Classifier**","metadata":{}},{"cell_type":"code","source":"tree = DecisionTreeClassifier()\nscoring = 'accuracy'\nscore = cross_val_score(tree, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.296695Z","iopub.execute_input":"2022-08-11T17:54:31.297448Z","iopub.status.idle":"2022-08-11T17:54:31.373616Z","shell.execute_reply.started":"2022-08-11T17:54:31.297402Z","shell.execute_reply":"2022-08-11T17:54:31.372381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.374865Z","iopub.execute_input":"2022-08-11T17:54:31.375435Z","iopub.status.idle":"2022-08-11T17:54:31.381665Z","shell.execute_reply.started":"2022-08-11T17:54:31.375392Z","shell.execute_reply":"2022-08-11T17:54:31.380809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Nearest Neighbors Classifier**","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 9)\nscoring = 'accuracy'\nscore = cross_val_score(knn, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.386809Z","iopub.execute_input":"2022-08-11T17:54:31.387178Z","iopub.status.idle":"2022-08-11T17:54:31.484606Z","shell.execute_reply.started":"2022-08-11T17:54:31.387140Z","shell.execute_reply":"2022-08-11T17:54:31.483454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.485955Z","iopub.execute_input":"2022-08-11T17:54:31.486328Z","iopub.status.idle":"2022-08-11T17:54:31.494251Z","shell.execute_reply.started":"2022-08-11T17:54:31.486295Z","shell.execute_reply":"2022-08-11T17:54:31.493021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Random Forest Classifier**","metadata":{}},{"cell_type":"code","source":"rand = RandomForestClassifier(n_estimators=12)\nscoring = 'accuracy'\nscore = cross_val_score(rand, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.496216Z","iopub.execute_input":"2022-08-11T17:54:31.496934Z","iopub.status.idle":"2022-08-11T17:54:31.814655Z","shell.execute_reply.started":"2022-08-11T17:54:31.496882Z","shell.execute_reply":"2022-08-11T17:54:31.813526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.816158Z","iopub.execute_input":"2022-08-11T17:54:31.816599Z","iopub.status.idle":"2022-08-11T17:54:31.824363Z","shell.execute_reply.started":"2022-08-11T17:54:31.816555Z","shell.execute_reply":"2022-08-11T17:54:31.823109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Naive Bayes Classifier**","metadata":{}},{"cell_type":"code","source":"naive = GaussianNB()\nscoring = 'accuracy'\nscore = cross_val_score(naive, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.825987Z","iopub.execute_input":"2022-08-11T17:54:31.826440Z","iopub.status.idle":"2022-08-11T17:54:31.889580Z","shell.execute_reply.started":"2022-08-11T17:54:31.826398Z","shell.execute_reply":"2022-08-11T17:54:31.888444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.891306Z","iopub.execute_input":"2022-08-11T17:54:31.891761Z","iopub.status.idle":"2022-08-11T17:54:31.899404Z","shell.execute_reply.started":"2022-08-11T17:54:31.891706Z","shell.execute_reply":"2022-08-11T17:54:31.898458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Support vector machine Classifier**","metadata":{}},{"cell_type":"code","source":"support= SVC()\nscoring = 'accuracy'\nscore = cross_val_score(support, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:31.901142Z","iopub.execute_input":"2022-08-11T17:54:31.901514Z","iopub.status.idle":"2022-08-11T17:54:32.243241Z","shell.execute_reply.started":"2022-08-11T17:54:31.901467Z","shell.execute_reply":"2022-08-11T17:54:32.242055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:32.245206Z","iopub.execute_input":"2022-08-11T17:54:32.245955Z","iopub.status.idle":"2022-08-11T17:54:32.253919Z","shell.execute_reply.started":"2022-08-11T17:54:32.245907Z","shell.execute_reply":"2022-08-11T17:54:32.252623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Logistic Regression Classifier**","metadata":{}},{"cell_type":"code","source":"#source : https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.LogisticRegression.html\nreg=LogisticRegression(max_iter=150)\nscoring = 'accuracy'\nscore = cross_val_score(reg, X, y, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:32.255681Z","iopub.execute_input":"2022-08-11T17:54:32.256033Z","iopub.status.idle":"2022-08-11T17:54:32.627052Z","shell.execute_reply.started":"2022-08-11T17:54:32.256002Z","shell.execute_reply":"2022-08-11T17:54:32.625791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"round(np.mean(score), 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:32.628495Z","iopub.execute_input":"2022-08-11T17:54:32.628853Z","iopub.status.idle":"2022-08-11T17:54:32.636769Z","shell.execute_reply.started":"2022-08-11T17:54:32.628819Z","shell.execute_reply":"2022-08-11T17:54:32.635514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***the most accurate classifier is RandomForestClassifier with accuracy of 81%***","metadata":{}},{"cell_type":"markdown","source":"# **Another Way to implement a classifier**","metadata":{}},{"cell_type":"code","source":"# Implementing on Logistic Regression Classifier. this way can run on any previous classifier\n# source : https://scikit-learn.org/stable/modules/generated/sklearn.model_selection.train_test_split.html\n\n# spliting the data\nX_train , X_test , y_train , y_test = train_test_split(X,y,test_size=0.2, random_state=2)\n\nlog_reg =LogisticRegression()\n#traing the the model with training data\nlog_reg.fit(X_train,y_train)\n\n# prediction the train data\nX_train_pred=log_reg.predict(X_train)\n\n# acuracy of the model \naccuracy_model_train = accuracy_score(y_train,X_train_pred)\n#accuracy_model_train= round(np.mean(accuracy_model_train), 2)\nprint(accuracy_model_train)\n\n# prediction the test data\nX_test_pred=log_reg.predict(X_test)\n\n# acuracy of the model \naccuracy_model_test = accuracy_score(y_test,X_test_pred)\n#accuracy_model_test= round(np.mean(accuracy_model_test), 2)\nprint(accuracy_model_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:32.638704Z","iopub.execute_input":"2022-08-11T17:54:32.639084Z","iopub.status.idle":"2022-08-11T17:54:32.688195Z","shell.execute_reply.started":"2022-08-11T17:54:32.639051Z","shell.execute_reply":"2022-08-11T17:54:32.686954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"observation : when changing random_state , accuracy will change.\nEx: \nrandom_state=1\noutput :\n0.7991573033707865\n0.7988826815642458","metadata":{}},{"cell_type":"markdown","source":"# **Deep Neural Networks**","metadata":{}},{"cell_type":"code","source":"#source1: https://www.kaggle.com/code/ryanholbrook/binary-classification\n#source2: https://www.kaggle.com/code/ahmedgabualnoor/exercise-binary-classification/edit\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\ninput_shape = [X_train.shape[1]]\n#Building Sequential Models\nmodel = keras.Sequential([\n    layers.BatchNormalization(input_shape=input_shape),\n    # the hidden ReLU layers\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(256, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    # the linear output layer \n    layers.Dense(1,activation='sigmoid'),\n])\n\n#add a loss function and optimizer with the model's compile\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy'],\n)\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    # how many epochs to wait before stopping\n    patience=10,\n    # minimium amount of change to count as an improvement\n    min_delta=0.001,\n    restore_best_weights=True,\n)\n\n# Now we're ready to start the training! We've told Keras to feed the optimizer 512 rows of the training data \n# at a time (the batch_size) and to do that 200 times all the way through the dataset (the epochs).\nhistory = model.fit(\n    X_train, y_train,\n    validation_data=(X_test, y_test),\n    batch_size=512,\n    epochs=200,\n    callbacks=[early_stopping],\n)\n\nhistory_df = pd.DataFrame(history.history)\n\nprint((\"Best Validation Loss: {:0.4f}\" +\\\n      \"\\nBest Validation Accuracy: {:0.4f}\")\\\n      .format(history_df['val_loss'].min(), \n              history_df['val_binary_accuracy'].max()))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:54:32.690074Z","iopub.execute_input":"2022-08-11T17:54:32.690519Z","iopub.status.idle":"2022-08-11T17:54:35.273326Z","shell.execute_reply.started":"2022-08-11T17:54:32.690487Z","shell.execute_reply":"2022-08-11T17:54:35.272125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Module Results**\n* Decision Tree Classifier         (0.77)\n* Nearest Neighbors Classifier     (0.73)\n* ***Ramdom Forest Classifier         (0.81)***\n* Naive Bayes Classifier           (0.79)\n* Support vector machine Classifier (0.68)\n* Logistic Regression Classifier (0.80)\n\n\n\n\n","metadata":{}}]}