{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# In this kernel i gonna try to do some feature analysis and compare between different classifiers \n# on dataset from Titanic - Machine Learning from Disaster Competition","metadata":{}},{"cell_type":"markdown","source":"* so our goal is to use machine learning to create a model that predicts which passengers survived the Titanic shipwreck\nusing data set saved in these csv files\n* training set (train.csv)\n* test set (test.csv)\n","metadata":{}},{"cell_type":"markdown","source":"more about the data \nhttps://www.kaggle.com/competitions/titanic/data","metadata":{}},{"cell_type":"markdown","source":"that is todo list to follow along this kernel\n* feature engineering , reduce features taken in training model , based on\n* number of unique categorical variables\n* number of missing elements in each feature\n* using correlation relation with that feature and trarget\n* fill missing entries in columns\n* use one-hot encoding for categorical columns\n* use the processed DataFrame in traning ,and get accuracy of the model\n* try to tune training model with parameters to get best results\n* try other training model\n* get prediction using the model with best accuracy and submit it to competition","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random as rnd\n\n#visulaization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#models\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC, LinearSVC\nfrom xgboost import XGBClassifier\nfrom sklearn.neural_network import MLPClassifier\n\n#files locations\n#/kaggle/input/titanic/train.csv\n#/kaggle/input/titanic/test.csv\n#/kaggle/input/titanic/gender_submission.csv\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:47.901473Z","iopub.execute_input":"2022-08-11T08:54:47.902537Z","iopub.status.idle":"2022-08-11T08:54:47.913047Z","shell.execute_reply.started":"2022-08-11T08:54:47.902488Z","shell.execute_reply":"2022-08-11T08:54:47.912218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loading training/test dataset from csv files\ntrain_df=pd.read_csv('/kaggle/input/titanic/train.csv')\ntest_df=pd.read_csv('/kaggle/input/titanic/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.019286Z","iopub.execute_input":"2022-08-11T08:54:48.020029Z","iopub.status.idle":"2022-08-11T08:54:48.046618Z","shell.execute_reply.started":"2022-08-11T08:54:48.019989Z","shell.execute_reply":"2022-08-11T08:54:48.045712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check features inside train_df\nprint(train_df.columns)\n#set PassengerId as index\ntrain_df.set_index('PassengerId',inplace=True)\ntest_df.set_index('PassengerId',inplace=True)\n\n#categorical columns : survived(0,1) , Pclass(1,2,3),Sex(male,female),Embarked(C,Q,S)\n#numerical columns : Age,SibSp,Parch ,Fare\n#mix columns : Ticket,Cabin\n#drop these ? : PassengerId , Name","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.226273Z","iopub.execute_input":"2022-08-11T08:54:48.226688Z","iopub.status.idle":"2022-08-11T08:54:48.235020Z","shell.execute_reply.started":"2022-08-11T08:54:48.226652Z","shell.execute_reply":"2022-08-11T08:54:48.233775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()\nprint('_'*40)\ntest_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.335322Z","iopub.execute_input":"2022-08-11T08:54:48.336066Z","iopub.status.idle":"2022-08-11T08:54:48.361133Z","shell.execute_reply.started":"2022-08-11T08:54:48.336004Z","shell.execute_reply":"2022-08-11T08:54:48.359896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"running that code we gonna get breif info about the training data\n* Cabin has 204 non-null out of 891 entries , so there is 687 entries that are missing\n* Age has 714 non-null out of 891 entries , so there is 177 entries that are missing\n* Embarked has 889 non-null out of 891 entries , so only 2 entries that are missing\n\nand we gonna run the same on testing data and we found that\n* Fare has 417 out of 418 , so only one entriy is missing\n* same observation as traning data set , about Cabin and Age and Embarked\n\nso after seeing these observation we gonna\n* exclude Cabin as it has lots of missing entries\n* fill missing entries in Embarked column\n* fill the missing entries in numberical columns like Age and Fare","metadata":{}},{"cell_type":"code","source":"train_df.describe(include=['O'])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.456305Z","iopub.execute_input":"2022-08-11T08:54:48.457057Z","iopub.status.idle":"2022-08-11T08:54:48.480516Z","shell.execute_reply.started":"2022-08-11T08:54:48.456996Z","shell.execute_reply":"2022-08-11T08:54:48.479167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* running the line of code down there will get breif info about the number of unique elements in categorical columns , \n* and this will show to us the features that will be suitable for encoding either using one-hot-encoding or ordinal encoding\n\nand after running the code we find that\n* features like Embarked, Sex have low unique entries and these are good candidate for further preprocessing , using one-hot-encoding\n* Name has 891 of unique entries out of 891 and it won't be good idea to use that feature as it , we might do some preprocessing on it\n* but i choosed to drop it as Name might not have huge impact on survival rate\n* Ticket feature has lots of unique entries so we might do some preprocessing on it as well\n* feature like Cabin we decided to drop it as it has lots of missing elements as we showed that before","metadata":{}},{"cell_type":"markdown","source":"# Discovering feature correlation with target feature Survived","metadata":{}},{"cell_type":"markdown","source":"**in this section i gonna try to find the correlation between each feature and Survived trarget**","metadata":{}},{"cell_type":"code","source":"#get table with average Survived value for each entry in feature\ndef get_survived_table(feature):\n    table=train_df[[feature,'Survived']].groupby([feature],as_index=False).mean().sort_values(by='Survived',ascending=False)\n    print(table)\n    print('-'*20)\n    \n#plot bar_plot of feature , it's same as get_survived_table but using plots\ndef get_survived_bar_plot(feature):\n    plt.figure(figsize = (6,4))\n    sns.barplot(data = train_df , x = feature , y = \"Survived\").set_title(f\"{feature} Vs Survived\")\n    plt.show()\n    \n#plot histogram of features according to the 2 values of Survived\ndef get_survived_histogram(feature):\n    g=sns.FacetGrid(train_df,col='Survived')\n    g.map(plt.hist,feature,bins=20)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.624414Z","iopub.execute_input":"2022-08-11T08:54:48.624851Z","iopub.status.idle":"2022-08-11T08:54:48.633249Z","shell.execute_reply.started":"2022-08-11T08:54:48.624815Z","shell.execute_reply":"2022-08-11T08:54:48.632007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_survived_table('Sex')\nget_survived_bar_plot('Sex')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.745576Z","iopub.execute_input":"2022-08-11T08:54:48.746338Z","iopub.status.idle":"2022-08-11T08:54:48.949975Z","shell.execute_reply.started":"2022-08-11T08:54:48.746290Z","shell.execute_reply":"2022-08-11T08:54:48.949048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we find from that plot and table taht female has higher survival rate with 74% , in other hand men have 18%\n* so we will choose Sex feature as it's correlate (contribute) in Survival rate","metadata":{}},{"cell_type":"code","source":"get_survived_table('Embarked')\nget_survived_bar_plot('Embarked')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:48.952044Z","iopub.execute_input":"2022-08-11T08:54:48.953202Z","iopub.status.idle":"2022-08-11T08:54:49.179468Z","shell.execute_reply.started":"2022-08-11T08:54:48.953154Z","shell.execute_reply":"2022-08-11T08:54:49.178036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we find that people who proted from C have higher survival rate ,\n* C with 55% survival rate\n* and Q with 38% survival rate\n* and S with 33% survival rate\n* so we will choose Embarked feature as it's correlate (contribute) in Survival rate","metadata":{}},{"cell_type":"code","source":"get_survived_table('Pclass')\nget_survived_bar_plot('Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:49.182940Z","iopub.execute_input":"2022-08-11T08:54:49.183705Z","iopub.status.idle":"2022-08-11T08:54:49.413231Z","shell.execute_reply.started":"2022-08-11T08:54:49.183658Z","shell.execute_reply":"2022-08-11T08:54:49.412332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we find that people who are from Pclass 1 have higher survival rate ,\n* Pclass=1 with 62% survival rate\n* Pclass=2 with 47% survival rate\n* Pclass=3 with 24% survival rate\n* so wealthy people have more survival rate\n* so we will choose Pclass feature as it's correlate (contribute) in Survival rate","metadata":{}},{"cell_type":"code","source":"get_survived_table('SibSp')\nget_survived_bar_plot('SibSp')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:49.416817Z","iopub.execute_input":"2022-08-11T08:54:49.417188Z","iopub.status.idle":"2022-08-11T08:54:49.774666Z","shell.execute_reply.started":"2022-08-11T08:54:49.417155Z","shell.execute_reply":"2022-08-11T08:54:49.773391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we find that SibSp has zero correlation for some values so we could make useful feature out of it\n* try to make new feature out of it so all of it's values correlate with Survival rate","metadata":{}},{"cell_type":"code","source":"get_survived_table('Parch')\nget_survived_bar_plot('Parch')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:49.777162Z","iopub.execute_input":"2022-08-11T08:54:49.777525Z","iopub.status.idle":"2022-08-11T08:54:50.164637Z","shell.execute_reply.started":"2022-08-11T08:54:49.777486Z","shell.execute_reply":"2022-08-11T08:54:50.163451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we find that Parch has zero correlation for some values so we could make useful feature out of it\n* try to make new feature out of it so all of it's values correlate with Survival rate","metadata":{}},{"cell_type":"code","source":"get_survived_histogram('Age')\nget_survived_histogram('Fare')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:50.167133Z","iopub.execute_input":"2022-08-11T08:54:50.168352Z","iopub.status.idle":"2022-08-11T08:54:51.367710Z","shell.execute_reply.started":"2022-08-11T08:54:50.168304Z","shell.execute_reply":"2022-08-11T08:54:51.366524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* we see that theres is some correlation with both Age ,and Fare features\n* so we gonna continue using those 2 features in making the model","metadata":{}},{"cell_type":"markdown","source":"# Discovering correlation features between other features using heat map","metadata":{}},{"cell_type":"code","source":"sns.set(rc = {'figure.figsize':(10,6)})\nsns.heatmap(train_df.corr(), annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.371445Z","iopub.execute_input":"2022-08-11T08:54:51.371882Z","iopub.status.idle":"2022-08-11T08:54:51.797057Z","shell.execute_reply.started":"2022-08-11T08:54:51.371837Z","shell.execute_reply":"2022-08-11T08:54:51.795880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* noticing the heatmap , we get that\n* Pclass has high correlation with Fare ,and Age and Survived\n* SibSp has high correlation with Age\n* Fare has small correlation with Survived , we might drop it as it has its features in Pclass and Age","metadata":{}},{"cell_type":"code","source":"#i tried to creat new feature but found out that it gets worest results\n# #making new feature from Parch and SibSp , FamilyMemebers = Parc+SibSp\n# train_df['FamilyM']=train_df['Parch']+train_df['SibSp']\n# test_df['FamilyM']=test_df['Parch']+test_df['SibSp']","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-11T08:54:51.798397Z","iopub.execute_input":"2022-08-11T08:54:51.798748Z","iopub.status.idle":"2022-08-11T08:54:51.803652Z","shell.execute_reply.started":"2022-08-11T08:54:51.798714Z","shell.execute_reply":"2022-08-11T08:54:51.802448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop Name ,Ticket ,as these has lots of unique values\n#drop : PassengerId,Cabin, so we get\n# columns_choosed=['Age','Pclass','Sex','SibSp','Parch','Fare','Embarked']\ncolumns_choosed=['Age','Pclass','Sex','SibSp','Parch','Embarked']\nX=train_df[columns_choosed].copy()\ny=train_df['Survived'].copy()\nX_test=test_df[columns_choosed]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.805316Z","iopub.execute_input":"2022-08-11T08:54:51.805637Z","iopub.status.idle":"2022-08-11T08:54:51.816251Z","shell.execute_reply.started":"2022-08-11T08:54:51.805608Z","shell.execute_reply":"2022-08-11T08:54:51.815119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"in this part of the code we choosed the feature we gonna run the model at , and we used features like\n* Age\n* Pclass\n* Sex\n* SibSp\n* Parch\n* Embarked","metadata":{}},{"cell_type":"code","source":"# #split X into , X_train,X_valid ,and y to y_train and y_valid\n# from sklearn.model_selection import train_test_split\n# X_train ,X_valid,y_train,y_valid=train_test_split(X,y,train_size=0.8,test_size=0.2,random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.821458Z","iopub.execute_input":"2022-08-11T08:54:51.822406Z","iopub.status.idle":"2022-08-11T08:54:51.827912Z","shell.execute_reply.started":"2022-08-11T08:54:51.822357Z","shell.execute_reply":"2022-08-11T08:54:51.827115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"you can uncomment that part of the code if you want to traning dataset into\n* traning dataset ,and validation dataset and use it to validate the model accuracy\n* i didn't use that as i used cross-validation method to validate the model ,as that is more acurate in getting model accuracy","metadata":{}},{"cell_type":"code","source":"#get categorical columns and numerical columns from columns_choosed\nnumerical_cols =[col for col in columns_choosed if( X[col].dtype in['int64','float64'])]\ncategorical_cols=list(set(columns_choosed)-set(numerical_cols))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.829665Z","iopub.execute_input":"2022-08-11T08:54:51.830132Z","iopub.status.idle":"2022-08-11T08:54:51.838322Z","shell.execute_reply.started":"2022-08-11T08:54:51.830049Z","shell.execute_reply":"2022-08-11T08:54:51.837149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"in this part of the code we split the columns into two variables\n* numerical_cols , and these columns with numerical dtype\n* categorical_cols , and these are columns which has entries filled with ojbect dtype","metadata":{}},{"cell_type":"code","source":"#preprocessing data using pipelining\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\n\n#fill missing entries for numerical columns with the mean value in column\nnumerical_transformer=SimpleImputer(strategy='mean')\n\n#creat categorical transforemer that will fill missing entires first with strategy = most_frequent\n#and the next step will use OneHotEncoder to convert categorical columns into several numerical columns\ncategorical_transformer=Pipeline(steps=[\n    ('imputer',SimpleImputer(strategy='most_frequent')),\n    ('onehot',OneHotEncoder(handle_unknown='ignore'))\n])\n\n#make ColumnTransformer that will apply the approbiate transformer to the DataFrame accoring to\n#the columns name\npreprocessor=ColumnTransformer(\n    transformers=[\n        ('num',numerical_transformer,numerical_cols),\n        ('cat',categorical_transformer,categorical_cols)\n    ]\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.839634Z","iopub.execute_input":"2022-08-11T08:54:51.840013Z","iopub.status.idle":"2022-08-11T08:54:51.854245Z","shell.execute_reply.started":"2022-08-11T08:54:51.839925Z","shell.execute_reply":"2022-08-11T08:54:51.853128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* in that part of the code i used pipleline as it make the code more clean and clear , and to organize the preprocessing phase","metadata":{}},{"cell_type":"code","source":"#function that return the score of the training model\n#get score using cross validation\nfrom sklearn.model_selection import cross_val_score\n\ndef score(X,y,model):\n    #creat pipeline for creating model for data\n    #any DataFrame will be passed to my_pipleline will be passed to preprocessor , then model\n    my_pipeline=Pipeline(steps=[('preprocessor',preprocessor),\n                               ('model',model)\n                               ])\n    #creating cross_val_score object ,and passing to it my_pipeline ,and features ,and target\n    scores = cross_val_score(my_pipeline, X, y,\n                             cv=5,\n                             scoring='accuracy')\n    return round(scores.mean()*100,2)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.875986Z","iopub.execute_input":"2022-08-11T08:54:51.877349Z","iopub.status.idle":"2022-08-11T08:54:51.886329Z","shell.execute_reply.started":"2022-08-11T08:54:51.877297Z","shell.execute_reply":"2022-08-11T08:54:51.885119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* this function is used to get the accuracy of model ,using cross-validation ,and it's being settup to get the accuracy of the model\n* and we used creat cross_val_score with cv=5 and it will split testing data into 5 segments ,and it will slice the features into 5 segments ,\n* 4 sections to train and 1 section part to validate with , so scoring will be taken 5 times and we will return the average scoring","metadata":{}},{"cell_type":"code","source":"#this part of the code will plot the accuracy of a model with changing traning parameter\n#and i use this to get the best point to stop the traning at ,to get to the early stop point to avoid overfitting\ndef plotScore(dic,xlabel,ylabel,model_name):\n    plt.plot(dic.keys(),dic.values())\n    plt.xlabel(xlabel)\n    plt.ylabel(ylabel)\n    plt.show()\n    #get the max accuracy\n    max_score=max(dic.values())\n    #get the max accuracy's parameter value\n    parameter_value = max(dic, key=dic.get)\n    print(\"best score is {} at {}={} for {} model \".format(  max_score , xlabel ,parameter_value ,model_name  ))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.887606Z","iopub.execute_input":"2022-08-11T08:54:51.888354Z","iopub.status.idle":"2022-08-11T08:54:51.900825Z","shell.execute_reply.started":"2022-08-11T08:54:51.888317Z","shell.execute_reply":"2022-08-11T08:54:51.899869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* in that part of the Kernel i'm trying to tune some of the traning paramters ,\n* for differenct classifiers , and  we gonna use the best traning parameter that will get best Accuracy\n* and we gonna use that parameter in making the final model","metadata":{}},{"cell_type":"code","source":"#getting best training parameters for DecisionTreeClassifier for earlystop point to that get best scores\n#parameter that i will tune are max_depth (values from 1,7)\n\ndef get_Best_Dtree_Para():\n    scores={}\n    for depth in range (1,7):\n        model=DecisionTreeClassifier(max_depth=depth,random_state=0)\n        scores[depth]=score(X,y,model)\n    plotScore(scores,'Depth','Score','DecisionTreeClassifier')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.902061Z","iopub.execute_input":"2022-08-11T08:54:51.903160Z","iopub.status.idle":"2022-08-11T08:54:51.911418Z","shell.execute_reply.started":"2022-08-11T08:54:51.903117Z","shell.execute_reply":"2022-08-11T08:54:51.910325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for the DecesionTreeClassifier\n* some i fixed some paramteres like random_state=0 ,\n* because i want to start from the same point every time i run the code ,to get same results","metadata":{}},{"cell_type":"code","source":"#getting best training parameters for KNeighborsClassifier that get best scores\n#parameter that i will tune are n_neighbors (values from 1,7)\ndef get_Best_KNN_Para():\n    scores={}\n    for neighbour in range (1,7):\n        model=KNeighborsClassifier(n_neighbors=neighbour)\n        scores[neighbour]=score(X,y,model)\n    plotScore(scores,'n_neighbours','Score','KNeighborsClassifier')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.912781Z","iopub.execute_input":"2022-08-11T08:54:51.913201Z","iopub.status.idle":"2022-08-11T08:54:51.922058Z","shell.execute_reply.started":"2022-08-11T08:54:51.913165Z","shell.execute_reply":"2022-08-11T08:54:51.921248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#getting best training parameters for RandomForestClassifier that get best scores\n#parameter that i will tune are n_estimators (values from 1 to 50 , with 1 step)\n\ndef get_Best_RandomF_Para():\n    scores={}\n    for estimators in range(1,51,1):\n        model=RandomForestClassifier(max_depth=4, n_estimators=estimators, max_features=1,random_state=0)\n        scores[estimators]=score(X,y,model)\n    plotScore(scores,'n_estimators','Score','RandomForestClassifier')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.923325Z","iopub.execute_input":"2022-08-11T08:54:51.924364Z","iopub.status.idle":"2022-08-11T08:54:51.937444Z","shell.execute_reply.started":"2022-08-11T08:54:51.924331Z","shell.execute_reply":"2022-08-11T08:54:51.936319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for RandomForestClassifier ,\n* there is some fixed paramters like\n* max_features=1 , because we are aiming on binray classification\n* random_state=0 ,so it gives us the same result when ever we run the code\n* max_depth=4 , i got that from previous tuning and that what gives best results","metadata":{}},{"cell_type":"code","source":"\n#getting best training parameters for SVC that get best scores\n#parameter that i will tune C which is Regularization parameter ,from values of (0.8 to 1)\n\ndef get_Best_SVC_Para():\n    scores={}\n    for c in [round(i*0.05+0.8,4) for i in range(0,5)]:\n        model=SVC(kernel=\"linear\",random_state=0,C=c)\n        scores[c]=score(X,y,model)\n    plotScore(scores,'Regularization c','Score','linearSVC')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.938900Z","iopub.execute_input":"2022-08-11T08:54:51.939249Z","iopub.status.idle":"2022-08-11T08:54:51.949181Z","shell.execute_reply.started":"2022-08-11T08:54:51.939209Z","shell.execute_reply":"2022-08-11T08:54:51.947995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#getting best training parameters for XGBoostClasifier that get best scores\n#parameter that i will tune are estimator(values from 500 to 2000 , with step = 500 )\n\ndef get_Best_XGBC_Para():\n    scores={}\n    for estimator in range(500,2500,500):\n        model=XGBClassifier(n_estimators=estimator,learning_rate=0.05,nthread=4)\n        scores[estimator]=score(X,y,model)\n    plotScore(scores,'n_estimators','Score','XGBClassifier')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.950682Z","iopub.execute_input":"2022-08-11T08:54:51.951579Z","iopub.status.idle":"2022-08-11T08:54:51.959273Z","shell.execute_reply.started":"2022-08-11T08:54:51.951544Z","shell.execute_reply":"2022-08-11T08:54:51.958442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for XGBoostClassifier , we fixed some paramters like\n* nthread =4 , number of threads (cores) that the model will use to train on the data\n* learning_rate=0.05 , we choosed small learning_rate with big number of n_estimators to get best scores","metadata":{}},{"cell_type":"code","source":"#get accuracy of neuralnetwork classifier\ndef get_Accuracy_Neural_Model():\n    model=MLPClassifier(random_state=0,max_iter=500)\n    scores=score(X,y,model)\n    print('-'*40)\n    print(\"NeuralNetworkClasifier has accuracy of {} \\nusing model : {}\".format(scores,model))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:51.963907Z","iopub.execute_input":"2022-08-11T08:54:51.964663Z","iopub.status.idle":"2022-08-11T08:54:51.971806Z","shell.execute_reply.started":"2022-08-11T08:54:51.964628Z","shell.execute_reply":"2022-08-11T08:54:51.970658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"for nuralnetwork model , i fixed the max_iter to 500 as it get's the best results","metadata":{}},{"cell_type":"code","source":"get_Best_Dtree_Para()\nget_Best_KNN_Para()\nget_Best_RandomF_Para()\nget_Best_SVC_Para()\nget_Best_XGBC_Para()\nget_Accuracy_Neural_Model()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:54:52.158744Z","iopub.execute_input":"2022-08-11T08:54:52.159169Z","iopub.status.idle":"2022-08-11T08:57:57.130827Z","shell.execute_reply.started":"2022-08-11T08:54:52.159131Z","shell.execute_reply":"2022-08-11T08:57:57.129254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this part of the code takes the model and output filename and save predictions into csv file\n#ready for submission\ndef saveCSVfile(model,output_name):\n    final_pipeline=Pipeline(steps=[('preprocessor',preprocessor),('model',model)])\n    final_pipeline.fit(X,y)\n    preds=final_pipeline.predict(X_test)\n    output = pd.DataFrame({'PassengerId': X_test.index, 'Survived': preds})\n    output.to_csv(output_name+'.csv', index=False)\n    print(\"Your {} was successfully saved!\".format(output_name))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:57:57.133796Z","iopub.execute_input":"2022-08-11T08:57:57.135182Z","iopub.status.idle":"2022-08-11T08:57:57.146370Z","shell.execute_reply.started":"2022-08-11T08:57:57.135115Z","shell.execute_reply":"2022-08-11T08:57:57.145295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* in this section we gonna make predictions using the different classifiers using \n* the best parameters for earlystop point","metadata":{}},{"cell_type":"code","source":"#DecisionTreeClassifier final model\nmodel=DecisionTreeClassifier(max_depth=4,random_state=0)\nsaveCSVfile(model,'DecisionTreePredicitions')\n\n#KNeighborsClassifier final model\nmodel=KNeighborsClassifier(n_neighbors=5)\nsaveCSVfile(model,'KNeighborsPredicitions')\n\n#RandomForestClassifier final model\nmodel=RandomForestClassifier(max_depth=4, n_estimators=24, max_features=1,random_state=0)\nsaveCSVfile(model,'RandomForestPredicitions')\n\n\n#SVC with linear kernel final model\nmodel=SVC(kernel=\"linear\",random_state=0,C=0.95)\nsaveCSVfile(model,'SVCPredictions')\n\n\n#XGBClassifier final model\nmodel=XGBClassifier(n_estimators=1000,learning_rate=0.05,nthread=4)\nsaveCSVfile(model,'XGBPredictions')\n\n\n#MLPClassifier final model\nmodel=MLPClassifier(random_state=0,max_iter=500)\nsaveCSVfile(model,'NeuralNetPredictions')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T08:57:57.148241Z","iopub.execute_input":"2022-08-11T08:57:57.155855Z","iopub.status.idle":"2022-08-11T08:58:03.155542Z","shell.execute_reply.started":"2022-08-11T08:57:57.155779Z","shell.execute_reply":"2022-08-11T08:58:03.153897Z"},"trusted":true},"execution_count":null,"outputs":[]}]}