{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T04:06:09.680347Z","iopub.execute_input":"2022-08-03T04:06:09.681195Z","iopub.status.idle":"2022-08-03T04:06:09.686494Z","shell.execute_reply.started":"2022-08-03T04:06:09.681143Z","shell.execute_reply":"2022-08-03T04:06:09.685665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:09.991988Z","iopub.execute_input":"2022-08-03T04:06:09.992307Z","iopub.status.idle":"2022-08-03T04:06:09.997136Z","shell.execute_reply.started":"2022-08-03T04:06:09.992272Z","shell.execute_reply":"2022-08-03T04:06:09.996148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.set()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.151975Z","iopub.execute_input":"2022-08-03T04:06:10.152734Z","iopub.status.idle":"2022-08-03T04:06:10.158387Z","shell.execute_reply.started":"2022-08-03T04:06:10.152698Z","shell.execute_reply":"2022-08-03T04:06:10.157420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Importing the data\n","metadata":{}},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.217818Z","iopub.execute_input":"2022-08-03T04:06:10.218259Z","iopub.status.idle":"2022-08-03T04:06:10.235865Z","shell.execute_reply.started":"2022-08-03T04:06:10.218228Z","shell.execute_reply":"2022-08-03T04:06:10.235204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest_set = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.424933Z","iopub.execute_input":"2022-08-03T04:06:10.425516Z","iopub.status.idle":"2022-08-03T04:06:10.444035Z","shell.execute_reply.started":"2022-08-03T04:06:10.425482Z","shell.execute_reply":"2022-08-03T04:06:10.443220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# descovering the data","metadata":{}},{"cell_type":"code","source":"combine = [train_set, test_set]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.616727Z","iopub.execute_input":"2022-08-03T04:06:10.617060Z","iopub.status.idle":"2022-08-03T04:06:10.622729Z","shell.execute_reply.started":"2022-08-03T04:06:10.617029Z","shell.execute_reply":"2022-08-03T04:06:10.621818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Datatype checking","metadata":{}},{"cell_type":"code","source":"for dataset in combine:\n    print(dataset.info())\n    print('--------------')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.778631Z","iopub.execute_input":"2022-08-03T04:06:10.778962Z","iopub.status.idle":"2022-08-03T04:06:10.804540Z","shell.execute_reply.started":"2022-08-03T04:06:10.778929Z","shell.execute_reply":"2022-08-03T04:06:10.803874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **Observations:** Some features like sex and Embarked are catogiral features\n\n- **Decisions:**  we can convert features which contain strings to numerical values. This is required by most model algorithms.","metadata":{}},{"cell_type":"markdown","source":"### Completness of the data checking","metadata":{}},{"cell_type":"code","source":"for data in combine:\n    print(data.isna().sum())\n    print('----------------')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:10.937935Z","iopub.execute_input":"2022-08-03T04:06:10.938620Z","iopub.status.idle":"2022-08-03T04:06:10.949915Z","shell.execute_reply.started":"2022-08-03T04:06:10.938571Z","shell.execute_reply":"2022-08-03T04:06:10.948886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:**\n- Some features have uncomplete features.\n\n**Decisions:**  we can either drop the column or empute it depinding on the column importance:\n- We may want to complete Age feature as it is definitely correlated to survival.\n- We may want to complete the Embarked feature as it may also correlate with survival or another important feature.\n- Cabin feature may be dropped as it is highly incomplete or contains many null values both in training and test dataset.","metadata":{}},{"cell_type":"markdown","source":"### Scaling and data distripution checking","metadata":{}},{"cell_type":"code","source":"combine[0].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:11.217057Z","iopub.execute_input":"2022-08-03T04:06:11.217729Z","iopub.status.idle":"2022-08-03T04:06:11.250116Z","shell.execute_reply.started":"2022-08-03T04:06:11.217688Z","shell.execute_reply":"2022-08-03T04:06:11.249111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:** \n- Some features wide range of values like Fare and age comparing to other features.\n\n**Decisions:**\n- we can normalize the data if the model is sensative to scaling of the data.\n- PassengerId may be dropped from training dataset as it does not contribute to survival.\n- We may also want to create a Fare range feature if it helps our analysis.","metadata":{}},{"cell_type":"code","source":"combine[0].describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:11.369009Z","iopub.execute_input":"2022-08-03T04:06:11.369446Z","iopub.status.idle":"2022-08-03T04:06:11.392347Z","shell.execute_reply.started":"2022-08-03T04:06:11.369415Z","shell.execute_reply":"2022-08-03T04:06:11.391414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:** \n- Cabin values have several dupicates across samples. Alternatively several passengers shared a cabin.\n- Ticket feature has high ratio (22%) of duplicate values (unique=681).\n\n**Decisions:**\n- Ticket feature may be dropped from our analysis as it contains high ratio of duplicates (22%) and there may not be a correlation between Ticket and survival.\n- Name feature is relatively non-standard, may not contribute directly to survival, so maybe dropped or converted to more usefull way by extracting information from it like nicknames.","metadata":{}},{"cell_type":"markdown","source":"### correlation between features¶\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(20,10))\nsns.heatmap(combine[0].corr(), vmin=-1, vmax=1,square=True,annot=True, cmap='BrBG')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:11.512919Z","iopub.execute_input":"2022-08-03T04:06:11.513386Z","iopub.status.idle":"2022-08-03T04:06:12.223185Z","shell.execute_reply.started":"2022-08-03T04:06:11.513355Z","shell.execute_reply":"2022-08-03T04:06:12.222053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- We may want to create a new feature called Family based on Parch and SibSp to get total count of family members on board as there is a relation between theme.\n- their is a relation between the fare and the Pclass","metadata":{}},{"cell_type":"markdown","source":"### Analyze by pivoting features","metadata":{}},{"cell_type":"code","source":"axis=sns.barplot(x ='Pclass' , y = 'Survived' , data= combine[0])\naxis.set_title('Pclass  v.s  Survived')\naxis.set_xlabel('Pclass')\naxis.set_ylabel('Survived Percentage')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:12.225506Z","iopub.execute_input":"2022-08-03T04:06:12.225865Z","iopub.status.idle":"2022-08-03T04:06:12.561714Z","shell.execute_reply.started":"2022-08-03T04:06:12.225823Z","shell.execute_reply":"2022-08-03T04:06:12.560843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:** \n- We observe significant correlation (>0.5) among Pclass=1 and Survived.\n\n**Decisions:**\n-  the feature sounds an important feature","metadata":{}},{"cell_type":"code","source":"axis=sns.barplot(x ='Sex' , y = 'Survived' , data= combine[0])\naxis.set_title('Gender  v.s  Survived')\naxis.set_xlabel('Gender')\naxis.set_ylabel('Survived Percentage')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:12.562904Z","iopub.execute_input":"2022-08-03T04:06:12.563166Z","iopub.status.idle":"2022-08-03T04:06:12.875439Z","shell.execute_reply.started":"2022-08-03T04:06:12.563134Z","shell.execute_reply":"2022-08-03T04:06:12.874710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:** \n- We confirm the observation during problem definition that Sex=female had very high survival rate at 74%.\n\n**Decisions:**\n-  the feature sounds an important feature","metadata":{}},{"cell_type":"code","source":"axis=sns.barplot(x ='Embarked' , y = 'Survived' , data= combine[0])\naxis.set_title('Twon  v.s  Survived')\naxis.set_xticklabels(['Southampton (S)' , 'Cherbourg (C)','Queenstown (Q)'])\naxis.set_xlabel('Town Name ')\naxis.set_ylabel('Survived Percentage')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:12.876911Z","iopub.execute_input":"2022-08-03T04:06:12.877544Z","iopub.status.idle":"2022-08-03T04:06:13.213628Z","shell.execute_reply.started":"2022-08-03T04:06:12.877511Z","shell.execute_reply":"2022-08-03T04:06:13.212597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:** \n- We confirm the observation during problem definition that most of people from 'Cherbourg (C) are survived , most of people from Southampton (S) and Queenstown unsurvived\n\n**Decisions:**\n-  the feature sounds an important feature","metadata":{}},{"cell_type":"code","source":"sns.distplot(combine[0][combine[0][\"Survived\"] ==  0]['Age']  ,label='Not Survived', color = '#e74c3c')\nsns.distplot(combine[0][combine[0][\"Survived\"] ==  1]['Age'] , label='Survived' ,color = '#2ecc71')\n\nplt.legend()\nplt.title('Distribtion of Survived with Age')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:13.215401Z","iopub.execute_input":"2022-08-03T04:06:13.215694Z","iopub.status.idle":"2022-08-03T04:06:13.670666Z","shell.execute_reply.started":"2022-08-03T04:06:13.215660Z","shell.execute_reply":"2022-08-03T04:06:13.669714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Observations:**\n- Infants (Age <=5) had high survival rate.\n- Oldest passengers (Age = 80) survived.\n- Large number of 15-25 year olds did not survive.\n\n**Decisions:**\n- We should consider Age in our model training.\n- Complete the Age feature for null values.\n-  We should band age groups.\n","metadata":{}},{"cell_type":"code","source":"bins = [ 0, 16, 32, 48,64, np.inf]\nlabels = [' 0-16', ' 16-32', ' 32-48', ' 48-64', ' 64-']\n\nage = pd.cut(combine[0][\"Age\"], bins, labels = labels)\n\naxis=sns.barplot(age , y=combine[0][\"Survived\"] )\naxis.set_title('Age  v.s  Survived')\naxis\naxis.set_xlabel('Age')\naxis.set_ylabel('Survived Percentage')\n_=axis.set_xticklabels(labels,rotation  = 30)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:13.672218Z","iopub.execute_input":"2022-08-03T04:06:13.673174Z","iopub.status.idle":"2022-08-03T04:06:14.077843Z","shell.execute_reply.started":"2022-08-03T04:06:13.673131Z","shell.execute_reply":"2022-08-03T04:06:14.076879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combine[0][combine[0][\"Survived\"] == 0]['Age']","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.079171Z","iopub.execute_input":"2022-08-03T04:06:14.079496Z","iopub.status.idle":"2022-08-03T04:06:14.090983Z","shell.execute_reply.started":"2022-08-03T04:06:14.079452Z","shell.execute_reply":"2022-08-03T04:06:14.090237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating a transformation pipline for the data","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.093434Z","iopub.execute_input":"2022-08-03T04:06:14.095708Z","iopub.status.idle":"2022-08-03T04:06:14.100122Z","shell.execute_reply.started":"2022-08-03T04:06:14.095670Z","shell.execute_reply":"2022-08-03T04:06:14.099346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creating a transformer of the feautures¶\n","metadata":{}},{"cell_type":"code","source":"class prepare_cat(BaseEstimator, TransformerMixin):\n    \n    def __init__(self,add_title = True, custom_fill_age = True): \n        self.add_title = add_title\n        self.custom_fill_age = custom_fill_age\n        \n    def fit(self, X, y=None):\n        return self  \n    \n    \n    def transform(self, X):\n    \n        # Converting a categorical feature\n        X['Sex'] = X['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n        \n        #fill uncomplete catogeral data\n        freq_port = X.Embarked.dropna().mode()[0]\n        X['Embarked'] = X['Embarked'].fillna(freq_port)\n        X['Embarked'] = X['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n        \n        if self.add_title:\n            X['Title'] = X.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n            # We can replace many titles with a more common name or classify them as Rare.\n            X['Title'] = X['Title'].replace(['Lady', 'Countess','Capt', 'Col','Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n            X['Title'] = X['Title'].replace('Mlle', 'Miss')\n            X['Title'] = X['Title'].replace('Ms', 'Miss')\n            X['Title'] = X['Title'].replace('Mme', 'Mrs')\n            # We can convert the categorical titles to ordinal.\n            title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\n            X['Title'] = X['Title'].map(title_mapping)\n            X['Title'] = X['Title'].fillna(0)\n            \n            if self.custom_fill_age:\n                #  way of guessing missing values is to use other correlated features (Age, Gender, and Pclass).\n                guess_ages = np.zeros((2,3))\n\n                X['Age'] = X.groupby(['Sex' , 'Pclass'])['Age'].apply(lambda x : x.fillna(x.median()))\n\n                X['Age'] = X['Age'].astype(int)\n\n                # replace Age with ordinals based on bands.\n                X.loc[ X['Age'] <= 16, 'Age'] = 0\n                X.loc[(X['Age'] > 16) & (X['Age'] <= 32), 'Age'] = 1\n                X.loc[(X['Age'] > 32) & (X['Age'] <= 48), 'Age'] = 2\n                X.loc[(X['Age'] > 48) & (X['Age'] <= 64), 'Age'] = 3\n                X.loc[ X['Age'] > 64, 'Age'] = 4\n            \n             \n        \n        # Drop useless columns \n        X = X.drop(columns = ['Cabin','Ticket','Name'])\n        \n        #impute any remaning missing values\n        imuter = SimpleImputer(strategy = 'most_frequent')\n        X = imuter.fit_transform(X)\n        \n        \n        return X\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.103005Z","iopub.execute_input":"2022-08-03T04:06:14.103798Z","iopub.status.idle":"2022-08-03T04:06:14.120437Z","shell.execute_reply.started":"2022-08-03T04:06:14.103729Z","shell.execute_reply":"2022-08-03T04:06:14.119347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class prepare_numeric(BaseEstimator, TransformerMixin):\n    \n    def __init__(self,handle_skewing = False, handel_company = False): \n        self.handel_company = handel_company\n        \n    def fit(self, X, y=None):\n        return self \n    \n    \n    def transform(self, X):\n\n        # scale the fare feature\n        X['Fare'] = X['Fare'].fillna(X['Fare'].median())\n        X.loc[ X['Fare'] <= 7.91, 'Fare'] = 0\n        X.loc[(X['Fare'] > 7.91) & (X['Fare'] <= 14.454), 'Fare'] = 1\n        X.loc[(X['Fare'] > 14.454) & (X['Fare'] <= 31), 'Fare']   = 2\n        X.loc[ X['Fare'] > 31, 'Fare'] = 3\n        X['Fare'] = X['Fare'].astype(int)\n        \n\n        if self.handel_company:\n            X['FamilySize'] = X['SibSp'] + X['Parch'] + 1\n            X['IsAlone'] = 0\n            X.loc[X['FamilySize'] == 1, 'IsAlone'] = 1\n            \n\n        \n        return X","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.122027Z","iopub.execute_input":"2022-08-03T04:06:14.122897Z","iopub.status.idle":"2022-08-03T04:06:14.137428Z","shell.execute_reply.started":"2022-08-03T04:06:14.122854Z","shell.execute_reply":"2022-08-03T04:06:14.136721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train_set.drop(columns = ['Survived'])\ntrain_y = train_set['Survived'].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.139085Z","iopub.execute_input":"2022-08-03T04:06:14.139648Z","iopub.status.idle":"2022-08-03T04:06:14.150827Z","shell.execute_reply.started":"2022-08-03T04:06:14.139567Z","shell.execute_reply":"2022-08-03T04:06:14.150038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_attribs = ['SibSp','Parch','Fare']\n\ncat_attribs = ['Name', 'Sex','Ticket','Cabin', 'Embarked', 'Age','Pclass']\n\nnum_pipeline = Pipeline([\n       ('prepare_numeric', prepare_numeric()),\n       ('num_imputer', SimpleImputer(strategy=\"median\")),\n    ])\n\nfull_pipeline = ColumnTransformer([\n        (\"num\", num_pipeline, num_attribs),\n        (\"cat\", prepare_cat(), cat_attribs),\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.152268Z","iopub.execute_input":"2022-08-03T04:06:14.152882Z","iopub.status.idle":"2022-08-03T04:06:14.160249Z","shell.execute_reply.started":"2022-08-03T04:06:14.152841Z","shell.execute_reply":"2022-08-03T04:06:14.159260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_pre = full_pipeline.fit_transform(train_X)\ntest_X_pre = full_pipeline.fit_transform(test_set)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.161612Z","iopub.execute_input":"2022-08-03T04:06:14.161872Z","iopub.status.idle":"2022-08-03T04:06:14.242627Z","shell.execute_reply.started":"2022-08-03T04:06:14.161843Z","shell.execute_reply":"2022-08-03T04:06:14.241786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model selection, hyperparameter tuning and cross validation ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier, ExtraTreesClassifier, VotingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV, cross_val_score, StratifiedKFold, learning_curve\nfrom sklearn.model_selection import cross_val_score\nfrom yellowbrick.model_selection import learning_curve\nfrom sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.243859Z","iopub.execute_input":"2022-08-03T04:06:14.244100Z","iopub.status.idle":"2022-08-03T04:06:14.249790Z","shell.execute_reply.started":"2022-08-03T04:06:14.244073Z","shell.execute_reply":"2022-08-03T04:06:14.248885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross validate model with Kfold stratified cross val\nkfold = StratifiedKFold(n_splits = 5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.252419Z","iopub.execute_input":"2022-08-03T04:06:14.252916Z","iopub.status.idle":"2022-08-03T04:06:14.260698Z","shell.execute_reply.started":"2022-08-03T04:06:14.252881Z","shell.execute_reply":"2022-08-03T04:06:14.260061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gradient boosting tunning\n\nGBC = GradientBoostingClassifier()\ngb_param_grid = {'loss' : [\"deviance\"],\n              'n_estimators' : [1000, 1500 ,2000],\n              'learning_rate': [0.1, 0.01, 0.001],\n              }\n\ngsGBC = GridSearchCV(GBC,param_grid = gb_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= -1, verbose = 1)\n\ngsGBC.fit(train_X_pre, train_y)\n\nGBC_best = gsGBC.best_estimator_\n\ngsGBC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:14.375899Z","iopub.execute_input":"2022-08-03T04:06:14.376595Z","iopub.status.idle":"2022-08-03T04:06:45.515759Z","shell.execute_reply.started":"2022-08-03T04:06:14.376555Z","shell.execute_reply":"2022-08-03T04:06:45.515030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(learning_curve(GBC_best, train_X_pre, train_y, cv=10, scoring='accuracy'))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:06:45.518331Z","iopub.execute_input":"2022-08-03T04:06:45.518919Z","iopub.status.idle":"2022-08-03T04:08:02.774881Z","shell.execute_reply.started":"2022-08-03T04:06:45.518866Z","shell.execute_reply":"2022-08-03T04:08:02.773763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RFC Parameters tunning \nRFC = RandomForestClassifier()\n\n\n## Search grid for optimal parameters\nrf_param_grid = {\"max_depth\": [7, 9],\n              \"max_features\": [3, 'auto'],\n              \"min_samples_split\": [6],\n              \"min_samples_leaf\": [6],\n              \"bootstrap\": [True],\n              \"n_estimators\" :[1500, 1750],\n              \"criterion\": [\"gini\"],\n              \"oob_score\":[True]        \n}\n\n\ngsRFC = GridSearchCV(RFC,param_grid = rf_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= -1, verbose = 1)\n\ngsRFC.fit(train_X_pre, train_y)\n\nRFC_best = gsRFC.best_estimator_\n\n# Best score\ngsRFC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:08:02.776461Z","iopub.execute_input":"2022-08-03T04:08:02.776825Z","iopub.status.idle":"2022-08-03T04:09:05.633577Z","shell.execute_reply.started":"2022-08-03T04:08:02.776765Z","shell.execute_reply":"2022-08-03T04:09:05.632525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(learning_curve(RFC_best, train_X_pre, train_y, cv=5, scoring='accuracy'))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:09:05.635396Z","iopub.execute_input":"2022-08-03T04:09:05.635644Z","iopub.status.idle":"2022-08-03T04:10:25.615286Z","shell.execute_reply.started":"2022-08-03T04:09:05.635617Z","shell.execute_reply":"2022-08-03T04:10:25.614246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### SVC classifier\nSVMC = SVC(probability=True)\nsvc_param_grid = {'kernel': ['rbf'], \n                  'gamma': [ 0.1,0.05 ,0.01 ,0.001],\n                  'C': [100,200,300, 500]}\n\ngsSVMC = GridSearchCV(SVMC,param_grid = svc_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= -1, verbose = 1)\n\ngsSVMC.fit(train_X_pre, train_y)\n\nSVMC_best = gsSVMC.best_estimator_\n\n# Best score\ngsSVMC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:10:25.624835Z","iopub.execute_input":"2022-08-03T04:10:25.625523Z","iopub.status.idle":"2022-08-03T04:10:33.761279Z","shell.execute_reply.started":"2022-08-03T04:10:25.625490Z","shell.execute_reply":"2022-08-03T04:10:33.760334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(learning_curve(SVMC_best, train_X_pre, train_y, cv=5, scoring='accuracy'))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:10:33.762322Z","iopub.execute_input":"2022-08-03T04:10:33.762574Z","iopub.status.idle":"2022-08-03T04:10:36.539318Z","shell.execute_reply.started":"2022-08-03T04:10:33.762542Z","shell.execute_reply":"2022-08-03T04:10:36.538415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"votingC = VotingClassifier(estimators=[('rfc', RFC_best), ('svc', SVMC_best), ('gbc',GBC_best)], voting='soft', n_jobs= -1)\nvotingC.fit(train_X_pre, train_y)\ncross_val_score(votingC, train_X_pre, train_y, cv=5, scoring=\"accuracy\").mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:10:36.540506Z","iopub.execute_input":"2022-08-03T04:10:36.540743Z","iopub.status.idle":"2022-08-03T04:11:00.391674Z","shell.execute_reply.started":"2022-08-03T04:10:36.540716Z","shell.execute_reply":"2022-08-03T04:11:00.390751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predict = votingC.predict(test_X_pre)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:11:00.394720Z","iopub.execute_input":"2022-08-03T04:11:00.394971Z","iopub.status.idle":"2022-08-03T04:11:00.618590Z","shell.execute_reply.started":"2022-08-03T04:11:00.394943Z","shell.execute_reply":"2022-08-03T04:11:00.617584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame(columns=['PassengerId', 'Survived'])\nsubmission_df['PassengerId'] = test_set['PassengerId']\nsubmission_df['Survived'] = y_predict\nsubmission_df.to_csv('submissions.csv', header=True, index=False)\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T04:11:00.621298Z","iopub.execute_input":"2022-08-03T04:11:00.621866Z","iopub.status.idle":"2022-08-03T04:11:00.641069Z","shell.execute_reply.started":"2022-08-03T04:11:00.621817Z","shell.execute_reply":"2022-08-03T04:11:00.639851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}