{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"markdown","source":"Hi Everyone! You can refer to the previous notebook for the EDA:\nhttps://www.kaggle.com/pohzixiang/titanic-eda\n\nThis notebook comprise of the following:\n* Feature Engineering\n* Data Preprocessing\n* Model training (Logistics regression, XGB, random forest, basic deep learning)\n* HyperParameter tuning\n\nDo leave your comments/feedback if you have any!","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T05:47:47.348124Z","iopub.execute_input":"2022-08-02T05:47:47.348584Z","iopub.status.idle":"2022-08-02T05:47:47.383370Z","shell.execute_reply.started":"2022-08-02T05:47:47.348496Z","shell.execute_reply":"2022-08-02T05:47:47.382096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature Engineering**\n\nObjective:\n* Create Surname column\n* Create Title column\n* Group Tickets together if they are equal \n* Combine Sibsp and Parch","metadata":{}},{"cell_type":"code","source":"#Read data\ntrain=pd.read_csv('/kaggle/input/titanic/train.csv')\ntest=pd.read_csv('/kaggle/input/titanic/test.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:28.688783Z","iopub.execute_input":"2022-08-02T05:52:28.689464Z","iopub.status.idle":"2022-08-02T05:52:28.726350Z","shell.execute_reply.started":"2022-08-02T05:52:28.689412Z","shell.execute_reply":"2022-08-02T05:52:28.725368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create surname column\ntrain['Surname']=train['Name'].apply(lambda x: x.split(',')[0])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:30.089949Z","iopub.execute_input":"2022-08-02T05:52:30.091160Z","iopub.status.idle":"2022-08-02T05:52:30.112357Z","shell.execute_reply.started":"2022-08-02T05:52:30.091118Z","shell.execute_reply":"2022-08-02T05:52:30.111479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create title column\nimport re\ndef look_title(x):\n    return re.findall(\"(?<=,\\s).+?(?=\\.)\", x)[0]\n\ndef filter_title(x):\n    if x==x:\n        if x not in ('Mr','Mrs','Miss','Master'):\n            return 'Others'\n        else:\n            return x\n    \ntrain['title']=train['Name'].apply(lambda x: look_title(x))\ntrain['title']=train['title'].apply(lambda x: filter_title(x))\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:31.574903Z","iopub.execute_input":"2022-08-02T05:52:31.575320Z","iopub.status.idle":"2022-08-02T05:52:31.602731Z","shell.execute_reply.started":"2022-08-02T05:52:31.575276Z","shell.execute_reply":"2022-08-02T05:52:31.601245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Group Size\ntrain['Ticketgrp']=train.groupby(by=['Ticket','Surname'])['PassengerId'].transform('count')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:34.692930Z","iopub.execute_input":"2022-08-02T05:52:34.693377Z","iopub.status.idle":"2022-08-02T05:52:34.717821Z","shell.execute_reply.started":"2022-08-02T05:52:34.693340Z","shell.execute_reply":"2022-08-02T05:52:34.716855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We are going to remove a few columns that we do not need, and remove Survived column to be used later**\n\nColumns we dont want: PassengerId, Name, Cabin, Surname, Survived (To be kept for later)","metadata":{}},{"cell_type":"code","source":"#Remove columns\nsurvived=train['Survived']\ntrain=train.drop(columns=['PassengerId', 'Name', 'Cabin', 'Survived','Ticket','Surname','SibSp','Parch'])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:36.570781Z","iopub.execute_input":"2022-08-02T05:52:36.571650Z","iopub.status.idle":"2022-08-02T05:52:36.596992Z","shell.execute_reply.started":"2022-08-02T05:52:36.571614Z","shell.execute_reply":"2022-08-02T05:52:36.595946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Preprocessing for training data**\n\n* Fill up missing data and perform encoding","metadata":{}},{"cell_type":"code","source":"#Check number of missing values in each columns for training data\nmissing=train.isnull().sum()\nprint(missing)\nprint('total data:', len(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:40.699449Z","iopub.execute_input":"2022-08-02T05:52:40.700129Z","iopub.status.idle":"2022-08-02T05:52:40.709079Z","shell.execute_reply.started":"2022-08-02T05:52:40.700093Z","shell.execute_reply":"2022-08-02T05:52:40.707535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fill with median values\n\nfrom sklearn.impute import SimpleImputer\n\nfilling_list=['Embarked']\nmean_val=['Age']\n\n#Imputer for Embarked\nCat_imputer = SimpleImputer(missing_values=np.nan, strategy='most_frequent')\nfor i in filling_list:\n    train[i] = Cat_imputer.fit_transform(train[i].to_numpy().reshape(-1,1))\n    \n#Imputer for age\na=train.groupby(by=['title'])['Age'].transform('median')\ntrain['Age']=train['Age'].fillna(a)\n    \nmissing=train.isnull().sum()\nmissing","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:43.481196Z","iopub.execute_input":"2022-08-02T05:52:43.482012Z","iopub.status.idle":"2022-08-02T05:52:44.249967Z","shell.execute_reply.started":"2022-08-02T05:52:43.481970Z","shell.execute_reply":"2022-08-02T05:52:44.248612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#One Hot Encoding\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nobject_cols=['Pclass', 'Sex', 'Embarked','title']\n\n# Apply one-hot encoder to each column with categorical data\nOH_encoder = OneHotEncoder(sparse=False)\nOH_cols_train = pd.DataFrame(OH_encoder.fit_transform(train[object_cols]))\nOH_cols_train.columns = OH_encoder.get_feature_names_out()\n\n# One-hot encoding removed index; put it back\nOH_cols_train.index = train.index\n\n# Remove categorical columns (will replace with one-hot encoding)\nnum_X_train = train.drop(object_cols, axis=1)\n\n# Add one-hot encoded columns to numerical features\ntrain = pd.concat([num_X_train, OH_cols_train], axis=1)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:46.308286Z","iopub.execute_input":"2022-08-02T05:52:46.308739Z","iopub.status.idle":"2022-08-02T05:52:46.349391Z","shell.execute_reply.started":"2022-08-02T05:52:46.308706Z","shell.execute_reply":"2022-08-02T05:52:46.348319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Preprocessing for test data**\n\n","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:52:49.362425Z","iopub.execute_input":"2022-08-02T05:52:49.363124Z","iopub.status.idle":"2022-08-02T05:52:49.381236Z","shell.execute_reply.started":"2022-08-02T05:52:49.363074Z","shell.execute_reply":"2022-08-02T05:52:49.380019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create surname column\ntest['Surname']=test['Name'].apply(lambda x: x.split(',')[0])\n#Create title column\ntest['title']=test['Name'].apply(lambda x: look_title(x))\ntest['title']=test['title'].apply(lambda x: filter_title(x))\n#Create Group Size\ntest['Ticketgrp']=test.groupby(by=['Ticket','Surname'])['PassengerId'].transform('count')\n\nId=test['PassengerId']\ntest=test.drop(columns=['PassengerId', 'Name', 'Cabin','Ticket','Surname','SibSp','Parch'])\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:28.136894Z","iopub.execute_input":"2022-08-02T05:53:28.137330Z","iopub.status.idle":"2022-08-02T05:53:28.163835Z","shell.execute_reply.started":"2022-08-02T05:53:28.137290Z","shell.execute_reply":"2022-08-02T05:53:28.162613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing=test.isnull().sum()\nmissing","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:30.937432Z","iopub.execute_input":"2022-08-02T05:53:30.938046Z","iopub.status.idle":"2022-08-02T05:53:30.954617Z","shell.execute_reply.started":"2022-08-02T05:53:30.937998Z","shell.execute_reply":"2022-08-02T05:53:30.952358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fill age and fare based on what we got from training data\ntest['Fare']=test['Fare'].fillna(train['Fare'].median())\ntest['Age']=test['Age'].fillna(a)\n\nmissing=test.isnull().sum()\nmissing","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:32.812141Z","iopub.execute_input":"2022-08-02T05:53:32.813281Z","iopub.status.idle":"2022-08-02T05:53:32.827950Z","shell.execute_reply.started":"2022-08-02T05:53:32.813222Z","shell.execute_reply":"2022-08-02T05:53:32.826848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply one-hot encoder to each column with categorical data\nOH_cols_test = pd.DataFrame(OH_encoder.transform(test[object_cols]))\nOH_cols_test.columns = OH_encoder.get_feature_names_out()\n\n# One-hot encoding removed index; put it back\nOH_cols_test.index = test.index\n\n# Remove categorical columns (will replace with one-hot encoding)\nnum_X_test = test.drop(object_cols, axis=1)\n\n# Add one-hot encoded columns to numerical features\ntest = pd.concat([num_X_test, OH_cols_test], axis=1)\n\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:34.574001Z","iopub.execute_input":"2022-08-02T05:53:34.574443Z","iopub.status.idle":"2022-08-02T05:53:34.607696Z","shell.execute_reply.started":"2022-08-02T05:53:34.574407Z","shell.execute_reply":"2022-08-02T05:53:34.606504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scaling\nfrom sklearn.preprocessing import MinMaxScaler\n\nscaler = MinMaxScaler()\ntrain = pd.DataFrame(scaler.fit_transform(train), index=train.index, columns=train.columns)\ntest = pd.DataFrame(scaler.transform(test), index=test.index, columns=test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:38.764602Z","iopub.execute_input":"2022-08-02T05:53:38.765008Z","iopub.status.idle":"2022-08-02T05:53:38.782200Z","shell.execute_reply.started":"2022-08-02T05:53:38.764973Z","shell.execute_reply":"2022-08-02T05:53:38.780857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model Training**\n\nWe try the following models:\n* logistics regression\n* xgboost\n* random forest\n* Deep learning","metadata":{}},{"cell_type":"code","source":"#logistics regression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import cross_val_score\nxtrain,xtest,ytrain,ytest=train_test_split(train, survived, test_size=0.2, random_state=42)\n\nfrom sklearn.linear_model import LogisticRegression\n\nclf=LogisticRegression(random_state=0)\nclf.fit(xtrain, ytrain)\nypred=clf.predict(xtest)\nfinal = np.round(ypred)\naccuracy_score(final, list(ytest))  #Accuracy is 0.821","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:40.801458Z","iopub.execute_input":"2022-08-02T05:53:40.802032Z","iopub.status.idle":"2022-08-02T05:53:40.860636Z","shell.execute_reply.started":"2022-08-02T05:53:40.801999Z","shell.execute_reply":"2022-08-02T05:53:40.859141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Stratified K Fold for logistics regression\nfrom sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=5)\n\ndef training(x_train, y_train, x_test, y_test, fold_no):\n    clf.fit(x_train, y_train)\n    score = clf.score(x_test,y_test)\n    print('For Fold {} the accuracy is {}'.format(str(fold_no),score))\n    return score\n\naccu=0\nfold_no = 1\nfor train_index,test_index in skf.split(train, survived):\n    train_copy = train.iloc[train_index,:]\n    test_copy = train.iloc[test_index,:]\n    survived_train=survived.iloc[train_index]\n    survived_test=survived.iloc[test_index]\n    accu=accu+training(train_copy,survived_train,test_copy, survived_test, fold_no)\n    fold_no += 1\n    \nprint('Average accuracy is {}'.format(accu/(fold_no-1)))  #Avg accuracy is 0.826","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:48.339265Z","iopub.execute_input":"2022-08-02T05:53:48.339791Z","iopub.status.idle":"2022-08-02T05:53:48.562018Z","shell.execute_reply.started":"2022-08-02T05:53:48.339750Z","shell.execute_reply":"2022-08-02T05:53:48.560506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot ROC curve\nimport matplotlib.pyplot as plt\nfrom sklearn import datasets, metrics, model_selection\nmetrics.plot_roc_curve(clf, train, survived)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:52.688433Z","iopub.execute_input":"2022-08-02T05:53:52.688841Z","iopub.status.idle":"2022-08-02T05:53:53.060073Z","shell.execute_reply.started":"2022-08-02T05:53:52.688806Z","shell.execute_reply":"2022-08-02T05:53:53.058870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#xgb\nimport xgboost as xgb\nxgb_model = xgb.XGBRegressor(random_state=42).fit(xtrain, ytrain)\nypred2=xgb_model.predict(xtest)\naccuracy_score(np.round(ypred2), list(ytest)) #Accuracy is 0.827","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:53:56.006923Z","iopub.execute_input":"2022-08-02T05:53:56.007370Z","iopub.status.idle":"2022-08-02T05:53:56.480195Z","shell.execute_reply.started":"2022-08-02T05:53:56.007334Z","shell.execute_reply":"2022-08-02T05:53:56.479229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\n\n# Hyperparam tuning for xgboost\n\n# params = {\n#         'min_child_weight': [1, 5, 10],\n#         'gamma': [0.5, 1, 1.5, 2, 5],\n#         'subsample': [0.6, 0.8, 1.0],\n#         'colsample_bytree': [0.6, 0.8, 1.0],\n#         'max_depth': [3, 4, 5]\n#         }\n\n# xg = xgb.XGBRegressor(random_state=42)\n# xg_random = RandomizedSearchCV(estimator = xg, param_distributions = params, n_iter = 100, cv = 3, verbose=0, random_state=42, n_jobs = -1)\n# xg_random.fit(xtrain,ytrain)\n# best_xgrandom = xg_random.best_estimator_\n# ypred=best_xgrandom.predict(xtest)\n# final = np.round(ypred)\n# accuracy_score(final, list(ytest))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:56:16.596577Z","iopub.execute_input":"2022-08-02T05:56:16.596988Z","iopub.status.idle":"2022-08-02T05:56:49.589853Z","shell.execute_reply.started":"2022-08-02T05:56:16.596952Z","shell.execute_reply":"2022-08-02T05:56:49.588956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ##Best parameters for xgboost after grid search\n\n# xg_best_param={'objective': 'reg:squarederror',\n#  'base_score': 0.5,\n#  'booster': 'gbtree',\n#  'callbacks': None,\n#  'colsample_bylevel': 1,\n#  'colsample_bynode': 1,\n#  'colsample_bytree': 0.6,\n#  'early_stopping_rounds': None,\n#  'enable_categorical': False,\n#  'eval_metric': None,\n#  'gamma': 1,\n#  'gpu_id': -1,\n#  'grow_policy': 'depthwise',\n#  'importance_type': None,\n#  'interaction_constraints': '',\n#  'learning_rate': 0.300000012,\n#  'max_bin': 256,\n#  'max_cat_to_onehot': 4,\n#  'max_delta_step': 0,\n#  'max_depth': 5,\n#  'max_leaves': 0,\n#  'min_child_weight': 1,\n#  'missing': nan,\n#  'monotone_constraints': '()',\n#  'n_estimators': 100,\n#  'n_jobs': 0,\n#  'num_parallel_tree': 1,\n#  'predictor': 'auto',\n#  'random_state': 42,\n#  'reg_alpha': 0,\n#  'reg_lambda': 1,\n#  'sampling_method': 'uniform',\n#  'scale_pos_weight': 1,\n#  'subsample': 0.8,\n#  'tree_method': 'exact',\n#  'validate_parameters': 1,\n#  'verbosity': None}\n\n\n# xgb_bestmodel = xgb.XGBRegressor(random_state=42)\n# xgb_bestmodel.set_params(**xg_best_param)\n# xgb_bestmodel.fit(xtrain,ytrain)\n# ypred2=xgb_bestmodel.predict(xtest)\n# accuracy_score(np.round(ypred2), list(ytest)) #Accuracy is 0.827","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:26:34.785949Z","iopub.execute_input":"2022-08-02T04:26:34.786513Z","iopub.status.idle":"2022-08-02T04:26:35.114152Z","shell.execute_reply.started":"2022-08-02T04:26:34.786467Z","shell.execute_reply":"2022-08-02T04:26:35.113031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_importances = xgb_model.feature_importances_\nprint(sorted(zip(feature_importances, list(train.columns)), reverse=True))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:54:03.078098Z","iopub.execute_input":"2022-08-02T05:54:03.078507Z","iopub.status.idle":"2022-08-02T05:54:03.088584Z","shell.execute_reply.started":"2022-08-02T05:54:03.078474Z","shell.execute_reply":"2022-08-02T05:54:03.086971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Random forest\nfrom sklearn.ensemble import RandomForestRegressor\n\nforest_model = RandomForestRegressor(random_state=1).fit(xtrain,ytrain)\nypred=forest_model.predict(xtest)\nfinal = np.round(ypred)\naccuracy_score(final, list(ytest))   #Accuracy is 0.821","metadata":{"execution":{"iopub.status.busy":"2022-08-02T05:54:16.515744Z","iopub.execute_input":"2022-08-02T05:54:16.516189Z","iopub.status.idle":"2022-08-02T05:54:16.872947Z","shell.execute_reply.started":"2022-08-02T05:54:16.516154Z","shell.execute_reply":"2022-08-02T05:54:16.871685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparam tuning for random forest\n\n# # Number of trees in random forest\n# n_estimators = [int(x) for x in np.linspace(start = 200, stop = 2000, num = 10)]\n# # Number of features to consider at every split\n# max_features = ['auto', 'sqrt']\n# # Maximum number of levels in tree\n# max_depth = [int(x) for x in np.linspace(10, 110, num = 11)]\n# max_depth.append(None)\n# # Minimum number of samples required to split a node\n# min_samples_split = [2, 5, 10]\n# # Minimum number of samples required at each leaf node\n# min_samples_leaf = [1, 2, 4]\n# # Method of selecting samples for training each tree\n# bootstrap = [True, False]\n# # Create the random grid\n# random_grid = {'n_estimators': n_estimators,\n#                'max_features': max_features,\n#                'max_depth': max_depth,\n#                'min_samples_split': min_samples_split,\n#                'min_samples_leaf': min_samples_leaf,\n#                'bootstrap': bootstrap}\n\n# rf=RandomForestRegressor(random_state=1)\n# rf_random = RandomizedSearchCV(estimator = rf, param_distributions = random_grid, n_iter = 100, cv = 3, verbose=0, random_state=42, n_jobs = -1)\n# rf_random.fit(xtrain,ytrain)\n# best_random = rf_random.best_estimator_\n# ypred=best_random.predict(xtest)\n# final = np.round(ypred)\n# accuracy_score(final, list(ytest)) ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:03:57.677226Z","iopub.execute_input":"2022-08-02T06:03:57.677622Z","iopub.status.idle":"2022-08-02T06:03:57.683321Z","shell.execute_reply.started":"2022-08-02T06:03:57.677590Z","shell.execute_reply":"2022-08-02T06:03:57.682376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Best parameters for random forest after grid search\n\nrf_best_param={'bootstrap': True,\n 'ccp_alpha': 0.0,\n 'criterion': 'squared_error',\n 'max_depth': 10,\n 'max_features': 'sqrt',\n 'max_leaf_nodes': None,\n 'max_samples': None,\n 'min_impurity_decrease': 0.0,\n 'min_samples_leaf': 2,\n 'min_samples_split': 10,\n 'min_weight_fraction_leaf': 0.0,\n 'n_estimators': 1000,\n 'n_jobs': None,\n 'oob_score': False,\n 'random_state': 1,\n 'verbose': 0,\n 'warm_start': False}\n\n\nrf_bestmodel = RandomForestRegressor(random_state=1)\nrf_bestmodel.set_params(**rf_best_param)\nrf_bestmodel.fit(xtrain,ytrain)\nypred2=rf_bestmodel.predict(xtest)\naccuracy_score(np.round(ypred2), list(ytest)) ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:04:03.993532Z","iopub.execute_input":"2022-08-02T06:04:03.994035Z","iopub.status.idle":"2022-08-02T06:04:05.679938Z","shell.execute_reply.started":"2022-08-02T06:04:03.993991Z","shell.execute_reply":"2022-08-02T06:04:05.678808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #Deep Learning\n\n# ncol=train.shape[1]\n\n# from tensorflow import keras\n# from tensorflow.keras import layers\n# from tensorflow.keras.callbacks import EarlyStopping\n\n# early_stopping = EarlyStopping(\n#     min_delta=0.001, # minimium amount of change to count as an improvement\n#     patience=20, # how many epochs to wait before stopping\n#     restore_best_weights=True,\n# )\n\n\n# #We try with 2 hidden layers\n\n# model = keras.Sequential([\n#     # the 2 hidden ReLU layers\n#     layers.Dense(units=3, activation='relu', input_shape=[ncol]),\n#     layers.Dense(units=3, activation='relu'),\n#     # the linear output layer \n#     layers.Dense(units=1,activation='sigmoid'),])\n\n# model.compile(\n#     optimizer='adam',\n#     loss='binary_crossentropy',\n#     metrics=['binary_accuracy'],)\n\n# history = model.fit(\n#     xtrain, ytrain,\n#     validation_data=(xtest, ytest),\n#     batch_size=20,\n#     epochs=150,\n#     callbacks=[early_stopping],\n#     verbose=1\n# )\n\n# history_df = pd.DataFrame(history.history)\n# history_df.loc[:, ['loss', 'val_loss']].plot();\n# print(\"Minimum validation loss: {}\".format(history_df['val_loss'].min()))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:05:06.306768Z","iopub.execute_input":"2022-08-02T04:05:06.307667Z","iopub.status.idle":"2022-08-02T04:05:06.314993Z","shell.execute_reply.started":"2022-08-02T04:05:06.307621Z","shell.execute_reply":"2022-08-02T04:05:06.313181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Submission with tuned rf model\nypred=rf_bestmodel.predict(test)\nfinal = np.round(ypred)\noutput = pd.DataFrame({'PassengerId': Id,'Survived': final.astype(int) })\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:41:36.250994Z","iopub.execute_input":"2022-08-02T04:41:36.251434Z","iopub.status.idle":"2022-08-02T04:41:36.323094Z","shell.execute_reply.started":"2022-08-02T04:41:36.251400Z","shell.execute_reply":"2022-08-02T04:41:36.322191Z"},"trusted":true},"execution_count":null,"outputs":[]}]}