{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport seaborn as sns\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"dataset = pd.read_csv(\"../input/train.csv\")\ndataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d486d630158473005f2adba6e4b4a0052115a921"},"cell_type":"code","source":"dataset[\"Ticket\"] = dataset[\"Ticket\"].apply(len)\ndataset[\"Name\"] = dataset[\"Name\"].apply(len)\ndataset[\"FamilySize\"] = dataset[\"SibSp\"] + dataset[\"Parch\"]\ndataset[\"IsAlone\"] = dataset[\"FamilySize\"].apply(lambda x:1 if x>0 else 0)\ndataset[\"Cabin\"] = dataset[\"Cabin\"].apply(lambda x:1 if type(x) == str else 0)\ndataset[\"Age\"] = dataset[\"Age\"].interpolate()\ndataset = pd.get_dummies(dataset, drop_first=True)\ndataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f164b95d81580c2508b56158656f1e66401b4240"},"cell_type":"code","source":"# Remove Outliers \nfrom scipy import stats\ndataset = dataset[(np.abs(stats.zscore(dataset)) < 4).all(axis=1)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8afde39d27ed7a9383f2e69f5602027493319aa9"},"cell_type":"code","source":"X = dataset.iloc[:,2:]\ny = dataset.iloc[:, dataset.columns==\"Survived\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26b052fcfd169051f6e8e8890446d34be8b08c0f"},"cell_type":"code","source":"dataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c280788584b16f21ab485b93dbfce4777d52f19"},"cell_type":"code","source":"#sns.pairplot(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"affae3671e78fd589cbc9fb9d29e985fee2a481c"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\ncolormap = plt.cm.RdBu\nplt.figure(figsize=(14,12))\nplt.title('Pearson Correlation of Features', y=1.05, size=15)\nsns.heatmap(dataset.iloc[:,1:].corr(),linewidths=0.1,vmax=1.0, \n            square=True, cmap=colormap, linecolor='white', annot=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"767389a212d520b83e76ce573bd139c50179755a"},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsc_x = StandardScaler()\nX = sc_x.fit_transform(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fedb21441f3890a031b2f82a29882dc20417c25d"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, stratify=y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3faf39a3b9e7f86c5a97a3008411170fd979f837"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier, ExtraTreesClassifier, VotingClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV, cross_val_score, StratifiedKFold, learning_curve\n\nrf = RandomForestClassifier(    \n    n_jobs= -1,\n    n_estimators= 50,\n    warm_start= True, \n    max_depth= 6,\n    min_samples_leaf= 2,\n    max_features = \"sqrt\",\n    verbose= 1)\n\nadaboost = AdaBoostClassifier(    \n            n_estimators= 50,\n            learning_rate = 0.3,\n            )\n\ngb = GradientBoostingClassifier(    \n    n_estimators= 50,\n    max_features= 0.2,\n    max_depth= 5,\n    min_samples_leaf= 2,\n    verbose= 1)\n\nlda = LinearDiscriminantAnalysis()\n\nlog_reg = LogisticRegression()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56107c48bb61cbd1d090869ebb065a51a683c66c"},"cell_type":"code","source":"params_rf = {\n    \"n_estimators\": [50],\n    \"criterion\": [\"gini\"],\n    \"max_depth\":[10,11,12, None],\n    \"max_features\": [\"log2\", None],\n    \"class_weight\":[\"balanced\",\"balanced_subsample\"]\n}\n\nrf_opt = GridSearchCV(rf, params_rf, scoring = \"accuracy\", cv=10, verbose=0, n_jobs=-1).fit(X_train,np.ravel(y_train))\nrf_opt = rf_opt.best_estimator_\nrf_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad358e95313a1307ddfb187944937349bb39641b"},"cell_type":"code","source":"dt = DecisionTreeClassifier()\ndt.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57e54317f6a19d1045e98614864d6a409d7388ef"},"cell_type":"code","source":"from sklearn.tree import export_graphviz\nexport_graphviz(dt,\n                feature_names=dataset.iloc[:,2:].columns,\n                filled=True,\n                rounded=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fec88d9b5f4fddd71747d290f9a4f9ce105f21c4"},"cell_type":"code","source":"params_adaboost = {\n    \"n_estimators\": [50],\n    \"learning_rate\": [0.51, 0.52,0.53],\n    \"algorithm\":[\"SAMME\"]\n}\n\n    adaboost_opt = GridSearchCV(adaboost, params_adaboost, scoring = \"accuracy\", cv=10, verbose=0, n_jobs=-1).fit(X_train,np.ravel(y_train))\nadaboost_opt = adaboost_opt.best_estimator_\nadaboost_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6226d995f9e317c6459323ef72b8dc21601230a0"},"cell_type":"code","source":"params_gb = {\n    \"n_estimators\": [50],\n    \"max_features\": [0.3],\n    \"max_depth\":[2,3,4,5, None],\n    \"min_samples_leaf\": [1,2,3,4],\n    \"loss\": [\"deviance\",\"exponential\"],\n    \"criterion\": [\"mse\", \"mae\"]\n}\n\ngb_opt = GridSearchCV(gb, params_gb, scoring = \"accuracy\", cv=10, verbose=0, n_jobs=-1).fit(X_train,np.ravel(y_train))\ngb_opt = gb_opt.best_estimator_\ngb_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8dca019a9524fae34407a0791f84108caa87570c"},"cell_type":"code","source":"params_lda = {\n    \"solver\": [\"svd\", \"lsqr\"]\n}\n\nlda_opt = GridSearchCV(lda, params_lda, scoring = \"accuracy\", cv=10, verbose=0, n_jobs=-1).fit(X_train,np.ravel(y_train))\nlda_opt = lda_opt.best_estimator_\nlda_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88df54cc27270e048cc217296c2d699f2dab54b7"},"cell_type":"code","source":"params_logreg = {\n    \"solver\": [\"newton-cg\", \"lbfgs\", \"liblinear\", \"sag\", \"saga\"],\n    \"C\": [0.1,0.5,0.8,1.0],    \n}\n\nlog_reg_opt = GridSearchCV(log_reg, params_logreg, scoring = \"accuracy\", cv=10, verbose=0, n_jobs=-1).fit(X_train,np.ravel(y_train))\nlog_reg_opt = log_reg_opt.best_estimator_\nlog_reg_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19db5a51e0eed1434b084c57c5eb533e4d59ab9b","scrolled":false},"cell_type":"code","source":"print(\"RF Score {}\".format(cross_val_score(rf_opt, X_test, np.ravel(y_test), scoring = \"accuracy\", cv= 15).mean()))\nprint(\"Adaboost Score {}\".format(cross_val_score(adaboost_opt, X_test, np.ravel(y_test), scoring = \"accuracy\", cv= 15).mean()))\nprint(\"Gradient Boosting Score {}\".format(cross_val_score(gb_opt, X_test, np.ravel(y_test), scoring = \"accuracy\", cv= 15).mean()))\nprint(\"LDA Score {}\".format(cross_val_score(lda_opt, X_test, np.ravel(y_test), scoring = \"accuracy\", cv= 15).mean()))\nprint(\"Log_Reg Score {}\".format(cross_val_score(log_reg_opt, X_test, np.ravel(y_test), scoring = \"accuracy\", cv= 15).mean()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e7f617bc9e1cdb2538c261ac5c7b6c1bb48b12c"},"cell_type":"code","source":"rf_feature = rf_opt.feature_importances_\nadaboost_feature = adaboost_opt.feature_importances_\ngb_feature = gb_opt.feature_importances_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"268433382fa08ca322e438cfd6721f2706a7fd96"},"cell_type":"code","source":"cols = pd.DataFrame(X).columns.values\n# Create a dataframe with features\nfeature_dataframe = pd.DataFrame( \n    {'features': cols,\n    'Random Forest feature importances': rf_feature,\n    'AdaBoost feature importances': adaboost_feature,\n    'Gradient Boost feature importances': gb_feature\n    })\n\nfeature_dataframe","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8bece9e6345d64f4383fb0b22e6b23bc98a1898"},"cell_type":"code","source":"votingC = VotingClassifier(estimators=[('rf', rf_opt), ('adaboost', adaboost_opt), ('gb', gb_opt), (\"lda\", lda_opt), (\"log_reg\", log_reg_opt)], voting='soft')\nvotingC.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc040827499b6db0bcf3d3987eefcaeef1b07ce4"},"cell_type":"code","source":"rf_pred = rf_opt.predict_proba(X)\nada_pred = adaboost_opt.predict_proba(X)\ngb_pred = gb_opt.predict_proba(X)\nlda_pred = lda_opt.predict_proba(X)\nlog_reg_pred = log_reg_opt.predict_proba(X)\nvotingC_pred = votingC.predict_proba(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40977f623ceff421b2a9d6739a04d2964497d75a"},"cell_type":"code","source":"predictions = pd.DataFrame({\"RandomForest_pred\": rf_pred[:,1].ravel(), \n                            \"AdaBoost_pred\": ada_pred[:,1].ravel(), \n                           \"GradientBoosting_pred\":gb_pred[:,1].ravel(),\n                           \"LDA_pred\": lda_pred[:,1].ravel(),\n                            \"log_reg_pred\": log_reg_pred[:,1].ravel(),\n                            \"votingC_pred\": votingC_pred[:,1].ravel()\n                           })\npredictions.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b48801b11b166e28d7d9fcd482a80b2a98378b12"},"cell_type":"code","source":"X = pd.concat([pd.DataFrame(X), predictions], axis=1)\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4376002abd8e542c6b54103639300561fc28e941"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25, stratify=y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63866ce400c9fa5edb8c66120c084f147e9e055b"},"cell_type":"code","source":"import xgboost as xgb\nxgboost = xgb.XGBClassifier(\n learning_rate = 0.01,\n max_depth= 4,\n min_child_weight= 2,\n gamma=0.9,                        \n subsample=0.8,\n colsample_bytree=0.8,\n nthread= -1,\n scale_pos_weight=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42ce1a9358aa65c688050fc1f398871777ef61a6"},"cell_type":"code","source":"params_xgb = {\n    \"learning_rate\": [0.01, 0.02],\n    \"max_depth\":[2,3,4,5],\n    \"gamma\": [0.7,0.8,0.9],\n}\n\nxgb_opt = GridSearchCV(xgboost, params_xgb, scoring = \"accuracy\", cv=10, verbose=1, n_jobs=-1).fit(X_train,y_train)\nxgb_opt = xgb_opt.best_estimator_\nxgb_opt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa1817ad65d3a9066ad28f4f8fb646245426bc3e"},"cell_type":"code","source":"#xgboost.fit(X_train, np.ravel(y_train))\ncross_val_score(xgb_opt, X_test, np.ravel(y_test), scoring = \"roc_auc\", cv= 15).mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e20642db4d9f44b9d39918fed7d2433b861ba852"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f34206c64ec9994d0718ec651715935f23de9c15"},"cell_type":"code","source":"X = test.iloc[:,1:]\nX[\"Name\"] = X[\"Name\"].apply(len)\nX[\"Ticket\"] = X[\"Ticket\"].apply(len)\nX[\"FamilySize\"] = X[\"SibSp\"] + X[\"Parch\"]\nX[\"IsAlone\"] = X[\"FamilySize\"].apply(lambda x:1 if x>0 else 0)\nX[\"Cabin\"] = X[\"Cabin\"].apply(lambda x:1 if type(x) == str else 0)\nX[\"Age\"] = X[\"Age\"].interpolate()\nX[\"Fare\"] = X[\"Fare\"].interpolate()\nX = pd.get_dummies(X, drop_first=True)\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0da6e48c5b5b1cd525dfa40d3bd1aa6bdc487fd"},"cell_type":"code","source":"X = sc_x.fit_transform(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42a36fd8094b098f2f7cc104514f636dec703c51"},"cell_type":"code","source":"rf_pred = rf_opt.predict_proba(X)\nada_pred = adaboost_opt.predict_proba(X)\ngb_pred = gb_opt.predict_proba(X)\nlda_pred = lda_opt.predict_proba(X)\nlog_reg_pred = log_reg_opt.predict_proba(X)\nvotingC_pred = votingC.predict_proba(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cca9acf5974fbf43aedea68978d072694ecd32eb"},"cell_type":"code","source":"predictions = pd.DataFrame({\"RandomForest_pred\": rf_pred[:,1].ravel(), \n                            \"AdaBoost_pred\": ada_pred[:,1].ravel(), \n                           \"GradientBoosting_pred\":gb_pred[:,1].ravel(),\n                           \"LDA_pred\": lda_pred[:,1].ravel(),\n                            \"log_reg_pred\": log_reg_pred[:,1].ravel(),\n                            \"votingC_pred\": votingC_pred[:,1].ravel()\n                           })\nX = pd.concat([pd.DataFrame(X), predictions], axis=1)\nX.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6677e40cc0566b6c0128ac4efa611b477ba0f3e3"},"cell_type":"code","source":"y_pred = xgb_opt.predict(X)\nsubmission = pd.concat([test.iloc[:,0], pd.DataFrame(y_pred, columns=[\"Survived\"])], axis=1)\nsubmission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff176ae12596fe493e5f803090ae5eaa6593e346"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}