{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport sklearn\n\nfrom sklearn.model_selection import KFold\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\ntrain.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afbabf0dfffb40b8a6edea6cd69e3f5782452520"},"cell_type":"code","source":"age = train['Age']\nage.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d10eee6952af0156918621864420563edbb709d6"},"cell_type":"code","source":"age.fillna(age.mean(), inplace=True)\nage.describe","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d205f14aef2a22c8cb1e4bd22431d5cc6b6df87"},"cell_type":"code","source":"ntrain = age.shape[0]\nSEED = 0 # for reproducibility\nNFOLDS = 2 # set folds for out-of-fold prediction\nkf = KFold(n_splits=NFOLDS, random_state=SEED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cdf12fd38c284e58e3d918c316c3de908c2fa8b6"},"cell_type":"code","source":"test_age = test['Age']\ntest_age.fillna(test_age.mean(), inplace=True)\nntest = test_age.shape[0]\nprint(ntest)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d9aca7bde5ae7092177c137219131ee26104e1c"},"cell_type":"code","source":"# Copied from https://www.kaggle.com/arthurtok/introduction-to-ensembling-stacking-in-python\ndef get_oof(clf, x_train, y_train, x_test):\n    oof_train = np.zeros((ntrain,))\n    oof_test = np.zeros((ntest,))\n    oof_test_skf = np.empty((NFOLDS, ntest))\n\n    for i, (train_index, test_index) in enumerate(kf.split(x_train,y_train)):\n        x_tr = x_train[train_index]\n        y_tr = y_train[train_index]\n        x_te = x_train[test_index]\n#         print(str(train_index) + \" : \" + str(test_index) + \" : \"+str(i))\n        clf.fit(x_tr, y_tr)\n\n        oof_train[test_index] = clf.predict(x_te)\n        oof_test_skf[i, :] = clf.predict(x_test)\n\n    oof_test[:] = oof_test_skf.mean(axis=0)\n    return oof_train.reshape(-1, 1), oof_test.reshape(-1, 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f68cd1f6bf25e1d30bc5ccaf40815e7c25d40bed"},"cell_type":"code","source":"# Support Vector Classifier parameters \nsvc_params = {\n    'kernel' : 'linear',\n    'C' : 0.025,\n    'random_state': SEED\n    }\n\n# AdaBoost parameters\nada_params = {\n    'n_estimators': 500,\n    'learning_rate' : 0.75,\n    'random_state': SEED\n}\n\n# Random Forest parameters\nrf_params = {\n    'n_jobs': -1,\n    'n_estimators': 500,\n     'warm_start': True, \n     #'max_features': 0.2,\n    'max_depth': 6,\n    'min_samples_leaf': 2,\n    'max_features' : 'sqrt',\n    'verbose': 0\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93814aee0e124e4c4fe3d376407dfb9488fbd459"},"cell_type":"code","source":"# Going to use these 2 base models for the stacking\nfrom sklearn.ensemble import (AdaBoostClassifier, RandomForestClassifier)\nfrom sklearn.svm import SVC","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea76ca1e2c4a2ee8e23da601e1deffd4277b414c"},"cell_type":"code","source":"# svc = SVC(**svc_params)\nada = AdaBoostClassifier(**ada_params)\nrf = RandomForestClassifier(**rf_params)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ddfa9db5beb96965d7a1d0191f1bc151302c2a89"},"cell_type":"code","source":"y_train = train['Survived'].ravel()\nprint(y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"276af5e8bc8d4c31beef4f97436707c82900bc01"},"cell_type":"code","source":"# Class to extend the Sklearn classifier\nclass SklearnHelper(object):\n    def __init__(self, clf, seed=0, params=None):\n        params['random_state'] = seed\n        self.clf = clf(**params)\n\n    def train(self, x_train, y_train):\n        self.clf.fit(x_train, y_train)\n\n    def predict(self, x):\n        return self.clf.predict(x)\n    \n    def fit(self,x,y):\n        return self.clf.fit(x,y)\n    \n    def feature_importances(self,x,y):\n        print(self.clf.fit(x,y).feature_importances_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"760b5ecfda1b6257d215244dea3f58543cb549b3"},"cell_type":"code","source":"# Create our OOF train and test predictions. These base results will be used as new features\nx_train = age.values\nx_test = test_age.values\n\nx_train = x_train.reshape(-1, 1)\nx_test = x_test.reshape(-1, 1)\n\n\nada_oof_train, ada_oof_test = get_oof(ada, x_train, y_train, x_test) # AdaBoost \nrf_oof_train, rf_oof_test = get_oof(rf,x_train, y_train, x_test) # Random Forest\n# svc_oof_train, svc_oof_test = get_oof(svc,x_train, y_train, x_test) # Support Vector Classifier\n\nprint(\"Training is complete\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de9383d0a885bc4c2ca1df0b9422ad088b36c2c4"},"cell_type":"code","source":"print(rf.fit(x_train,y_train).feature_importances_)\n# rf_feature = rf.feature_importances(x_train,y_train)\nprint(ada.fit(x_train,y_train).feature_importances_)\n# svc_feature = svc.feature_importances(x_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9350a59d3b2105efd968178eb26077c5fafa15a3"},"cell_type":"code","source":"rf_features = rf.fit(x_train,y_train).feature_importances_\nada_features = ada.fit(x_train,y_train).feature_importances_\nfeature_dataframe = pd.DataFrame( {\n     'Random Forest feature importances': rf_features,\n      'AdaBoost feature importances': ada_features,\n    })","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db0ce846153183c12e682a4f52600b2a27957acb"},"cell_type":"code","source":"print(feature_dataframe)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0659332bd84d182e33770bfabaeacce301b9820"},"cell_type":"code","source":"x_train = np.concatenate(( rf_oof_train, ada_oof_train), axis=1)\nx_test = np.concatenate(( rf_oof_test, ada_oof_test), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb9867cb117f89fe0db83b7209ec4b8b049115bb"},"cell_type":"code","source":"import xgboost as xgb\n\ngbm = xgb.XGBClassifier(\n    #learning_rate = 0.02,\n n_estimators= 2000,\n max_depth= 4,\n min_child_weight= 2,\n #gamma=1,\n gamma=0.9,                        \n subsample=0.8,\n colsample_bytree=0.8,\n objective= 'binary:logistic',\n nthread= -1,\n scale_pos_weight=1).fit(x_train, y_train)\npredictions = gbm.predict(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23248495fc0af3de0c9e2dcb6d5045a1a4a8eb1e"},"cell_type":"code","source":"print(predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51b04d4ac4080e6c579f934c823cb70c5abdb062"},"cell_type":"code","source":"StackingSubmission = pd.DataFrame({ 'PassengerId': test['PassengerId'],\n                            'Survived': predictions })\nStackingSubmission.to_csv(\"StackingSubmission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6856dbe6e09c16825f29511e030ba55ac38518d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}