{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"\ntrain_set = pd.read_csv(\"../input/train.csv\")\n\ntest_set = pd.read_csv(\"../input/test.csv\")\n\ngender_set = pd.read_csv(\"../input/gender_submission.csv\")\n\ntrain_set.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d05aea85007a7b946c6d922c12744ba5a61bc56a"},"cell_type":"code","source":"test_set.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f04e5fcb0a859e62db9c77c07c21aa10fe992fc"},"cell_type":"code","source":"gender_set.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70bc0ce6a21388292e6eb490f56d82f879e223c4"},"cell_type":"code","source":"train_label = train_set['Survived'].copy()\ntrain_label.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f98e107dd352249977d40c3b6a52d6afacc724ac"},"cell_type":"code","source":"train_set = train_set.drop(['PassengerId', 'Survived', 'Cabin'], axis=1)\ntrain_set.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5496df8623e6581d72923802a9fb03c112440fd4"},"cell_type":"code","source":"train_set.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec0322318e0ec6059e5abe17bda65f2cc5f89630"},"cell_type":"code","source":"train_set.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f78376624b3e32b154cb6341ded0695e8d2b4b0c"},"cell_type":"code","source":"test_set = test_set.drop(['PassengerId', 'Cabin'], axis=1)\ntest_label = gender_set.drop(['PassengerId'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3073e2bacc8809173f572805ceb844af21c8825e"},"cell_type":"code","source":"num_features = ['Age', 'Fare']\ncat_features = ['Pclass', 'Sex', 'SibSp', 'Parch', 'Embarked']\nstr_features = ['Name', 'Ticket']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b63e44073337ee8c1f62cba190f37a3073fedb6"},"cell_type":"code","source":"train_label.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cecbae3d32c6e49bb09b156e04688ee729f39db"},"cell_type":"code","source":"train_label.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a4daafb954449cd2a7a0dd83d06d2c834bbe8e4"},"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\nclass DataFrameSelector(BaseEstimator, TransformerMixin):\n    def __init__(self, attribute_names):\n        self.attribute_names = attribute_names\n    def fit(self, X, y=None):\n        return self\n    def transform(self, X):\n        return X[self.attribute_names]\n      \n      \nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\n\nnum_pipeline = Pipeline([\n        (\"select_numeric\", DataFrameSelector(num_features)),\n        (\"imputer\", SimpleImputer(strategy=\"median\")),\n    ])\n\nnum_pipeline.fit_transform(train_set)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e977530c0b1d16c33cb31a8c2318fb474011bd14"},"cell_type":"code","source":"class MostFrequentImputer(BaseEstimator, TransformerMixin):\n    def fit(self, X, y=None):\n        self.most_frequent_ = pd.Series([X[c].value_counts().index[0] for c in X],\n                                        index=X.columns)\n        return self\n    def transform(self, X, y=None):\n        return X.fillna(self.most_frequent_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbcf7bc408382d75c02cdbe8cd39415c3b795c69"},"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\ncat_pipeline = Pipeline([\n    (\"cat_features_selection\", DataFrameSelector(cat_features)),\n    (\"fill_with_most_frequent\", MostFrequentImputer()),\n    (\"one_hot_encoder\", OneHotEncoder(sparse=False))\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d8fd2cbae49163d99eb055a4fe3d0b3b45ef0d5"},"cell_type":"code","source":"both_set = train_set.append(test_set, ignore_index=True)\nboth_set.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db7663a1d3d17f09935228627031aa92ae44a582"},"cell_type":"code","source":"from sklearn.pipeline import FeatureUnion\n\npreprocess_pipeline = FeatureUnion(transformer_list=[\n        (\"num_pipeline\", num_pipeline),\n        #(\"most_frequent_pipeline\", most_frequent_pipeline),\n        (\"cat_pipeline\", cat_pipeline),\n    ], n_jobs=1)\n\npreprocess_pipeline.fit(both_set)\nX_train = preprocess_pipeline.transform(train_set)\nX_test = preprocess_pipeline.transform(test_set)\ny_test = test_label.values\ny_train = train_label.values\nX_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0946dcb277b5802d6a470eb72829e3c437e0474"},"cell_type":"code","source":"from sklearn.svm import SVC\n\n\nsvm_clf = SVC()\nsvm_clf.fit(X_train, y_train)\nsurvived = svm_clf.predict([X_train[42]])\nprint('this guy survided? {}, we predicted {}'.format(y_train[42], survived))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86e0231e8174ab40faef8fd12d282fcb12a8dcbd"},"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\nsvm_clf_cross_val = cross_val_score(estimator=svm_clf, X=X_train, y=y_train, cv=5)\n\nsvm_clf_cross_val.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc05e1ff7cc28e594f6a88995f871cd26a24b831"},"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\nGNB_clf = GaussianNB()\n\nGNB_clf_cross_val = cross_val_score(estimator=GNB_clf, X=X_train, y=y_train, cv=5)\n\nGNB_clf_cross_val.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45f4cc8b7d7e40f1a1c2021b79ebabe4819a6c11"},"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nKNN_clf = KNeighborsClassifier()\n\nKNN_clf_cross_val = cross_val_score(estimator=KNN_clf, X=X_train, y=y_train, cv=5)\nKNN_clf_cross_val.mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bbe0dafe52b8585abcf2e3fa44dd204e2ac9124c"},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nRF_clf = RandomForestClassifier()\n\nRF_clf_cross_val = cross_val_score(estimator=RF_clf, X=X_train, y=y_train, cv=5)\nprint('score: {}'.format(RF_clf_cross_val.mean()))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3734edcbdc180294320311caed53106868d1ba4"},"cell_type":"code","source":"# let's focus on RandomForest classifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc2c3a049340962b27a546fd44b054b4e327f9c2"},"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nimport numpy as np\nRF_clf.fit(X_train, y_train)\n\ny_predict = RF_clf.predict(X_test)\n\nprint('score on test_set: {}'.format(accuracy_score(y_test, y_predict)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34935bc8dc138ae1c493335d9f95538440a68b94"},"cell_type":"code","source":"from sklearn.model_selection import RandomizedSearchCV\nimport scipy as sp\n\nparam_distributions={'n_estimators': sp.stats.randint(2, 10),\n                     'max_features': sp.stats.randint(3, 10),\n                     #'bootstrap': [True, False],\n}\n\n\n\nsearch = RandomizedSearchCV(RF_clf, param_distributions, cv = 10, n_iter=100, scoring = 'accuracy', random_state=42)\n\nsearch.fit(X_train, y_train)\n\n\nprint(search.best_score_)\nprint(search.best_params_)\n\ny_predict = search.best_estimator_.predict(X_test)\nsum_good_predict = sum(y_predict == y_test.reshape([-1]))\nscore = (sum_good_predict/len(y_predict))\nprint('score on test_set: {}'.format(score))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe235084941739dbdb15f9fbe6701e987c4255bf"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}