{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport scipy as sp\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"_kg_hide-output":false,"_kg_hide-input":true,"scrolled":false},"cell_type":"code","source":"titanic_test = pd.read_csv(\"../input/test.csv\")\ntitanic_test\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9f437107f8b9306443409f5066d3c72d24dc24f8"},"cell_type":"markdown","source":""},{"metadata":{"_kg_hide-output":false,"_kg_hide-input":true,"trusted":true,"_uuid":"cf1b310aa083e544afa211f9e587c134359d36d5","scrolled":false},"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.linear_model import LogisticRegression\nohe = OneHotEncoder()\ntitanic = pd.read_csv(\"../input/train.csv\")\ntitanic[\"CabinGroup\"] = titanic[\"Cabin\"].fillna(\"\").str[:1]\ny_test = titanic[\"Survived\"].values\nX_numeric = titanic[[\"Age\", \"Fare\"]].fillna(0).values\nX_categorical = ohe.fit_transform(titanic[[\"Pclass\", \"Sex\", \"Embarked\", \"CabinGroup\"]].fillna(\"\").values)\nX_train = sp.hstack([X_numeric,X_categorical.todense()])\nmodel = LogisticRegression()\nmodel.fit(X_train, y_train)\nmodel.score(X_train, y_train)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ceeb864c2d4383abfb2767b9bdb6f9a847177be"},"cell_type":"code","source":"%pylab inline\nscatter(titanic[\"Age\"], titanic[\"Fare\"], c=titanic[\"Survived\"])\nxlabel(\"Age\")\nylabel(\"Fare\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5027f3bebb81cebc0bc1aea02cdecc1b0a776260"},"cell_type":"code","source":"from sklearn import linear_model\nmodel = linear_model.LogisticRegression(solver=\"lbfgs\")\nmodel.fit(titanic[[\"Age\",\"Fare\"]].fillna(0), titanic[\"Survived\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6662aa23428e2b66d7d07292b1d4a397cf6ab6ba"},"cell_type":"code","source":"scatter(titanic[\"Age\"], titanic[\"Fare\"], c=titanic[\"Survived\"])\nxlabel(\"Age\")\nylabel(\"Fare\")\nx_min, x_max = 0, 80\ny_min, y_max = 0, 500\nxx, yy = np.meshgrid(np.arange(x_min, x_max, 5),\n                     np.arange(y_min, y_max, 50))\nZ = model.predict(np.c_[xx.ravel(), yy.ravel()])\nZ = Z.reshape(xx.shape)\ncontour(xx, yy, Z, colors=\"teal\", levels=[0.5], linewidths=5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a2e6126979d9ac9113241745e38c992e850de12"},"cell_type":"code","source":"from sklearn import neighbors\nknn_model = neighbors.KNeighborsClassifier(5)\nknn_model.fit(titanic[[\"Age\",\"Fare\"]].fillna(0), titanic[\"Survived\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39ee414f19523b8346120350440425620273939e"},"cell_type":"code","source":"{\"logreg\": model.score(titanic[[\"Age\",\"Fare\"]].fillna(0), titanic[\"Survived\"]), \n \"knn\": knn_model.score(titanic[[\"Age\",\"Fare\"]].fillna(0), titanic[\"Survived\"])}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a9fbb49f691c63f41e7e4bb7df71134c404364e"},"cell_type":"code","source":"from sklearn import compose, impute, pipeline, preprocessing\n\nnumeric_features = [\"Age\", \"Fare\"]\ncategorical_features = [\"Pclass\", \"Sex\"]\n\nnumeric_transformer = pipeline.make_pipeline(\n  impute.SimpleImputer(strategy=\"median\"),\n  preprocessing.StandardScaler())\ncategorical_transformer = pipeline.make_pipeline(\n  impute.SimpleImputer(strategy=\"constant\", fill_value=\"NA\"),\n  preprocessing.OneHotEncoder(handle_unknown=\"ignore\"))\n\npreprocessor = compose.make_column_transformer(\n  (numeric_transformer, numeric_features),\n  (categorical_transformer, categorical_features))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b462610cc68ed423502a56a75352a1e0f20cffd6"},"cell_type":"code","source":"preprocessor.fit(titanic)\nX = preprocessor.transform(titanic)\ny = titanic[\"Survived\"]\n\nmodel.fit(X, y)\nknn_model.fit(X, y)\n\n{\"logreg\": model.score(X, y), \"knn\": knn_model.score(X, y)}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac9d457f45bf7bdb7b8c13ecf3cccdb4bfbecf59"},"cell_type":"code","source":"X_test = preprocessor.transform(titanic_test)\n\ny_pred = knn_model.predict(X_test)\n\nsubmission = pd.DataFrame({\n    \"PassengerId\": titanic_test[\"PassengerId\"],\n    \"Survived\": y_pred\n})\nsubmission.to_csv(\"submission.csv\", index=False)\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}