{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T02:19:44.175053Z","iopub.execute_input":"2022-07-27T02:19:44.175827Z","iopub.status.idle":"2022-07-27T02:19:44.209331Z","shell.execute_reply.started":"2022-07-27T02:19:44.175704Z","shell.execute_reply":"2022-07-27T02:19:44.208330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read data and some exploration","metadata":{}},{"cell_type":"code","source":"## read data \ntrain_df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\n\ngender_submission = pd.read_csv(\"/kaggle/input/titanic/gender_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:21:57.742614Z","iopub.execute_input":"2022-07-27T02:21:57.743067Z","iopub.status.idle":"2022-07-27T02:21:57.807779Z","shell.execute_reply.started":"2022-07-27T02:21:57.743024Z","shell.execute_reply":"2022-07-27T02:21:57.806566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## take a look\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:22:10.525216Z","iopub.execute_input":"2022-07-27T02:22:10.525681Z","iopub.status.idle":"2022-07-27T02:22:10.561239Z","shell.execute_reply.started":"2022-07-27T02:22:10.525647Z","shell.execute_reply":"2022-07-27T02:22:10.560446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## summarization.....someing missing?\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:23:14.006601Z","iopub.execute_input":"2022-07-27T02:23:14.007043Z","iopub.status.idle":"2022-07-27T02:23:14.052437Z","shell.execute_reply.started":"2022-07-27T02:23:14.007008Z","shell.execute_reply":"2022-07-27T02:23:14.051636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## only the numeric columns?\ntrain_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:25:18.751062Z","iopub.execute_input":"2022-07-27T02:25:18.751763Z","iopub.status.idle":"2022-07-27T02:25:18.763702Z","shell.execute_reply.started":"2022-07-27T02:25:18.751715Z","shell.execute_reply":"2022-07-27T02:25:18.762043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## here you are....\ntrain_df.describe(include=[\"object\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:25:59.310425Z","iopub.execute_input":"2022-07-27T02:25:59.310857Z","iopub.status.idle":"2022-07-27T02:25:59.338437Z","shell.execute_reply.started":"2022-07-27T02:25:59.310824Z","shell.execute_reply":"2022-07-27T02:25:59.337192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## everything is correlated\ntrain_df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:27:24.191321Z","iopub.execute_input":"2022-07-27T02:27:24.191796Z","iopub.status.idle":"2022-07-27T02:27:24.212686Z","shell.execute_reply.started":"2022-07-27T02:27:24.191757Z","shell.execute_reply":"2022-07-27T02:27:24.211443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://external-preview.redd.it/i61pQn1_S6CFegwvzGC0K_ZuKATtzmSh_1IH5_-yfuU.png?width=960&crop=smart&auto=webp&s=55894f55f9bb5a323810be76f52a7095721a2754)","metadata":{}},{"cell_type":"markdown","source":"## A simple guess\n\nHow about a logisitic regression model with 2 features: \n$$ y = \\sigma(\\alpha + \\beta_1  x_1 + \\beta_2  x_2 + \\epsilon)$$","metadata":{}},{"cell_type":"code","source":"## set up the target\ny = train_df[\"Survived\"]\n\n## select the features\nfeatures_in_model = [\"Age\", \"Sex\"]\nX = train_df[features_in_model]\n\n## but the \"Sex\" is not numeric...... not allowed in a logistic regression\nX","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:30:16.543677Z","iopub.execute_input":"2022-07-27T02:30:16.544124Z","iopub.status.idle":"2022-07-27T02:30:16.564191Z","shell.execute_reply.started":"2022-07-27T02:30:16.544092Z","shell.execute_reply":"2022-07-27T02:30:16.563371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.Sex.replace([\"male\", \"female\"], [0, 1])","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:46.112969Z","iopub.execute_input":"2022-07-26T07:25:46.114096Z","iopub.status.idle":"2022-07-26T07:25:46.125900Z","shell.execute_reply.started":"2022-07-26T07:25:46.114052Z","shell.execute_reply":"2022-07-26T07:25:46.124311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## replace the original column with facotorized one\nX[\"Sex\"] = X.Sex.replace([\"male\", \"female\"], [0, 1])\n\n## Still..... there are NA in \"Age\"\nX","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:32:13.739979Z","iopub.execute_input":"2022-07-27T02:32:13.740501Z","iopub.status.idle":"2022-07-27T02:32:13.761149Z","shell.execute_reply.started":"2022-07-27T02:32:13.740461Z","shell.execute_reply":"2022-07-27T02:32:13.759762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## fill the NA in Age with mean value\nX[\"Age\"] = X.Age.fillna(29.7)\n\nX\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:33:30.753258Z","iopub.execute_input":"2022-07-27T02:33:30.753997Z","iopub.status.idle":"2022-07-27T02:33:30.773521Z","shell.execute_reply.started":"2022-07-27T02:33:30.753953Z","shell.execute_reply":"2022-07-27T02:33:30.772101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Scikit-learn \nfrom sklearn.linear_model import LogisticRegression\n\n## set up a logistic regression model\nlr_model = LogisticRegression()\n\n## train the model with X and y\nlr_model.fit(X, y)\n\n## the evaluation method in the competition is \"accuracy\"\nlr_model.score(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:35:55.127016Z","iopub.execute_input":"2022-07-27T02:35:55.128169Z","iopub.status.idle":"2022-07-27T02:35:55.802680Z","shell.execute_reply.started":"2022-07-27T02:35:55.128113Z","shell.execute_reply":"2022-07-27T02:35:55.801440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## do all the preprocessing works on the test set\n\nX_test = test_df[features_in_model]\nX_test[\"Sex\"] = X_test.Sex.replace([\"male\", \"female\"], [0, 1])\nX_test[\"Age\"] = X_test.Age.fillna(29.7)\n\n## predict with test set\ngender_submission[\"Survived\"] = lr_model.predict(X_test)\ngender_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:37:47.615962Z","iopub.execute_input":"2022-07-27T02:37:47.616468Z","iopub.status.idle":"2022-07-27T02:37:47.637384Z","shell.execute_reply.started":"2022-07-27T02:37:47.616430Z","shell.execute_reply":"2022-07-27T02:37:47.636140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## save as csv\n## gender_submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:38:43.464578Z","iopub.execute_input":"2022-07-27T02:38:43.465058Z","iopub.status.idle":"2022-07-27T02:38:43.475488Z","shell.execute_reply.started":"2022-07-27T02:38:43.465020Z","shell.execute_reply":"2022-07-27T02:38:43.474481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## More and More","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:39:05.204811Z","iopub.execute_input":"2022-07-26T07:39:05.205382Z","iopub.status.idle":"2022-07-26T07:39:05.210983Z","shell.execute_reply.started":"2022-07-26T07:39:05.205326Z","shell.execute_reply":"2022-07-26T07:39:05.209588Z"}}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\ntree_model = DecisionTreeClassifier()\n\n## train the model with X and y\ntree_model.fit(X, y)\n\n## the evaluation method in the competition is \"accuracy\"\ntree_model.score(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:42:12.484188Z","iopub.execute_input":"2022-07-27T02:42:12.484681Z","iopub.status.idle":"2022-07-27T02:42:12.574280Z","shell.execute_reply.started":"2022-07-27T02:42:12.484641Z","shell.execute_reply":"2022-07-27T02:42:12.573026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## predict with test set\ngender_submission[\"Survived\"] = tree_model.predict(X_test)\ngender_submission\n\n## save as csv\ngender_submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T02:42:21.847633Z","iopub.execute_input":"2022-07-27T02:42:21.848072Z","iopub.status.idle":"2022-07-27T02:42:21.856568Z","shell.execute_reply.started":"2022-07-27T02:42:21.848033Z","shell.execute_reply":"2022-07-27T02:42:21.855427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ","metadata":{}}]}