{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.linear_model import LogisticRegression\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T17:18:54.608058Z","iopub.execute_input":"2022-07-27T17:18:54.609179Z","iopub.status.idle":"2022-07-27T17:18:55.153558Z","shell.execute_reply.started":"2022-07-27T17:18:54.609063Z","shell.execute_reply":"2022-07-27T17:18:55.152204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.155367Z","iopub.execute_input":"2022-07-27T17:18:55.155751Z","iopub.status.idle":"2022-07-27T17:18:55.173643Z","shell.execute_reply.started":"2022-07-27T17:18:55.155715Z","shell.execute_reply":"2022-07-27T17:18:55.172356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.175267Z","iopub.execute_input":"2022-07-27T17:18:55.175744Z","iopub.status.idle":"2022-07-27T17:18:55.201829Z","shell.execute_reply.started":"2022-07-27T17:18:55.175698Z","shell.execute_reply":"2022-07-27T17:18:55.200425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.205916Z","iopub.execute_input":"2022-07-27T17:18:55.206509Z","iopub.status.idle":"2022-07-27T17:18:55.222437Z","shell.execute_reply.started":"2022-07-27T17:18:55.206474Z","shell.execute_reply":"2022-07-27T17:18:55.221433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.223657Z","iopub.execute_input":"2022-07-27T17:18:55.224662Z","iopub.status.idle":"2022-07-27T17:18:55.239467Z","shell.execute_reply.started":"2022-07-27T17:18:55.224607Z","shell.execute_reply":"2022-07-27T17:18:55.237975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### NaN Values","metadata":{}},{"cell_type":"code","source":"median_age = train_df[\"Age\"].median()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.241506Z","iopub.execute_input":"2022-07-27T17:18:55.242269Z","iopub.status.idle":"2022-07-27T17:18:55.251675Z","shell.execute_reply.started":"2022-07-27T17:18:55.242222Z","shell.execute_reply":"2022-07-27T17:18:55.250615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[train_df[\"Age\"].isna(), \"Age\"] = median_age","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.253113Z","iopub.execute_input":"2022-07-27T17:18:55.253721Z","iopub.status.idle":"2022-07-27T17:18:55.263394Z","shell.execute_reply.started":"2022-07-27T17:18:55.253684Z","shell.execute_reply":"2022-07-27T17:18:55.262093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.loc[test_df[\"Age\"].isna(), \"Age\"] = median_age","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.265122Z","iopub.execute_input":"2022-07-27T17:18:55.265628Z","iopub.status.idle":"2022-07-27T17:18:55.275781Z","shell.execute_reply.started":"2022-07-27T17:18:55.265578Z","shell.execute_reply":"2022-07-27T17:18:55.274476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Visualization","metadata":{}},{"cell_type":"markdown","source":"### Feature Engineering","metadata":{}},{"cell_type":"code","source":"cat_features = [\n    \"Sex\",\n    # \"Embarked\",\n]\n\nnum_features = [\n    \"Pclass\",\n    \"Age\",\n]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.277561Z","iopub.execute_input":"2022-07-27T17:18:55.278446Z","iopub.status.idle":"2022-07-27T17:18:55.287512Z","shell.execute_reply.started":"2022-07-27T17:18:55.278383Z","shell.execute_reply":"2022-07-27T17:18:55.286296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.get_dummies(train_df[cat_features], drop_first=True)\nX_test = pd.get_dummies(test_df[cat_features], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.289077Z","iopub.execute_input":"2022-07-27T17:18:55.290263Z","iopub.status.idle":"2022-07-27T17:18:55.306706Z","shell.execute_reply.started":"2022-07-27T17:18:55.290213Z","shell.execute_reply":"2022-07-27T17:18:55.305257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = pd.concat([X_train, train_df[num_features]], axis=1)\nX_test = pd.concat([X_test, test_df[num_features]], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.324141Z","iopub.execute_input":"2022-07-27T17:18:55.324720Z","iopub.status.idle":"2022-07-27T17:18:55.343967Z","shell.execute_reply.started":"2022-07-27T17:18:55.324684Z","shell.execute_reply":"2022-07-27T17:18:55.342394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df[\"Survived\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.349802Z","iopub.execute_input":"2022-07-27T17:18:55.350246Z","iopub.status.idle":"2022-07-27T17:18:55.356084Z","shell.execute_reply.started":"2022-07-27T17:18:55.350212Z","shell.execute_reply":"2022-07-27T17:18:55.354840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"code","source":"log_reg = LogisticRegression()\nlog_reg.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.357983Z","iopub.execute_input":"2022-07-27T17:18:55.358720Z","iopub.status.idle":"2022-07-27T17:18:55.385112Z","shell.execute_reply.started":"2022-07-27T17:18:55.358673Z","shell.execute_reply":"2022-07-27T17:18:55.383823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"test_df[\"Survived\"] = log_reg.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.386488Z","iopub.execute_input":"2022-07-27T17:18:55.386902Z","iopub.status.idle":"2022-07-27T17:18:55.396359Z","shell.execute_reply.started":"2022-07-27T17:18:55.386848Z","shell.execute_reply":"2022-07-27T17:18:55.394890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.398114Z","iopub.execute_input":"2022-07-27T17:18:55.398634Z","iopub.status.idle":"2022-07-27T17:18:55.423717Z","shell.execute_reply.started":"2022-07-27T17:18:55.398578Z","shell.execute_reply":"2022-07-27T17:18:55.422211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[[\"PassengerId\", \"Survived\"]].to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T17:18:55.425241Z","iopub.execute_input":"2022-07-27T17:18:55.426058Z","iopub.status.idle":"2022-07-27T17:18:55.437743Z","shell.execute_reply.started":"2022-07-27T17:18:55.426004Z","shell.execute_reply":"2022-07-27T17:18:55.436282Z"},"trusted":true},"execution_count":null,"outputs":[]}]}