{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T04:45:54.186494Z","iopub.execute_input":"2022-08-02T04:45:54.187949Z","iopub.status.idle":"2022-08-02T04:45:55.542716Z","shell.execute_reply.started":"2022-08-02T04:45:54.187825Z","shell.execute_reply":"2022-08-02T04:45:55.541817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wrangle(filepath):\n    df = pd.read_csv(filepath)\n    \n    # Removing unnecessary columns \n    df.drop(columns=[\"Cabin\", \"Ticket\", \"Name\", \"Embarked\"], inplace=True)\n    # Change Sex from male to 1 and female to 0\n    df[\"Sex\"] = (df[\"Sex\"].str[0]==\"m\").replace(\"m\", 1).astype(int)\n    # forward filling those columns which have NaN values\n    df[\"Age\"] = df[\"Age\"].ffill()\n    df[\"Fare\"] = df[\"Fare\"].ffill()\n    \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:46:55.711443Z","iopub.execute_input":"2022-08-02T04:46:55.711908Z","iopub.status.idle":"2022-08-02T04:46:55.719357Z","shell.execute_reply.started":"2022-08-02T04:46:55.711870Z","shell.execute_reply":"2022-08-02T04:46:55.718492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = wrangle(\"/kaggle/input/titanic/train.csv\")\nprint(train.info())","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:49:10.458335Z","iopub.execute_input":"2022-08-02T04:49:10.458769Z","iopub.status.idle":"2022-08-02T04:49:10.513340Z","shell.execute_reply.started":"2022-08-02T04:49:10.458735Z","shell.execute_reply":"2022-08-02T04:49:10.512451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = wrangle(\"/kaggle/input/titanic/test.csv\")\ntest_ids = test[\"PassengerId\"]\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:50:04.387056Z","iopub.execute_input":"2022-08-02T04:50:04.387822Z","iopub.status.idle":"2022-08-02T04:50:04.411609Z","shell.execute_reply.started":"2022-08-02T04:50:04.387756Z","shell.execute_reply":"2022-08-02T04:50:04.410582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = \"Survived\"\nX = train.drop(columns=\"Survived\")\ny= train[target]\nprint(\"X shape: \", X.shape)\nprint(\"y shape: \", y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:51:12.923432Z","iopub.execute_input":"2022-08-02T04:51:12.923907Z","iopub.status.idle":"2022-08-02T04:51:12.932876Z","shell.execute_reply.started":"2022-08-02T04:51:12.923869Z","shell.execute_reply":"2022-08-02T04:51:12.931862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val= train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:51:26.840507Z","iopub.execute_input":"2022-08-02T04:51:26.840959Z","iopub.status.idle":"2022-08-02T04:51:26.851915Z","shell.execute_reply.started":"2022-08-02T04:51:26.840923Z","shell.execute_reply":"2022-08-02T04:51:26.850026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_baseline = y.value_counts(normalize=True).max()\nprint(\"Baseline Accuracy\", round(acc_baseline, 4))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:51:35.372219Z","iopub.execute_input":"2022-08-02T04:51:35.372673Z","iopub.status.idle":"2022-08-02T04:51:35.380461Z","shell.execute_reply.started":"2022-08-02T04:51:35.372633Z","shell.execute_reply":"2022-08-02T04:51:35.379396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LogisticRegression(max_iter=1000)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:51:45.742101Z","iopub.execute_input":"2022-08-02T04:51:45.742578Z","iopub.status.idle":"2022-08-02T04:51:45.824688Z","shell.execute_reply.started":"2022-08-02T04:51:45.742542Z","shell.execute_reply":"2022-08-02T04:51:45.823316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_train = model.score(X_train, y_train)\nacc_train","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:52:06.822278Z","iopub.execute_input":"2022-08-02T04:52:06.822698Z","iopub.status.idle":"2022-08-02T04:52:06.835444Z","shell.execute_reply.started":"2022-08-02T04:52:06.822660Z","shell.execute_reply":"2022-08-02T04:52:06.834048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc_val = model.score(X_val, y_val)\nacc_val","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:52:15.627295Z","iopub.execute_input":"2022-08-02T04:52:15.627699Z","iopub.status.idle":"2022-08-02T04:52:15.638282Z","shell.execute_reply.started":"2022-08-02T04:52:15.627666Z","shell.execute_reply":"2022-08-02T04:52:15.637232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test)\ndf = pd.DataFrame({\"PassengerId\":test_ids.values,\n                  \"Survived\":predictions})\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:52:32.156888Z","iopub.execute_input":"2022-08-02T04:52:32.157314Z","iopub.status.idle":"2022-08-02T04:52:32.176441Z","shell.execute_reply.started":"2022-08-02T04:52:32.157273Z","shell.execute_reply":"2022-08-02T04:52:32.175203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"predictions.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T04:52:47.852575Z","iopub.execute_input":"2022-08-02T04:52:47.853010Z","iopub.status.idle":"2022-08-02T04:52:47.862524Z","shell.execute_reply.started":"2022-08-02T04:52:47.852971Z","shell.execute_reply":"2022-08-02T04:52:47.861229Z"},"trusted":true},"execution_count":null,"outputs":[]}]}