{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T06:04:01.973246Z","iopub.execute_input":"2022-07-14T06:04:01.973622Z","iopub.status.idle":"2022-07-14T06:04:02.267722Z","shell.execute_reply.started":"2022-07-14T06:04:01.973568Z","shell.execute_reply":"2022-07-14T06:04:02.266697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing necessary libraries & dependencies\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import confusion_matrix\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:02:25.710490Z","iopub.execute_input":"2022-07-14T07:02:25.711056Z","iopub.status.idle":"2022-07-14T07:02:25.716237Z","shell.execute_reply.started":"2022-07-14T07:02:25.711006Z","shell.execute_reply":"2022-07-14T07:02:25.715508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load train data into a dataframe\n\ndf = pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-07-14T06:04:03.837555Z","iopub.execute_input":"2022-07-14T06:04:03.837982Z","iopub.status.idle":"2022-07-14T06:04:03.861733Z","shell.execute_reply.started":"2022-07-14T06:04:03.837903Z","shell.execute_reply":"2022-07-14T06:04:03.860857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the info of all the features\n\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.864708Z","iopub.execute_input":"2022-07-14T06:04:03.865017Z","iopub.status.idle":"2022-07-14T06:04:03.882571Z","shell.execute_reply.started":"2022-07-14T06:04:03.864969Z","shell.execute_reply":"2022-07-14T06:04:03.881554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Numerical description of the features in dataframe\n\ndf.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.886092Z","iopub.execute_input":"2022-07-14T06:04:03.886427Z","iopub.status.idle":"2022-07-14T06:04:03.938552Z","shell.execute_reply.started":"2022-07-14T06:04:03.886350Z","shell.execute_reply":"2022-07-14T06:04:03.937551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for empty/null values in the dataframe\n\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.942069Z","iopub.execute_input":"2022-07-14T06:04:03.942402Z","iopub.status.idle":"2022-07-14T06:04:03.952776Z","shell.execute_reply.started":"2022-07-14T06:04:03.942330Z","shell.execute_reply":"2022-07-14T06:04:03.952024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Missing Data Handling**\n","metadata":{}},{"cell_type":"code","source":"# Majority of 'Cabin' column is missing so it is better to remove it.\n\ndf = df.drop(columns='Cabin', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.953992Z","iopub.execute_input":"2022-07-14T06:04:03.954444Z","iopub.status.idle":"2022-07-14T06:04:03.967900Z","shell.execute_reply.started":"2022-07-14T06:04:03.954372Z","shell.execute_reply":"2022-07-14T06:04:03.966922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.969239Z","iopub.execute_input":"2022-07-14T06:04:03.969821Z","iopub.status.idle":"2022-07-14T06:04:03.996828Z","shell.execute_reply.started":"2022-07-14T06:04:03.969760Z","shell.execute_reply":"2022-07-14T06:04:03.996024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling \"Age\" column with the mean value of it. \n\ndf['Age'].fillna(df['Age'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:03.998033Z","iopub.execute_input":"2022-07-14T06:04:03.998516Z","iopub.status.idle":"2022-07-14T06:04:04.005628Z","shell.execute_reply.started":"2022-07-14T06:04:03.998466Z","shell.execute_reply":"2022-07-14T06:04:04.004742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling \"Embarked\" column with its mode.\n\ndf['Embarked'].fillna((df['Embarked'].mode()[0]), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.007004Z","iopub.execute_input":"2022-07-14T06:04:04.007428Z","iopub.status.idle":"2022-07-14T06:04:04.021013Z","shell.execute_reply.started":"2022-07-14T06:04:04.007253Z","shell.execute_reply":"2022-07-14T06:04:04.020054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.022618Z","iopub.execute_input":"2022-07-14T06:04:04.022962Z","iopub.status.idle":"2022-07-14T06:04:04.038988Z","shell.execute_reply.started":"2022-07-14T06:04:04.022903Z","shell.execute_reply":"2022-07-14T06:04:04.037816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have zero entries in our data set that are null/empty.","metadata":{}},{"cell_type":"code","source":"# Correlation of Survived with every other feature in Data set\n\ndf.corr()['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.040558Z","iopub.execute_input":"2022-07-14T06:04:04.040950Z","iopub.status.idle":"2022-07-14T06:04:04.058176Z","shell.execute_reply.started":"2022-07-14T06:04:04.040908Z","shell.execute_reply":"2022-07-14T06:04:04.056980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describing categorical columns\n\ndf.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.059760Z","iopub.execute_input":"2022-07-14T06:04:04.060216Z","iopub.status.idle":"2022-07-14T06:04:04.098821Z","shell.execute_reply.started":"2022-07-14T06:04:04.060134Z","shell.execute_reply":"2022-07-14T06:04:04.097972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Value counts of respective features**","metadata":{}},{"cell_type":"code","source":"df['Embarked'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.100176Z","iopub.execute_input":"2022-07-14T06:04:04.100645Z","iopub.status.idle":"2022-07-14T06:04:04.110206Z","shell.execute_reply.started":"2022-07-14T06:04:04.100428Z","shell.execute_reply":"2022-07-14T06:04:04.109490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Parch'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.111599Z","iopub.execute_input":"2022-07-14T06:04:04.112328Z","iopub.status.idle":"2022-07-14T06:04:04.125354Z","shell.execute_reply.started":"2022-07-14T06:04:04.112235Z","shell.execute_reply":"2022-07-14T06:04:04.124186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.127337Z","iopub.execute_input":"2022-07-14T06:04:04.127925Z","iopub.status.idle":"2022-07-14T06:04:04.138998Z","shell.execute_reply.started":"2022-07-14T06:04:04.127863Z","shell.execute_reply":"2022-07-14T06:04:04.138183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bar graph representing the number of people survived or not.\n\nplt.figure(figsize=(4,6))\nax = sns.countplot(df[\"Survived\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.140297Z","iopub.execute_input":"2022-07-14T06:04:04.140789Z","iopub.status.idle":"2022-07-14T06:04:04.395023Z","shell.execute_reply.started":"2022-07-14T06:04:04.140739Z","shell.execute_reply":"2022-07-14T06:04:04.393699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bar grpah representing the number of people survived or not, based on their Passenger Class.\n\nplt.figure(figsize=(6,8))\nax = sns.countplot(df[\"Pclass\"],hue=df[\"Survived\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.396913Z","iopub.execute_input":"2022-07-14T06:04:04.397565Z","iopub.status.idle":"2022-07-14T06:04:04.749763Z","shell.execute_reply.started":"2022-07-14T06:04:04.397500Z","shell.execute_reply":"2022-07-14T06:04:04.748612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Bar grpah representing the number of people survived or not, based on their Gender.\n\nplt.figure(figsize=(6,8))\nax = sns.countplot(df[\"Sex\"],hue=df[\"Survived\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:04.751764Z","iopub.execute_input":"2022-07-14T06:04:04.752488Z","iopub.status.idle":"2022-07-14T06:04:05.130816Z","shell.execute_reply.started":"2022-07-14T06:04:04.752414Z","shell.execute_reply":"2022-07-14T06:04:05.129286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Histogram of the distribution of Ages of people who boarded the Titanic\n\nplt.figure(figsize=(6,7))\ndf[\"Age\"].plot.hist()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:20:40.434869Z","iopub.execute_input":"2022-07-14T06:20:40.435184Z","iopub.status.idle":"2022-07-14T06:20:40.755318Z","shell.execute_reply.started":"2022-07-14T06:20:40.435137Z","shell.execute_reply":"2022-07-14T06:20:40.754420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation of Survival column with respect to all other features.\n\ndf.corr()[\"Survived\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:05.133301Z","iopub.execute_input":"2022-07-14T06:04:05.133794Z","iopub.status.idle":"2022-07-14T06:04:05.147473Z","shell.execute_reply.started":"2022-07-14T06:04:05.133714Z","shell.execute_reply":"2022-07-14T06:04:05.146298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:04:05.149250Z","iopub.execute_input":"2022-07-14T06:04:05.149887Z","iopub.status.idle":"2022-07-14T06:04:05.207545Z","shell.execute_reply.started":"2022-07-14T06:04:05.149821Z","shell.execute_reply":"2022-07-14T06:04:05.206739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Encoding the categorical columns**","metadata":{}},{"cell_type":"code","source":"# Convert values using one hot encoding\n\npd.get_dummies(df[\"Sex\"]).head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:26:25.786003Z","iopub.execute_input":"2022-07-14T06:26:25.786583Z","iopub.status.idle":"2022-07-14T06:26:25.797878Z","shell.execute_reply.started":"2022-07-14T06:26:25.786335Z","shell.execute_reply":"2022-07-14T06:26:25.797046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# To reduce redundant columns in the data set.\n\nsex = pd.get_dummies(df[\"Sex\"], drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:28:37.468463Z","iopub.execute_input":"2022-07-14T06:28:37.468800Z","iopub.status.idle":"2022-07-14T06:28:37.475940Z","shell.execute_reply.started":"2022-07-14T06:28:37.468748Z","shell.execute_reply":"2022-07-14T06:28:37.475078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One hot encoding  and reduction of redundant rows for \"Embarked\" feature.\n\nembark = pd.get_dummies(df[\"Embarked\"], drop_first=True)\nembark.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:31:00.511989Z","iopub.execute_input":"2022-07-14T06:31:00.512367Z","iopub.status.idle":"2022-07-14T06:31:00.524993Z","shell.execute_reply.started":"2022-07-14T06:31:00.512302Z","shell.execute_reply":"2022-07-14T06:31:00.524078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One hot encoding  and reduction of redundant rows for \"PClass\" feature.\n\npclass = pd.get_dummies(df[\"Pclass\"], drop_first=True)\npclass.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:46:12.775167Z","iopub.execute_input":"2022-07-14T06:46:12.775587Z","iopub.status.idle":"2022-07-14T06:46:12.787707Z","shell.execute_reply.started":"2022-07-14T06:46:12.775518Z","shell.execute_reply":"2022-07-14T06:46:12.786669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df = pd.concat([df, sex, embark, pclass], axis=1)\ntitanic_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:47:35.189971Z","iopub.execute_input":"2022-07-14T06:47:35.190354Z","iopub.status.idle":"2022-07-14T06:47:35.219500Z","shell.execute_reply.started":"2022-07-14T06:47:35.190293Z","shell.execute_reply":"2022-07-14T06:47:35.217181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df.drop([\"PassengerId\",\"Pclass\",\"Name\",\"Sex\",\"Ticket\",\"Embarked\"], axis=1, inplace=True)\ntitanic_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:49:34.515061Z","iopub.execute_input":"2022-07-14T06:49:34.515427Z","iopub.status.idle":"2022-07-14T06:49:34.539018Z","shell.execute_reply.started":"2022-07-14T06:49:34.515360Z","shell.execute_reply":"2022-07-14T06:49:34.537521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training the Data using Logistic Regression**","metadata":{}},{"cell_type":"code","source":"# Creating X and Y values from Dataset\n\nX = titanic_df.drop(\"Survived\", axis=1)\ny = titanic_df[\"Survived\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:51:28.921019Z","iopub.execute_input":"2022-07-14T06:51:28.921608Z","iopub.status.idle":"2022-07-14T06:51:28.928908Z","shell.execute_reply.started":"2022-07-14T06:51:28.921538Z","shell.execute_reply":"2022-07-14T06:51:28.927631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train & Test data splitting\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:53:01.414323Z","iopub.execute_input":"2022-07-14T06:53:01.414962Z","iopub.status.idle":"2022-07-14T06:53:01.423907Z","shell.execute_reply.started":"2022-07-14T06:53:01.414913Z","shell.execute_reply":"2022-07-14T06:53:01.422798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Implementing the model\n\nmodel = LogisticRegression()\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T06:55:05.976537Z","iopub.execute_input":"2022-07-14T06:55:05.976883Z","iopub.status.idle":"2022-07-14T06:55:06.096533Z","shell.execute_reply.started":"2022-07-14T06:55:05.976830Z","shell.execute_reply":"2022-07-14T06:55:06.095828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict based on training\n\npred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:00:16.395592Z","iopub.execute_input":"2022-07-14T07:00:16.395945Z","iopub.status.idle":"2022-07-14T07:00:16.403094Z","shell.execute_reply.started":"2022-07-14T07:00:16.395903Z","shell.execute_reply":"2022-07-14T07:00:16.401543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the True Positives, False negatives, etc. using confusion matrix.\n\nconfusion_matrix(y_test, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:02:59.478277Z","iopub.execute_input":"2022-07-14T07:02:59.478622Z","iopub.status.idle":"2022-07-14T07:02:59.489547Z","shell.execute_reply.started":"2022-07-14T07:02:59.478573Z","shell.execute_reply":"2022-07-14T07:02:59.488511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check our Accuracy score\n\naccuracy_score(y_test, pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:03:14.306767Z","iopub.execute_input":"2022-07-14T07:03:14.307136Z","iopub.status.idle":"2022-07-14T07:03:14.314630Z","shell.execute_reply.started":"2022-07-14T07:03:14.307072Z","shell.execute_reply":"2022-07-14T07:03:14.313859Z"},"trusted":true},"execution_count":null,"outputs":[]}]}