{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T09:45:44.165228Z","iopub.execute_input":"2022-07-27T09:45:44.165855Z","iopub.status.idle":"2022-07-27T09:45:44.181458Z","shell.execute_reply.started":"2022-07-27T09:45:44.165702Z","shell.execute_reply":"2022-07-27T09:45:44.180308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:45:44.923725Z","iopub.execute_input":"2022-07-27T09:45:44.924542Z","iopub.status.idle":"2022-07-27T09:45:44.935712Z","shell.execute_reply.started":"2022-07-27T09:45:44.924499Z","shell.execute_reply":"2022-07-27T09:45:44.934744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()\ndf.head()\ndf.count()\n\n# print(df.groupby(['ColName'])['ColName'].count())\n# df.describe()\n# print(df.count())\n# print(df['ColName'].isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:45:47.858537Z","iopub.execute_input":"2022-07-27T09:45:47.859183Z","iopub.status.idle":"2022-07-27T09:45:47.888553Z","shell.execute_reply.started":"2022-07-27T09:45:47.859142Z","shell.execute_reply":"2022-07-27T09:45:47.887672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(columns=['Name','Ticket','Cabin','Embarked'])\n\ndf.head()\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:46:13.323451Z","iopub.execute_input":"2022-07-27T09:46:13.323765Z","iopub.status.idle":"2022-07-27T09:46:13.329731Z","shell.execute_reply.started":"2022-07-27T09:46:13.323718Z","shell.execute_reply":"2022-07-27T09:46:13.328402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Cleaning the attribute (Sex)**\n\nMale = 0\nFemale = 1","metadata":{}},{"cell_type":"code","source":"df['Sex'] = df['Sex'].replace(['male','female'],[0,1])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:46:23.350582Z","iopub.execute_input":"2022-07-27T09:46:23.350859Z","iopub.status.idle":"2022-07-27T09:46:23.367391Z","shell.execute_reply.started":"2022-07-27T09:46:23.350829Z","shell.execute_reply":"2022-07-27T09:46:23.366291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['Sex'])['Sex'].count().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:46:25.476356Z","iopub.execute_input":"2022-07-27T09:46:25.477080Z","iopub.status.idle":"2022-07-27T09:46:25.678489Z","shell.execute_reply.started":"2022-07-27T09:46:25.477030Z","shell.execute_reply":"2022-07-27T09:46:25.676594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Cleaning the attribute (Age)**","metadata":{}},{"cell_type":"code","source":"plt.hist(df['Age'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:46:32.499275Z","iopub.execute_input":"2022-07-27T09:46:32.499932Z","iopub.status.idle":"2022-07-27T09:46:32.812394Z","shell.execute_reply.started":"2022-07-27T09:46:32.499891Z","shell.execute_reply":"2022-07-27T09:46:32.811762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Remove Missing Values\ndf['Age'] = df['Age'].fillna(df['Age'].mean())\n\n#Remove Outliers\nq1 = df['Age'].quantile(0.25)\nq3 = df['Age'].quantile(0.75)\niqr = q3-q1\nlower = q1 - (1.5*iqr)\nupper = q3 + (1.5*iqr)\ndf = df[(df['Age']>= lower)&(df['Age']<= upper)]\n\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:46:41.868176Z","iopub.execute_input":"2022-07-27T09:46:41.868848Z","iopub.status.idle":"2022-07-27T09:46:41.882469Z","shell.execute_reply.started":"2022-07-27T09:46:41.868777Z","shell.execute_reply":"2022-07-27T09:46:41.881587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.boxplot(df['Age'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:47:12.909398Z","iopub.execute_input":"2022-07-27T09:47:12.909766Z","iopub.status.idle":"2022-07-27T09:47:13.056130Z","shell.execute_reply.started":"2022-07-27T09:47:12.909728Z","shell.execute_reply":"2022-07-27T09:47:13.055190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['Age'].isnull().sum())\nprint(df.count())","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:47:20.458721Z","iopub.execute_input":"2022-07-27T09:47:20.459040Z","iopub.status.idle":"2022-07-27T09:47:20.467996Z","shell.execute_reply.started":"2022-07-27T09:47:20.459009Z","shell.execute_reply":"2022-07-27T09:47:20.467330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(df['Age'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:47:34.975221Z","iopub.execute_input":"2022-07-27T09:47:34.975675Z","iopub.status.idle":"2022-07-27T09:47:35.168573Z","shell.execute_reply.started":"2022-07-27T09:47:34.975634Z","shell.execute_reply":"2022-07-27T09:47:35.167643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Features and Labels (Modelling)**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_val_score\n\n#Dataset Split\nx = df.drop(['Survived'],1)\ny = df['Survived']\n\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:58:09.219772Z","iopub.execute_input":"2022-07-27T09:58:09.220798Z","iopub.status.idle":"2022-07-27T09:58:09.232242Z","shell.execute_reply.started":"2022-07-27T09:58:09.220742Z","shell.execute_reply":"2022-07-27T09:58:09.231310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\n#Regression Model\nreg1 = LinearRegression().fit(x_train, y_train)\nprint('Accuracy Score: ',reg1.score(x_test, y_test))\nprint('Cross Validation Score: ', cross_val_score(reg1, x_test, y_test, cv=5))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:58:10.309430Z","iopub.execute_input":"2022-07-27T09:58:10.309706Z","iopub.status.idle":"2022-07-27T09:58:10.345540Z","shell.execute_reply.started":"2022-07-27T09:58:10.309677Z","shell.execute_reply":"2022-07-27T09:58:10.344653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\n\n#Decision Tree\nclf1 = DecisionTreeClassifier().fit(x_train, y_train)\nprint('Accuracy Score: ',clf1.score(x_test, y_test))\nprint('Cross Validation Score: ', cross_val_score(clf1, x_test, y_test, cv=5))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:58:25.958264Z","iopub.execute_input":"2022-07-27T09:58:25.958614Z","iopub.status.idle":"2022-07-27T09:58:26.000608Z","shell.execute_reply.started":"2022-07-27T09:58:25.958567Z","shell.execute_reply":"2022-07-27T09:58:25.999649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\n#Logistic Regression Model\nreg2 = LogisticRegression().fit(x_train, y_train)\nprint('Accuracy Score: ',reg2.score(x_test, y_test))\nprint('Cross Validation Score: ', cross_val_score(reg2, x_test, y_test, cv=5))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T09:58:44.382477Z","iopub.execute_input":"2022-07-27T09:58:44.382790Z","iopub.status.idle":"2022-07-27T09:58:44.554284Z","shell.execute_reply.started":"2022-07-27T09:58:44.382752Z","shell.execute_reply":"2022-07-27T09:58:44.553480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\n#Random Forests Classifier\nclf2 = RandomForestClassifier().fit(x_train, y_train)\nprint('Accuracy Score: ',clf2.score(x_test, y_test))\nprint('Cross Validation Score: ', cross_val_score(clf2, x_test, y_test, cv=5))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:03:23.348579Z","iopub.execute_input":"2022-07-27T10:03:23.348864Z","iopub.status.idle":"2022-07-27T10:03:24.493431Z","shell.execute_reply.started":"2022-07-27T10:03:23.348828Z","shell.execute_reply":"2022-07-27T10:03:24.492557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\n#Extra Tree Classifier\nclf3 = KNeighborsClassifier(algorithm='auto', leaf_size=26, metric='minkowski',metric_params=None, n_jobs=1, n_neighbors=6, p=2, weights='uniform').fit(x_train, y_train)\nprint('Accuracy Score: ',clf3.score(x_test, y_test))\nprint('Cross Validation Score: ', cross_val_score(clf3, x_test, y_test, cv=5))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:09:07.695259Z","iopub.execute_input":"2022-07-27T10:09:07.695598Z","iopub.status.idle":"2022-07-27T10:09:07.749675Z","shell.execute_reply.started":"2022-07-27T10:09:07.695561Z","shell.execute_reply":"2022-07-27T10:09:07.748930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Test Modelling**","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"../input/titanic/test.csv\")\n\ntest_df['Sex'] = test_df['Sex'].replace(['male','female'],[0,1])\ntest_df['Age'] = test_df['Age'].fillna(test_df['Age'].mean())\ntest_df['Fare'] = test_df['Fare'].fillna(test_df['Fare'].mean())\ntest_df = test_df.drop(columns=['Name','Ticket','Cabin','Embarked'])\n\ntest_df[\"Survived\"] = clf2.predict(test_df)\ngender_submission = test_df[['PassengerId','Survived']]\ngender_submission.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T10:03:50.690531Z","iopub.execute_input":"2022-07-27T10:03:50.690834Z","iopub.status.idle":"2022-07-27T10:03:50.733105Z","shell.execute_reply.started":"2022-07-27T10:03:50.690798Z","shell.execute_reply":"2022-07-27T10:03:50.732467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}