{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# data analysis and wrangling\nimport pandas as pd\nimport numpy as np\nimport random as rnd\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# find NULL Data\nimport missingno as msno\n\n# machine learning\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom lightgbm import LGBMClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.633504Z","iopub.execute_input":"2022-07-07T06:31:12.634284Z","iopub.status.idle":"2022-07-07T06:31:12.645661Z","shell.execute_reply.started":"2022-07-07T06:31:12.634251Z","shell.execute_reply":"2022-07-07T06:31:12.644385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/titanic/train.csv')\ndf_test = pd.read_csv('../input/titanic/test.csv')\ncombine = [df_train, df_test]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.647821Z","iopub.execute_input":"2022-07-07T06:31:12.648912Z","iopub.status.idle":"2022-07-07T06:31:12.674063Z","shell.execute_reply.started":"2022-07-07T06:31:12.648860Z","shell.execute_reply":"2022-07-07T06:31:12.673108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.675588Z","iopub.execute_input":"2022-07-07T06:31:12.676429Z","iopub.status.idle":"2022-07-07T06:31:12.696733Z","shell.execute_reply.started":"2022-07-07T06:31:12.676372Z","shell.execute_reply":"2022-07-07T06:31:12.695733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NaN data\nfor col in df_train.columns:\n    msg = 'columns: {:>10}\\t Percent of NaN value: {:.2f}%'.format(col, 100 * (df_train[col].isnull().sum() / df_train[col].shape[0]))\n    print(msg)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.698065Z","iopub.execute_input":"2022-07-07T06:31:12.698909Z","iopub.status.idle":"2022-07-07T06:31:12.714126Z","shell.execute_reply.started":"2022-07-07T06:31:12.698875Z","shell.execute_reply":"2022-07-07T06:31:12.712537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NaN data\nfor col in df_test.columns:\n    msg = 'columns: {:>10}\\t Percent of NaN value: {:.2f}%'.format(col, 100 * (df_test[col].isnull().sum() / df_test[col].shape[0]))\n    print(msg)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.716704Z","iopub.execute_input":"2022-07-07T06:31:12.717484Z","iopub.status.idle":"2022-07-07T06:31:12.730119Z","shell.execute_reply.started":"2022-07-07T06:31:12.717447Z","shell.execute_reply":"2022-07-07T06:31:12.728879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df=df_train.iloc[:, :], figsize=(8, 8), color=(0.1, 0.6, 0.8))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:12.732986Z","iopub.execute_input":"2022-07-07T06:31:12.733513Z","iopub.status.idle":"2022-07-07T06:31:13.146516Z","shell.execute_reply.started":"2022-07-07T06:31:12.733467Z","shell.execute_reply":"2022-07-07T06:31:13.144734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(df=df_test.iloc[:, :], figsize=(8, 8), color=(0.1, 0.6, 0.8))","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.148817Z","iopub.execute_input":"2022-07-07T06:31:13.149162Z","iopub.status.idle":"2022-07-07T06:31:13.524985Z","shell.execute_reply.started":"2022-07-07T06:31:13.149133Z","shell.execute_reply":"2022-07-07T06:31:13.523437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()\nprint('_'*40)\ndf_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.527089Z","iopub.execute_input":"2022-07-07T06:31:13.527836Z","iopub.status.idle":"2022-07-07T06:31:13.554538Z","shell.execute_reply.started":"2022-07-07T06:31:13.527788Z","shell.execute_reply":"2022-07-07T06:31:13.553147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.556772Z","iopub.execute_input":"2022-07-07T06:31:13.557872Z","iopub.status.idle":"2022-07-07T06:31:13.600336Z","shell.execute_reply.started":"2022-07-07T06:31:13.557831Z","shell.execute_reply":"2022-07-07T06:31:13.598992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.601978Z","iopub.execute_input":"2022-07-07T06:31:13.602405Z","iopub.status.idle":"2022-07-07T06:31:13.629810Z","shell.execute_reply.started":"2022-07-07T06:31:13.602311Z","shell.execute_reply":"2022-07-07T06:31:13.628414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['Pclass', 'Survived']].groupby(['Pclass'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.631905Z","iopub.execute_input":"2022-07-07T06:31:13.633416Z","iopub.status.idle":"2022-07-07T06:31:13.653014Z","shell.execute_reply.started":"2022-07-07T06:31:13.633185Z","shell.execute_reply":"2022-07-07T06:31:13.651874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['Sex', 'Survived']].groupby(['Sex'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.654486Z","iopub.execute_input":"2022-07-07T06:31:13.654876Z","iopub.status.idle":"2022-07-07T06:31:13.672855Z","shell.execute_reply.started":"2022-07-07T06:31:13.654846Z","shell.execute_reply":"2022-07-07T06:31:13.671786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['SibSp', 'Survived']].groupby(['SibSp'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.674495Z","iopub.execute_input":"2022-07-07T06:31:13.674862Z","iopub.status.idle":"2022-07-07T06:31:13.694889Z","shell.execute_reply.started":"2022-07-07T06:31:13.674830Z","shell.execute_reply":"2022-07-07T06:31:13.693621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[[\"Parch\", \"Survived\"]].groupby(['Parch'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.697000Z","iopub.execute_input":"2022-07-07T06:31:13.697612Z","iopub.status.idle":"2022-07-07T06:31:13.716314Z","shell.execute_reply.started":"2022-07-07T06:31:13.697576Z","shell.execute_reply":"2022-07-07T06:31:13.715113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = sns.FacetGrid(df_train, col='Survived', size=3)\ns.map(plt.hist, 'Age', bins=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:13.720418Z","iopub.execute_input":"2022-07-07T06:31:13.721312Z","iopub.status.idle":"2022-07-07T06:31:14.217231Z","shell.execute_reply.started":"2022-07-07T06:31:13.721273Z","shell.execute_reply":"2022-07-07T06:31:14.216019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(df_train, col='Survived', row='Pclass', size=3, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:14.218917Z","iopub.execute_input":"2022-07-07T06:31:14.220127Z","iopub.status.idle":"2022-07-07T06:31:15.918551Z","shell.execute_reply.started":"2022-07-07T06:31:14.220076Z","shell.execute_reply":"2022-07-07T06:31:15.917383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(df_train, row='Embarked', size=4, aspect=1.6)\ngrid.map(sns.pointplot, 'Pclass', 'Survived', 'Sex', palette='deep')\ngrid.add_legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:15.920039Z","iopub.execute_input":"2022-07-07T06:31:15.920412Z","iopub.status.idle":"2022-07-07T06:31:17.546286Z","shell.execute_reply.started":"2022-07-07T06:31:15.920379Z","shell.execute_reply":"2022-07-07T06:31:17.545183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(df_train, row='Embarked', col='Survived', size=3, aspect=1.6)\ngrid.map(sns.barplot, 'Sex', 'Fare', alpha=.5, ci=None)\ngrid.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:17.547799Z","iopub.execute_input":"2022-07-07T06:31:17.548124Z","iopub.status.idle":"2022-07-07T06:31:18.589879Z","shell.execute_reply.started":"2022-07-07T06:31:17.548095Z","shell.execute_reply":"2022-07-07T06:31:18.588619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before\", df_train.shape, df_test.shape, combine[0].shape, combine[1].shape)\n\ndf_train = df_train.drop(['Ticket', 'Cabin'], axis=1)\ndf_test = df_test.drop(['Ticket', 'Cabin'], axis=1)\ncombine = [df_train, df_test]\n\nprint(\"After\", df_train.shape, df_test.shape, combine[0].shape, combine[1].shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.591417Z","iopub.execute_input":"2022-07-07T06:31:18.591854Z","iopub.status.idle":"2022-07-07T06:31:18.602095Z","shell.execute_reply.started":"2022-07-07T06:31:18.591819Z","shell.execute_reply":"2022-07-07T06:31:18.600709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Inital'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.603690Z","iopub.execute_input":"2022-07-07T06:31:18.604060Z","iopub.status.idle":"2022-07-07T06:31:18.619681Z","shell.execute_reply.started":"2022-07-07T06:31:18.604028Z","shell.execute_reply":"2022-07-07T06:31:18.618331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.crosstab(df_train['Inital'], df_train['Sex']).T.style.background_gradient(cmap='summer_r')","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.620953Z","iopub.execute_input":"2022-07-07T06:31:18.621542Z","iopub.status.idle":"2022-07-07T06:31:18.678903Z","shell.execute_reply.started":"2022-07-07T06:31:18.621497Z","shell.execute_reply":"2022-07-07T06:31:18.677282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Inital'] = dataset['Inital'].replace(['Lady', 'Countess','Capt', 'Col',\\\n                                                 'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Inital'] = dataset['Inital'].replace('Mlle', 'Miss')\n    dataset['Inital'] = dataset['Inital'].replace('Ms', 'Miss')\n    dataset['Inital'] = dataset['Inital'].replace('Mme', 'Mrs')\n    \ndf_train[['Inital', 'Survived']].groupby(['Inital'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.681624Z","iopub.execute_input":"2022-07-07T06:31:18.682082Z","iopub.status.idle":"2022-07-07T06:31:18.711489Z","shell.execute_reply.started":"2022-07-07T06:31:18.682042Z","shell.execute_reply":"2022-07-07T06:31:18.709871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Inital'] = dataset['Inital'].map(title_mapping)\n    dataset['Inital'] = dataset['Inital'].fillna(0)\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.713100Z","iopub.execute_input":"2022-07-07T06:31:18.713542Z","iopub.status.idle":"2022-07-07T06:31:18.740077Z","shell.execute_reply.started":"2022-07-07T06:31:18.713506Z","shell.execute_reply":"2022-07-07T06:31:18.739072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['Name', 'PassengerId'], axis=1)\ndf_test = df_test.drop(['Name'], axis=1)\ncombine = [df_train, df_test]\ndf_train.shape, df_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.741837Z","iopub.execute_input":"2022-07-07T06:31:18.742309Z","iopub.status.idle":"2022-07-07T06:31:18.756849Z","shell.execute_reply.started":"2022-07-07T06:31:18.742263Z","shell.execute_reply":"2022-07-07T06:31:18.755450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map({'female': 1, 'male': 0}).astype(int)\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.759169Z","iopub.execute_input":"2022-07-07T06:31:18.760135Z","iopub.status.idle":"2022-07-07T06:31:18.783218Z","shell.execute_reply.started":"2022-07-07T06:31:18.760086Z","shell.execute_reply":"2022-07-07T06:31:18.781925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid = sns.FacetGrid(df_train, row='Pclass', col='Sex', size=3, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:18.785069Z","iopub.execute_input":"2022-07-07T06:31:18.785522Z","iopub.status.idle":"2022-07-07T06:31:20.420670Z","shell.execute_reply.started":"2022-07-07T06:31:18.785487Z","shell.execute_reply":"2022-07-07T06:31:20.419423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def category_age(x):\n    if x < 10:\n        return 0\n    elif x < 20:\n        return 1\n    elif x < 30:\n        return 2\n    elif x < 40:\n        return 3\n    elif x < 50:\n        return 4\n    elif x < 60:\n        return 5\n    elif x < 70:\n        return 6\n    else:\n        return 7","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.422050Z","iopub.execute_input":"2022-07-07T06:31:20.422384Z","iopub.status.idle":"2022-07-07T06:31:20.430224Z","shell.execute_reply.started":"2022-07-07T06:31:20.422342Z","shell.execute_reply":"2022-07-07T06:31:20.428588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Age'] = dataset['Age'].apply(category_age)\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.432490Z","iopub.execute_input":"2022-07-07T06:31:20.432914Z","iopub.status.idle":"2022-07-07T06:31:20.456414Z","shell.execute_reply.started":"2022-07-07T06:31:20.432879Z","shell.execute_reply":"2022-07-07T06:31:20.455093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch']\n\ndf_train[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.458744Z","iopub.execute_input":"2022-07-07T06:31:20.459403Z","iopub.status.idle":"2022-07-07T06:31:20.478475Z","shell.execute_reply.started":"2022-07-07T06:31:20.459368Z","shell.execute_reply":"2022-07-07T06:31:20.477593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ndf_test = df_test.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ncombine = [df_train, df_test]\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.479794Z","iopub.execute_input":"2022-07-07T06:31:20.480855Z","iopub.status.idle":"2022-07-07T06:31:20.497736Z","shell.execute_reply.started":"2022-07-07T06:31:20.480819Z","shell.execute_reply":"2022-07-07T06:31:20.496291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Age*Class'] = dataset.Age * dataset.Pclass\n\ndf_train.loc[:, ['Age*Class', 'Age', 'Pclass']].head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.499410Z","iopub.execute_input":"2022-07-07T06:31:20.499910Z","iopub.status.idle":"2022-07-07T06:31:20.521250Z","shell.execute_reply.started":"2022-07-07T06:31:20.499874Z","shell.execute_reply":"2022-07-07T06:31:20.520070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(['Age*Class'], axis=1, inplace=True)\ndf_test.drop(['Age*Class'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.523008Z","iopub.execute_input":"2022-07-07T06:31:20.524045Z","iopub.status.idle":"2022-07-07T06:31:20.532311Z","shell.execute_reply.started":"2022-07-07T06:31:20.523995Z","shell.execute_reply":"2022-07-07T06:31:20.531442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# freq_port=\ncombine[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.539602Z","iopub.execute_input":"2022-07-07T06:31:20.540336Z","iopub.status.idle":"2022-07-07T06:31:20.556806Z","shell.execute_reply.started":"2022-07-07T06:31:20.540300Z","shell.execute_reply":"2022-07-07T06:31:20.555901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Embarked'].value_counts()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-07T06:31:20.558039Z","iopub.execute_input":"2022-07-07T06:31:20.559194Z","iopub.status.idle":"2022-07-07T06:31:20.569036Z","shell.execute_reply.started":"2022-07-07T06:31:20.559139Z","shell.execute_reply":"2022-07-07T06:31:20.567976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq_port = df_train.Embarked.dropna().mode()[0]\nfreq_port","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.570603Z","iopub.execute_input":"2022-07-07T06:31:20.570980Z","iopub.status.idle":"2022-07-07T06:31:20.585696Z","shell.execute_reply.started":"2022-07-07T06:31:20.570947Z","shell.execute_reply":"2022-07-07T06:31:20.584490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].fillna(freq_port)\n    \ndf_train[['Embarked', 'Survived']].groupby(['Embarked'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.587250Z","iopub.execute_input":"2022-07-07T06:31:20.587685Z","iopub.status.idle":"2022-07-07T06:31:20.607501Z","shell.execute_reply.started":"2022-07-07T06:31:20.587652Z","shell.execute_reply":"2022-07-07T06:31:20.606447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.608702Z","iopub.execute_input":"2022-07-07T06:31:20.609251Z","iopub.status.idle":"2022-07-07T06:31:20.626245Z","shell.execute_reply.started":"2022-07-07T06:31:20.609214Z","shell.execute_reply":"2022-07-07T06:31:20.625263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Fare'].fillna(df_test['Fare'].dropna().median(), inplace=True)\ndf_test['Fare'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.627948Z","iopub.execute_input":"2022-07-07T06:31:20.628604Z","iopub.status.idle":"2022-07-07T06:31:20.640202Z","shell.execute_reply.started":"2022-07-07T06:31:20.628562Z","shell.execute_reply":"2022-07-07T06:31:20.638806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['FareBand'] = pd.qcut(df_train['Fare'], 4)\ndf_train[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.643051Z","iopub.execute_input":"2022-07-07T06:31:20.643969Z","iopub.status.idle":"2022-07-07T06:31:20.665814Z","shell.execute_reply.started":"2022-07-07T06:31:20.643925Z","shell.execute_reply":"2022-07-07T06:31:20.664756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(['FareBand'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.667165Z","iopub.execute_input":"2022-07-07T06:31:20.667790Z","iopub.status.idle":"2022-07-07T06:31:20.673669Z","shell.execute_reply.started":"2022-07-07T06:31:20.667756Z","shell.execute_reply":"2022-07-07T06:31:20.672452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def category_fare(x):\n    if x < 7.91:\n        return 0\n    elif x < 14.454:\n        return 1\n    elif x < 31.0:\n        return 2\n    else:\n        return 3","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.675226Z","iopub.execute_input":"2022-07-07T06:31:20.676444Z","iopub.status.idle":"2022-07-07T06:31:20.685654Z","shell.execute_reply.started":"2022-07-07T06:31:20.676385Z","shell.execute_reply":"2022-07-07T06:31:20.684553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Fare'] = dataset['Fare'].apply(category_fare)\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.687241Z","iopub.execute_input":"2022-07-07T06:31:20.688545Z","iopub.status.idle":"2022-07-07T06:31:20.710861Z","shell.execute_reply.started":"2022-07-07T06:31:20.688496Z","shell.execute_reply":"2022-07-07T06:31:20.709290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.get_dummies(df_train, columns=['Inital', 'Embarked', 'Sex'], prefix=['Inital', 'Embarked', 'Sex'])\ndf_test = pd.get_dummies(df_test, columns=['Inital', 'Embarked', 'Sex'], prefix=['Inital', 'Embarked', 'Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.713643Z","iopub.execute_input":"2022-07-07T06:31:20.714135Z","iopub.status.idle":"2022-07-07T06:31:20.733089Z","shell.execute_reply.started":"2022-07-07T06:31:20.714097Z","shell.execute_reply":"2022-07-07T06:31:20.731547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.735540Z","iopub.execute_input":"2022-07-07T06:31:20.735933Z","iopub.status.idle":"2022-07-07T06:31:20.753472Z","shell.execute_reply.started":"2022-07-07T06:31:20.735903Z","shell.execute_reply":"2022-07-07T06:31:20.751818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df_train.drop(\"Survived\", axis=1)\nY_train = df_train[\"Survived\"]\nX_test  = df_test.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.755115Z","iopub.execute_input":"2022-07-07T06:31:20.756782Z","iopub.status.idle":"2022-07-07T06:31:20.773666Z","shell.execute_reply.started":"2022-07-07T06:31:20.756740Z","shell.execute_reply":"2022-07-07T06:31:20.771922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Support Vector Machines\n\nsvc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.796462Z","iopub.execute_input":"2022-07-07T06:31:20.797672Z","iopub.status.idle":"2022-07-07T06:31:20.886216Z","shell.execute_reply.started":"2022-07-07T06:31:20.797624Z","shell.execute_reply":"2022-07-07T06:31:20.885078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.888064Z","iopub.execute_input":"2022-07-07T06:31:20.888450Z","iopub.status.idle":"2022-07-07T06:31:20.960833Z","shell.execute_reply.started":"2022-07-07T06:31:20.888419Z","shell.execute_reply":"2022-07-07T06:31:20.959040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gaussian Naive Bayes\n\ngaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train) * 100, 2)\nacc_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.962216Z","iopub.execute_input":"2022-07-07T06:31:20.963378Z","iopub.status.idle":"2022-07-07T06:31:20.980656Z","shell.execute_reply.started":"2022-07-07T06:31:20.963311Z","shell.execute_reply":"2022-07-07T06:31:20.979471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perceptron\n\nperceptron = Perceptron()\nperceptron.fit(X_train, Y_train)\nY_pred = perceptron.predict(X_test)\nacc_perceptron = round(perceptron.score(X_train, Y_train) * 100, 2)\nacc_perceptron","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:20.982283Z","iopub.execute_input":"2022-07-07T06:31:20.982629Z","iopub.status.idle":"2022-07-07T06:31:21.005634Z","shell.execute_reply.started":"2022-07-07T06:31:20.982600Z","shell.execute_reply":"2022-07-07T06:31:21.004075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Linear SVC\n\nlinear_svc = LinearSVC()\nlinear_svc.fit(X_train, Y_train)\nY_pred = linear_svc.predict(X_test)\nacc_linear_svc = round(linear_svc.score(X_train, Y_train) * 100, 2)\nacc_linear_svc","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.011159Z","iopub.execute_input":"2022-07-07T06:31:21.016234Z","iopub.status.idle":"2022-07-07T06:31:21.122253Z","shell.execute_reply.started":"2022-07-07T06:31:21.016162Z","shell.execute_reply":"2022-07-07T06:31:21.120895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\n\nsgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train) * 100, 2)\nacc_sgd\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.129209Z","iopub.execute_input":"2022-07-07T06:31:21.134708Z","iopub.status.idle":"2022-07-07T06:31:21.172855Z","shell.execute_reply.started":"2022-07-07T06:31:21.134628Z","shell.execute_reply":"2022-07-07T06:31:21.171111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stochastic Gradient Descent\n\nsgd = SGDClassifier()\nsgd.fit(X_train, Y_train)\nY_pred = sgd.predict(X_test)\nacc_sgd = round(sgd.score(X_train, Y_train) * 100, 2)\nacc_sgd\n","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.180825Z","iopub.execute_input":"2022-07-07T06:31:21.184793Z","iopub.status.idle":"2022-07-07T06:31:21.225644Z","shell.execute_reply.started":"2022-07-07T06:31:21.184720Z","shell.execute_reply":"2022-07-07T06:31:21.224164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decision Tree\n\ndecision_tree = DecisionTreeClassifier()\ndecision_tree.fit(X_train, Y_train)\nY_pred = decision_tree.predict(X_test)\nacc_decision_tree = round(decision_tree.score(X_train, Y_train) * 100, 2)\nacc_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.233614Z","iopub.execute_input":"2022-07-07T06:31:21.237713Z","iopub.status.idle":"2022-07-07T06:31:21.274058Z","shell.execute_reply.started":"2022-07-07T06:31:21.237636Z","shell.execute_reply":"2022-07-07T06:31:21.272624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GBC = GradientBoostingClassifier()\nGBC.fit(X_train, Y_train)\nY_pred = GBC.predict(X_test)\nGBC.score(X_train, Y_train)\nacc_GBC = round(GBC.score(X_train, Y_train) * 100, 2)\nacc_GBC","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.281604Z","iopub.execute_input":"2022-07-07T06:31:21.285340Z","iopub.status.idle":"2022-07-07T06:31:21.446911Z","shell.execute_reply.started":"2022-07-07T06:31:21.285269Z","shell.execute_reply":"2022-07-07T06:31:21.445621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression\n\nlogreg = LogisticRegression(max_iter=1000)\nlogreg.fit(X_train, Y_train)\nY_pred = logreg.predict(X_test)\nacc_log = round(logreg.score(X_train, Y_train) * 100, 2)\nacc_log","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.841598Z","iopub.execute_input":"2022-07-07T06:31:21.841995Z","iopub.status.idle":"2022-07-07T06:31:21.925847Z","shell.execute_reply.started":"2022-07-07T06:31:21.841961Z","shell.execute_reply":"2022-07-07T06:31:21.924447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\nrandom_forest = RandomForestClassifier(n_estimators=100, max_depth=3, random_state=2)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.554162Z","iopub.execute_input":"2022-07-07T06:31:21.556556Z","iopub.status.idle":"2022-07-07T06:31:21.839767Z","shell.execute_reply.started":"2022-07-07T06:31:21.556508Z","shell.execute_reply":"2022-07-07T06:31:21.838512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LGBM = LGBMClassifier()\nLGBM.fit(X_train, Y_train)\nY_pred = LGBM.predict(X_test)\nLGBM.score(X_train, Y_train)\nacc_LGBM = round(LGBM.score(X_train, Y_train) * 100, 2)\nacc_LGBM","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.448373Z","iopub.execute_input":"2022-07-07T06:31:21.448823Z","iopub.status.idle":"2022-07-07T06:31:21.549537Z","shell.execute_reply.started":"2022-07-07T06:31:21.448778Z","shell.execute_reply":"2022-07-07T06:31:21.548465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Logistic Regression', \n              'Random Forest', 'Naive Bayes', 'Perceptron', \n              'Stochastic Gradient Decent', 'Linear SVC', \n              'Decision Tree', 'Gradient Boosting Classifier'],\n    'Score': [acc_svc, acc_knn, acc_log, \n              acc_random_forest, acc_gaussian, acc_perceptron, \n              acc_sgd, acc_linear_svc, acc_decision_tree, acc_GBC]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.927727Z","iopub.execute_input":"2022-07-07T06:31:21.928995Z","iopub.status.idle":"2022-07-07T06:31:21.956054Z","shell.execute_reply.started":"2022-07-07T06:31:21.928933Z","shell.execute_reply":"2022-07-07T06:31:21.954243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": df_test[\"PassengerId\"],\n        \"Survived\": Y_pred\n    })\nsubmission.to_csv('./my_7th_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-07T06:31:21.958142Z","iopub.execute_input":"2022-07-07T06:31:21.958997Z","iopub.status.idle":"2022-07-07T06:31:21.970411Z","shell.execute_reply.started":"2022-07-07T06:31:21.958941Z","shell.execute_reply":"2022-07-07T06:31:21.968849Z"},"trusted":true},"execution_count":null,"outputs":[]}]}