{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T14:50:24.077828Z","iopub.execute_input":"2022-07-23T14:50:24.078534Z","iopub.status.idle":"2022-07-23T14:50:24.109531Z","shell.execute_reply.started":"2022-07-23T14:50:24.078431Z","shell.execute_reply":"2022-07-23T14:50:24.108289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:24.111584Z","iopub.execute_input":"2022-07-23T14:50:24.111930Z","iopub.status.idle":"2022-07-23T14:50:25.236180Z","shell.execute_reply.started":"2022-07-23T14:50:24.111898Z","shell.execute_reply":"2022-07-23T14:50:25.235342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv('/kaggle/input/titanic/train.csv')\ntest=pd.read_csv('/kaggle/input/titanic/test.csv')\ncombine = [train, test]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.238626Z","iopub.execute_input":"2022-07-23T14:50:25.239042Z","iopub.status.idle":"2022-07-23T14:50:25.264533Z","shell.execute_reply.started":"2022-07-23T14:50:25.239002Z","shell.execute_reply":"2022-07-23T14:50:25.263423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.268001Z","iopub.execute_input":"2022-07-23T14:50:25.268487Z","iopub.status.idle":"2022-07-23T14:50:25.302110Z","shell.execute_reply.started":"2022-07-23T14:50:25.268442Z","shell.execute_reply":"2022-07-23T14:50:25.301039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.303492Z","iopub.execute_input":"2022-07-23T14:50:25.304018Z","iopub.status.idle":"2022-07-23T14:50:25.312763Z","shell.execute_reply.started":"2022-07-23T14:50:25.303987Z","shell.execute_reply":"2022-07-23T14:50:25.311737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Before\", train.shape, test.shape, combine[0].shape, combine[1].shape)\n\ntrain = train.drop(['Ticket', 'Cabin'], axis=1)\ntest = test.drop(['Ticket', 'Cabin'], axis=1)\ncombine = [train, test]\n\"After\", train.shape, test.shape, combine[0].shape, combine[1].shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.313979Z","iopub.execute_input":"2022-07-23T14:50:25.314757Z","iopub.status.idle":"2022-07-23T14:50:25.337293Z","shell.execute_reply.started":"2022-07-23T14:50:25.314725Z","shell.execute_reply":"2022-07-23T14:50:25.336393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train['Title'], train['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.338719Z","iopub.execute_input":"2022-07-23T14:50:25.339061Z","iopub.status.idle":"2022-07-23T14:50:25.379984Z","shell.execute_reply.started":"2022-07-23T14:50:25.339026Z","shell.execute_reply":"2022-07-23T14:50:25.379229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.381313Z","iopub.execute_input":"2022-07-23T14:50:25.381820Z","iopub.status.idle":"2022-07-23T14:50:25.409876Z","shell.execute_reply.started":"2022-07-23T14:50:25.381791Z","shell.execute_reply":"2022-07-23T14:50:25.408704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.415355Z","iopub.execute_input":"2022-07-23T14:50:25.416357Z","iopub.status.idle":"2022-07-23T14:50:25.444587Z","shell.execute_reply.started":"2022-07-23T14:50:25.416307Z","shell.execute_reply":"2022-07-23T14:50:25.443264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(['Name', 'PassengerId'], axis=1)\ntest = test.drop(['Name'], axis=1)\ncombine = [train, test]\ntrain.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.447972Z","iopub.execute_input":"2022-07-23T14:50:25.448737Z","iopub.status.idle":"2022-07-23T14:50:25.464449Z","shell.execute_reply.started":"2022-07-23T14:50:25.448690Z","shell.execute_reply":"2022-07-23T14:50:25.463264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.465867Z","iopub.execute_input":"2022-07-23T14:50:25.467123Z","iopub.status.idle":"2022-07-23T14:50:25.486296Z","shell.execute_reply.started":"2022-07-23T14:50:25.467079Z","shell.execute_reply":"2022-07-23T14:50:25.485437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# grid = sns.FacetGrid(train_df, col='Pclass', hue='Gender')\ngrid = sns.FacetGrid(train, row='Pclass', col='Sex', size=2.2, aspect=1.6)\ngrid.map(plt.hist, 'Age', alpha=.5, bins=20)\ngrid.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:25.487516Z","iopub.execute_input":"2022-07-23T14:50:25.487994Z","iopub.status.idle":"2022-07-23T14:50:27.111834Z","shell.execute_reply.started":"2022-07-23T14:50:25.487966Z","shell.execute_reply":"2022-07-23T14:50:27.110266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"guess_ages = np.zeros((2,3))\nguess_ages","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.113223Z","iopub.execute_input":"2022-07-23T14:50:27.114359Z","iopub.status.idle":"2022-07-23T14:50:27.123485Z","shell.execute_reply.started":"2022-07-23T14:50:27.114294Z","shell.execute_reply":"2022-07-23T14:50:27.122263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['Sex'] == i) & \\\n                                  (dataset['Pclass'] == j+1)]['Age'].dropna()\n\n            \n\n            age_guess = guess_df.median()\n\n            \n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex == i) & (dataset.Pclass == j+1),\\\n                    'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.125105Z","iopub.execute_input":"2022-07-23T14:50:27.125718Z","iopub.status.idle":"2022-07-23T14:50:27.189802Z","shell.execute_reply.started":"2022-07-23T14:50:27.125683Z","shell.execute_reply":"2022-07-23T14:50:27.189029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:    \n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.191168Z","iopub.execute_input":"2022-07-23T14:50:27.191719Z","iopub.status.idle":"2022-07-23T14:50:27.215169Z","shell.execute_reply.started":"2022-07-23T14:50:27.191687Z","shell.execute_reply":"2022-07-23T14:50:27.214015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n\ntrain[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.216580Z","iopub.execute_input":"2022-07-23T14:50:27.217677Z","iopub.status.idle":"2022-07-23T14:50:27.237945Z","shell.execute_reply.started":"2022-07-23T14:50:27.217634Z","shell.execute_reply":"2022-07-23T14:50:27.236737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['IsAlone'] = 0\n    dataset.loc[dataset['FamilySize'] == 1, 'IsAlone'] = 1\n\ntrain[['IsAlone', 'Survived']].groupby(['IsAlone'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.239352Z","iopub.execute_input":"2022-07-23T14:50:27.239706Z","iopub.status.idle":"2022-07-23T14:50:27.258326Z","shell.execute_reply.started":"2022-07-23T14:50:27.239677Z","shell.execute_reply":"2022-07-23T14:50:27.257091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ntest = test.drop(['Parch', 'SibSp', 'FamilySize'], axis=1)\ncombine = [train, test]\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.260133Z","iopub.execute_input":"2022-07-23T14:50:27.261300Z","iopub.status.idle":"2022-07-23T14:50:27.282979Z","shell.execute_reply.started":"2022-07-23T14:50:27.261258Z","shell.execute_reply":"2022-07-23T14:50:27.281766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Age*Class'] = dataset.Age * dataset.Pclass\n\ntrain.loc[:, ['Age*Class', 'Age', 'Pclass']].head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.284234Z","iopub.execute_input":"2022-07-23T14:50:27.284921Z","iopub.status.idle":"2022-07-23T14:50:27.301198Z","shell.execute_reply.started":"2022-07-23T14:50:27.284883Z","shell.execute_reply":"2022-07-23T14:50:27.300262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freq_port = train.Embarked.dropna().mode()[0]\nfreq_port","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.302713Z","iopub.execute_input":"2022-07-23T14:50:27.303419Z","iopub.status.idle":"2022-07-23T14:50:27.312388Z","shell.execute_reply.started":"2022-07-23T14:50:27.303376Z","shell.execute_reply":"2022-07-23T14:50:27.311391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].fillna(freq_port)\n    \ntrain[['Embarked', 'Survived']].groupby(['Embarked'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.314063Z","iopub.execute_input":"2022-07-23T14:50:27.314988Z","iopub.status.idle":"2022-07-23T14:50:27.332227Z","shell.execute_reply.started":"2022-07-23T14:50:27.314944Z","shell.execute_reply":"2022-07-23T14:50:27.331055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.333722Z","iopub.execute_input":"2022-07-23T14:50:27.334418Z","iopub.status.idle":"2022-07-23T14:50:27.352763Z","shell.execute_reply.started":"2022-07-23T14:50:27.334375Z","shell.execute_reply":"2022-07-23T14:50:27.351486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Fare'].fillna(test['Fare'].dropna().median(), inplace=True)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.354631Z","iopub.execute_input":"2022-07-23T14:50:27.355087Z","iopub.status.idle":"2022-07-23T14:50:27.372264Z","shell.execute_reply.started":"2022-07-23T14:50:27.355044Z","shell.execute_reply":"2022-07-23T14:50:27.371081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['FareBand'] = pd.qcut(train['Fare'], 4)\ntrain[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.379255Z","iopub.execute_input":"2022-07-23T14:50:27.380039Z","iopub.status.idle":"2022-07-23T14:50:27.412610Z","shell.execute_reply.started":"2022-07-23T14:50:27.379987Z","shell.execute_reply":"2022-07-23T14:50:27.411387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3\n    dataset['Fare'] = dataset['Fare'].astype(int)\n\ntrain = train.drop(['FareBand'], axis=1)\ncombine = [train, test]\n    \ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.414349Z","iopub.execute_input":"2022-07-23T14:50:27.415075Z","iopub.status.idle":"2022-07-23T14:50:27.452145Z","shell.execute_reply.started":"2022-07-23T14:50:27.415020Z","shell.execute_reply":"2022-07-23T14:50:27.451008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.453899Z","iopub.execute_input":"2022-07-23T14:50:27.454626Z","iopub.status.idle":"2022-07-23T14:50:27.474824Z","shell.execute_reply.started":"2022-07-23T14:50:27.454584Z","shell.execute_reply":"2022-07-23T14:50:27.473885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train.drop(\"Survived\", axis=1)\nY_train = train[\"Survived\"]\nX_test  = test.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.476083Z","iopub.execute_input":"2022-07-23T14:50:27.476442Z","iopub.status.idle":"2022-07-23T14:50:27.486941Z","shell.execute_reply.started":"2022-07-23T14:50:27.476411Z","shell.execute_reply":"2022-07-23T14:50:27.486059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC, LinearSVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.488267Z","iopub.execute_input":"2022-07-23T14:50:27.488589Z","iopub.status.idle":"2022-07-23T14:50:27.833439Z","shell.execute_reply.started":"2022-07-23T14:50:27.488560Z","shell.execute_reply":"2022-07-23T14:50:27.832459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svc = SVC()\nsvc.fit(X_train, Y_train)\nY_pred = svc.predict(X_test)\nacc_svc = round(svc.score(X_train, Y_train) * 100, 2)\nacc_svc","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.834757Z","iopub.execute_input":"2022-07-23T14:50:27.835588Z","iopub.status.idle":"2022-07-23T14:50:27.970218Z","shell.execute_reply.started":"2022-07-23T14:50:27.835539Z","shell.execute_reply":"2022-07-23T14:50:27.968906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaussian = GaussianNB()\ngaussian.fit(X_train, Y_train)\nY_pred = gaussian.predict(X_test)\nacc_gaussian = round(gaussian.score(X_train, Y_train) * 100, 2)\nacc_gaussian","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.971994Z","iopub.execute_input":"2022-07-23T14:50:27.972769Z","iopub.status.idle":"2022-07-23T14:50:27.991802Z","shell.execute_reply.started":"2022-07-23T14:50:27.972723Z","shell.execute_reply":"2022-07-23T14:50:27.991007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors = 3)\nknn.fit(X_train, Y_train)\nY_pred = knn.predict(X_test)\nacc_knn = round(knn.score(X_train, Y_train) * 100, 2)\nacc_knn","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:27.993530Z","iopub.execute_input":"2022-07-23T14:50:27.994325Z","iopub.status.idle":"2022-07-23T14:50:28.063396Z","shell.execute_reply.started":"2022-07-23T14:50:27.994277Z","shell.execute_reply":"2022-07-23T14:50:28.062243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN', 'Naive Bayes'],\n    'Score': [acc_svc, acc_knn, acc_gaussian]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:50:28.064963Z","iopub.execute_input":"2022-07-23T14:50:28.065631Z","iopub.status.idle":"2022-07-23T14:50:28.080038Z","shell.execute_reply.started":"2022-07-23T14:50:28.065590Z","shell.execute_reply":"2022-07-23T14:50:28.078597Z"},"trusted":true},"execution_count":null,"outputs":[]}]}