{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T18:50:06.565979Z","iopub.execute_input":"2022-08-11T18:50:06.566380Z","iopub.status.idle":"2022-08-11T18:50:06.580023Z","shell.execute_reply.started":"2022-08-11T18:50:06.566348Z","shell.execute_reply":"2022-08-11T18:50:06.579088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reading the data sets \ntrain = pd.read_csv('../input/titanic/train.csv')\ntest = pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:06.636235Z","iopub.execute_input":"2022-08-11T18:50:06.636963Z","iopub.status.idle":"2022-08-11T18:50:06.653911Z","shell.execute_reply.started":"2022-08-11T18:50:06.636916Z","shell.execute_reply":"2022-08-11T18:50:06.652941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#printing first 5 rows of train data set\ntrain.head() ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:06.739132Z","iopub.execute_input":"2022-08-11T18:50:06.739908Z","iopub.status.idle":"2022-08-11T18:50:06.758780Z","shell.execute_reply.started":"2022-08-11T18:50:06.739845Z","shell.execute_reply":"2022-08-11T18:50:06.757321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#printing first 5 rows of test data set\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:06.837455Z","iopub.execute_input":"2022-08-11T18:50:06.837856Z","iopub.status.idle":"2022-08-11T18:50:06.854756Z","shell.execute_reply.started":"2022-08-11T18:50:06.837822Z","shell.execute_reply":"2022-08-11T18:50:06.853462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#printing number of columns and rows in the training data set of training \ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:06.907319Z","iopub.execute_input":"2022-08-11T18:50:06.907741Z","iopub.status.idle":"2022-08-11T18:50:06.914912Z","shell.execute_reply.started":"2022-08-11T18:50:06.907708Z","shell.execute_reply":"2022-08-11T18:50:06.913592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#printing number of columns and rows in the testing data set of training \ntest.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:06.951518Z","iopub.execute_input":"2022-08-11T18:50:06.952734Z","iopub.status.idle":"2022-08-11T18:50:06.959249Z","shell.execute_reply.started":"2022-08-11T18:50:06.952690Z","shell.execute_reply":"2022-08-11T18:50:06.958100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#determine the missing data in train data set \ntrain.isnull().sum()\n#Age column is missing 177 values \n#Cabin column is missing 687 values\n#Embraked column is missing 2 values ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.046349Z","iopub.execute_input":"2022-08-11T18:50:07.047107Z","iopub.status.idle":"2022-08-11T18:50:07.057299Z","shell.execute_reply.started":"2022-08-11T18:50:07.047064Z","shell.execute_reply":"2022-08-11T18:50:07.056323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#determine the missing data in test data set \ntest.isnull().sum()\n#Age column is missing 86 values \n#Cabin column is missing 327 values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.118547Z","iopub.execute_input":"2022-08-11T18:50:07.119207Z","iopub.status.idle":"2022-08-11T18:50:07.130569Z","shell.execute_reply.started":"2022-08-11T18:50:07.119159Z","shell.execute_reply":"2022-08-11T18:50:07.129289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_test_data = [train, test] # combining train and test dataset\n\nfor dataset in train_test_data:\n    dataset['Title'] = dataset['Name'].str.extract(' ([A-Za-z]+)\\.', expand=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.195134Z","iopub.execute_input":"2022-08-11T18:50:07.196275Z","iopub.status.idle":"2022-08-11T18:50:07.205007Z","shell.execute_reply.started":"2022-08-11T18:50:07.196231Z","shell.execute_reply":"2022-08-11T18:50:07.203800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.293687Z","iopub.execute_input":"2022-08-11T18:50:07.294951Z","iopub.status.idle":"2022-08-11T18:50:07.305957Z","shell.execute_reply.started":"2022-08-11T18:50:07.294884Z","shell.execute_reply":"2022-08-11T18:50:07.304580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Title'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.337677Z","iopub.execute_input":"2022-08-11T18:50:07.338283Z","iopub.status.idle":"2022-08-11T18:50:07.348313Z","shell.execute_reply.started":"2022-08-11T18:50:07.338250Z","shell.execute_reply":"2022-08-11T18:50:07.346995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Title map**\n\nMr : 0\n\nMiss : 1\n\nMrs: 2\n\nOthers: 3\n\nwe are doing feature engineering first we map the features to nominal values to the classifier can work with it ","metadata":{}},{"cell_type":"code","source":"title_mapping = {\"Mr\": 0, \"Miss\": 1, \"Mrs\": 2, \n                 \"Master\": 3, \"Dr\": 3, \"Rev\": 3, \"Col\": 3, \"Major\": 3, \"Mlle\": 3,\"Countess\": 3,\n                 \"Ms\": 3, \"Lady\": 3, \"Jonkheer\": 3, \"Don\": 3, \"Dona\" : 3, \"Mme\": 3,\"Capt\": 3,\"Sir\": 3 }\nfor dataset in train_test_data:\n    dataset['Title'] = dataset['Title'].map(title_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.378361Z","iopub.execute_input":"2022-08-11T18:50:07.379877Z","iopub.status.idle":"2022-08-11T18:50:07.392145Z","shell.execute_reply.started":"2022-08-11T18:50:07.379804Z","shell.execute_reply":"2022-08-11T18:50:07.391040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.429701Z","iopub.execute_input":"2022-08-11T18:50:07.430339Z","iopub.status.idle":"2022-08-11T18:50:07.451511Z","shell.execute_reply.started":"2022-08-11T18:50:07.430301Z","shell.execute_reply":"2022-08-11T18:50:07.449970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.463221Z","iopub.execute_input":"2022-08-11T18:50:07.463824Z","iopub.status.idle":"2022-08-11T18:50:07.482489Z","shell.execute_reply.started":"2022-08-11T18:50:07.463788Z","shell.execute_reply":"2022-08-11T18:50:07.481262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete unnecessary feature from dataset\ntrain.drop('Name', axis=1, inplace=True)\ntest.drop('Name', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.502994Z","iopub.execute_input":"2022-08-11T18:50:07.503676Z","iopub.status.idle":"2022-08-11T18:50:07.513102Z","shell.execute_reply.started":"2022-08-11T18:50:07.503635Z","shell.execute_reply":"2022-08-11T18:50:07.512149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.543777Z","iopub.execute_input":"2022-08-11T18:50:07.544737Z","iopub.status.idle":"2022-08-11T18:50:07.562740Z","shell.execute_reply.started":"2022-08-11T18:50:07.544698Z","shell.execute_reply":"2022-08-11T18:50:07.561518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.584127Z","iopub.execute_input":"2022-08-11T18:50:07.584846Z","iopub.status.idle":"2022-08-11T18:50:07.602897Z","shell.execute_reply.started":"2022-08-11T18:50:07.584796Z","shell.execute_reply":"2022-08-11T18:50:07.601726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sex Mapping**\n\n","metadata":{}},{"cell_type":"code","source":"sex_mapping = {\"male\": 0, \"female\": 1}\nfor dataset in train_test_data:\n    dataset['Sex'] = dataset['Sex'].map(sex_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.621886Z","iopub.execute_input":"2022-08-11T18:50:07.623404Z","iopub.status.idle":"2022-08-11T18:50:07.632661Z","shell.execute_reply.started":"2022-08-11T18:50:07.623348Z","shell.execute_reply":"2022-08-11T18:50:07.631658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now to deal with missing data such in age column**","metadata":{}},{"cell_type":"code","source":"train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.658864Z","iopub.execute_input":"2022-08-11T18:50:07.659865Z","iopub.status.idle":"2022-08-11T18:50:07.680490Z","shell.execute_reply.started":"2022-08-11T18:50:07.659827Z","shell.execute_reply":"2022-08-11T18:50:07.679130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing age with median age for each title (Mr, Mrs, Miss, Others)\ntrain[\"Age\"].fillna(train.groupby(\"Title\")[\"Age\"].transform(\"median\"), inplace=True)\ntest[\"Age\"].fillna(test.groupby(\"Title\")[\"Age\"].transform(\"median\"), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.702089Z","iopub.execute_input":"2022-08-11T18:50:07.703127Z","iopub.status.idle":"2022-08-11T18:50:07.715998Z","shell.execute_reply.started":"2022-08-11T18:50:07.703073Z","shell.execute_reply":"2022-08-11T18:50:07.714695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Converting Numerical Age to Categorical Variable**\n\nfeature vector map:\n\nchild: 0\n\nyoung: 1\n\nadult: 2\n\nmid-age: 3\n\nsenior: 4","metadata":{}},{"cell_type":"code","source":"train.info()\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.740617Z","iopub.execute_input":"2022-08-11T18:50:07.741416Z","iopub.status.idle":"2022-08-11T18:50:07.769988Z","shell.execute_reply.started":"2022-08-11T18:50:07.741365Z","shell.execute_reply":"2022-08-11T18:50:07.768515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:    \n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.773563Z","iopub.execute_input":"2022-08-11T18:50:07.773952Z","iopub.status.idle":"2022-08-11T18:50:07.793042Z","shell.execute_reply.started":"2022-08-11T18:50:07.773921Z","shell.execute_reply":"2022-08-11T18:50:07.791416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.800282Z","iopub.execute_input":"2022-08-11T18:50:07.801193Z","iopub.status.idle":"2022-08-11T18:50:07.823408Z","shell.execute_reply.started":"2022-08-11T18:50:07.801145Z","shell.execute_reply":"2022-08-11T18:50:07.822046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filling the Nan values of Embarked column with most_frequent value ('s')\nfor dataset in train_test_data:\n    dataset['Embarked'] = dataset['Embarked'].fillna('S')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.825596Z","iopub.execute_input":"2022-08-11T18:50:07.825969Z","iopub.status.idle":"2022-08-11T18:50:07.833473Z","shell.execute_reply.started":"2022-08-11T18:50:07.825938Z","shell.execute_reply":"2022-08-11T18:50:07.832363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.835305Z","iopub.execute_input":"2022-08-11T18:50:07.836170Z","iopub.status.idle":"2022-08-11T18:50:07.862835Z","shell.execute_reply.started":"2022-08-11T18:50:07.836114Z","shell.execute_reply":"2022-08-11T18:50:07.861955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using map funcion to change the Embarked column S = 1, C = 2, Q = 0\nembarked_mapping = {\"S\": 0, \"C\": 1, \"Q\": 2}\nfor dataset in train_test_data:\n    dataset['Embarked'] = dataset['Embarked'].map(embarked_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.864334Z","iopub.execute_input":"2022-08-11T18:50:07.865112Z","iopub.status.idle":"2022-08-11T18:50:07.872694Z","shell.execute_reply.started":"2022-08-11T18:50:07.865078Z","shell.execute_reply":"2022-08-11T18:50:07.871747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing Fare with median fare for each Pclass\ntrain[\"Fare\"].fillna(train.groupby(\"Pclass\")[\"Fare\"].transform(\"median\"), inplace=True)\ntest[\"Fare\"].fillna(test.groupby(\"Pclass\")[\"Fare\"].transform(\"median\"), inplace=True)\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:50:07.874243Z","iopub.execute_input":"2022-08-11T18:50:07.874650Z","iopub.status.idle":"2022-08-11T18:50:07.904893Z","shell.execute_reply.started":"2022-08-11T18:50:07.874616Z","shell.execute_reply":"2022-08-11T18:50:07.904000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:53:06.549159Z","iopub.execute_input":"2022-08-11T18:53:06.549599Z","iopub.status.idle":"2022-08-11T18:53:06.564869Z","shell.execute_reply.started":"2022-08-11T18:53:06.549565Z","shell.execute_reply":"2022-08-11T18:53:06.563569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:53:22.230022Z","iopub.execute_input":"2022-08-11T18:53:22.230443Z","iopub.status.idle":"2022-08-11T18:53:22.248340Z","shell.execute_reply.started":"2022-08-11T18:53:22.230391Z","shell.execute_reply":"2022-08-11T18:53:22.246729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Cabin.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:53:39.943586Z","iopub.execute_input":"2022-08-11T18:53:39.944206Z","iopub.status.idle":"2022-08-11T18:53:39.954287Z","shell.execute_reply.started":"2022-08-11T18:53:39.944172Z","shell.execute_reply":"2022-08-11T18:53:39.952946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Cabin'] = dataset['Cabin'].str[:1]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:53:51.911156Z","iopub.execute_input":"2022-08-11T18:53:51.911598Z","iopub.status.idle":"2022-08-11T18:53:51.919402Z","shell.execute_reply.started":"2022-08-11T18:53:51.911565Z","shell.execute_reply":"2022-08-11T18:53:51.918100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cabin_mapping = {\"A\": 0, \"B\": 0.4, \"C\": 0.8, \"D\": 1.2, \"E\": 1.6, \"F\": 2, \"G\": 2.4, \"T\": 2.8}\nfor dataset in train_test_data:\n    dataset['Cabin'] = dataset['Cabin'].map(cabin_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:54:06.579563Z","iopub.execute_input":"2022-08-11T18:54:06.579984Z","iopub.status.idle":"2022-08-11T18:54:06.589831Z","shell.execute_reply.started":"2022-08-11T18:54:06.579949Z","shell.execute_reply":"2022-08-11T18:54:06.588515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fill missing Fare with median fare for each Pclass\ntrain[\"Cabin\"].fillna(train.groupby(\"Pclass\")[\"Cabin\"].transform(\"median\"), inplace=True)\ntest[\"Cabin\"].fillna(test.groupby(\"Pclass\")[\"Cabin\"].transform(\"median\"), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:55:38.189738Z","iopub.execute_input":"2022-08-11T18:55:38.190121Z","iopub.status.idle":"2022-08-11T18:55:38.200503Z","shell.execute_reply.started":"2022-08-11T18:55:38.190090Z","shell.execute_reply":"2022-08-11T18:55:38.199316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"FamilySize\"] = train[\"SibSp\"] + train[\"Parch\"] + 1\ntest[\"FamilySize\"] = test[\"SibSp\"] + test[\"Parch\"] + 1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:56:14.270896Z","iopub.execute_input":"2022-08-11T18:56:14.271291Z","iopub.status.idle":"2022-08-11T18:56:14.280178Z","shell.execute_reply.started":"2022-08-11T18:56:14.271259Z","shell.execute_reply":"2022-08-11T18:56:14.278969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_mapping = {1: 0, 2: 0.4, 3: 0.8, 4: 1.2, 5: 1.6, 6: 2, 7: 2.4, 8: 2.8, 9: 3.2, 10: 3.6, 11: 4}\nfor dataset in train_test_data:\n    dataset['FamilySize'] = dataset['FamilySize'].map(family_mapping)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:56:16.441894Z","iopub.execute_input":"2022-08-11T18:56:16.442315Z","iopub.status.idle":"2022-08-11T18:56:16.452148Z","shell.execute_reply.started":"2022-08-11T18:56:16.442281Z","shell.execute_reply":"2022-08-11T18:56:16.451089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:56:31.121370Z","iopub.execute_input":"2022-08-11T18:56:31.121974Z","iopub.status.idle":"2022-08-11T18:56:31.143454Z","shell.execute_reply.started":"2022-08-11T18:56:31.121939Z","shell.execute_reply":"2022-08-11T18:56:31.142257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop columns \nfeatures_drop = ['Ticket', 'SibSp', 'Parch']\ntrain = train.drop(features_drop, axis=1)\ntest = test.drop(features_drop, axis=1)\ntrain = train.drop(['PassengerId'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:56:58.109750Z","iopub.execute_input":"2022-08-11T18:56:58.110144Z","iopub.status.idle":"2022-08-11T18:56:58.120157Z","shell.execute_reply.started":"2022-08-11T18:56:58.110112Z","shell.execute_reply":"2022-08-11T18:56:58.119003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train.drop('Survived', axis=1)\ntarget = train['Survived']\n\ntrain_data.shape, target.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:57:08.541105Z","iopub.execute_input":"2022-08-11T18:57:08.541518Z","iopub.status.idle":"2022-08-11T18:57:08.551540Z","shell.execute_reply.started":"2022-08-11T18:57:08.541485Z","shell.execute_reply":"2022-08-11T18:57:08.550268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T18:57:19.682416Z","iopub.execute_input":"2022-08-11T18:57:19.682880Z","iopub.status.idle":"2022-08-11T18:57:19.701750Z","shell.execute_reply.started":"2022-08-11T18:57:19.682845Z","shell.execute_reply":"2022-08-11T18:57:19.700419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modelling** ","metadata":{}},{"cell_type":"code","source":"# Importing Classifier Modules\nfrom sklearn.neighbors import KNeighborsClassifier #KNN K-nearest neighbor \nfrom sklearn.tree import DecisionTreeClassifier # Decision Tree\nfrom sklearn.ensemble import RandomForestClassifier # Random Forest  \nfrom sklearn.svm import SVC #support vector machine","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:00:58.437649Z","iopub.execute_input":"2022-08-11T19:00:58.438091Z","iopub.status.idle":"2022-08-11T19:00:59.156102Z","shell.execute_reply.started":"2022-08-11T19:00:58.438054Z","shell.execute_reply":"2022-08-11T19:00:59.154801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Cross Validation (K-fold)**\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.model_selection import cross_val_score\nk_fold = KFold(n_splits=10, shuffle=True, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:01:40.739467Z","iopub.execute_input":"2022-08-11T19:01:40.739869Z","iopub.status.idle":"2022-08-11T19:01:40.746484Z","shell.execute_reply.started":"2022-08-11T19:01:40.739837Z","shell.execute_reply":"2022-08-11T19:01:40.744967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1.KNN Classifier**","metadata":{}},{"cell_type":"code","source":"clf = KNeighborsClassifier(n_neighbors = 13)\nscoring = 'accuracy'\nscore = cross_val_score(clf, train_data, target, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)\n\n# kNN Score\nround(np.mean(score)*100, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:02:52.168361Z","iopub.execute_input":"2022-08-11T19:02:52.168817Z","iopub.status.idle":"2022-08-11T19:02:52.285845Z","shell.execute_reply.started":"2022-08-11T19:02:52.168781Z","shell.execute_reply":"2022-08-11T19:02:52.285003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2.Decision Tree**","metadata":{}},{"cell_type":"code","source":"clf = DecisionTreeClassifier()\nscoring = 'accuracy'\nscore = cross_val_score(clf, train_data, target, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score) \n\n# decision tree Score\nround(np.mean(score)*100, 2) \n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:04:03.935668Z","iopub.execute_input":"2022-08-11T19:04:03.936086Z","iopub.status.idle":"2022-08-11T19:04:04.007353Z","shell.execute_reply.started":"2022-08-11T19:04:03.936050Z","shell.execute_reply":"2022-08-11T19:04:04.006520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**3.Random Forest**","metadata":{}},{"cell_type":"code","source":"clf = RandomForestClassifier(n_estimators=13)\nscoring = 'accuracy'\nscore = cross_val_score(clf, train_data, target, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)\n\n# Random Forest Score\nround(np.mean(score)*100, 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:05:05.487555Z","iopub.execute_input":"2022-08-11T19:05:05.487968Z","iopub.status.idle":"2022-08-11T19:05:05.837057Z","shell.execute_reply.started":"2022-08-11T19:05:05.487935Z","shell.execute_reply":"2022-08-11T19:05:05.836104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**4.SVR**","metadata":{}},{"cell_type":"code","source":"clf = SVC()\nscoring = 'accuracy'\nscore = cross_val_score(clf, train_data, target, cv=k_fold, n_jobs=1, scoring=scoring)\nprint(score)\n\nround(np.mean(score)*100,2)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:06:12.321550Z","iopub.execute_input":"2022-08-11T19:06:12.321949Z","iopub.status.idle":"2022-08-11T19:06:12.609657Z","shell.execute_reply.started":"2022-08-11T19:06:12.321907Z","shell.execute_reply":"2022-08-11T19:06:12.608619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Testing **","metadata":{}},{"cell_type":"code","source":"clf = SVC()\nclf.fit(train_data, target)\n\ntest_data = test.drop(\"PassengerId\", axis=1).copy()\nprediction = clf.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:08:39.051656Z","iopub.execute_input":"2022-08-11T19:08:39.052086Z","iopub.status.idle":"2022-08-11T19:08:39.099904Z","shell.execute_reply.started":"2022-08-11T19:08:39.052052Z","shell.execute_reply":"2022-08-11T19:08:39.098871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": test[\"PassengerId\"],\n        \"Survived\": prediction\n    })\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:08:41.380338Z","iopub.execute_input":"2022-08-11T19:08:41.381582Z","iopub.status.idle":"2022-08-11T19:08:41.393104Z","shell.execute_reply.started":"2022-08-11T19:08:41.381499Z","shell.execute_reply":"2022-08-11T19:08:41.391840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:08:50.869012Z","iopub.execute_input":"2022-08-11T19:08:50.869447Z","iopub.status.idle":"2022-08-11T19:08:50.881593Z","shell.execute_reply.started":"2022-08-11T19:08:50.869395Z","shell.execute_reply":"2022-08-11T19:08:50.880686Z"},"trusted":true},"execution_count":null,"outputs":[]}]}