{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Required libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns      # Make Graph Visuals\nfrom sklearn.ensemble import RandomForestClassifier\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T16:51:46.600357Z","iopub.execute_input":"2022-07-23T16:51:46.600816Z","iopub.status.idle":"2022-07-23T16:51:46.613449Z","shell.execute_reply.started":"2022-07-23T16:51:46.600786Z","shell.execute_reply":"2022-07-23T16:51:46.612356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Required Files","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ngender_submission = pd.read_csv(\"/kaggle/input/titanic/gender_submission.csv\")\nactual_output = pd.read_csv(\"/kaggle/input/titanic-real-1-0/submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:47.186633Z","iopub.execute_input":"2022-07-23T16:51:47.187459Z","iopub.status.idle":"2022-07-23T16:51:47.211299Z","shell.execute_reply.started":"2022-07-23T16:51:47.187410Z","shell.execute_reply":"2022-07-23T16:51:47.209999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make Categorial Labels","metadata":{}},{"cell_type":"markdown","source":"1. Machine Learning Models Only Accept Numericals Values\n2. So, Change Alphabet to Numerical Values","metadata":{}},{"cell_type":"code","source":"train['Embarked'] =  train.Embarked.map({'S':1, 'C':3, 'Q':2})\ntest['Embarked']  = test.Embarked.map({'S':1, 'C':3, 'Q':2})\n\ntrain['Sex'] = train.Sex.map({'male': 0, 'female': 1})\ntest['Sex']  = test.Sex.map( {'male': 0, 'female': 1})","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:47.906559Z","iopub.execute_input":"2022-07-23T16:51:47.907780Z","iopub.status.idle":"2022-07-23T16:51:47.919406Z","shell.execute_reply.started":"2022-07-23T16:51:47.907714Z","shell.execute_reply":"2022-07-23T16:51:47.918739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check Null Values","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:48.377109Z","iopub.execute_input":"2022-07-23T16:51:48.377508Z","iopub.status.idle":"2022-07-23T16:51:48.390081Z","shell.execute_reply.started":"2022-07-23T16:51:48.377478Z","shell.execute_reply":"2022-07-23T16:51:48.388882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fill Null Values","metadata":{}},{"cell_type":"markdown","source":"1. Null Values Decrease the accuracy\n2. So, remove the null values","metadata":{}},{"cell_type":"code","source":"train['Age'].fillna(train.Age.median(), inplace=True)\ntest['Age'].fillna(train.Age.median(), inplace=True)\n\ntrain['Fare'].fillna(train.Fare.mean(), inplace=True)\ntest['Fare'].fillna(train.Fare.mean(), inplace=True)\n\ntrain['Embarked'].fillna(train.Embarked.median(), inplace=True)\ntest['Embarked'].fillna(test.Embarked.median(), inplace=True)\n\ntrain.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:48.866989Z","iopub.execute_input":"2022-07-23T16:51:48.868056Z","iopub.status.idle":"2022-07-23T16:51:48.886708Z","shell.execute_reply.started":"2022-07-23T16:51:48.868024Z","shell.execute_reply":"2022-07-23T16:51:48.885488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## See All Relations","metadata":{}},{"cell_type":"code","source":"sns.heatmap(train.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:49.471426Z","iopub.execute_input":"2022-07-23T16:51:49.471773Z","iopub.status.idle":"2022-07-23T16:51:49.753277Z","shell.execute_reply.started":"2022-07-23T16:51:49.471749Z","shell.execute_reply":"2022-07-23T16:51:49.752217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Make New Req Columns","metadata":{}},{"cell_type":"markdown","source":"### Family Size\n1. SibSp and Parch Column both Belongs to Family\n2. Create FamilySize Column from SibSp And Parch","metadata":{}},{"cell_type":"code","source":"train['FamilySize'] = train['SibSp'] + train['Parch']\ntest['FamilySize'] = test['SibSp'] + test['Parch']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:50.174730Z","iopub.execute_input":"2022-07-23T16:51:50.175105Z","iopub.status.idle":"2022-07-23T16:51:50.181931Z","shell.execute_reply.started":"2022-07-23T16:51:50.175077Z","shell.execute_reply":"2022-07-23T16:51:50.181241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Name Category","metadata":{}},{"cell_type":"markdown","source":"1. All Name Contains Some Title\n2. Extract Title From Name\n3. Specify Title For Male and Female","metadata":{}},{"cell_type":"code","source":"train['name_title'] = train.Name.apply(lambda x: x.split(',')[1].split('.')[0].strip())\ntest['name_title'] = test.Name.apply(lambda x: x.split(',')[1].split('.')[0].strip())\ntrain.name_title.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:50.649791Z","iopub.execute_input":"2022-07-23T16:51:50.650752Z","iopub.status.idle":"2022-07-23T16:51:50.662221Z","shell.execute_reply.started":"2022-07-23T16:51:50.650692Z","shell.execute_reply":"2022-07-23T16:51:50.661271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['name_cat'] = train.name_title.map({'Mrs':1,'Miss': 1,'Ms':1,'Lady':1,'Mlle': 1,'the Countess': 1,\n               'Mr': 0, 'Don':0,'Master':0,'Rev':0,'Dr':0,'Mme':0,'Major':0,\n               'Sir':0,'Col':0,'Capt':0,'Jonkheer':1})\ntest['name_cat'] = test.name_title.map({'Mrs':1,'Miss':1,'Master':0,'Ms':1,'Col':0,\n                                       'Rev':0, 'Dr': 0, 'Dona': 1, 'Mr': 0})","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:50.901780Z","iopub.execute_input":"2022-07-23T16:51:50.904702Z","iopub.status.idle":"2022-07-23T16:51:50.915188Z","shell.execute_reply.started":"2022-07-23T16:51:50.904661Z","shell.execute_reply":"2022-07-23T16:51:50.913995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cabin_multiple'] = train.Cabin.apply(lambda x: 0 if pd.isna(x) else len(x.split(' ')))\ntrain['cabin_adv'] = train.Cabin.apply(lambda x: str(x)[0])\n\ntest['cabin_multiple'] = test.Cabin.apply(lambda x: 0 if pd.isna(x) else len(x.split(' ')))\ntest['cabin_adv'] = test.Cabin.apply(lambda x: str(x)[0])\n\ntrain['cabin_adv'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:51.084609Z","iopub.execute_input":"2022-07-23T16:51:51.085178Z","iopub.status.idle":"2022-07-23T16:51:51.105951Z","shell.execute_reply.started":"2022-07-23T16:51:51.085146Z","shell.execute_reply":"2022-07-23T16:51:51.104905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['cabin_adv'] = train.cabin_adv.map({'n':0, 'C':1, 'E':2, 'G':3, 'D':4, 'A':5, 'B':6, \n                                          'F':7, 'T':8})\ntest['cabin_adv'] = test.cabin_adv.map({'n':0, 'B':6, 'E':2, 'A':5, 'C':1, 'D':4, 'F':7, 'G':3})","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:51.186980Z","iopub.execute_input":"2022-07-23T16:51:51.187352Z","iopub.status.idle":"2022-07-23T16:51:51.196898Z","shell.execute_reply.started":"2022-07-23T16:51:51.187324Z","shell.execute_reply":"2022-07-23T16:51:51.195985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['numeric_ticket'] = train.Ticket.apply(lambda x: 1 if x.isnumeric() else 0)\ntrain['ticket_letters'] = train.Ticket.apply(lambda x: ''.join(x.split(' ')[:-1]).replace('.','').replace('/','').lower() if len(x.split(' ')[:-1]) >0 else 0)\ntest['numeric_ticket'] = test.Ticket.apply(lambda x: 1 if x.isnumeric() else 0)\ntest['ticket_letters'] = test.Ticket.apply(lambda x: ''.join(x.split(' ')[:-1]).replace('.','').replace('/','').lower() if len(x.split(' ')[:-1]) >0 else 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:51.332728Z","iopub.execute_input":"2022-07-23T16:51:51.333286Z","iopub.status.idle":"2022-07-23T16:51:51.349093Z","shell.execute_reply.started":"2022-07-23T16:51:51.333256Z","shell.execute_reply":"2022-07-23T16:51:51.348267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['ticket_letters'].unique()\ntrain['ticket_letter_cat'] = train.ticket_letters.map({'a5': 7, 'pc': 5, 'stono2': 4, 'pp': 15, \n                                                          'ca': 9, 'scparis': 3, 'sca4': 17, 'a4': 1,\n                                                          'sp': 26, 'soc':19, 'wc': 10, 'sotonoq': 11,\n                                                          'wep': 2, 'c':6, 'sop': 27, 'fa': 28, 'fcc':13,\n                                                          'swpp': 29, 'scow': 30, 'ppp': 31, 'sc': 23,\n                                                          'scah': 8, 'as': 32, 'scahbasle': 33, 'sopp': 18,\n                                                          'fc': 14, 'sotono2': 20, 'casoton': 34, 0:0})\ntest['ticket_letter_cat'] = test.ticket_letters.map({'a4': 1, 'wep': 2, 'scparis': 3, 'stono2': 4,\n                                                        'pc': 5, 'c': 6, 'a5': 7, 'scah': 8, 'ca': 9,\n                                                        'wc': 10, 'sotonoq': 11, 'sca3':12, 'fcc': 13,\n                                                        'fc': 14, 'pp': 15, 'stonoq': 16, 'sca4': 17,\n                                                        'sopp': 18, 'soc': 19, 'sotono2': 20, 'aq4': 21,\n                                                        'a2': 22, 'sc': 23, 'lp': 24, 'aq3': 25,0:0})","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:51.483963Z","iopub.execute_input":"2022-07-23T16:51:51.484538Z","iopub.status.idle":"2022-07-23T16:51:51.499245Z","shell.execute_reply.started":"2022-07-23T16:51:51.484507Z","shell.execute_reply":"2022-07-23T16:51:51.497811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:51.641608Z","iopub.execute_input":"2022-07-23T16:51:51.642192Z","iopub.status.idle":"2022-07-23T16:51:51.652282Z","shell.execute_reply.started":"2022-07-23T16:51:51.642161Z","shell.execute_reply":"2022-07-23T16:51:51.651581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Model","metadata":{}},{"cell_type":"code","source":"y = train[\"Survived\"]\n\nfeatures = [\"Sex\" , \"Fare\", \"Embarked\",'Pclass',\"Age\", \"FamilySize\",'name_cat']\nX = train[features]\nX_test = test[features]\n\nmodel = RandomForestClassifier(n_estimators= 100, max_depth= 5, random_state= 1)\nmodel.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:52:19.910165Z","iopub.execute_input":"2022-07-23T16:52:19.910485Z","iopub.status.idle":"2022-07-23T16:52:20.055401Z","shell.execute_reply.started":"2022-07-23T16:52:19.910452Z","shell.execute_reply":"2022-07-23T16:52:20.054360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test Model","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(X_test)\nfrom sklearn.metrics import accuracy_score\nprint(\"Accuracy Score: \"+ str(accuracy_score(actual_output['Survived'], predictions, normalize=True, sample_weight=None) * 100)) \nprint(\"Model Score: \"+ str(model.score(X_test,predictions)))\n# 891 - 418\n\n# Answer should be\n# 0,1,0,0,1,1,0,1,1","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:52:22.994026Z","iopub.execute_input":"2022-07-23T16:52:22.994351Z","iopub.status.idle":"2022-07-23T16:52:23.039286Z","shell.execute_reply.started":"2022-07-23T16:52:22.994324Z","shell.execute_reply":"2022-07-23T16:52:23.038474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save Output File","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': test.PassengerId, 'Survived': predictions})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:51:53.433607Z","iopub.execute_input":"2022-07-23T16:51:53.434144Z","iopub.status.idle":"2022-07-23T16:51:53.442335Z","shell.execute_reply.started":"2022-07-23T16:51:53.434114Z","shell.execute_reply":"2022-07-23T16:51:53.441468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}