{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn import tree\n\ntest = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntrain = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_test_data = [train, test]\n\n# Split into validation and training data\n#train_X, val_X, train_y, val_y = train_test_split(X, y, random_state=1)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T19:26:36.591730Z","iopub.execute_input":"2022-07-31T19:26:36.592494Z","iopub.status.idle":"2022-07-31T19:26:38.119667Z","shell.execute_reply.started":"2022-07-31T19:26:36.592397Z","shell.execute_reply":"2022-07-31T19:26:38.118371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndeck = {\"A\": 1, \"B\": 2, \"C\": 3, \"D\": 4, \"E\": 5, \"F\": 6, \"G\": 7, \"U\": 8}\n\nfor dataset in train_test_data:\n    dataset['Cabin'] = dataset['Cabin'].fillna(\"U0\")\n    dataset['Deck'] = dataset['Cabin'].map(lambda x: re.compile(\"([a-zA-Z]+)\").search(x).group())\n    dataset['Deck'] = dataset['Deck'].map(deck)\n    dataset['Deck'] = dataset['Deck'].fillna(0)\n    dataset['Deck'] = dataset['Deck'].astype(int)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:26:38.978538Z","iopub.execute_input":"2022-07-31T19:26:38.978987Z","iopub.status.idle":"2022-07-31T19:26:39.008595Z","shell.execute_reply.started":"2022-07-31T19:26:38.978950Z","shell.execute_reply":"2022-07-31T19:26:39.007234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Age']=dataset['Age'].fillna(dataset['Age'].mean())\n    dataset['Age']=dataset['Age'].round()\n    dataset['Age']=dataset['Age'].astype(int)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:26:45.870743Z","iopub.execute_input":"2022-07-31T19:26:45.871132Z","iopub.status.idle":"2022-07-31T19:26:45.882382Z","shell.execute_reply.started":"2022-07-31T19:26:45.871100Z","shell.execute_reply":"2022-07-31T19:26:45.881526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Fare']=dataset['Fare'].fillna(dataset['Age'].mean())\n    dataset['Fare']=dataset['Fare'].round()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:26:47.346750Z","iopub.execute_input":"2022-07-31T19:26:47.347172Z","iopub.status.idle":"2022-07-31T19:26:47.355865Z","shell.execute_reply.started":"2022-07-31T19:26:47.347137Z","shell.execute_reply":"2022-07-31T19:26:47.354975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:26:48.582415Z","iopub.execute_input":"2022-07-31T19:26:48.583203Z","iopub.status.idle":"2022-07-31T19:26:48.592437Z","shell.execute_reply.started":"2022-07-31T19:26:48.583156Z","shell.execute_reply":"2022-07-31T19:26:48.591416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Embarked'] = dataset['Embarked'].fillna('S')\nfor dataset in train_test_data:\n    #print(dataset.Embarked.unique())\n    dataset['Embarked'] = dataset['Embarked'].map( {'S': 0, 'C': 1, 'Q': 2} ).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:26:57.049157Z","iopub.execute_input":"2022-07-31T19:26:57.049698Z","iopub.status.idle":"2022-07-31T19:26:57.060311Z","shell.execute_reply.started":"2022-07-31T19:26:57.049654Z","shell.execute_reply":"2022-07-31T19:26:57.059194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in train_test_data:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.')\n\ntitle_map = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Other\": 5}\nfor dataset in train_test_data:\n    dataset['Title'] = dataset['Title'].map(title_map)\n    dataset['Title'] = dataset['Title'].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:27:00.363694Z","iopub.execute_input":"2022-07-31T19:27:00.364139Z","iopub.status.idle":"2022-07-31T19:27:00.382424Z","shell.execute_reply.started":"2022-07-31T19:27:00.364101Z","shell.execute_reply":"2022-07-31T19:27:00.381210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_drop = ['Name', 'SibSp', 'Parch', 'Ticket', 'Cabin']\ntrain = train.drop(features_drop, axis=1)\ntest = test.drop(features_drop, axis=1)\ntrain = train.drop(['PassengerId'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:27:03.277512Z","iopub.execute_input":"2022-07-31T19:27:03.278499Z","iopub.status.idle":"2022-07-31T19:27:03.289487Z","shell.execute_reply.started":"2022-07-31T19:27:03.278460Z","shell.execute_reply":"2022-07-31T19:27:03.288356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train.drop('Survived', axis=1)\ny = train['Survived']\nX_test = test.drop(\"PassengerId\", axis=1).copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:27:09.639584Z","iopub.execute_input":"2022-07-31T19:27:09.639984Z","iopub.status.idle":"2022-07-31T19:27:09.648145Z","shell.execute_reply.started":"2022-07-31T19:27:09.639953Z","shell.execute_reply":"2022-07-31T19:27:09.646949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_model = DecisionTreeRegressor(random_state=100)\n\n# Fit model\ntitanic_model.fit(x, y)\npredict_titanic = titanic_model.predict(x)\n\ny_random_forest = titanic_model.predict(X_test)\nacc_random_forest = round(titanic_model.score(x,y) * 100, 2)\nprint (acc_random_forest)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:27:40.077732Z","iopub.execute_input":"2022-07-31T19:27:40.078164Z","iopub.status.idle":"2022-07-31T19:27:40.098084Z","shell.execute_reply.started":"2022-07-31T19:27:40.078128Z","shell.execute_reply":"2022-07-31T19:27:40.097117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": test[\"PassengerId\"],\n        \"Survived\": y_random_forest.round().astype(int)\n    })\nsubmission.to_csv('submission.csv', index = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T19:37:05.093355Z","iopub.execute_input":"2022-07-31T19:37:05.093741Z","iopub.status.idle":"2022-07-31T19:37:05.102128Z","shell.execute_reply.started":"2022-07-31T19:37:05.093707Z","shell.execute_reply":"2022-07-31T19:37:05.101142Z"},"trusted":true},"execution_count":null,"outputs":[]}]}