{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T13:59:11.453736Z","iopub.execute_input":"2022-07-26T13:59:11.454796Z","iopub.status.idle":"2022-07-26T13:59:11.485255Z","shell.execute_reply.started":"2022-07-26T13:59:11.45469Z","shell.execute_reply":"2022-07-26T13:59:11.48409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:59:16.098502Z","iopub.execute_input":"2022-07-26T13:59:16.098879Z","iopub.status.idle":"2022-07-26T13:59:16.123473Z","shell.execute_reply.started":"2022-07-26T13:59:16.098848Z","shell.execute_reply":"2022-07-26T13:59:16.122145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#quantidade de nulos em cada coluna\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:43:23.218744Z","iopub.execute_input":"2022-07-22T17:43:23.219186Z","iopub.status.idle":"2022-07-22T17:43:23.229728Z","shell.execute_reply.started":"2022-07-22T17:43:23.219149Z","shell.execute_reply":"2022-07-22T17:43:23.228622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:43:23.267042Z","iopub.execute_input":"2022-07-22T17:43:23.267798Z","iopub.status.idle":"2022-07-22T17:43:23.277245Z","shell.execute_reply.started":"2022-07-22T17:43:23.267757Z","shell.execute_reply":"2022-07-22T17:43:23.27548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#escolher as colunas que serão usadas no treinamento\ncolunas = ['Pclass','SibSp','Parch','Fare']\nX = train[colunas]\ny = train.Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:59:25.066952Z","iopub.execute_input":"2022-07-26T13:59:25.06737Z","iopub.status.idle":"2022-07-26T13:59:25.083306Z","shell.execute_reply.started":"2022-07-26T13:59:25.067335Z","shell.execute_reply":"2022-07-26T13:59:25.081851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"há a necessidade de um pré-processamento nos outros campos","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_X, val_X, train_y, val_y = train_test_split(X,y,random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:59:31.826071Z","iopub.execute_input":"2022-07-26T13:59:31.826489Z","iopub.status.idle":"2022-07-26T13:59:32.373772Z","shell.execute_reply.started":"2022-07-26T13:59:31.826457Z","shell.execute_reply":"2022-07-26T13:59:32.372619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nmodelo = DecisionTreeClassifier(random_state=1)\nmodelo.fit(train_X,train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:59:38.983594Z","iopub.execute_input":"2022-07-26T13:59:38.983992Z","iopub.status.idle":"2022-07-26T13:59:39.154866Z","shell.execute_reply.started":"2022-07-26T13:59:38.983956Z","shell.execute_reply":"2022-07-26T13:59:39.153685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#testando nos dados de teste\npredicoes = modelo.predict(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:54:16.419571Z","iopub.execute_input":"2022-07-22T17:54:16.420288Z","iopub.status.idle":"2022-07-22T17:54:16.426752Z","shell.execute_reply.started":"2022-07-22T17:54:16.420249Z","shell.execute_reply":"2022-07-22T17:54:16.425704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_y","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:54:18.209777Z","iopub.execute_input":"2022-07-22T17:54:18.210519Z","iopub.status.idle":"2022-07-22T17:54:18.220829Z","shell.execute_reply.started":"2022-07-22T17:54:18.210464Z","shell.execute_reply":"2022-07-22T17:54:18.219728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:54:20.905378Z","iopub.execute_input":"2022-07-22T17:54:20.905758Z","iopub.status.idle":"2022-07-22T17:54:20.913716Z","shell.execute_reply.started":"2022-07-22T17:54:20.905726Z","shell.execute_reply":"2022-07-22T17:54:20.912418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn.metrics as metrics\nmetrics.accuracy_score(val_y, predicoes)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:54:22.75682Z","iopub.execute_input":"2022-07-22T17:54:22.757256Z","iopub.status.idle":"2022-07-22T17:54:22.764491Z","shell.execute_reply.started":"2022-07-22T17:54:22.757221Z","shell.execute_reply":"2022-07-22T17:54:22.763426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(val_y,predicoes))","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:54:26.007262Z","iopub.execute_input":"2022-07-22T17:54:26.007645Z","iopub.status.idle":"2022-07-22T17:54:26.019381Z","shell.execute_reply.started":"2022-07-22T17:54:26.007613Z","shell.execute_reply":"2022-07-22T17:54:26.018329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fazer a previsão do conjunto de testes e enviar para o placar","metadata":{}},{"cell_type":"code","source":"dados_teste = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ndados_teste","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:00:06.98904Z","iopub.execute_input":"2022-07-26T14:00:06.98947Z","iopub.status.idle":"2022-07-26T14:00:07.025104Z","shell.execute_reply.started":"2022-07-26T14:00:06.98943Z","shell.execute_reply":"2022-07-26T14:00:07.023969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dados_teste[colunas]\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:00:16.145696Z","iopub.execute_input":"2022-07-26T14:00:16.146107Z","iopub.status.idle":"2022-07-26T14:00:16.162238Z","shell.execute_reply.started":"2022-07-26T14:00:16.146071Z","shell.execute_reply":"2022-07-26T14:00:16.16116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:50:33.320165Z","iopub.execute_input":"2022-07-22T17:50:33.320573Z","iopub.status.idle":"2022-07-22T17:50:33.331416Z","shell.execute_reply.started":"2022-07-22T17:50:33.320536Z","shell.execute_reply":"2022-07-22T17:50:33.330054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preencher os valores nulos com a mediana\nmedia = test.Fare.median()\ntest.fillna(media,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:00:28.250115Z","iopub.execute_input":"2022-07-26T14:00:28.250625Z","iopub.status.idle":"2022-07-26T14:00:28.259698Z","shell.execute_reply.started":"2022-07-26T14:00:28.250582Z","shell.execute_reply":"2022-07-26T14:00:28.258467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T17:50:36.705698Z","iopub.execute_input":"2022-07-22T17:50:36.706482Z","iopub.status.idle":"2022-07-22T17:50:36.717325Z","shell.execute_reply.started":"2022-07-22T17:50:36.706441Z","shell.execute_reply":"2022-07-22T17:50:36.715984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes = modelo.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:00:43.840698Z","iopub.execute_input":"2022-07-26T14:00:43.841039Z","iopub.status.idle":"2022-07-26T14:00:43.857818Z","shell.execute_reply.started":"2022-07-26T14:00:43.841008Z","shell.execute_reply":"2022-07-26T14:00:43.856727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#produzir o arquivo de resposta\noutput = pd.DataFrame({'PassengerId': dados_teste.PassengerId,\n                       'Survived': predicoes})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:01:16.459596Z","iopub.execute_input":"2022-07-26T14:01:16.459973Z","iopub.status.idle":"2022-07-26T14:01:16.469126Z","shell.execute_reply.started":"2022-07-26T14:01:16.459944Z","shell.execute_reply":"2022-07-26T14:01:16.468368Z"},"trusted":true},"execution_count":null,"outputs":[]}]}