{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T13:42:02.049346Z","iopub.execute_input":"2022-07-28T13:42:02.049848Z","iopub.status.idle":"2022-07-28T13:42:02.096312Z","shell.execute_reply.started":"2022-07-28T13:42:02.049749Z","shell.execute_reply":"2022-07-28T13:42:02.095133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:02.098793Z","iopub.execute_input":"2022-07-28T13:42:02.099503Z","iopub.status.idle":"2022-07-28T13:42:02.123357Z","shell.execute_reply.started":"2022-07-28T13:42:02.099463Z","shell.execute_reply":"2022-07-28T13:42:02.122228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#quantidade de nulos em cada coluna\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:02.125021Z","iopub.execute_input":"2022-07-28T13:42:02.125470Z","iopub.status.idle":"2022-07-28T13:42:02.147221Z","shell.execute_reply.started":"2022-07-28T13:42:02.125429Z","shell.execute_reply":"2022-07-28T13:42:02.145009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:02.150570Z","iopub.execute_input":"2022-07-28T13:42:02.151437Z","iopub.status.idle":"2022-07-28T13:42:02.163155Z","shell.execute_reply.started":"2022-07-28T13:42:02.151387Z","shell.execute_reply":"2022-07-28T13:42:02.161776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#escolher as colunas que serão usadas no treinamento\ncolunas = ['Pclass','SibSp','Parch','Fare']\nX = train[colunas]\ny = train.Survived","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:02.165043Z","iopub.execute_input":"2022-07-28T13:42:02.165767Z","iopub.status.idle":"2022-07-28T13:42:02.186957Z","shell.execute_reply.started":"2022-07-28T13:42:02.165725Z","shell.execute_reply":"2022-07-28T13:42:02.185385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"há a necessidade de um pré-processamento nos outros campos","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_X, val_X, train_y, val_y = train_test_split(X,y,random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:02.189132Z","iopub.execute_input":"2022-07-28T13:42:02.190217Z","iopub.status.idle":"2022-07-28T13:42:03.524922Z","shell.execute_reply.started":"2022-07-28T13:42:02.190169Z","shell.execute_reply":"2022-07-28T13:42:03.523752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nmodelo = DecisionTreeClassifier(random_state=1)\nmodelo.fit(train_X,train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.526984Z","iopub.execute_input":"2022-07-28T13:42:03.528233Z","iopub.status.idle":"2022-07-28T13:42:03.711667Z","shell.execute_reply.started":"2022-07-28T13:42:03.528182Z","shell.execute_reply":"2022-07-28T13:42:03.710757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#testando nos dados de teste\npredicoes = modelo.predict(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.713281Z","iopub.execute_input":"2022-07-28T13:42:03.714389Z","iopub.status.idle":"2022-07-28T13:42:03.723500Z","shell.execute_reply.started":"2022-07-28T13:42:03.714342Z","shell.execute_reply":"2022-07-28T13:42:03.722355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_y","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.725442Z","iopub.execute_input":"2022-07-28T13:42:03.726206Z","iopub.status.idle":"2022-07-28T13:42:03.740514Z","shell.execute_reply.started":"2022-07-28T13:42:03.726161Z","shell.execute_reply":"2022-07-28T13:42:03.738969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.746026Z","iopub.execute_input":"2022-07-28T13:42:03.746996Z","iopub.status.idle":"2022-07-28T13:42:03.757172Z","shell.execute_reply.started":"2022-07-28T13:42:03.746945Z","shell.execute_reply":"2022-07-28T13:42:03.755717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn.metrics as metrics\nmetrics.accuracy_score(val_y, predicoes)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.759871Z","iopub.execute_input":"2022-07-28T13:42:03.760990Z","iopub.status.idle":"2022-07-28T13:42:03.772089Z","shell.execute_reply.started":"2022-07-28T13:42:03.760942Z","shell.execute_reply":"2022-07-28T13:42:03.770932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(val_y,predicoes))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.775361Z","iopub.execute_input":"2022-07-28T13:42:03.775963Z","iopub.status.idle":"2022-07-28T13:42:03.794179Z","shell.execute_reply.started":"2022-07-28T13:42:03.775926Z","shell.execute_reply":"2022-07-28T13:42:03.792725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fazer a previsão do conjunto de testes e enviar para o placar","metadata":{}},{"cell_type":"code","source":"dados_teste = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ndados_teste","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.795769Z","iopub.execute_input":"2022-07-28T13:42:03.797084Z","iopub.status.idle":"2022-07-28T13:42:03.839621Z","shell.execute_reply.started":"2022-07-28T13:42:03.797026Z","shell.execute_reply":"2022-07-28T13:42:03.838171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dados_teste[colunas]\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.843691Z","iopub.execute_input":"2022-07-28T13:42:03.844594Z","iopub.status.idle":"2022-07-28T13:42:03.866194Z","shell.execute_reply.started":"2022-07-28T13:42:03.844547Z","shell.execute_reply":"2022-07-28T13:42:03.864604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.867973Z","iopub.execute_input":"2022-07-28T13:42:03.868404Z","iopub.status.idle":"2022-07-28T13:42:03.877849Z","shell.execute_reply.started":"2022-07-28T13:42:03.868361Z","shell.execute_reply":"2022-07-28T13:42:03.876785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preencher os valores nulos com a mediana\nmedia = test.Fare.median()\ntest.fillna(media,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.879002Z","iopub.execute_input":"2022-07-28T13:42:03.879961Z","iopub.status.idle":"2022-07-28T13:42:03.895002Z","shell.execute_reply.started":"2022-07-28T13:42:03.879929Z","shell.execute_reply":"2022-07-28T13:42:03.893581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.896892Z","iopub.execute_input":"2022-07-28T13:42:03.897408Z","iopub.status.idle":"2022-07-28T13:42:03.909478Z","shell.execute_reply.started":"2022-07-28T13:42:03.897364Z","shell.execute_reply":"2022-07-28T13:42:03.908016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes = modelo.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.911521Z","iopub.execute_input":"2022-07-28T13:42:03.912238Z","iopub.status.idle":"2022-07-28T13:42:03.920465Z","shell.execute_reply.started":"2022-07-28T13:42:03.912194Z","shell.execute_reply":"2022-07-28T13:42:03.919554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#produzir o arquivo de resposta\noutput = pd.DataFrame({'PassengerId': dados_teste.PassengerId,\n                       'Survived': predicoes})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T13:42:03.921780Z","iopub.execute_input":"2022-07-28T13:42:03.922252Z","iopub.status.idle":"2022-07-28T13:42:03.939931Z","shell.execute_reply.started":"2022-07-28T13:42:03.922218Z","shell.execute_reply":"2022-07-28T13:42:03.938538Z"},"trusted":true},"execution_count":null,"outputs":[]}]}