{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T21:59:06.191607Z","iopub.execute_input":"2022-08-11T21:59:06.192029Z","iopub.status.idle":"2022-08-11T21:59:06.202550Z","shell.execute_reply.started":"2022-08-11T21:59:06.191999Z","shell.execute_reply":"2022-08-11T21:59:06.201232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importação dos dados de treino\ntrain = pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:06.371997Z","iopub.execute_input":"2022-08-11T21:59:06.372893Z","iopub.status.idle":"2022-08-11T21:59:06.382197Z","shell.execute_reply.started":"2022-08-11T21:59:06.372854Z","shell.execute_reply":"2022-08-11T21:59:06.381221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tratamento de dados para a coluna Cabin**","metadata":{}},{"cell_type":"code","source":"# Salvar os valores unicos de Cabin\nitens = []\nfor item in train['Cabin']:\n    if item not in itens:\n        itens.append(item)\nprint(itens)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:06.551194Z","iopub.execute_input":"2022-08-11T21:59:06.552057Z","iopub.status.idle":"2022-08-11T21:59:06.559044Z","shell.execute_reply.started":"2022-08-11T21:59:06.552018Z","shell.execute_reply":"2022-08-11T21:59:06.557821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Substituindo os valores(strings) para numéricos\nfor i in range(len(itens)):\n    train.loc[itens[i] == train.Cabin, 'Cabin'] = i\ntrain.Cabin.values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:06.685607Z","iopub.execute_input":"2022-08-11T21:59:06.686406Z","iopub.status.idle":"2022-08-11T21:59:06.773782Z","shell.execute_reply.started":"2022-08-11T21:59:06.686368Z","shell.execute_reply":"2022-08-11T21:59:06.772950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tratando valores nulos da coluna\ntrain['Cabin'].fillna(0, inplace=True)\ntrain.Cabin.values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:06.776086Z","iopub.execute_input":"2022-08-11T21:59:06.776529Z","iopub.status.idle":"2022-08-11T21:59:06.787366Z","shell.execute_reply.started":"2022-08-11T21:59:06.776487Z","shell.execute_reply":"2022-08-11T21:59:06.786162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Adicionar coluna Sex aos dados de treino**","metadata":{}},{"cell_type":"code","source":"# Tratar os dados da coluna Sex\ntrain = train.replace('male', 1).replace('female', 2)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:06.910044Z","iopub.execute_input":"2022-08-11T21:59:06.910466Z","iopub.status.idle":"2022-08-11T21:59:06.940222Z","shell.execute_reply.started":"2022-08-11T21:59:06.910432Z","shell.execute_reply":"2022-08-11T21:59:06.938602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Salvar os valores unicos de Ticket\nitens = []\nfor item in train['Ticket']:\n    if item not in itens:\n        itens.append(item)\n\n# Substituindo os valores(strings) para numéricos\nfor i in range(len(itens)):\n    train.loc[itens[i] == train.Ticket, 'Ticket'] = i\ntrain.Cabin.values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.022587Z","iopub.execute_input":"2022-08-11T21:59:07.023043Z","iopub.status.idle":"2022-08-11T21:59:07.436735Z","shell.execute_reply.started":"2022-08-11T21:59:07.023008Z","shell.execute_reply":"2022-08-11T21:59:07.435469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Escolher as colunas que serão usadas no treinamento\ncolunas = ['Pclass', 'Sex', 'SibSp', 'Parch', 'Fare', 'Cabin', 'Ticket']\n\n# Verificar a existência de dados nulos nas colunas que utilizaremos\ntrain[colunas].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.439180Z","iopub.execute_input":"2022-08-11T21:59:07.439887Z","iopub.status.idle":"2022-08-11T21:59:07.452378Z","shell.execute_reply.started":"2022-08-11T21:59:07.439846Z","shell.execute_reply":"2022-08-11T21:59:07.451012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separando dados de treino e alvo\nX = train[colunas]\ny = train.Survived","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.454482Z","iopub.execute_input":"2022-08-11T21:59:07.455019Z","iopub.status.idle":"2022-08-11T21:59:07.462960Z","shell.execute_reply.started":"2022-08-11T21:59:07.454974Z","shell.execute_reply":"2022-08-11T21:59:07.461698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modelo preditivo**","metadata":{}},{"cell_type":"code","source":"# Treinando modelo\nfrom sklearn.model_selection import train_test_split\ntrain_X, val_X, train_y, val_y = train_test_split(X,y,random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.465117Z","iopub.execute_input":"2022-08-11T21:59:07.465479Z","iopub.status.idle":"2022-08-11T21:59:07.479407Z","shell.execute_reply.started":"2022-08-11T21:59:07.465448Z","shell.execute_reply":"2022-08-11T21:59:07.478193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross Validation\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_validate\nn_folders = 5\nnome_metricas = ['accuracy', 'precision_macro', 'recall_macro','f1_macro']\ncross_val = StratifiedKFold(n_splits=n_folders, shuffle=True, random_state=32)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.568137Z","iopub.execute_input":"2022-08-11T21:59:07.568930Z","iopub.status.idle":"2022-08-11T21:59:07.575929Z","shell.execute_reply.started":"2022-08-11T21:59:07.568883Z","shell.execute_reply":"2022-08-11T21:59:07.574431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Utilizando algoritmo classificativo\nfrom sklearn.tree import DecisionTreeClassifier\nmodelo = DecisionTreeClassifier(random_state=3)\nmodelo.fit(train_X,train_y)\nmetricas = cross_validate(modelo, X, y, cv=cross_val, scoring=nome_metricas)\nfor met in metricas:\n    print(f\"- {met}:\")\n    print(f\"-- {metricas[met]}\")\n    print(f\"-- {np.mean(metricas[met])} +- {np.std(metricas[met])}\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.748303Z","iopub.execute_input":"2022-08-11T21:59:07.748710Z","iopub.status.idle":"2022-08-11T21:59:07.815311Z","shell.execute_reply.started":"2022-08-11T21:59:07.748678Z","shell.execute_reply":"2022-08-11T21:59:07.813900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Testando nos dados de teste\npredicoes = modelo.predict(val_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:07.997170Z","iopub.execute_input":"2022-08-11T21:59:07.997588Z","iopub.status.idle":"2022-08-11T21:59:08.005448Z","shell.execute_reply.started":"2022-08-11T21:59:07.997554Z","shell.execute_reply":"2022-08-11T21:59:08.004253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:08.175070Z","iopub.execute_input":"2022-08-11T21:59:08.175470Z","iopub.status.idle":"2022-08-11T21:59:08.183300Z","shell.execute_reply.started":"2022-08-11T21:59:08.175440Z","shell.execute_reply":"2022-08-11T21:59:08.182057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Medindo acurácea\nimport sklearn.metrics as metrics\nmetrics.accuracy_score(val_y, predicoes)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:08.354604Z","iopub.execute_input":"2022-08-11T21:59:08.354995Z","iopub.status.idle":"2022-08-11T21:59:08.363352Z","shell.execute_reply.started":"2022-08-11T21:59:08.354963Z","shell.execute_reply":"2022-08-11T21:59:08.362167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Utilização de outras métricas\nfrom sklearn.metrics import classification_report\nprint(classification_report(val_y,predicoes))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:08.529947Z","iopub.execute_input":"2022-08-11T21:59:08.530362Z","iopub.status.idle":"2022-08-11T21:59:08.541628Z","shell.execute_reply.started":"2022-08-11T21:59:08.530328Z","shell.execute_reply":"2022-08-11T21:59:08.540486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importando dados de teste\ndados_teste = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\n\n# Tratamento de dados da coluna Cabin(O mesmo feito na base de dados train)\nitens = []\nfor item in dados_teste['Cabin']:\n    if item not in itens:\n        itens.append(item)\n\n# Substituindo os valores(strings) para numéricos\nfor i in range(len(itens)):\n    dados_teste.loc[itens[i] == dados_teste.Cabin, 'Cabin'] = i\n\n# Tratando valores nulos da coluna\ndados_teste['Cabin'].fillna(0, inplace=True)\n\n# Tratamento de dados da coluna Ticket\n# Salvar os valores unicos de Ticket\nitens = []\nfor item in dados_teste['Ticket']:\n    if item not in itens:\n        itens.append(item)\n\n# Substituindo os valores(strings) para numéricos\nfor i in range(len(itens)):\n    dados_teste.loc[itens[i] == dados_teste.Ticket, 'Ticket'] = i\ndados_teste.Ticket.values\n\n# Tratamento de dados da coluna Sex\ndados_teste = dados_teste.replace('male', 1).replace('female', 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:08.706278Z","iopub.execute_input":"2022-08-11T21:59:08.707482Z","iopub.status.idle":"2022-08-11T21:59:08.948665Z","shell.execute_reply.started":"2022-08-11T21:59:08.707440Z","shell.execute_reply":"2022-08-11T21:59:08.947416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dados_teste[colunas]\n# Verificando por valores nulos\ntest.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:08.951550Z","iopub.execute_input":"2022-08-11T21:59:08.952046Z","iopub.status.idle":"2022-08-11T21:59:08.961870Z","shell.execute_reply.started":"2022-08-11T21:59:08.952009Z","shell.execute_reply":"2022-08-11T21:59:08.960919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tratando os valores nulos com a mediana\ntest.fillna(0,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:09.181502Z","iopub.execute_input":"2022-08-11T21:59:09.182377Z","iopub.status.idle":"2022-08-11T21:59:09.189514Z","shell.execute_reply.started":"2022-08-11T21:59:09.182337Z","shell.execute_reply":"2022-08-11T21:59:09.188184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verificando existência de valores nulos\ntest.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:09.465423Z","iopub.execute_input":"2022-08-11T21:59:09.466288Z","iopub.status.idle":"2022-08-11T21:59:09.479127Z","shell.execute_reply.started":"2022-08-11T21:59:09.466241Z","shell.execute_reply":"2022-08-11T21:59:09.477734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicoes = modelo.predict(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:09.639464Z","iopub.execute_input":"2022-08-11T21:59:09.640349Z","iopub.status.idle":"2022-08-11T21:59:09.648577Z","shell.execute_reply.started":"2022-08-11T21:59:09.640299Z","shell.execute_reply":"2022-08-11T21:59:09.647732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Produzir o arquivo de resposta\noutput = pd.DataFrame({'PassengerId': dados_teste.PassengerId,\n                       'Survived': predicoes})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:59:09.812857Z","iopub.execute_input":"2022-08-11T21:59:09.813616Z","iopub.status.idle":"2022-08-11T21:59:09.822177Z","shell.execute_reply.started":"2022-08-11T21:59:09.813576Z","shell.execute_reply":"2022-08-11T21:59:09.820847Z"},"trusted":true},"execution_count":null,"outputs":[]}]}