{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.147292Z","iopub.execute_input":"2022-07-10T13:20:36.147658Z","iopub.status.idle":"2022-07-10T13:20:36.157815Z","shell.execute_reply.started":"2022-07-10T13:20:36.147630Z","shell.execute_reply":"2022-07-10T13:20:36.156557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\ndf = pd.read_csv('../input/titanic/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.196820Z","iopub.execute_input":"2022-07-10T13:20:36.197281Z","iopub.status.idle":"2022-07-10T13:20:36.219323Z","shell.execute_reply.started":"2022-07-10T13:20:36.197242Z","shell.execute_reply":"2022-07-10T13:20:36.218196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/titanic/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.251803Z","iopub.execute_input":"2022-07-10T13:20:36.252822Z","iopub.status.idle":"2022-07-10T13:20:36.272460Z","shell.execute_reply.started":"2022-07-10T13:20:36.252777Z","shell.execute_reply":"2022-07-10T13:20:36.271445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['PassengerId','Name','SibSp','Parch','Ticket','Cabin','Embarked'],axis='columns', inplace=True)\ntest.drop(['PassengerId','Name','SibSp','Parch','Ticket','Cabin','Embarked'],axis='columns', inplace=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.286632Z","iopub.execute_input":"2022-07-10T13:20:36.286990Z","iopub.status.idle":"2022-07-10T13:20:36.301843Z","shell.execute_reply.started":"2022-07-10T13:20:36.286960Z","shell.execute_reply":"2022-07-10T13:20:36.300773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.341608Z","iopub.execute_input":"2022-07-10T13:20:36.342368Z","iopub.status.idle":"2022-07-10T13:20:36.354333Z","shell.execute_reply.started":"2022-07-10T13:20:36.342318Z","shell.execute_reply":"2022-07-10T13:20:36.353218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ndf['Sex'] = le.fit_transform(df['Sex'])\ntest['Sex'] = le.fit_transform(test['Sex'])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.372513Z","iopub.execute_input":"2022-07-10T13:20:36.373138Z","iopub.status.idle":"2022-07-10T13:20:36.387484Z","shell.execute_reply.started":"2022-07-10T13:20:36.373089Z","shell.execute_reply":"2022-07-10T13:20:36.386240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.fillna(method = 'ffill',inplace=True)\ntest.fillna(method = 'ffill',inplace=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.421699Z","iopub.execute_input":"2022-07-10T13:20:36.422370Z","iopub.status.idle":"2022-07-10T13:20:36.441271Z","shell.execute_reply.started":"2022-07-10T13:20:36.422317Z","shell.execute_reply":"2022-07-10T13:20:36.440130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df.drop(['Survived'], axis='columns')\ny_train = df['Survived']\ntarget = pd.read_csv('../input/titanic/gender_submission.csv')\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.487288Z","iopub.execute_input":"2022-07-10T13:20:36.488306Z","iopub.status.idle":"2022-07-10T13:20:36.501481Z","shell.execute_reply.started":"2022-07-10T13:20:36.488268Z","shell.execute_reply":"2022-07-10T13:20:36.500606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndt = DecisionTreeClassifier()\ndt.fit(X_train,y_train)\ntarget.drop(['PassengerId'],axis='columns', inplace=True)\ntarget.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.551789Z","iopub.execute_input":"2022-07-10T13:20:36.552602Z","iopub.status.idle":"2022-07-10T13:20:36.567906Z","shell.execute_reply.started":"2022-07-10T13:20:36.552563Z","shell.execute_reply":"2022-07-10T13:20:36.566677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt.score(test,target)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T13:20:36.625123Z","iopub.execute_input":"2022-07-10T13:20:36.625506Z","iopub.status.idle":"2022-07-10T13:20:36.634789Z","shell.execute_reply.started":"2022-07-10T13:20:36.625476Z","shell.execute_reply":"2022-07-10T13:20:36.633504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}