{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Read, Skim and Pre-process data","metadata":{}},{"cell_type":"code","source":"# 1.0 Initial Codes given from Kaggle\n\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-31T05:36:23.329978Z","iopub.execute_input":"2022-07-31T05:36:23.330413Z","iopub.status.idle":"2022-07-31T05:36:23.362600Z","shell.execute_reply.started":"2022-07-31T05:36:23.330316Z","shell.execute_reply":"2022-07-31T05:36:23.361279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import cross_validate\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.inspection import permutation_importance\n\n# 1.1 Read and Skim data\n\ndf = pd.read_csv('/kaggle/input/titanic/train.csv')\n\nprint(df.head())\ndf.info()\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:23.365065Z","iopub.execute_input":"2022-07-31T05:36:23.366410Z","iopub.status.idle":"2022-07-31T05:36:24.738201Z","shell.execute_reply.started":"2022-07-31T05:36:23.366343Z","shell.execute_reply":"2022-07-31T05:36:24.736350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1.2 Find where to pre-processing\n\nprint(df[\"Embarked\"].unique())                                       # ['S' 'C' 'Q' nan]\nprint(df[\"Embarked\"].value_counts())                                 # mode : 'S' (644/891)\n\n# Remove : 1 PassengerId, 3 Name, 8 Ticket (useless) / 10 Cabin (too many NaN)\n# Replace : 4 Sex(categorical) 5 Age(fill NaN) 11 Embarked(some NaN, categorical)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:24.739910Z","iopub.execute_input":"2022-07-31T05:36:24.740237Z","iopub.status.idle":"2022-07-31T05:36:24.750226Z","shell.execute_reply.started":"2022-07-31T05:36:24.740208Z","shell.execute_reply":"2022-07-31T05:36:24.748978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1.3 Pre-processing : Remove or replace NaN\n\n# Remove : 1 PassengerId, 3 Name, 8 Ticket (useless) / 10 Cabin (too many NaN)\n# Replace : 4 Sex (categorical) 5 Age (fill NaN) 11 Embarked (some NaN, categorical)\n#           + 2 Pclass (categorical) - added since Version 3\n\ndf.drop([\"PassengerId\", \"Name\", \"Ticket\", \"Cabin\"], axis=1, inplace=True)\ndf[\"Age\"].fillna(df.Age.mean(), inplace=True)\ndf[\"Embarked\"].fillna(\"S\", inplace=True)                  # \"S\" : mode\ndf = pd.get_dummies(df, columns=[\"Pclass\", \"Embarked\", \"Sex\"])\n# df[\"Sex\"].replace(to_replace=\"male\", value=1, inplace=True)\n# df[\"Sex\"].replace(to_replace=\"female\", value=0, inplace=True)\n\nprint(df.head())\ndf.info()\ndf.describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:24.753152Z","iopub.execute_input":"2022-07-31T05:36:24.753541Z","iopub.status.idle":"2022-07-31T05:36:24.861797Z","shell.execute_reply.started":"2022-07-31T05:36:24.753508Z","shell.execute_reply":"2022-07-31T05:36:24.860535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. HGB","metadata":{}},{"cell_type":"code","source":"# 2.1 Split input and target data\n\ndata = df.iloc[:,1:].to_numpy()                   # except 0 : Survived (target)\ntarget = df.iloc[:,0].to_numpy()\n\nprint(len(data))                                  # 891\nprint(len(target))                                # 891\n\nprint(data[:5,])\nprint(target[:5])","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:24.863768Z","iopub.execute_input":"2022-07-31T05:36:24.864518Z","iopub.status.idle":"2022-07-31T05:36:24.874964Z","shell.execute_reply.started":"2022-07-31T05:36:24.864469Z","shell.execute_reply":"2022-07-31T05:36:24.873783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2.2 HGB\n\ntrain_input, valid_input, train_target, valid_target = train_test_split(data, target, test_size=0.2, random_state=604)\n\nhgb = HistGradientBoostingClassifier(max_leaf_nodes=5, learning_rate=0.01, max_iter=3000, random_state=604)\nhgb.fit(train_input, train_target)\n\nprint(hgb.score(valid_input, valid_target))","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:24.876810Z","iopub.execute_input":"2022-07-31T05:36:24.877646Z","iopub.status.idle":"2022-07-31T05:36:28.314537Z","shell.execute_reply.started":"2022-07-31T05:36:24.877599Z","shell.execute_reply":"2022-07-31T05:36:28.312988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Submit","metadata":{}},{"cell_type":"code","source":"# 3.1 Read and pre-process the test data\n\ntest = pd.read_csv('/kaggle/input/titanic/test.csv')\n\n# print(test.head())\ntest.drop([\"Name\", \"Ticket\", \"Cabin\"], axis=1, inplace=True)            # \"PassengerId\" should be remained\ntest[\"Age\"].fillna(test.Age.mean(), inplace=True)\ntest[\"Fare\"].fillna(test.Fare.mean(), inplace=True)\ntest[\"Embarked\"].fillna(\"S\", inplace=True)\ntest = pd.get_dummies(test, columns=[\"Pclass\", \"Embarked\", \"Sex\"])\n\nprint(test.head())\ntest.info()\n\ntest_input = test.iloc[:,1:].to_numpy() ","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:28.319786Z","iopub.execute_input":"2022-07-31T05:36:28.322504Z","iopub.status.idle":"2022-07-31T05:36:28.365867Z","shell.execute_reply.started":"2022-07-31T05:36:28.322456Z","shell.execute_reply":"2022-07-31T05:36:28.364585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3.2 Generate the submission file\n\ntest_id = test[\"PassengerId\"]\ntest_output = hgb.predict(test_input)\nsubmission = pd.DataFrame({\"PassengerId\": test_id, \"Survived\": test_output})\nsubmission.to_csv(\"./submission_hgb_3.csv\", index=False)\n\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T05:36:28.366953Z","iopub.execute_input":"2022-07-31T05:36:28.367321Z","iopub.status.idle":"2022-07-31T05:36:28.518835Z","shell.execute_reply.started":"2022-07-31T05:36:28.367292Z","shell.execute_reply":"2022-07-31T05:36:28.517689Z"},"trusted":true},"execution_count":null,"outputs":[]}]}