{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T16:23:38.537713Z","iopub.execute_input":"2022-07-25T16:23:38.538107Z","iopub.status.idle":"2022-07-25T16:23:38.547503Z","shell.execute_reply.started":"2022-07-25T16:23:38.538075Z","shell.execute_reply":"2022-07-25T16:23:38.546542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"# Import models\nimport pandas as pd\nimport numpy as np\nimport plotly.express as px\n\n# Read file\ndf = pd.read_csv(\"../input/titanic/train.csv\")\ndf_result = pd.read_csv(\"../input/titanic/gender_submission.csv\")\ndf_test = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:38.675824Z","iopub.execute_input":"2022-07-25T16:23:38.676232Z","iopub.status.idle":"2022-07-25T16:23:38.696720Z","shell.execute_reply.started":"2022-07-25T16:23:38.676199Z","shell.execute_reply":"2022-07-25T16:23:38.695522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# EDA\n\n# Find the class or value - \"class\" of feature: Pclass, Sex, Cabin, Embarked\n# print(df.head(10))\n\n# check null value - Remove column of \"Cabin\"\ndf.isnull().sum() / df.shape[0]\ndf.drop(\"Cabin\", axis=1, inplace=True)\ndf_test.drop(\"Cabin\", axis=1, inplace=True)\n\n# see the static of each feature\ndf.describe()\n\n# see the non-value feature : name, sex, embarked (object)\ndf.info()\n\n# see the corr : important features - Pclass, Fare\ndf.corr()\n\n# see the numbers of unique - the PassengerId, Name, Ticket are too much, dropping these feature might improve the accuracy\nprint(\"Numbers of unique\", df.nunique())\ndf.drop([\"PassengerId\", \"Name\", \"Ticket\"], axis=1, inplace=True)\ndf_test.drop([\"Name\", \"Ticket\"], axis=1, inplace=True)\n\n# Visualization of box - non-class type\nfig = px.box(df, x=\"Survived\", y=\"Age\") # the age and survived is not sensity\nfig.show()\n\nfig = px.box(df, x=\"Survived\", y=\"Fare\") # the fare and survived is sensity\nfig.show()\n\n# Visualization of count plot - class type\ngrouped_col = [\"Pclass\", \"Sex\", \"Embarked\"]\n\nprint(\"Plot of Counter\")\nfor col in grouped_col:\n    fig = px.histogram(df, x=\"Survived\", color=col, barmode='group')\n    fig.show()\n\n# Insight from the box plot:\n# Age : the outlier is (60~80) which still reasonable, we'll keep the data\n# Fare : the higher fare will have more probability to survived\n\n# Insight from the count plot:\n# Pclass!=1 will more likely to survived\n# Sex=Female will more likely to survived\n# Embarked!=S will more likely to survivied","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:38.816396Z","iopub.execute_input":"2022-07-25T16:23:38.817442Z","iopub.status.idle":"2022-07-25T16:23:39.160373Z","shell.execute_reply.started":"2022-07-25T16:23:38.817388Z","shell.execute_reply":"2022-07-25T16:23:39.159326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Feature Engineering\n# 1. SibSp, Parch : check \"isAlone\" (create new feature)\n# 2. Encode all categorical columns\n# 3. Drop unnecessary columns\n\n# \"isAlone\" : we can see the \"isAlone\" have corr with \"Survived\"\ndf[\"isAlone\"] = (df[\"SibSp\"] + df[\"Parch\"]) == 0\ndf_test[\"isAlone\"] = (df_test[\"SibSp\"] + df_test[\"Parch\"]) == 0\n\ndf.corr()\n\n# Encode - sex\ndf[\"Sex\"] = df[\"Sex\"].apply(lambda x: 0 if x==\"female\" else 1)\ndf_test[\"Sex\"] = df_test[\"Sex\"].apply(lambda x: 0 if x==\"female\" else 1)\n\n# Encode - embarked\ndf[\"Embarked\"] = df[\"Embarked\"].map({\"S\":0, \"C\":1, \"Q\":2})\ndf_test[\"Embarked\"] = df_test[\"Embarked\"].map({\"S\":0, \"C\":1, \"Q\":2})\n\n# drop sibsp, parch\ndf.drop([\"SibSp\", \"Parch\"], axis=1, inplace=True)\ndf_test.drop([\"SibSp\", \"Parch\"], axis=1, inplace=True)\n\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.162885Z","iopub.execute_input":"2022-07-25T16:23:39.163311Z","iopub.status.idle":"2022-07-25T16:23:39.202614Z","shell.execute_reply.started":"2022-07-25T16:23:39.163259Z","shell.execute_reply":"2022-07-25T16:23:39.201344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check corr : we can find the \"Age\" need to improve : 試著利用age的特徵建立新特徵 (unfinished)\ndf.corr()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.204520Z","iopub.execute_input":"2022-07-25T16:23:39.205296Z","iopub.status.idle":"2022-07-25T16:23:39.228981Z","shell.execute_reply.started":"2022-07-25T16:23:39.205262Z","shell.execute_reply":"2022-07-25T16:23:39.226735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imputation","metadata":{}},{"cell_type":"code","source":"# Impute the missing value\nprint(\"NULL : \", df.isnull().sum()) # check the null value in each feature (only age and embarked)\n\n# Impute-Age : we can use \"median\" but this is not good way, because we can see the corr between Age and Pclass\n# Impute(missing value) Age based on Pclass having higher corr - 利用 Pclass 分組計算 age 的中位數, 進行 missing value imputation\nprint(df.groupby(\"Pclass\")[\"Age\"].median())\ndf[\"Age\"] = df.groupby(\"Pclass\")[\"Age\"].apply(lambda x: x.fillna(x.median()))\ndf_test[\"Age\"] = df_test.groupby(\"Pclass\")[\"Age\"].apply(lambda x: x.fillna(x.median()))\n\n# Impute-Embarked : just use median\ndf[\"Embarked\"].replace(np.nan, df[\"Embarked\"].median(), inplace=True)\ndf_test[\"Embarked\"].replace(np.nan, df_test[\"Embarked\"].median(), inplace=True)\n\n# Impute-Fare for df_test\ndf_test[\"Fare\"].replace(np.nan, df_test[\"Fare\"].median(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.232543Z","iopub.execute_input":"2022-07-25T16:23:39.233889Z","iopub.status.idle":"2022-07-25T16:23:39.262824Z","shell.execute_reply.started":"2022-07-25T16:23:39.233840Z","shell.execute_reply":"2022-07-25T16:23:39.261632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalization","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import normalize\n\ndf[[\"Age\", \"Fare\"]] = normalize(df[[\"Age\", \"Fare\"]])\ndf_test[[\"Age\", \"Fare\"]] = normalize(df_test[[\"Age\", \"Fare\"]])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.264496Z","iopub.execute_input":"2022-07-25T16:23:39.265139Z","iopub.status.idle":"2022-07-25T16:23:39.279103Z","shell.execute_reply.started":"2022-07-25T16:23:39.265094Z","shell.execute_reply":"2022-07-25T16:23:39.277698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modeling**","metadata":{}},{"cell_type":"code","source":"# Modeling\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost.sklearn import XGBClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\n\nfrom sklearn import metrics\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.281322Z","iopub.execute_input":"2022-07-25T16:23:39.282167Z","iopub.status.idle":"2022-07-25T16:23:39.290266Z","shell.execute_reply.started":"2022-07-25T16:23:39.282077Z","shell.execute_reply":"2022-07-25T16:23:39.289122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(\"Survived\", axis=1)\nX = pd.get_dummies(X, drop_first=True) # OneHot encoding\n\ny = df[\"Survived\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.307894Z","iopub.execute_input":"2022-07-25T16:23:39.308568Z","iopub.status.idle":"2022-07-25T16:23:39.322069Z","shell.execute_reply.started":"2022-07-25T16:23:39.308520Z","shell.execute_reply":"2022-07-25T16:23:39.321165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_best_model(X_train, X_test, y_train, y_test):\n    # Logistic Regression\n    logreg = LogisticRegression(max_iter=600, random_state=42)\n    logreg.fit(X_train, y_train)\n    y_pred = logreg.predict(X_test)\n    logreg_acc = round(metrics.accuracy_score(y_test, y_pred), 2) # for the classified accuracy\n    logreg_cv_score = cross_val_score(logreg, X_train, y_train, cv=5).mean()\n\n    # Decision Tree\n    dt = DecisionTreeClassifier(random_state=42)\n    dt.fit(X_train, y_train)\n    y_pred = dt.predict(X_test)\n    dt_acc = round(metrics.accuracy_score(y_test, y_pred), 2)\n    dt_cv_score = cross_val_score(dt, X_train, y_train, cv=5).mean()\n\n    # Random Forest\n    rf = RandomForestClassifier(random_state=42)\n    rf.fit(X_train, y_train)\n    y_pred = rf.predict(X_test)\n    rf_acc = round(metrics.accuracy_score(y_test, y_pred), 2)\n    rf_cv_score = cross_val_score(rf, X_train, y_train, cv=5).mean()\n\n    # XGB\n    xgb = XGBClassifier(random_state=42)\n    xgb.fit(X_train, y_train)\n    y_pred = xgb.predict(X_test)\n    xgb_acc = round(metrics.accuracy_score(y_test, y_pred), 2)\n    xgb_cv_score = cross_val_score(xgb, X_train, y_train, cv=5).mean()\n\n    # Gradient Boosting\n    gb = GradientBoostingClassifier(random_state=42)\n    gb.fit(X_train, y_train)\n    y_pred = gb.predict(X_test)\n    gb_acc = round(metrics.accuracy_score(y_test, y_pred), 2)\n    gb_cv_score = cross_val_score(gb, X_train, y_train, cv=5).mean()\n\n    df_model = pd.DataFrame({\"Model\":[\"LogisticRegression\", \"DecisionTree\", \"RandomForest\", \"XGBoost\", \"GBM\"],\n                           \"Acc\": [logreg_acc, dt_acc, rf_acc, xgb_acc, gb_acc],\n                            \"CvScore\": [logreg_cv_score, dt_cv_score, rf_cv_score, xgb_cv_score, gb_cv_score]})\n    print(df_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.386348Z","iopub.execute_input":"2022-07-25T16:23:39.386983Z","iopub.status.idle":"2022-07-25T16:23:39.401014Z","shell.execute_reply.started":"2022-07-25T16:23:39.386950Z","shell.execute_reply":"2022-07-25T16:23:39.400119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"find_best_model(X_train, X_test, y_train, y_test) # Select XGB\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:39.456283Z","iopub.execute_input":"2022-07-25T16:23:39.456681Z","iopub.status.idle":"2022-07-25T16:23:43.373620Z","shell.execute_reply.started":"2022-07-25T16:23:39.456634Z","shell.execute_reply":"2022-07-25T16:23:43.372324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGB","metadata":{}},{"cell_type":"code","source":"xgb = XGBClassifier(random_state=42, n_jobs=-1) # n_job - use all cores on machine\n\npara_grid = {\n    \"n_estimators\":[100, 200, 300], # 決策數的個數, default=100\n    \"max_depth\":[4, 6, 8], # the maximu depth of the individual estimators(tree), default=6\n    \"eta\":[0.01, 0.1, 0.3], # the learning rate, default=0.3\n}\n\nCV_xgb = GridSearchCV(estimator=xgb, param_grid=para_grid, cv=5, scoring=\"accuracy\", verbose=10)\nCV_xgb.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:23:43.376155Z","iopub.execute_input":"2022-07-25T16:23:43.376843Z","iopub.status.idle":"2022-07-25T16:25:23.028805Z","shell.execute_reply.started":"2022-07-25T16:23:43.376798Z","shell.execute_reply":"2022-07-25T16:25:23.027878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"print(\"Best hyperparameters:\", CV_xgb.best_params_)\nprint(\"Best Score:\", CV_xgb.best_score_)\n\nX_test = df_test.drop(\"PassengerId\", axis=1)\n\npredictions = CV_xgb.predict(X_test)\n\noutput = pd.DataFrame({\"PassengerId\": df_test[\"PassengerId\"],\n                      \"Survived\": predictions})\n\noutput.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T16:26:00.834224Z","iopub.execute_input":"2022-07-25T16:26:00.834612Z","iopub.status.idle":"2022-07-25T16:26:00.856462Z","shell.execute_reply.started":"2022-07-25T16:26:00.834579Z","shell.execute_reply":"2022-07-25T16:26:00.855539Z"},"trusted":true},"execution_count":null,"outputs":[]}]}