{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport numpy as np \nimport pandas as pd\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, accuracy_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:12.230615Z","iopub.execute_input":"2022-08-11T10:58:12.231125Z","iopub.status.idle":"2022-08-11T10:58:12.244635Z","shell.execute_reply.started":"2022-08-11T10:58:12.231082Z","shell.execute_reply":"2022-08-11T10:58:12.243427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read training data\ndf = pd.read_csv(\"/kaggle/input/titanic/train.csv\", index_col=\"PassengerId\")\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:12.250955Z","iopub.execute_input":"2022-08-11T10:58:12.251928Z","iopub.status.idle":"2022-08-11T10:58:12.285821Z","shell.execute_reply.started":"2022-08-11T10:58:12.251881Z","shell.execute_reply":"2022-08-11T10:58:12.284781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the proportion of people survived\nproportion = df[\"Survived\"].value_counts(normalize=True)\nsns.barplot(x=df[\"Survived\"].unique(), y=proportion)\nplt.ylabel(\"Frequency of Survived (%)\")\nplt.xlabel(\"Survived\")\nplt.title(\"Survived Target Proportion\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:12.287541Z","iopub.execute_input":"2022-08-11T10:58:12.288683Z","iopub.status.idle":"2022-08-11T10:58:12.471470Z","shell.execute_reply.started":"2022-08-11T10:58:12.288639Z","shell.execute_reply":"2022-08-11T10:58:12.470256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot target againsts continous features\nlength = 2\nfig , axs = plt.subplots(ncols=length, nrows=length, figsize=(15, 10))\n\ncolumnsBoxplot = [\"Age\", \"SibSp\", \"Parch\", \"Fare\"]\n\nfor i, column in enumerate(columnsBoxplot):\n    y = i % length\n    x = int(i / length)\n    sns.boxplot(data=df, x=\"Survived\", y=f'{column}', ax=axs[x, y])\n    axs[x, y].set_title(f'{column} VS Survived Boxplot')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:12.473427Z","iopub.execute_input":"2022-08-11T10:58:12.473961Z","iopub.status.idle":"2022-08-11T10:58:13.034619Z","shell.execute_reply.started":"2022-08-11T10:58:12.473915Z","shell.execute_reply":"2022-08-11T10:58:13.033115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove outlier plot for continous features\nfig , axs = plt.subplots(ncols=length, nrows=length, figsize=(15, 10))\n\ncolumnsBoxplot = [\"Age\", \"SibSp\", \"Parch\", \"Fare\"]\n\nfor i, column in enumerate(columnsBoxplot):\n    y = i % length\n    x = int(i / length)\n    low, high = df[column].quantile([0.05, 0.95])\n    dfNoOutlier = df[df[column].between(low, high)]\n    sns.boxplot(data=dfNoOutlier, x=\"Survived\", y=f'{column}', ax=axs[x, y])\n    axs[x, y].set_title(f'{column} VS Survived Boxplot')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:13.038191Z","iopub.execute_input":"2022-08-11T10:58:13.038630Z","iopub.status.idle":"2022-08-11T10:58:13.602363Z","shell.execute_reply.started":"2022-08-11T10:58:13.038597Z","shell.execute_reply":"2022-08-11T10:58:13.600742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The Age VS Survived plot indicates no significant differences in their distribution. The SibSp VS Survived and Fare VS Survived plot show differences between the distribution when the outliers are removed. Lastly, the Parch vs Survived plot shows differences between the distribution even before removing the outliers","metadata":{}},{"cell_type":"code","source":"# Plot categorical data against frequency of people survived\ncolumnsCategory = [\"Pclass\", \"Sex\", \"Embarked\"]\n\nfig , axs = plt.subplots(nrows=3, figsize=(10, 20))\n\nfor i, column in enumerate(columnsCategory):\n    survivalCategory = df.groupby(column)[\"Survived\"].value_counts(normalize=True).rename(\"frequency\").reset_index()\n    sns.barplot(data=survivalCategory, x=column, y=\"frequency\", hue=\"Survived\", ax=axs[i])\n    axs[i].set_title(f'Barplot of {column} and Survived')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:13.604194Z","iopub.execute_input":"2022-08-11T10:58:13.604703Z","iopub.status.idle":"2022-08-11T10:58:14.801479Z","shell.execute_reply.started":"2022-08-11T10:58:13.604655Z","shell.execute_reply":"2022-08-11T10:58:14.800139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The plots between categortical data and the target feature show differences between their categories. Pclass has a downward trend of the target feature following 1st, 2nd, and 3rd class. Embarked also shows the same trend following C, Q, and S. Lastly, for the Sex feature, female feature tends to have a higher number of survivor.","metadata":{}},{"cell_type":"code","source":"# Show correlation grid\ncorrelationMap = df.drop(columns=\"Survived\").corr()\nsns.heatmap(correlationMap)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:14.803087Z","iopub.execute_input":"2022-08-11T10:58:14.803507Z","iopub.status.idle":"2022-08-11T10:58:15.078590Z","shell.execute_reply.started":"2022-08-11T10:58:14.803455Z","shell.execute_reply":"2022-08-11T10:58:15.077039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove outlier for SibSp and Fare\nthresholdPairs = []\ncolumnsOutliers = [\"SibSp\", \"Parch\"]\nfor column in columnsOutliers:\n    thresholdPairs.append(df[column].quantile([0.05, 0.95]))\n\nfor thresholdPair, column in zip(thresholdPairs, columnsOutliers):\n    low, high = thresholdPair\n    print(low, high)\n    df = df[df[column].between(low, high)]\n    \ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.080194Z","iopub.execute_input":"2022-08-11T10:58:15.080615Z","iopub.status.idle":"2022-08-11T10:58:15.101442Z","shell.execute_reply.started":"2022-08-11T10:58:15.080571Z","shell.execute_reply":"2022-08-11T10:58:15.099996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check which feature has large missing data\nbadFeatures = []\nprint(df.isnull().sum() / df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.103145Z","iopub.execute_input":"2022-08-11T10:58:15.103555Z","iopub.status.idle":"2022-08-11T10:58:15.116450Z","shell.execute_reply.started":"2022-08-11T10:58:15.103524Z","shell.execute_reply":"2022-08-11T10:58:15.115386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Features which has more than 50% missing data will be dropped\nbadFeatures.append(\"Cabin\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.120532Z","iopub.execute_input":"2022-08-11T10:58:15.120924Z","iopub.status.idle":"2022-08-11T10:58:15.126003Z","shell.execute_reply.started":"2022-08-11T10:58:15.120884Z","shell.execute_reply":"2022-08-11T10:58:15.125023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check which feature has high or low cardinality (cardinality of 1)\nprint(df.select_dtypes(include=\"object\").nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.127523Z","iopub.execute_input":"2022-08-11T10:58:15.127909Z","iopub.status.idle":"2022-08-11T10:58:15.142457Z","shell.execute_reply.started":"2022-08-11T10:58:15.127863Z","shell.execute_reply":"2022-08-11T10:58:15.141322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# High cardinality would not be useful to the model\nbadFeatures.extend([\"Name\", \"Ticket\", \"Cabin\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.144171Z","iopub.execute_input":"2022-08-11T10:58:15.144861Z","iopub.status.idle":"2022-08-11T10:58:15.155273Z","shell.execute_reply.started":"2022-08-11T10:58:15.144797Z","shell.execute_reply":"2022-08-11T10:58:15.154217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop bad feature\ndfGoodFeatures = df.drop(columns=badFeatures)\ndfGoodFeatures.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.157435Z","iopub.execute_input":"2022-08-11T10:58:15.158734Z","iopub.status.idle":"2022-08-11T10:58:15.179607Z","shell.execute_reply.started":"2022-08-11T10:58:15.158676Z","shell.execute_reply":"2022-08-11T10:58:15.177920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split feature and target\nX = dfGoodFeatures.drop([\"Survived\"], axis=1)\ny = dfGoodFeatures[\"Survived\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.181448Z","iopub.execute_input":"2022-08-11T10:58:15.182961Z","iopub.status.idle":"2022-08-11T10:58:15.200433Z","shell.execute_reply.started":"2022-08-11T10:58:15.182906Z","shell.execute_reply":"2022-08-11T10:58:15.199167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create grid, classifier, along with preprocessor\ncategorical_transformer = Pipeline(steps=[\n    (\"simpleimputer\", SimpleImputer(strategy=\"most_frequent\")),\n    (\"onehotencoder\", OneHotEncoder(handle_unknown=\"ignore\"))\n])\n\nnumerical_transformer = Pipeline(steps=[\n    (\"simpleimputer\", SimpleImputer(strategy=\"mean\")),\n])\n\nnum_col = X.select_dtypes(include=\"number\").columns\ncat_col = X.select_dtypes(exclude=\"number\").columns\n\nprint(num_col, cat_col)\n\ntransformer = ColumnTransformer(\n    transformers=[\n        (\"cat\", categorical_transformer, cat_col),\n        (\"num\", numerical_transformer, num_col)\n    ]\n)\n\nclf = Pipeline(steps=[\n    (\"preprocessor\", transformer),\n    (\"model\", GradientBoostingClassifier())\n])\n\nprint(clf.named_steps[\"preprocessor\"])\n\nparam_grid = {\n    'model__learning_rate': [0.2, 0.25, 0.3],\n    'model__criterion': [\"friedman_mse\", \"squared_error\"],\n    'model__max_depth': [2, 3],\n    'model__n_estimators': [80, 95, 100]\n}\n\ngrid = GridSearchCV(clf, param_grid, n_jobs=-1, cv=10, verbose=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.204119Z","iopub.execute_input":"2022-08-11T10:58:15.205067Z","iopub.status.idle":"2022-08-11T10:58:15.234297Z","shell.execute_reply.started":"2022-08-11T10:58:15.205016Z","shell.execute_reply":"2022-08-11T10:58:15.233100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the grid with training data\ngrid.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:15.235798Z","iopub.execute_input":"2022-08-11T10:58:15.236979Z","iopub.status.idle":"2022-08-11T10:58:31.828408Z","shell.execute_reply.started":"2022-08-11T10:58:15.236935Z","shell.execute_reply":"2022-08-11T10:58:31.826891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GridCV results\npd.DataFrame(grid.cv_results_).sort_values(by=[\"rank_test_score\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:31.830000Z","iopub.execute_input":"2022-08-11T10:58:31.830376Z","iopub.status.idle":"2022-08-11T10:58:31.893442Z","shell.execute_reply.started":"2022-08-11T10:58:31.830345Z","shell.execute_reply":"2022-08-11T10:58:31.892217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and predict test data\ndf_test = pd.read_csv(\"/kaggle/input/titanic/test.csv\", index_col=\"PassengerId\")\nX_fin = df_test.drop(columns=badFeatures)\n\ny_fin_pred = grid.predict(X_fin)\n\ndf_final = pd.DataFrame({\n    \"PassengerId\": df_test.index,\n    \"Survived\": y_fin_pred\n})","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:31.895083Z","iopub.execute_input":"2022-08-11T10:58:31.896129Z","iopub.status.idle":"2022-08-11T10:58:31.919739Z","shell.execute_reply.started":"2022-08-11T10:58:31.896087Z","shell.execute_reply":"2022-08-11T10:58:31.918084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:58:31.921735Z","iopub.execute_input":"2022-08-11T10:58:31.922605Z","iopub.status.idle":"2022-08-11T10:58:31.931303Z","shell.execute_reply.started":"2022-08-11T10:58:31.922552Z","shell.execute_reply":"2022-08-11T10:58:31.930034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}