{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T10:21:17.104769Z","iopub.execute_input":"2022-07-30T10:21:17.105170Z","iopub.status.idle":"2022-07-30T10:21:17.113921Z","shell.execute_reply.started":"2022-07-30T10:21:17.105137Z","shell.execute_reply":"2022-07-30T10:21:17.112515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/titanic/train.csv').drop('PassengerId', axis=1)\ndata","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:17.643785Z","iopub.execute_input":"2022-07-30T10:21:17.644238Z","iopub.status.idle":"2022-07-30T10:21:17.676816Z","shell.execute_reply.started":"2022-07-30T10:21:17.644201Z","shell.execute_reply":"2022-07-30T10:21:17.675631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Importing some additional modules","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport sklearn\nimport math\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:17.689767Z","iopub.execute_input":"2022-07-30T10:21:17.690809Z","iopub.status.idle":"2022-07-30T10:21:17.696386Z","shell.execute_reply.started":"2022-07-30T10:21:17.690758Z","shell.execute_reply":"2022-07-30T10:21:17.695318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Handling numerical columns","metadata":{}},{"cell_type":"code","source":"nums = data[['Survived', 'Fare', 'Age', 'SibSp', 'Parch', 'Pclass']]\nnums","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:17.772864Z","iopub.execute_input":"2022-07-30T10:21:17.773253Z","iopub.status.idle":"2022-07-30T10:21:17.792433Z","shell.execute_reply.started":"2022-07-30T10:21:17.773221Z","shell.execute_reply":"2022-07-30T10:21:17.791224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(16, 15))\ncols = nums.columns\nfor i in range(len(cols)):\n    ax[math.floor(i / 2), i % 2].set_title(f'Hist of {cols[i]}')\n    ax[math.floor(i / 2), i % 2].hist(nums[cols[i]])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:17.794478Z","iopub.execute_input":"2022-07-30T10:21:17.795640Z","iopub.status.idle":"2022-07-30T10:21:18.726540Z","shell.execute_reply.started":"2022-07-30T10:21:17.795563Z","shell.execute_reply":"2022-07-30T10:21:18.725364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(16, 15))\ncols = nums.columns\nfor i in range(len(cols)):\n    ax[math.floor(i / 2), i % 2].set_title(f'Boxplot of {cols[i]}')\n    ax[math.floor(i / 2), i % 2].boxplot(nums[cols[i]].dropna())","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:18.728670Z","iopub.execute_input":"2022-07-30T10:21:18.729865Z","iopub.status.idle":"2022-07-30T10:21:19.393125Z","shell.execute_reply.started":"2022-07-30T10:21:18.729812Z","shell.execute_reply":"2022-07-30T10:21:19.391693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Handle NaNs","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:12:35.711648Z","iopub.execute_input":"2022-07-28T15:12:35.712050Z","iopub.status.idle":"2022-07-28T15:12:35.718103Z","shell.execute_reply.started":"2022-07-28T15:12:35.712023Z","shell.execute_reply":"2022-07-28T15:12:35.716732Z"}}},{"cell_type":"code","source":"data.loc[data['Age'].isna(), 'Age'] = data['Age'].median()\ndata[data['Age'].isna()]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.394287Z","iopub.execute_input":"2022-07-30T10:21:19.394665Z","iopub.status.idle":"2022-07-30T10:21:19.410691Z","shell.execute_reply.started":"2022-07-30T10:21:19.394632Z","shell.execute_reply":"2022-07-30T10:21:19.409808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remove outliers","metadata":{}},{"cell_type":"code","source":"norm_data = data[\n    (data['Fare'] < 300) &\n    (data['Age'] < 70) &\n    (data['SibSp'] < 2) &\n    (data['Parch'] < 3)\n]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.412743Z","iopub.execute_input":"2022-07-30T10:21:19.413834Z","iopub.status.idle":"2022-07-30T10:21:19.421646Z","shell.execute_reply.started":"2022-07-30T10:21:19.413789Z","shell.execute_reply":"2022-07-30T10:21:19.420387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Handling categorical columns","metadata":{}},{"cell_type":"code","source":"def ReverseClass(df):\n    def reverse_class(x):\n        if x == 3:\n            return 1\n        if x == 1:\n            return 3\n        return x\n\n    df['Pclass'] = df['Pclass'].apply(reverse_class)\n    return df\nnorm_data = ReverseClass(norm_data)\nnorm_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.423533Z","iopub.execute_input":"2022-07-30T10:21:19.424681Z","iopub.status.idle":"2022-07-30T10:21:19.458871Z","shell.execute_reply.started":"2022-07-30T10:21:19.424627Z","shell.execute_reply":"2022-07-30T10:21:19.457681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`Name` and `Ticket` are pretty useless. Even if we apply feature engineering for them, it's really hard to fit into model. I decide to remove them.","metadata":{}},{"cell_type":"code","source":"def RemoveColumns(df):\n    return df.loc[:, ~df.columns.isin(['Name', 'Ticket'])]\nnorm_data = RemoveColumns(norm_data)\nnorm_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.460441Z","iopub.execute_input":"2022-07-30T10:21:19.461793Z","iopub.status.idle":"2022-07-30T10:21:19.487580Z","shell.execute_reply.started":"2022-07-30T10:21:19.461738Z","shell.execute_reply":"2022-07-30T10:21:19.485976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabin Featuring","metadata":{}},{"cell_type":"code","source":"def CabinFeature(df):\n    df.loc[:, 'Cabin'] = df['Cabin'].str[0]\n    cabin_map = {'G': 0, 'F': 1, 'E': 2, 'D': 3, 'C': 4, 'B': 5, 'A': 6}\n    df.loc[:, 'Cabin'] = df['Cabin'].apply(lambda x: cabin_map.get(x, x))\n    return df\n\nnorm_data = CabinFeature(norm_data)\nnorm_data['Cabin'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.489568Z","iopub.execute_input":"2022-07-30T10:21:19.490095Z","iopub.status.idle":"2022-07-30T10:21:19.507286Z","shell.execute_reply.started":"2022-07-30T10:21:19.490047Z","shell.execute_reply":"2022-07-30T10:21:19.506039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[data['Cabin'].str[0] == 'T']","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.508847Z","iopub.execute_input":"2022-07-30T10:21:19.509245Z","iopub.status.idle":"2022-07-30T10:21:19.529128Z","shell.execute_reply.started":"2022-07-30T10:21:19.509209Z","shell.execute_reply":"2022-07-30T10:21:19.527849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cabin \"T\" is too few, we will consider it an outlier to remove","metadata":{}},{"cell_type":"code","source":"norm_data = norm_data[norm_data['Cabin'] != 'T']\ndata = data[data['Cabin'] != 'T']\nnorm_data['Cabin'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.530402Z","iopub.execute_input":"2022-07-30T10:21:19.530776Z","iopub.status.idle":"2022-07-30T10:21:19.546228Z","shell.execute_reply.started":"2022-07-30T10:21:19.530740Z","shell.execute_reply":"2022-07-30T10:21:19.544876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we need to fill NaN of `Cabin`. Looks like it relate strongly to the group `Pclass`.","metadata":{}},{"cell_type":"markdown","source":"Let's try fill the `Cabin`'s mode value of each `Pclass` group to all these NaN rows then...","metadata":{}},{"cell_type":"code","source":"def FillCabin(df):\n    group_by_class = df.groupby(['Pclass'])\n    def fill_na(df):\n        df.loc[df['Cabin'].isna(), 'Cabin'] = df['Cabin'].mode().values[0]\n        return df\n\n    df = group_by_class.apply(fill_na)\n    return df\n    \nnorm_data = FillCabin(norm_data)\nnorm_data['Cabin'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.551207Z","iopub.execute_input":"2022-07-30T10:21:19.551838Z","iopub.status.idle":"2022-07-30T10:21:19.575741Z","shell.execute_reply.started":"2022-07-30T10:21:19.551796Z","shell.execute_reply":"2022-07-30T10:21:19.574785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"norm_data['Embarked'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.577177Z","iopub.execute_input":"2022-07-30T10:21:19.577544Z","iopub.status.idle":"2022-07-30T10:21:19.586178Z","shell.execute_reply.started":"2022-07-30T10:21:19.577510Z","shell.execute_reply":"2022-07-30T10:21:19.585239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group = norm_data.groupby('Pclass')\nfig, ax = plt.subplots(1, 3, figsize=(16, 10))\n\nfor name, df in group:\n    print(name)\n    ax[name - 1].set_title(f'Group tier {df[\"Pclass\"].values[0]}')\n    ax[name - 1].pie(df['Embarked'].value_counts(), labels = df['Embarked'].value_counts().index)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.587216Z","iopub.execute_input":"2022-07-30T10:21:19.587606Z","iopub.status.idle":"2022-07-30T10:21:19.861003Z","shell.execute_reply.started":"2022-07-30T10:21:19.587549Z","shell.execute_reply":"2022-07-30T10:21:19.859573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def OneHot(df, to_do):\n    dummies = pd.get_dummies(df[to_do])\n    df = pd.concat([df,dummies], axis=1).drop(to_do, axis=1)\n    return df\n\nnorm_data = OneHot(norm_data, ['Sex', 'Embarked'])\nnorm_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.863133Z","iopub.execute_input":"2022-07-30T10:21:19.864046Z","iopub.status.idle":"2022-07-30T10:21:19.904540Z","shell.execute_reply.started":"2022-07-30T10:21:19.863993Z","shell.execute_reply":"2022-07-30T10:21:19.903674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def SimpleImputer(df, to_do):\n    for i in to_do:\n        df.loc[df[i].isna(), i] = df[i].median()\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.905911Z","iopub.execute_input":"2022-07-30T10:21:19.906424Z","iopub.status.idle":"2022-07-30T10:21:19.911272Z","shell.execute_reply.started":"2022-07-30T10:21:19.906389Z","shell.execute_reply":"2022-07-30T10:21:19.910359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Pipeline(df):\n    df = SimpleImputer(df, ['Age', 'Fare'])\n    df = ReverseClass(df)\n    df = RemoveColumns(df)\n    df = CabinFeature(df)\n    df = FillCabin(df)\n    encoder = OneHotEncoder(categories=['Sex', 'Embarked'])\n    df = OneHot(df, ['Sex', 'Embarked'])\n    \n    return df\n\nPipeline(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.912388Z","iopub.execute_input":"2022-07-30T10:21:19.913410Z","iopub.status.idle":"2022-07-30T10:21:19.963329Z","shell.execute_reply.started":"2022-07-30T10:21:19.913369Z","shell.execute_reply":"2022-07-30T10:21:19.962564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = Pipeline(data)\nX = train_data.loc[:, train_data.columns != 'Survived']\ny = train_data[['Survived']]\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.964406Z","iopub.execute_input":"2022-07-30T10:21:19.965389Z","iopub.status.idle":"2022-07-30T10:21:19.995423Z","shell.execute_reply.started":"2022-07-30T10:21:19.965352Z","shell.execute_reply":"2022-07-30T10:21:19.993910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.tree import plot_tree\nclf = DecisionTreeClassifier(\n    criterion='entropy',\n    splitter='best',\n    max_depth=20,\n    max_leaf_nodes=40,\n    max_features='sqrt'\n)\n\nclf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:19.996606Z","iopub.execute_input":"2022-07-30T10:21:19.997409Z","iopub.status.idle":"2022-07-30T10:21:20.010158Z","shell.execute_reply.started":"2022-07-30T10:21:19.997373Z","shell.execute_reply":"2022-07-30T10:21:20.008976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.score(X_train, y_train), clf.score(X_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:20.011506Z","iopub.execute_input":"2022-07-30T10:21:20.012146Z","iopub.status.idle":"2022-07-30T10:21:20.026666Z","shell.execute_reply.started":"2022-07-30T10:21:20.012109Z","shell.execute_reply":"2022-07-30T10:21:20.025820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/titanic/test.csv')\ntest_data = Pipeline(test)\nX_test = test_data.loc[:, test_data.columns != 'PassengerId']\ny_pred = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:20.028060Z","iopub.execute_input":"2022-07-30T10:21:20.028617Z","iopub.status.idle":"2022-07-30T10:21:20.063830Z","shell.execute_reply.started":"2022-07-30T10:21:20.028572Z","shell.execute_reply":"2022-07-30T10:21:20.062613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Survived'] = y_pred\ntest_data","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:20.065365Z","iopub.execute_input":"2022-07-30T10:21:20.066965Z","iopub.status.idle":"2022-07-30T10:21:20.091319Z","shell.execute_reply.started":"2022-07-30T10:21:20.066915Z","shell.execute_reply":"2022-07-30T10:21:20.090194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data[['PassengerId', 'Survived']]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:21:20.092806Z","iopub.execute_input":"2022-07-30T10:21:20.093164Z","iopub.status.idle":"2022-07-30T10:21:20.099814Z","shell.execute_reply.started":"2022-07-30T10:21:20.093130Z","shell.execute_reply":"2022-07-30T10:21:20.098791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T10:24:11.032471Z","iopub.execute_input":"2022-07-30T10:24:11.032923Z","iopub.status.idle":"2022-07-30T10:24:11.040847Z","shell.execute_reply.started":"2022-07-30T10:24:11.032886Z","shell.execute_reply":"2022-07-30T10:24:11.039497Z"},"trusted":true},"execution_count":null,"outputs":[]}]}