{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T18:11:41.322903Z","iopub.execute_input":"2022-07-06T18:11:41.323329Z","iopub.status.idle":"2022-07-06T18:11:41.360961Z","shell.execute_reply.started":"2022-07-06T18:11:41.323240Z","shell.execute_reply":"2022-07-06T18:11:41.359384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing required libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom xgboost import XGBClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.feature_selection import mutual_info_classif\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:38:32.382782Z","iopub.execute_input":"2022-07-06T18:38:32.383207Z","iopub.status.idle":"2022-07-06T18:38:32.393365Z","shell.execute_reply.started":"2022-07-06T18:38:32.383173Z","shell.execute_reply":"2022-07-06T18:38:32.391937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Approach\n1. Load data\n2. Replace outliers with mean/median\n3. Find Mutual Information\n3. Select Useful features\n4. Extract numerical and categorical columns\n5. Make pipeline\n6. Do hyperparameter tuning with cross-validation (`XGBRegressor`)\n7. Check MAE\n8. Best parameters\n8. Train model\n9. Predict `Survived` on `test_data`\n10. Submit predictions","metadata":{}},{"cell_type":"markdown","source":"## 1. Loading data","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"../input/titanic/train.csv\", index_col=\"PassengerId\")\ntest_data = pd.read_csv(\"../input/titanic/test.csv\", index_col=\"PassengerId\")\n\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:11:43.044892Z","iopub.execute_input":"2022-07-06T18:11:43.045337Z","iopub.status.idle":"2022-07-06T18:11:43.107189Z","shell.execute_reply.started":"2022-07-06T18:11:43.045294Z","shell.execute_reply":"2022-07-06T18:11:43.105939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Replacing outliers with median","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.scatterplot(x=\"Fare\", y=data.index, hue=\"Survived\", data=data)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:17:23.352388Z","iopub.execute_input":"2022-07-06T18:17:23.353720Z","iopub.status.idle":"2022-07-06T18:17:23.653232Z","shell.execute_reply.started":"2022-07-06T18:17:23.353675Z","shell.execute_reply":"2022-07-06T18:17:23.651811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outliers = data[data[\"Fare\"] > 200].index\ndata.loc[outliers, \"Fare\"] = data[\"Fare\"].median()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:21:01.666506Z","iopub.execute_input":"2022-07-06T18:21:01.666954Z","iopub.status.idle":"2022-07-06T18:21:01.676379Z","shell.execute_reply.started":"2022-07-06T18:21:01.666901Z","shell.execute_reply":"2022-07-06T18:21:01.675324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.scatterplot(x=\"Fare\", y=data.index, hue=\"Survived\", data=data)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:21:02.804458Z","iopub.execute_input":"2022-07-06T18:21:02.804879Z","iopub.status.idle":"2022-07-06T18:21:03.117142Z","shell.execute_reply.started":"2022-07-06T18:21:02.804846Z","shell.execute_reply":"2022-07-06T18:21:03.115870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Finding Mutual Information","metadata":{}},{"cell_type":"code","source":"def impute_data(data, *, num_strategy=\"mean\", cat_strategy=\"most_frequent\"):\n    data = data.copy()\n    index = data.index\n    num_cols = data.select_dtypes(exclude=\"object\").columns\n    cat_cols = data.select_dtypes(\"object\").columns\n    \n    num_imputer = SimpleImputer(strategy=num_strategy)\n    cat_imputer = SimpleImputer(strategy=cat_strategy)\n    \n    data[num_cols] = pd.DataFrame(num_imputer.fit_transform(data[num_cols]), index=index, columns=num_cols)\n    data[cat_cols] = pd.DataFrame(cat_imputer.fit_transform(data[cat_cols]), index=index, columns=cat_cols)\n    return data","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:22:09.045721Z","iopub.execute_input":"2022-07-06T18:22:09.046132Z","iopub.status.idle":"2022-07-06T18:22:09.054766Z","shell.execute_reply.started":"2022-07-06T18:22:09.046099Z","shell.execute_reply":"2022-07-06T18:22:09.053499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_mi_score(X, y):\n    X = impute_data(X.copy(), num_strategy=\"median\")\n    # Converting values of discrete features to numerical values\n    for col in X.select_dtypes([\"object\"]):\n        X[col] = X[col].factorize()[0]\n    \n    discrete_features = X.dtypes == int\n    mi_scores = mutual_info_classif(X, y, discrete_features=discrete_features, random_state=0)\n    mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=X.columns).sort_values(ascending=False)\n    return mi_scores","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:36:24.330853Z","iopub.execute_input":"2022-07-06T18:36:24.331613Z","iopub.status.idle":"2022-07-06T18:36:24.338996Z","shell.execute_reply.started":"2022-07-06T18:36:24.331576Z","shell.execute_reply":"2022-07-06T18:36:24.337667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data.copy()\nY = X.pop(\"Survived\")\nmi_scores = get_mi_score(X, Y)\nmi_scores","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:36:48.525669Z","iopub.execute_input":"2022-07-06T18:36:48.526056Z","iopub.status.idle":"2022-07-06T18:36:48.590088Z","shell.execute_reply.started":"2022-07-06T18:36:48.526019Z","shell.execute_reply":"2022-07-06T18:36:48.588861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Selecting useful features","metadata":{}},{"cell_type":"code","source":"useful_features = mi_scores[mi_scores > 0.05].index\nX = X[useful_features]\ntest_data = test_data[useful_features]\nuseful_features","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:37:05.598128Z","iopub.execute_input":"2022-07-06T18:37:05.598493Z","iopub.status.idle":"2022-07-06T18:37:05.609948Z","shell.execute_reply.started":"2022-07-06T18:37:05.598464Z","shell.execute_reply":"2022-07-06T18:37:05.608748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Extracting numerical and categorical columns","metadata":{}},{"cell_type":"code","source":"num_cols = X.select_dtypes(exclude=\"object\").columns\ncat_cols = X.select_dtypes(\"object\").columns\nnum_cols, cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:37:06.119340Z","iopub.execute_input":"2022-07-06T18:37:06.120576Z","iopub.status.idle":"2022-07-06T18:37:06.130895Z","shell.execute_reply.started":"2022-07-06T18:37:06.120509Z","shell.execute_reply":"2022-07-06T18:37:06.129607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Making pipeline","metadata":{}},{"cell_type":"code","source":"class CreatePipeline:\n    \"\"\"Create Pipeline\n    methods:\n        pipeline: Create Final Pipeline\n        \n        create_model: Create the provided model\n        \n        numerical_transformer: Transform numerical cols\n        \n        categorical_transformer: Transform categorical cols \\\n        OneHotEncoding / OrdinalEncoding\n        \n        data_preprocessor: Preprocess the data using ColumnTransformer     \n        \"\"\"\n    \n    def pipeline(self, *, preprocessor, model, verbose=False):\n        \"\"\"Creates pipeline\n        params:\n            preprocessor\n            model\n        \"\"\"\n        steps = [(\"preprocessor\", preprocessor),\n                 (\"model\", model)]\n        return Pipeline(steps=steps, verbose=verbose)\n    \n    \n    def numerical_transformer(self, *, strategy=\"mean\", **params):\n        \"\"\"Transform numerical columns using `SimpleImputer`.\n        params:\n            strategy: \"mean\" | \"median\" | \"most_frequent\" | \"constant\"\n            **params: extra keyword args for SimpleImputer\"\"\"\n        \n        transformer = SimpleImputer(strategy=strategy, **params)\n        return transformer\n\n    \n    def categorical_transformer(self, *, \n                                imp_strategy=\"most_frequent\", \n                                encoder_type=\"Ordinal\", \n                                imp_params={}, encoder_params={}):\n        \"\"\"Transform categorical columns by making Pipeline\n        `SimpleImputer` | `OneHotEncoder` | `OrdinalEncoder`.\n        args:\n            imp_strategy: strategy for imputer values can be\n                \"most_frequent\" | \"constant\"\n            encoder_type: encoder type,\n                \"Ordinal\" | \"OneHot\"\n        kwargs:\n            imp_params: keyword args for `SimpleImputer`.\n            encoder_params: keyword args for encoder.`\n        \"\"\"\n        if not encoder_type in (\"Ordinal\", \"OneHot\"):\n            raise ValueError(f\"Inappropriate value for encoder_type passed: {encoder_type}\\\n            Takes one of 'Ordinal' | 'OneHot'.\")\n        \n        encoder = OrdinalEncoder if encoder_type==\"Ordinal\" else OneHotEncoder\n        transformer = Pipeline(steps=[\n            (\"imputer\", SimpleImputer(strategy=imp_strategy, **imp_params)),\n            (encoder_type, encoder(**encoder_params))\n        ])\n        return transformer\n    \n    \n    def data_preprocessor(self, *, transformers):\n        \"\"\"Preprocess the data using `ColumnTransformer`.\n        Pass extact list of transformers\n        to be passed in `ColumnTransformer`.\n        each tuple consist of: (transformer_name,\n                                transformer,\n                                list_of_columns).\"\"\"\n        preprocessor = ColumnTransformer(transformers=transformers)\n        return preprocessor\n    \n    \n    def create_model(self, *, model, random_state=0, n_estimators=1000, **kwargs):\n        \"\"\"Creates the model.\n        **kwargs: keyword args for model.\"\"\"\n        my_model = model(random_state=random_state, n_estimators=n_estimators, **kwargs)\n        return my_model","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:37:07.750109Z","iopub.execute_input":"2022-07-06T18:37:07.750839Z","iopub.status.idle":"2022-07-06T18:37:07.764142Z","shell.execute_reply.started":"2022-07-06T18:37:07.750804Z","shell.execute_reply":"2022-07-06T18:37:07.762705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp = CreatePipeline()\nnum_transformer = cp.numerical_transformer(strategy=\"median\")\ncat_transformer = cp.categorical_transformer(encoder_type=\"OneHot\", encoder_params={\"handle_unknown\": \"ignore\"})\npreprocessor = cp.data_preprocessor(\n                    transformers=[(\"num\", num_transformer, num_cols),\n                                  (\"cat\", cat_transformer, cat_cols)\n                                 ])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:37:40.366916Z","iopub.execute_input":"2022-07-06T18:37:40.368047Z","iopub.status.idle":"2022-07-06T18:37:40.374212Z","shell.execute_reply.started":"2022-07-06T18:37:40.368007Z","shell.execute_reply":"2022-07-06T18:37:40.372946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Doing hyperparameter tuning with cross-validation (`XGBRegressor`)","metadata":{}},{"cell_type":"code","source":"n_estimators = [500, 750, 1000]\nmax_depths = [5, 10]\nlearning_rate = [0.05, 0.1]\nmaes = {}\ni = 0\nfor n in n_estimators:\n    for md in max_depths:\n        for rate in learning_rate:\n            i += 1\n            model = cp.create_model(model=XGBClassifier, n_estimators=n, max_depth=md, learning_rate=rate)\n            pipeline = cp.pipeline(preprocessor=preprocessor, model=model)\n            scores = -1 * cross_val_score(pipeline, X, Y, cv=10, verbose=True,\n                                    scoring=\"neg_mean_absolute_error\")\n            mae = scores.mean()\n            maes[i] = [n, md, rate, mae]","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:48:42.808475Z","iopub.execute_input":"2022-07-06T18:48:42.809067Z","iopub.status.idle":"2022-07-06T18:55:44.681038Z","shell.execute_reply.started":"2022-07-06T18:48:42.809034Z","shell.execute_reply":"2022-07-06T18:55:44.680009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Checking MAE","metadata":{}},{"cell_type":"code","source":"for i in maes:\n    n, md, rate, mae = maes[i]\n    print(f\"{i}.\\tN_estimators: {n}\\tmax_depth: {md}\\tlearning_rate: {rate}\\tMAE: {mae}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:55:44.682651Z","iopub.execute_input":"2022-07-06T18:55:44.683331Z","iopub.status.idle":"2022-07-06T18:55:44.691578Z","shell.execute_reply.started":"2022-07-06T18:55:44.683293Z","shell.execute_reply":"2022-07-06T18:55:44.690345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(maes, key=lambda x: maes[x][3])","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:56:00.129714Z","iopub.execute_input":"2022-07-06T18:56:00.130139Z","iopub.status.idle":"2022-07-06T18:56:00.138469Z","shell.execute_reply.started":"2022-07-06T18:56:00.130103Z","shell.execute_reply":"2022-07-06T18:56:00.137226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Best parameters\n#### `n_estimators: 500`\n#### `max_depth: 5`\n#### `learning_rate: 0.05`","metadata":{}},{"cell_type":"code","source":"best_n_estimators = 500\nbest_max_depth = 5\nbest_rate = 0.05","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:56:32.478370Z","iopub.execute_input":"2022-07-06T18:56:32.478770Z","iopub.status.idle":"2022-07-06T18:56:32.484000Z","shell.execute_reply.started":"2022-07-06T18:56:32.478737Z","shell.execute_reply":"2022-07-06T18:56:32.482759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. Training model on best parameters","metadata":{}},{"cell_type":"code","source":"model = cp.create_model(model=XGBClassifier, n_estimators=best_n_estimators, learning_rate=best_rate, max_depth=best_max_depth)\npipeline = cp.pipeline(preprocessor=preprocessor, model=model)\npipeline.fit(X, Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:56:35.376126Z","iopub.execute_input":"2022-07-06T18:56:35.376505Z","iopub.status.idle":"2022-07-06T18:56:37.859468Z","shell.execute_reply.started":"2022-07-06T18:56:35.376474Z","shell.execute_reply":"2022-07-06T18:56:37.858665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 11. Predicting `Survived` on `test_data`","metadata":{}},{"cell_type":"code","source":"test_preds = pipeline.predict(test_data)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:56:47.562134Z","iopub.execute_input":"2022-07-06T18:56:47.562539Z","iopub.status.idle":"2022-07-06T18:56:47.597184Z","shell.execute_reply.started":"2022-07-06T18:56:47.562489Z","shell.execute_reply":"2022-07-06T18:56:47.595948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 12. Submitting predictions","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": test_data.index, \"Survived\": test_preds})\noutput.to_csv(\"./submission_3.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T18:56:50.945824Z","iopub.execute_input":"2022-07-06T18:56:50.946837Z","iopub.status.idle":"2022-07-06T18:56:50.958750Z","shell.execute_reply.started":"2022-07-06T18:56:50.946793Z","shell.execute_reply":"2022-07-06T18:56:50.957172Z"},"trusted":true},"execution_count":null,"outputs":[]}]}