{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T16:55:58.950394Z","iopub.execute_input":"2022-07-05T16:55:58.950858Z","iopub.status.idle":"2022-07-05T16:55:58.987389Z","shell.execute_reply.started":"2022-07-05T16:55:58.950772Z","shell.execute_reply":"2022-07-05T16:55:58.986432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing required libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom xgboost import XGBClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:58:27.227737Z","iopub.execute_input":"2022-07-05T16:58:27.228160Z","iopub.status.idle":"2022-07-05T16:58:27.234169Z","shell.execute_reply.started":"2022-07-05T16:58:27.228127Z","shell.execute_reply":"2022-07-05T16:58:27.233178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Approach\n1. Load data\n2. Select useful features\n3. Extract numerical and categorical columns\n4. Make pipeline\n5. Do Hyperparameter tuning with cross-validation\n6. Check MAE\n7. Find best parameters\n8. Train model\n9. Predicting `survived` on `test_data`\n10. Submit predictions","metadata":{}},{"cell_type":"markdown","source":"## 1. Loading data","metadata":{}},{"cell_type":"code","source":"titanic_data = pd.read_csv(\"../input/titanic/train.csv\", index_col=\"PassengerId\")\ntest_data = pd.read_csv(\"../input/titanic/test.csv\", index_col=\"PassengerId\")\n\ntitanic_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:45.454217Z","iopub.execute_input":"2022-07-05T17:40:45.454628Z","iopub.status.idle":"2022-07-05T17:40:45.482309Z","shell.execute_reply.started":"2022-07-05T17:40:45.454593Z","shell.execute_reply":"2022-07-05T17:40:45.481588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Selecting useful features","metadata":{}},{"cell_type":"code","source":"useful_features = [\"Pclass\", \"Name\", \"Sex\", \"Age\", \"SibSp\", \"Parch\", \"Ticket\", \"Fare\", \"Embarked\"]\nX = titanic_data[useful_features]\nY = titanic_data[\"Survived\"]\ntest_data = test_data[useful_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:45.848398Z","iopub.execute_input":"2022-07-05T17:40:45.849150Z","iopub.status.idle":"2022-07-05T17:40:45.860402Z","shell.execute_reply.started":"2022-07-05T17:40:45.849101Z","shell.execute_reply":"2022-07-05T17:40:45.859485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Extracting numerical and categorical columns","metadata":{}},{"cell_type":"code","source":"num_cols = X.select_dtypes(exclude=\"object\").columns\ncat_cols = X.select_dtypes(\"object\").columns\nnum_cols, cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:47.179073Z","iopub.execute_input":"2022-07-05T17:40:47.179697Z","iopub.status.idle":"2022-07-05T17:40:47.192233Z","shell.execute_reply.started":"2022-07-05T17:40:47.179663Z","shell.execute_reply":"2022-07-05T17:40:47.191445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Making pipeline","metadata":{}},{"cell_type":"code","source":"class CreatePipeline:\n    \"\"\"Create Pipeline\n    methods:\n        pipeline: Create Final Pipeline\n        \n        create_model: Create the provided model\n        \n        numerical_transformer: Transform numerical cols\n        \n        categorical_transformer: Transform categorical cols \\\n        OneHotEncoding / OrdinalEncoding\n        \n        data_preprocessor: Preprocess the data using ColumnTransformer     \n        \"\"\"\n    \n    def pipeline(self, *, preprocessor, model, verbose=False):\n        \"\"\"Creates pipeline\n        params:\n            preprocessor\n            model\n        \"\"\"\n        steps = [(\"preprocessor\", preprocessor),\n                 (\"model\", model)]\n        return Pipeline(steps=steps, verbose=verbose)\n    \n    \n    def numerical_transformer(self, *, strategy=\"mean\", **params):\n        \"\"\"Transform numerical columns using `SimpleImputer`.\n        params:\n            strategy: \"mean\" | \"median\" | \"most_frequent\" | \"constant\"\n            **params: extra keyword args for SimpleImputer\"\"\"\n        \n        transformer = SimpleImputer(strategy=strategy, **params)\n        return transformer\n\n    \n    def categorical_transformer(self, *, \n                                imp_strategy=\"most_frequent\", \n                                encoder_type=\"Ordinal\", \n                                imp_params={}, encoder_params={}):\n        \"\"\"Transform categorical columns by making Pipeline\n        `SimpleImputer` | `OneHotEncoder` | `OrdinalEncoder`.\n        args:\n            imp_strategy: strategy for imputer values can be\n                \"most_frequent\" | \"constant\"\n            encoder_type: encoder type,\n                \"Ordinal\" | \"OneHot\"\n        kwargs:\n            imp_params: keyword args for `SimpleImputer`.\n            encoder_params: keyword args for encoder.`\n        \"\"\"\n        if not encoder_type in (\"Ordinal\", \"OneHot\"):\n            raise ValueError(f\"Inappropriate value for encoder_type passed: {encoder_type}\\\n            Takes one of 'Ordinal' | 'OneHot'.\")\n        \n        encoder = OrdinalEncoder if encoder_type==\"Ordinal\" else OneHotEncoder\n        transformer = Pipeline(steps=[\n            (\"imputer\", SimpleImputer(strategy=imp_strategy, **imp_params)),\n            (encoder_type, encoder(**encoder_params))\n        ])\n        return transformer\n    \n    \n    def data_preprocessor(self, *, transformers):\n        \"\"\"Preprocess the data using `ColumnTransformer`.\n        Pass extact list of transformers\n        to be passed in `ColumnTransformer`.\n        each tuple consist of: (transformer_name,\n                                transformer,\n                                list_of_columns).\"\"\"\n        preprocessor = ColumnTransformer(transformers=transformers)\n        return preprocessor\n    \n    \n    def create_model(self, *, model, random_state=0, n_estimators=1000, **kwargs):\n        \"\"\"Creates the model.\n        **kwargs: keyword args for model.\"\"\"\n        my_model = model(random_state=random_state, n_estimators=n_estimators, **kwargs)\n        return my_model","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:47.603313Z","iopub.execute_input":"2022-07-05T17:40:47.604027Z","iopub.status.idle":"2022-07-05T17:40:47.618461Z","shell.execute_reply.started":"2022-07-05T17:40:47.603990Z","shell.execute_reply":"2022-07-05T17:40:47.616751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp = CreatePipeline()\nnum_transformer = cp.numerical_transformer()\ncat_transformer = cp.categorical_transformer(encoder_type=\"OneHot\", encoder_params={\"handle_unknown\": \"ignore\"})\npreprocessor = cp.data_preprocessor(\n                    transformers=[(\"num\", num_transformer, num_cols),\n                                  (\"cat\", cat_transformer, cat_cols)\n                                 ])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:47.802117Z","iopub.execute_input":"2022-07-05T17:40:47.802786Z","iopub.status.idle":"2022-07-05T17:40:47.809036Z","shell.execute_reply.started":"2022-07-05T17:40:47.802743Z","shell.execute_reply":"2022-07-05T17:40:47.807940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Doing hyperparameter tuning with Cross-validation using `XGBClassifier`","metadata":{}},{"cell_type":"code","source":"n_estimators = [350, 500, 750]\nmax_depths = [5, 10, 20]\nlearning_rate = [0.05, 0.1]\nmaes = {}\ni = 0\nfor n in n_estimators:\n    for md in max_depths:\n        for rate in learning_rate:\n            i += 1\n            model = cp.create_model(model=XGBClassifier, n_estimators=n, max_depth=md, learning_rate=rate)\n            pipeline = cp.pipeline(preprocessor=preprocessor, model=model)\n            scores = -1 * cross_val_score(pipeline, X, Y, cv=10, verbose=True,\n                                    scoring=\"neg_mean_absolute_error\")\n            mae = scores.mean()\n            maes[i] = [n, md, rate, mae]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:40:49.092152Z","iopub.execute_input":"2022-07-05T17:40:49.093144Z","iopub.status.idle":"2022-07-05T17:48:57.215791Z","shell.execute_reply.started":"2022-07-05T17:40:49.093085Z","shell.execute_reply":"2022-07-05T17:48:57.214704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Checking MAE","metadata":{}},{"cell_type":"code","source":"for i in maes:\n    n, md, rate, mae = maes[i]\n    print(f\"{i}.\\tN_estimators: {n}\\tmax_depth: {md}\\tlearning_rate: {rate}\\tMAE: {mae}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:48:57.217450Z","iopub.execute_input":"2022-07-05T17:48:57.218420Z","iopub.status.idle":"2022-07-05T17:48:57.225866Z","shell.execute_reply.started":"2022-07-05T17:48:57.218380Z","shell.execute_reply":"2022-07-05T17:48:57.224715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Best parameters\n#### `n_estimators: 350`\n#### `max_depth: 5`\n#### `learning_rate: 0.05`\n#### `MAE: 0.1615730337078652`","metadata":{}},{"cell_type":"code","source":"best_n_estimators = 350\nbest_max_depth = 5\nbest_rate = 0.05","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:50:55.652279Z","iopub.execute_input":"2022-07-05T17:50:55.653209Z","iopub.status.idle":"2022-07-05T17:50:55.657374Z","shell.execute_reply.started":"2022-07-05T17:50:55.653167Z","shell.execute_reply":"2022-07-05T17:50:55.656580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Training `XGBClassifier` model with best parameters","metadata":{}},{"cell_type":"code","source":"model = cp.create_model(model=XGBClassifier, n_estimators=best_n_estimators, learning_rate=best_rate, max_depth=best_max_depth)\npipeline = cp.pipeline(preprocessor=preprocessor, model=model)\npipeline.fit(X, Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:50:57.064632Z","iopub.execute_input":"2022-07-05T17:50:57.065396Z","iopub.status.idle":"2022-07-05T17:50:58.635698Z","shell.execute_reply.started":"2022-07-05T17:50:57.065351Z","shell.execute_reply":"2022-07-05T17:50:58.634603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. Predicting `Survived` on `test_data`","metadata":{}},{"cell_type":"code","source":"test_preds = pipeline.predict(test_data)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:51:03.121275Z","iopub.execute_input":"2022-07-05T17:51:03.122016Z","iopub.status.idle":"2022-07-05T17:51:03.187775Z","shell.execute_reply.started":"2022-07-05T17:51:03.121979Z","shell.execute_reply":"2022-07-05T17:51:03.186450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 11. Submitting predictions","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": test_data.index, \"Survived\": test_preds})\noutput.to_csv(\"./submission_3.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T17:51:06.513247Z","iopub.execute_input":"2022-07-05T17:51:06.513721Z","iopub.status.idle":"2022-07-05T17:51:06.523897Z","shell.execute_reply.started":"2022-07-05T17:51:06.513685Z","shell.execute_reply":"2022-07-05T17:51:06.522698Z"},"trusted":true},"execution_count":null,"outputs":[]}]}