{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T16:01:07.273559Z","iopub.execute_input":"2022-07-05T16:01:07.274362Z","iopub.status.idle":"2022-07-05T16:01:07.310109Z","shell.execute_reply.started":"2022-07-05T16:01:07.274253Z","shell.execute_reply":"2022-07-05T16:01:07.308737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing required libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:03:37.219145Z","iopub.execute_input":"2022-07-05T16:03:37.219596Z","iopub.status.idle":"2022-07-05T16:03:37.229292Z","shell.execute_reply.started":"2022-07-05T16:03:37.219559Z","shell.execute_reply":"2022-07-05T16:03:37.228001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Approach\n1. Load data\n2. Select useful features\n3. Extract numerical and categorical columns\n4. Make pipeline\n5. Do Hyperparameter tuning with cross-validation\n- a. One with Ordinal Encoding\n- b. One with OneHot Encoding, use best\n6. Check MAE\n7. Find best parameters\n8. Train model\n9. Predicting `survived` on `test_data`\n10. Submit predictions","metadata":{}},{"cell_type":"markdown","source":"## 1. Load data","metadata":{}},{"cell_type":"code","source":"titanic_data = pd.read_csv(\"../input/titanic/train.csv\", index_col=\"PassengerId\")\ntest_data = pd.read_csv(\"../input/titanic/test.csv\", index_col=\"PassengerId\")\n\ntitanic_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:06:01.292217Z","iopub.execute_input":"2022-07-05T16:06:01.292771Z","iopub.status.idle":"2022-07-05T16:06:01.358668Z","shell.execute_reply.started":"2022-07-05T16:06:01.292725Z","shell.execute_reply":"2022-07-05T16:06:01.357490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Selecting useful features","metadata":{}},{"cell_type":"code","source":"useful_features = [\"Pclass\", \"Sex\", \"Age\", \"SibSp\", \"Parch\", \"Ticket\", \"Fare\", \"Cabin\", \"Embarked\"]\nX = titanic_data[useful_features]\nY = titanic_data[\"Survived\"]\ntest_data = test_data[useful_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:09:01.609620Z","iopub.execute_input":"2022-07-05T16:09:01.610048Z","iopub.status.idle":"2022-07-05T16:09:01.621674Z","shell.execute_reply.started":"2022-07-05T16:09:01.610015Z","shell.execute_reply":"2022-07-05T16:09:01.620338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Extract numerical and categorical columns","metadata":{}},{"cell_type":"code","source":"num_cols = X.select_dtypes(exclude=\"object\").columns\ncat_cols = X.select_dtypes(\"object\").columns\nnum_cols, cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:10:04.346722Z","iopub.execute_input":"2022-07-05T16:10:04.347118Z","iopub.status.idle":"2022-07-05T16:10:04.361438Z","shell.execute_reply.started":"2022-07-05T16:10:04.347086Z","shell.execute_reply":"2022-07-05T16:10:04.360343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Make Pipeline","metadata":{}},{"cell_type":"code","source":"class CreatePipeline:\n    \"\"\"Create Pipeline\n    methods:\n        pipeline: Create Final Pipeline\n        \n        create_model: Create the provided model\n        \n        numerical_transformer: Transform numerical cols\n        \n        categorical_transformer: Transform categorical cols \\\n        OneHotEncoding / OrdinalEncoding\n        \n        data_preprocessor: Preprocess the data using ColumnTransformer     \n        \"\"\"\n    \n    def pipeline(self, *, preprocessor, model, verbose=False):\n        \"\"\"Creates pipeline\n        params:\n            preprocessor\n            model\n        \"\"\"\n        steps = [(\"preprocessor\", preprocessor),\n                 (\"model\", model)]\n        return Pipeline(steps=steps, verbose=verbose)\n    \n    \n    def numerical_transformer(self, *, strategy=\"mean\", **params):\n        \"\"\"Transform numerical columns using `SimpleImputer`.\n        params:\n            strategy: \"mean\" | \"median\" | \"most_frequent\" | \"constant\"\n            **params: extra keyword args for SimpleImputer\"\"\"\n        \n        transformer = SimpleImputer(strategy=strategy, **params)\n        return transformer\n\n    \n    def categorical_transformer(self, *, \n                                imp_strategy=\"most_frequent\", \n                                encoder_type=\"Ordinal\", \n                                imp_params={}, encoder_params={}):\n        \"\"\"Transform categorical columns by making Pipeline\n        `SimpleImputer` | `OneHotEncoder` | `OrdinalEncoder`.\n        args:\n            imp_strategy: strategy for imputer values can be\n                \"most_frequent\" | \"constant\"\n            encoder_type: encoder type,\n                \"Ordinal\" | \"OneHot\"\n        kwargs:\n            imp_params: keyword args for `SimpleImputer`.\n            encoder_params: keyword args for encoder.`\n        \"\"\"\n        if not encoder_type in (\"Ordinal\", \"OneHot\"):\n            raise ValueError(f\"Inappropriate value for encoder_type passed: {encoder_type}\\\n            Takes one of 'Ordinal' | 'OneHot'.\")\n        \n        encoder = OrdinalEncoder if encoder_type==\"Ordinal\" else OneHotEncoder\n        transformer = Pipeline(steps=[\n            (\"imputer\", SimpleImputer(strategy=imp_strategy, **imp_params)),\n            (encoder_type, encoder(**encoder_params))\n        ])\n        return transformer\n    \n    \n    def data_preprocessor(self, *, transformers):\n        \"\"\"Preprocess the data using `ColumnTransformer`.\n        Pass extact list of transformers\n        to be passed in `ColumnTransformer`.\n        each tuple consist of: (transformer_name,\n                                transformer,\n                                list_of_columns).\"\"\"\n        preprocessor = ColumnTransformer(transformers=transformers)\n        return preprocessor\n    \n    \n    def create_model(self, *, model, random_state=0, n_estimators=1000, **kwargs):\n        \"\"\"Creates the model.\n        **kwargs: keyword args for model.\"\"\"\n        my_model = model(random_state=random_state, n_estimators=n_estimators, **kwargs)\n        return my_model","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:11:00.803401Z","iopub.execute_input":"2022-07-05T16:11:00.803792Z","iopub.status.idle":"2022-07-05T16:11:00.818988Z","shell.execute_reply.started":"2022-07-05T16:11:00.803761Z","shell.execute_reply":"2022-07-05T16:11:00.817877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp = CreatePipeline()\nnum_transformer = cp.numerical_transformer()\ncat_oh_transformer = cp.categorical_transformer(encoder_type=\"OneHot\", encoder_params={\"handle_unknown\": \"ignore\"})\ncat_ord_transformer = cp.categorical_transformer(encoder_type=\"Ordinal\", encoder_params={\"handle_unknown\":\"use_encoded_value\", \"unknown_value\":-1})\noh_preprocessor = cp.data_preprocessor(\n                    transformers=[(\"num\", num_transformer, num_cols),\n                                  (\"cat\", cat_oh_transformer, cat_cols)\n                                 ])\nord_preprocessor = cp.data_preprocessor(\n                    transformers=[(\"num\", num_transformer, num_cols),\n                                  (\"cat\", cat_ord_transformer, cat_cols)\n                                 ])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:33:59.838554Z","iopub.execute_input":"2022-07-05T16:33:59.838952Z","iopub.status.idle":"2022-07-05T16:33:59.846390Z","shell.execute_reply.started":"2022-07-05T16:33:59.838920Z","shell.execute_reply":"2022-07-05T16:33:59.845595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Hyperparamter tuning with Cross-validation\n### a. with `OneHot Encoding`","metadata":{}},{"cell_type":"code","source":"n_estimators = [100, 250, 500, 750]\nmax_depths = [5, 10, 20]\noh_maes = {}\ni = 0\nfor n in n_estimators:\n    for md in max_depths:\n        i += 1\n        model = cp.create_model(model=RandomForestClassifier, n_estimators=n, max_depth=md)\n        pipeline = cp.pipeline(preprocessor=oh_preprocessor, model=model)\n        scores = -1 * cross_val_score(pipeline, X, Y, cv=10, verbose=True,\n                                scoring=\"neg_mean_absolute_error\")\n        mae = scores.mean()\n        oh_maes[i] = [n, md, mae]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:34:01.732511Z","iopub.execute_input":"2022-07-05T16:34:01.732925Z","iopub.status.idle":"2022-07-05T16:35:48.628557Z","shell.execute_reply.started":"2022-07-05T16:34:01.732889Z","shell.execute_reply":"2022-07-05T16:35:48.627371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. with `Ordinal Encoding`","metadata":{}},{"cell_type":"code","source":"n_estimators = [100, 250, 500, 750]\nmax_depths = [5, 10, 20]\nord_maes = {}\ni = 0\nfor n in n_estimators:\n    for md in max_depths:\n        i += 1\n        model = cp.create_model(model=RandomForestClassifier, n_estimators=n, max_depth=md)\n        pipeline = cp.pipeline(preprocessor=ord_preprocessor, model=model)\n        scores = -1 * cross_val_score(pipeline, X, Y, cv=10, verbose=True,\n                                scoring=\"neg_mean_absolute_error\")\n        mae = scores.mean()\n        ord_maes[i] = [n, md, mae]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:35:48.630870Z","iopub.execute_input":"2022-07-05T16:35:48.631208Z","iopub.status.idle":"2022-07-05T16:37:43.039012Z","shell.execute_reply.started":"2022-07-05T16:35:48.631177Z","shell.execute_reply":"2022-07-05T16:37:43.037960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Checking MAEs\n### a. `OneHot Encoding`","metadata":{}},{"cell_type":"code","source":"for i in oh_maes:\n    n, md, mae = oh_maes[i]\n    print(f\"{i}.\\tN_estimators: {n}\\tmax_depth: {md}\\tMAE: {mae}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:37:43.040949Z","iopub.execute_input":"2022-07-05T16:37:43.041266Z","iopub.status.idle":"2022-07-05T16:37:43.047799Z","shell.execute_reply.started":"2022-07-05T16:37:43.041236Z","shell.execute_reply":"2022-07-05T16:37:43.046658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### b. `Ordinal Encoding`","metadata":{}},{"cell_type":"code","source":"for i in ord_maes:\n    n, md, mae = ord_maes[i]\n    print(f\"{i}.\\tN_estimators: {n}\\tmax_depth: {md}\\tMAE: {mae}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:37:43.049356Z","iopub.execute_input":"2022-07-05T16:37:43.049950Z","iopub.status.idle":"2022-07-05T16:37:43.062078Z","shell.execute_reply.started":"2022-07-05T16:37:43.049900Z","shell.execute_reply":"2022-07-05T16:37:43.061044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(oh_maes, key=lambda x: oh_maes[x][2])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:39:33.906063Z","iopub.execute_input":"2022-07-05T16:39:33.906491Z","iopub.status.idle":"2022-07-05T16:39:33.914606Z","shell.execute_reply.started":"2022-07-05T16:39:33.906456Z","shell.execute_reply":"2022-07-05T16:39:33.913336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min(ord_maes, key=lambda x: ord_maes[x][2])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:39:00.916216Z","iopub.execute_input":"2022-07-05T16:39:00.917043Z","iopub.status.idle":"2022-07-05T16:39:00.924100Z","shell.execute_reply.started":"2022-07-05T16:39:00.917002Z","shell.execute_reply":"2022-07-05T16:39:00.923345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Best parameters:\n- a. `Ordinal Encoding`\n- - `N_estimators: 250` & `max_depth:10` & `MAE:0.17280898876404496`\n- b. `One Hot Encoding`\n- - `N_estimators: 500` &\t`max_depth: 20` & `MAE: 0.1683395755305868`\n\n### `OneHotEncoding` did better","metadata":{}},{"cell_type":"code","source":"# using onehot encoding\nbest_n_estimators = 500\nbest_max_depth = 20","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:43:06.354108Z","iopub.execute_input":"2022-07-05T16:43:06.354802Z","iopub.status.idle":"2022-07-05T16:43:06.360789Z","shell.execute_reply.started":"2022-07-05T16:43:06.354755Z","shell.execute_reply":"2022-07-05T16:43:06.359485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Training model on best parameters with `OneHotEncoding` as preprocessor","metadata":{}},{"cell_type":"code","source":"model = cp.create_model(model=RandomForestClassifier, n_estimators=best_n_estimators, max_depth=best_max_depth)\npipeline = cp.pipeline(preprocessor=oh_preprocessor, model=model)\npipeline.fit(X, Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:44:48.964941Z","iopub.execute_input":"2022-07-05T16:44:48.965349Z","iopub.status.idle":"2022-07-05T16:44:50.199307Z","shell.execute_reply.started":"2022-07-05T16:44:48.965316Z","shell.execute_reply":"2022-07-05T16:44:50.198234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Predicting `survived` on `test_data`","metadata":{}},{"cell_type":"code","source":"test_preds = pipeline.predict(test_data)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:45:29.803687Z","iopub.execute_input":"2022-07-05T16:45:29.804093Z","iopub.status.idle":"2022-07-05T16:45:30.324990Z","shell.execute_reply.started":"2022-07-05T16:45:29.804059Z","shell.execute_reply":"2022-07-05T16:45:30.323781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": test_data.index, \"Survived\": test_preds})\noutput.to_csv(\"./submission_2.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T16:46:25.881872Z","iopub.execute_input":"2022-07-05T16:46:25.882392Z","iopub.status.idle":"2022-07-05T16:46:25.894234Z","shell.execute_reply.started":"2022-07-05T16:46:25.882346Z","shell.execute_reply":"2022-07-05T16:46:25.892952Z"},"trusted":true},"execution_count":null,"outputs":[]}]}