{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-04T20:40:35.981152Z","iopub.execute_input":"2022-07-04T20:40:35.981548Z","iopub.status.idle":"2022-07-04T20:40:35.992892Z","shell.execute_reply.started":"2022-07-04T20:40:35.981518Z","shell.execute_reply":"2022-07-04T20:40:35.992254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Importing required modules","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.092147Z","iopub.execute_input":"2022-07-04T20:40:36.093530Z","iopub.status.idle":"2022-07-04T20:40:36.102586Z","shell.execute_reply.started":"2022-07-04T20:40:36.093457Z","shell.execute_reply":"2022-07-04T20:40:36.101309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Approach\n1. Load data\n2. Find useful features\n3. select useful features\n4. Extract numerical and categorical columns\n5. Make pipeline\n6. Do Hyperparameter tuning with cross-validation\n7. Check MAE\n8. Find best parameters\n9. Train model\n10. Predict `survived` on `test_data`\n11. Submit predictions","metadata":{}},{"cell_type":"markdown","source":"## 1. Load data","metadata":{}},{"cell_type":"code","source":"titanic_data = pd.read_csv(\"../input/titanic/train.csv\", index_col=\"PassengerId\")\ntest_data = pd.read_csv(\"../input/titanic/test.csv\", index_col=\"PassengerId\")\n\npd.set_option(\"display.max_columns\", titanic_data.shape[1])\ntitanic_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.161669Z","iopub.execute_input":"2022-07-04T20:40:36.162920Z","iopub.status.idle":"2022-07-04T20:40:36.198400Z","shell.execute_reply.started":"2022-07-04T20:40:36.162861Z","shell.execute_reply":"2022-07-04T20:40:36.197513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Finding useful features","metadata":{}},{"cell_type":"code","source":"titanic_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.225918Z","iopub.execute_input":"2022-07-04T20:40:36.226528Z","iopub.status.idle":"2022-07-04T20:40:36.241266Z","shell.execute_reply.started":"2022-07-04T20:40:36.226497Z","shell.execute_reply":"2022-07-04T20:40:36.239153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.292536Z","iopub.execute_input":"2022-07-04T20:40:36.292855Z","iopub.status.idle":"2022-07-04T20:40:36.320058Z","shell.execute_reply.started":"2022-07-04T20:40:36.292830Z","shell.execute_reply":"2022-07-04T20:40:36.318700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.scatterplot(x=titanic_data.index, y=\"Age\", hue=\"Survived\", data=titanic_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.361035Z","iopub.execute_input":"2022-07-04T20:40:36.361936Z","iopub.status.idle":"2022-07-04T20:40:36.571324Z","shell.execute_reply.started":"2022-07-04T20:40:36.361908Z","shell.execute_reply":"2022-07-04T20:40:36.570221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.swarmplot(x=\"Sex\", y=titanic_data.index, hue=\"Survived\", data=titanic_data, dodge=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.573105Z","iopub.execute_input":"2022-07-04T20:40:36.573665Z","iopub.status.idle":"2022-07-04T20:40:36.933209Z","shell.execute_reply.started":"2022-07-04T20:40:36.573631Z","shell.execute_reply":"2022-07-04T20:40:36.932202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like, more males are die, and more females have survived!","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.swarmplot(x=\"Pclass\", y=titanic_data.index, hue=\"Survived\", data=titanic_data, dodge=True)\nplt.xticks([0, 1, 2], [\"Upper\", \"Middle\", \"Lower\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:36.934780Z","iopub.execute_input":"2022-07-04T20:40:36.935835Z","iopub.status.idle":"2022-07-04T20:40:37.312829Z","shell.execute_reply.started":"2022-07-04T20:40:36.935799Z","shell.execute_reply":"2022-07-04T20:40:37.311810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"More people of Lower class (3rd class) have died","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.swarmplot(x=\"Embarked\", y=titanic_data.index, hue=\"Survived\", data=titanic_data, dodge=True)\nplt.xticks([0, 1, 2], [\"Southampton\", \"Cherbourg\", \"Queenstown\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:37.315761Z","iopub.execute_input":"2022-07-04T20:40:37.316037Z","iopub.status.idle":"2022-07-04T20:40:37.704374Z","shell.execute_reply.started":"2022-07-04T20:40:37.316009Z","shell.execute_reply":"2022-07-04T20:40:37.703221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- S = Southampton\n- C = Cherbourg\n- Q = Queen\n\n- More People boarded from `Southampton` died\n- Less People die from rest both of embarked port\n### But I don't think embarkation port have something related with Survived or not (according to domain knowledge)","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(x=titanic_data[\"Fare\"], y=titanic_data.index, hue=\"Survived\", data=titanic_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:37.705918Z","iopub.execute_input":"2022-07-04T20:40:37.706399Z","iopub.status.idle":"2022-07-04T20:40:37.913278Z","shell.execute_reply.started":"2022-07-04T20:40:37.706363Z","shell.execute_reply":"2022-07-04T20:40:37.912099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Removing outliers","metadata":{}},{"cell_type":"code","source":"outliers = titanic_data[(titanic_data[\"Survived\"]==1) & (titanic_data[\"Fare\"] > 400)].index\ntitanic_data.drop(labels=outliers, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:37.914840Z","iopub.execute_input":"2022-07-04T20:40:37.915143Z","iopub.status.idle":"2022-07-04T20:40:37.923863Z","shell.execute_reply.started":"2022-07-04T20:40:37.915112Z","shell.execute_reply":"2022-07-04T20:40:37.922678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x=titanic_data[\"Fare\"], y=titanic_data.index, hue=\"Survived\", data=titanic_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:37.925457Z","iopub.execute_input":"2022-07-04T20:40:37.925795Z","iopub.status.idle":"2022-07-04T20:40:38.174761Z","shell.execute_reply.started":"2022-07-04T20:40:37.925765Z","shell.execute_reply":"2022-07-04T20:40:38.173538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Selecting useful features","metadata":{}},{"cell_type":"code","source":"useful_features = [\"Pclass\", \"Name\", \"Sex\", \"Age\", \"SibSp\", \"Parch\", \"Ticket\", \"Fare\", \"Embarked\"]\nX = titanic_data[useful_features]\nY = titanic_data[\"Survived\"]\ntest_data = test_data[useful_features]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:38.176418Z","iopub.execute_input":"2022-07-04T20:40:38.176817Z","iopub.status.idle":"2022-07-04T20:40:38.185570Z","shell.execute_reply.started":"2022-07-04T20:40:38.176786Z","shell.execute_reply":"2022-07-04T20:40:38.184203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Extracting numerical and categorical columns","metadata":{}},{"cell_type":"code","source":"num_cols = X.select_dtypes(exclude=\"object\").columns\ncat_cols = X.select_dtypes(\"object\").columns\nnum_cols, cat_cols","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:38.187826Z","iopub.execute_input":"2022-07-04T20:40:38.188724Z","iopub.status.idle":"2022-07-04T20:40:38.201824Z","shell.execute_reply.started":"2022-07-04T20:40:38.188682Z","shell.execute_reply":"2022-07-04T20:40:38.200390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Making Pipeline","metadata":{}},{"cell_type":"code","source":"class CreatePipeline:\n    \"\"\"Create Pipeline\n    methods:\n        pipeline: Create Final Pipeline\n        \n        create_model: Create the provided model\n        \n        numerical_transformer: Transform numerical cols\n        \n        categorical_transformer: Transform categorical cols \\\n        OneHotEncoding / OrdinalEncoding\n        \n        data_preprocessor: Preprocess the data using ColumnTransformer     \n        \"\"\"\n    \n    def pipeline(self, *, preprocessor, model, verbose=False):\n        \"\"\"Creates pipeline\n        params:\n            preprocessor\n            model\n        \"\"\"\n        steps = [(\"preprocessor\", preprocessor),\n                 (\"model\", model)]\n        return Pipeline(steps=steps, verbose=verbose)\n    \n    \n    def numerical_transformer(self, *, strategy=\"mean\", **params):\n        \"\"\"Transform numerical columns using `SimpleImputer`.\n        params:\n            strategy: \"mean\" | \"median\" | \"most_frequent\" | \"constant\"\n            **params: extra keyword args for SimpleImputer\"\"\"\n        \n        transformer = SimpleImputer(strategy=strategy, **params)\n        return transformer\n\n    \n    def categorical_transformer(self, *, \n                                imp_strategy=\"most_frequent\", \n                                encoder_type=\"Ordinal\", \n                                imp_params={}, encoder_params={}):\n        \"\"\"Transform categorical columns by making Pipeline\n        `SimpleImputer` | `OneHotEncoder` | `OrdinalEncoder`.\n        args:\n            imp_strategy: strategy for imputer values can be\n                \"most_frequent\" | \"constant\"\n            encoder_type: encoder type,\n                \"Ordinal\" | \"OneHot\"\n        kwargs:\n            imp_params: keyword args for `SimpleImputer`.\n            encoder_params: keyword args for encoder.`\n        \"\"\"\n        if not encoder_type in (\"Ordinal\", \"OneHot\"):\n            raise ValueError(f\"Inappropriate value for encoder_type passed: {encoder_type}\\\n            Takes one of 'Ordinal' | 'OneHot'.\")\n        \n        encoder = OrdinalEncoder if encoder_type==\"Ordinal\" else OneHotEncoder\n        transformer = Pipeline(steps=[\n            (\"imputer\", SimpleImputer(strategy=imp_strategy, **imp_params)),\n            (encoder_type, encoder(**encoder_params))\n        ])\n        return transformer\n    \n    \n    def data_preprocessor(self, *, transformers):\n        \"\"\"Preprocess the data using `ColumnTransformer`.\n        Pass extact list of transformers\n        to be passed in `ColumnTransformer`.\n        each tuple consist of: (transformer_name,\n                                transformer,\n                                list_of_columns).\"\"\"\n        preprocessor = ColumnTransformer(transformers=transformers)\n        return preprocessor\n    \n    \n    def create_model(self, *, model, random_state=0, n_estimators=1000, **kwargs):\n        \"\"\"Creates the model.\n        **kwargs: keyword args for model.\"\"\"\n        my_model = model(random_state=random_state, n_estimators=n_estimators, **kwargs)\n        return my_model","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:38.205215Z","iopub.execute_input":"2022-07-04T20:40:38.205678Z","iopub.status.idle":"2022-07-04T20:40:38.221208Z","shell.execute_reply.started":"2022-07-04T20:40:38.205648Z","shell.execute_reply":"2022-07-04T20:40:38.220278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cp = CreatePipeline()\nnum_transformer = cp.numerical_transformer()\ncat_transformer = cp.categorical_transformer(encoder_params={\"handle_unknown\":\"use_encoded_value\", \"unknown_value\":-1})\npreprocessor = cp.data_preprocessor(\n    transformers=[(\"num\", num_transformer, num_cols),\n                  (\"cat\", cat_transformer, cat_cols)\n                 ])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:38.222358Z","iopub.execute_input":"2022-07-04T20:40:38.222835Z","iopub.status.idle":"2022-07-04T20:40:38.233836Z","shell.execute_reply.started":"2022-07-04T20:40:38.222796Z","shell.execute_reply":"2022-07-04T20:40:38.232827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Hyperparamter tuning with Cross-validation","metadata":{}},{"cell_type":"code","source":"n_estimators = [100, 250, 500, 750, 1000]\nmax_depths = [5, 10, 20]\nmaes = {}\ni = 0\nfor n in n_estimators:\n    for md in max_depths:\n        i += 1\n        model = cp.create_model(model=RandomForestClassifier, n_estimators=n, max_depth=md)\n        pipeline = cp.pipeline(preprocessor=preprocessor, model=model)\n        scores = -1 * cross_val_score(pipeline, X, Y, cv=10, verbose=True,\n                                scoring=\"neg_mean_absolute_error\")\n        mae = scores.mean()\n        maes[i] = [n, md, mae]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:40:38.235034Z","iopub.execute_input":"2022-07-04T20:40:38.235616Z","iopub.status.idle":"2022-07-04T20:43:02.443154Z","shell.execute_reply.started":"2022-07-04T20:40:38.235592Z","shell.execute_reply":"2022-07-04T20:43:02.442180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Checking MAE","metadata":{}},{"cell_type":"code","source":"for i in maes:\n    n, md, mae = maes[i]\n    print(f\"{i}.\\tN_estimators: {n}\\tMax depth: {md}\\tMAE: {mae}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.443998Z","iopub.execute_input":"2022-07-04T20:43:02.444213Z","iopub.status.idle":"2022-07-04T20:43:02.449639Z","shell.execute_reply.started":"2022-07-04T20:43:02.444191Z","shell.execute_reply":"2022-07-04T20:43:02.448423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = min(maes, key=lambda x: maes[x][2])\nmaes[index]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.451011Z","iopub.execute_input":"2022-07-04T20:43:02.451295Z","iopub.status.idle":"2022-07-04T20:43:02.463314Z","shell.execute_reply.started":"2022-07-04T20:43:02.451271Z","shell.execute_reply":"2022-07-04T20:43:02.462344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Finding best parameters\n### `n_estimators=100` and `max_depth=5`","metadata":{}},{"cell_type":"code","source":"best_n_estimators = 100\nbest_max_depth = 5","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.464778Z","iopub.execute_input":"2022-07-04T20:43:02.465079Z","iopub.status.idle":"2022-07-04T20:43:02.476974Z","shell.execute_reply.started":"2022-07-04T20:43:02.465048Z","shell.execute_reply":"2022-07-04T20:43:02.475613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Training Model on best parameters","metadata":{}},{"cell_type":"code","source":"model = cp.create_model(model=RandomForestClassifier, n_estimators=best_n_estimators, max_depth=best_max_depth)\npipeline = cp.pipeline(preprocessor=preprocessor, model=model)\npipeline.fit(X, Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.478241Z","iopub.execute_input":"2022-07-04T20:43:02.478522Z","iopub.status.idle":"2022-07-04T20:43:02.681894Z","shell.execute_reply.started":"2022-07-04T20:43:02.478499Z","shell.execute_reply":"2022-07-04T20:43:02.680981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. Predicting `Survived` on `test_data`","metadata":{}},{"cell_type":"code","source":"test_preds = pipeline.predict(test_data)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.683032Z","iopub.execute_input":"2022-07-04T20:43:02.683257Z","iopub.status.idle":"2022-07-04T20:43:02.722521Z","shell.execute_reply.started":"2022-07-04T20:43:02.683236Z","shell.execute_reply":"2022-07-04T20:43:02.721577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 11. Submitting predictions","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame({\"PassengerId\": test_data.index, \"Survived\": test_preds})\noutput.to_csv(\"./submission_1.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T20:43:02.723737Z","iopub.execute_input":"2022-07-04T20:43:02.724012Z","iopub.status.idle":"2022-07-04T20:43:02.731511Z","shell.execute_reply.started":"2022-07-04T20:43:02.723985Z","shell.execute_reply":"2022-07-04T20:43:02.730556Z"},"trusted":true},"execution_count":null,"outputs":[]}]}