{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T21:44:39.703794Z","iopub.execute_input":"2022-08-07T21:44:39.704320Z","iopub.status.idle":"2022-08-07T21:44:39.716467Z","shell.execute_reply.started":"2022-08-07T21:44:39.704281Z","shell.execute_reply":"2022-08-07T21:44:39.715117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import Normalizer","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:41.432420Z","iopub.execute_input":"2022-08-07T21:44:41.433723Z","iopub.status.idle":"2022-08-07T21:44:42.029613Z","shell.execute_reply.started":"2022-08-07T21:44:41.433661Z","shell.execute_reply":"2022-08-07T21:44:42.028507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain.head()\ntest = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ny = train['Survived']\nX = train.copy()\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state = 1)\n\ncategorical_cols = [cname for cname in X_train.columns if\n                    X_train[cname].nunique() < 10 and \n                    X_train[cname].dtype == \"object\"]\n\nnumerical_cols = [cname for cname in X_train.columns if \n                X_train[cname].dtype in ['int64', 'float64']]\n\nmy_cols = categorical_cols + numerical_cols\nX_train1 = X_train[my_cols].copy()\nX_valid = X_test[my_cols].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:46:20.373091Z","iopub.execute_input":"2022-08-07T21:46:20.373592Z","iopub.status.idle":"2022-08-07T21:46:20.402985Z","shell.execute_reply.started":"2022-08-07T21:46:20.373555Z","shell.execute_reply":"2022-08-07T21:46:20.401742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\n\n# Preprocessing for numerical data\nnumerical_transformer = SimpleImputer(strategy='constant')\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])\n\n# Define model\nmodel = RandomForestRegressor(n_estimators=100, random_state=0)\n\n# Bundle preprocessing and modeling code in a pipeline\nclf = Pipeline(steps=[('preprocessor', preprocessor),\n                      ('model', model)\n                     ])\n\n# Preprocessing of training data, fit model \nclf.fit(X_train1, y_train)\n\n# Preprocessing of validation data, get predictions\npreds = clf.predict(X_valid)\n\nprint('MAE:', mean_absolute_error(y_test, preds))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:47:41.574848Z","iopub.execute_input":"2022-08-07T21:47:41.575313Z","iopub.status.idle":"2022-08-07T21:47:41.748690Z","shell.execute_reply.started":"2022-08-07T21:47:41.575277Z","shell.execute_reply":"2022-08-07T21:47:41.747022Z"},"trusted":true},"execution_count":null,"outputs":[]}]}