{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T14:11:53.432036Z","iopub.execute_input":"2022-07-26T14:11:53.432423Z","iopub.status.idle":"2022-07-26T14:11:53.443017Z","shell.execute_reply.started":"2022-07-26T14:11:53.432379Z","shell.execute_reply":"2022-07-26T14:11:53.441828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nX_full = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv', index_col='PassengerId')\nX_test_full = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv', index_col='PassengerId')\ntest_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\n# Remove rows with missing target, separate target from predictors\nX_full.dropna(axis=0, subset=['Transported'], inplace=True)\ny = X_full.Transported\nX_full.drop(['Transported'], axis=1, inplace=True)\n# Break off validation set from training data\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X_full, y, \n                                                                train_size=0.8, test_size=0.2,\n                                                                random_state=0)\n# \"Cardinality\" means the number of unique values in a column\n# Select categorical columns with relatively low cardinality (convenient but arbitrary)\ncategorical_cols = [cname for cname in X_train_full.columns if\n                    X_train_full[cname].nunique() < 10 and \n                    X_train_full[cname].dtype == \"object\"]\n# Select numerical columns\nnumerical_cols = [cname for cname in X_train_full.columns if \n                X_train_full[cname].dtype in ['int64', 'float64']]\n# Keep selected columns only\nmy_cols = categorical_cols + numerical_cols\nX_train = X_train_full[my_cols].copy()\nX_valid = X_valid_full[my_cols].copy()\nX_test = X_test_full[my_cols].copy()\nX_train.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T15:37:21.612523Z","iopub.execute_input":"2022-07-26T15:37:21.612881Z","iopub.status.idle":"2022-07-26T15:37:21.737587Z","shell.execute_reply.started":"2022-07-26T15:37:21.612852Z","shell.execute_reply":"2022-07-26T15:37:21.736408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, confusion_matrix\nnumerical_transformer = SimpleImputer(strategy='constant')\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])\nmodel = XGBClassifier(n_estimators=1000, max_depth=10, random_state=0)\n\n# Bundle preprocessing and modeling code in a pipeline\nclf = Pipeline(steps=[('preprocessor', preprocessor),\n                      ('model', model)\n                     ])\nclf.fit(X_train, y_train)\ny_pred = clf.predict(X_test)\nprint(y_pred)\noutput = pd.DataFrame({'PassengerId': test_data.PassengerId, 'Transported': y_pred})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T15:48:06.629686Z","iopub.execute_input":"2022-07-26T15:48:06.630087Z","iopub.status.idle":"2022-07-26T15:48:17.569620Z","shell.execute_reply.started":"2022-07-26T15:48:06.630055Z","shell.execute_reply":"2022-07-26T15:48:17.568621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}