{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport scipy.stats as st \n\nimport sklearn #preprocessing data and getting classifiers\nfrom sklearn import preprocessing\n\nimport matplotlib.pyplot as plt #plotting\nimport seaborn as sns\n\n#import files\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T14:41:40.232098Z","iopub.execute_input":"2022-07-08T14:41:40.232507Z","iopub.status.idle":"2022-07-08T14:41:40.242113Z","shell.execute_reply.started":"2022-07-08T14:41:40.232472Z","shell.execute_reply":"2022-07-08T14:41:40.241206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First steps: \n-Read in the data\n-Get an idea of what the features are\n-Take a quick look at their distributions","metadata":{}},{"cell_type":"code","source":"#Read in training data\ntrain_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")\nprint(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:40.356723Z","iopub.execute_input":"2022-07-08T14:41:40.357162Z","iopub.status.idle":"2022-07-08T14:41:40.414436Z","shell.execute_reply.started":"2022-07-08T14:41:40.357124Z","shell.execute_reply":"2022-07-08T14:41:40.413244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:40.545919Z","iopub.execute_input":"2022-07-08T14:41:40.546518Z","iopub.status.idle":"2022-07-08T14:41:40.577572Z","shell.execute_reply.started":"2022-07-08T14:41:40.546485Z","shell.execute_reply":"2022-07-08T14:41:40.576372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP']\n\nfor feature in cat_features:\n    data = train_data[str(feature)].value_counts()\n    print(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:40.739692Z","iopub.execute_input":"2022-07-08T14:41:40.740982Z","iopub.status.idle":"2022-07-08T14:41:40.756622Z","shell.execute_reply.started":"2022-07-08T14:41:40.740937Z","shell.execute_reply":"2022-07-08T14:41:40.754683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 20))\nna_features = ['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\nna_colors = ['red', 'orange', 'yellow', 'green', 'blue', 'purple']\n\nfor idx, feature in enumerate(na_features):\n    plt.subplot(2, 3, (idx+1, idx+1))\n    sns.kdeplot(train_data[str(feature)], color=na_colors[idx])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:40.882008Z","iopub.execute_input":"2022-07-08T14:41:40.882414Z","iopub.status.idle":"2022-07-08T14:41:42.005044Z","shell.execute_reply.started":"2022-07-08T14:41:40.882380Z","shell.execute_reply":"2022-07-08T14:41:42.003955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Step 2:** Impute missing values and preprocessing training and test data","metadata":{}},{"cell_type":"code","source":"#Check for nulls\nprint(train_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.006845Z","iopub.execute_input":"2022-07-08T14:41:42.007214Z","iopub.status.idle":"2022-07-08T14:41:42.022221Z","shell.execute_reply.started":"2022-07-08T14:41:42.007182Z","shell.execute_reply":"2022-07-08T14:41:42.021117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Figure out data types\nprint(train_data.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.025595Z","iopub.execute_input":"2022-07-08T14:41:42.026076Z","iopub.status.idle":"2022-07-08T14:41:42.047542Z","shell.execute_reply.started":"2022-07-08T14:41:42.026021Z","shell.execute_reply":"2022-07-08T14:41:42.046731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Group categorical versus numerical data\nX = train_data.drop(labels=['PassengerId', 'Name', 'Transported'], axis=1)\ncat_cols = ['HomePlanet', 'CryoSleep', 'Destination', 'VIP', 'IsAdult', 'Deck', 'Number', 'Side']\nnum_cols = ['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'TotalAmenities', 'GroupSize']","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.049215Z","iopub.execute_input":"2022-07-08T14:41:42.050116Z","iopub.status.idle":"2022-07-08T14:41:42.057446Z","shell.execute_reply.started":"2022-07-08T14:41:42.050071Z","shell.execute_reply":"2022-07-08T14:41:42.056223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create pipeline to impute missing values\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder\n\n\n#Categorical\ncat_imputer = Pipeline(steps = [('imputer', SimpleImputer(strategy='most_frequent')), \n                                ('onehot', OneHotEncoder(handle_unknown='ignore'))])\n\n#Numerical\nnum_imputer = SimpleImputer(strategy='median')\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_imputer, num_cols),\n        ('cat', cat_imputer, cat_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.059177Z","iopub.execute_input":"2022-07-08T14:41:42.060003Z","iopub.status.idle":"2022-07-08T14:41:42.073001Z","shell.execute_reply.started":"2022-07-08T14:41:42.059959Z","shell.execute_reply":"2022-07-08T14:41:42.071573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we've created our pipelines, let's do some feature engineering.","metadata":{}},{"cell_type":"code","source":"#Create IsAdult feature\nis_adult = np.where(X['Age']<18, 'Child', 'Adult')\nX['IsAdult'] = is_adult","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.074497Z","iopub.execute_input":"2022-07-08T14:41:42.074836Z","iopub.status.idle":"2022-07-08T14:41:42.092041Z","shell.execute_reply.started":"2022-07-08T14:41:42.074807Z","shell.execute_reply":"2022-07-08T14:41:42.090821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get groups sizes based on number of people in each cabin\ngroups = X['Cabin'].value_counts().rename_axis('Cabin').reset_index(name='GroupSize')\nX = X.merge(groups, on='Cabin', how='left')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.151568Z","iopub.execute_input":"2022-07-08T14:41:42.152418Z","iopub.status.idle":"2022-07-08T14:41:42.177871Z","shell.execute_reply.started":"2022-07-08T14:41:42.152367Z","shell.execute_reply":"2022-07-08T14:41:42.176748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get more information from Cabin\ncabins = X['Cabin'].str.split(pat='/', expand=True)\ncabins = cabins.rename(columns={0: 'Deck', 1:'Number', 2:'Side'})\nX = pd.concat([X, cabins], axis=1)\nX = X.drop(labels=['Cabin'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.337678Z","iopub.execute_input":"2022-07-08T14:41:42.338109Z","iopub.status.idle":"2022-07-08T14:41:42.369254Z","shell.execute_reply.started":"2022-07-08T14:41:42.338073Z","shell.execute_reply":"2022-07-08T14:41:42.367954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Normalize the linear variables\n#Age, using min max scaling\nX['Age'] = sklearn.preprocessing.minmax_scale(X['Age'])\n\nplt.figure()\nsns.kdeplot(X['Age'], color='red')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.514835Z","iopub.execute_input":"2022-07-08T14:41:42.515494Z","iopub.status.idle":"2022-07-08T14:41:42.729502Z","shell.execute_reply.started":"2022-07-08T14:41:42.515454Z","shell.execute_reply":"2022-07-08T14:41:42.728342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scale Group Size, using min max scaling\nX['GroupSize'] = sklearn.preprocessing.minmax_scale(X['GroupSize'])\n\nplt.figure()\nsns.kdeplot(X['GroupSize'], color='red')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.731377Z","iopub.execute_input":"2022-07-08T14:41:42.731717Z","iopub.status.idle":"2022-07-08T14:41:42.954558Z","shell.execute_reply.started":"2022-07-08T14:41:42.731686Z","shell.execute_reply":"2022-07-08T14:41:42.953042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a total spent variable\ntotal_amenities = np.array(X['RoomService'])+np.array(X['FoodCourt'])+np.array(X['ShoppingMall'])+np.array(X['Spa'])+np.array(X['VRDeck'])\nX['TotalAmenities'] = total_amenities","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:42.956952Z","iopub.execute_input":"2022-07-08T14:41:42.957432Z","iopub.status.idle":"2022-07-08T14:41:42.964849Z","shell.execute_reply.started":"2022-07-08T14:41:42.957388Z","shell.execute_reply":"2022-07-08T14:41:42.964066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shopping variables\nfrom sklearn.preprocessing import MaxAbsScaler\nma = MaxAbsScaler()\nX.loc[:, ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'TotalAmenities']] = ma.fit_transform(X.loc[:, ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'TotalAmenities']])\n\namenity_features = ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck', 'TotalAmenities']\n\nplt.figure(figsize=(20, 20))\nfor idx, feature in enumerate(amenity_features):\n    plt.subplot(2, 3, (idx+1, idx+1))\n    sns.kdeplot(X[str(feature)], color=na_colors[idx])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:43.048972Z","iopub.execute_input":"2022-07-08T14:41:43.049546Z","iopub.status.idle":"2022-07-08T14:41:44.110373Z","shell.execute_reply.started":"2022-07-08T14:41:43.049514Z","shell.execute_reply":"2022-07-08T14:41:44.109383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:44.112009Z","iopub.execute_input":"2022-07-08T14:41:44.112503Z","iopub.status.idle":"2022-07-08T14:41:44.133249Z","shell.execute_reply.started":"2022-07-08T14:41:44.112473Z","shell.execute_reply":"2022-07-08T14:41:44.132155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#combine dummy variables, normed categorical variables, and define y\nX = pd.concat([X.loc[:, num_cols], X.loc[:, cat_cols]], axis=1)\ny = train_data['Transported']\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(X, y)\n\nfrom sklearn.ensemble import RandomForestClassifier\nrf_model = RandomForestClassifier(n_estimators = 100)\nrf_pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', rf_model)])\nrf_pipeline.fit(X_train, y_train)\n\nfrom xgboost import XGBClassifier\nxgb_model = XGBClassifier(n_estimators=100)\nxgb_pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', xgb_model)])\nxgb_pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:44.134449Z","iopub.execute_input":"2022-07-08T14:41:44.134738Z","iopub.status.idle":"2022-07-08T14:41:48.808157Z","shell.execute_reply.started":"2022-07-08T14:41:44.134711Z","shell.execute_reply":"2022-07-08T14:41:48.807013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\nrf_predictions = rf_pipeline.predict(X_valid)\nxgb_predictions = xgb_pipeline.predict(X_valid)\n\nprint(\"Random Forest\")\nprint(accuracy_score(rf_predictions, y_valid.astype('int')))\nprint(\"XGBooster\")\nprint(accuracy_score(xgb_predictions, y_valid.astype('int')))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:48.810296Z","iopub.execute_input":"2022-07-08T14:41:48.810650Z","iopub.status.idle":"2022-07-08T14:41:48.972460Z","shell.execute_reply.started":"2022-07-08T14:41:48.810618Z","shell.execute_reply":"2022-07-08T14:41:48.971314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = test_data.drop(labels=['PassengerId', 'Name'], axis=1)\n\npreprocessor_test = ColumnTransformer(\n    transformers=[\n        ('num', num_imputer, num_cols),\n        ('cat', cat_imputer, cat_cols)\n    ])\n\n#Create IsAdult Feature\nis_adult = np.where(X_test['Age']<18, 'Child', 'Adult')\nX_test['IsAdult'] = is_adult\n\n#Scale Age\nX_test['Age'] = sklearn.preprocessing.minmax_scale(X_test['Age'])\n\n#Get groups sizes based on number of people in each cabin\ngroups = X_test['Cabin'].value_counts().rename_axis('Cabin').reset_index(name='GroupSize')\nX_test = X_test.merge(groups, on='Cabin', how='left')\n\n#Scale Group Size, using min max scaling\nX_test['GroupSize'] = sklearn.preprocessing.minmax_scale(X_test['GroupSize'])\n\n#Get Side, Deck, and Number features\ncabins = X_test['Cabin'].str.split(pat='/', expand=True)\ncabins = cabins.rename(columns={0: 'Deck', 1:'Number', 2:'Side'})\nX_test = pd.concat([X_test, cabins], axis=1)\nX_test = X_test.drop(labels=['Cabin'], axis=1)\n\n#Create TotalAmenities Column\ntotal_amenities = np.array(X_test['RoomService'])+np.array(X_test['FoodCourt'])+np.array(X_test['ShoppingMall'])+np.array(X_test['Spa'])+np.array(X_test['VRDeck'])\nX_test['TotalAmenities'] = total_amenities\n\n#Scale Amenities\nX_test.loc[:, ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] = ma.fit_transform(X_test.loc[:, ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']])\n\nX_test = pd.concat([X_test.loc[:, num_cols], X_test.loc[:, cat_cols]], axis=1)\nxgb_predictions = xgb_pipeline.predict(X_test)\n\npredictions = pd.DataFrame(np.where(xgb_predictions==1, \"True\", \"False\"))\nsubmission = pd.concat([test_data['PassengerId'], predictions], axis=1)\nsubmission = submission.rename(columns = {0: \"Transported\"})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:48.973914Z","iopub.execute_input":"2022-07-08T14:41:48.974470Z","iopub.status.idle":"2022-07-08T14:41:49.084378Z","shell.execute_reply.started":"2022-07-08T14:41:48.974428Z","shell.execute_reply":"2022-07-08T14:41:49.083221Z"},"trusted":true},"execution_count":null,"outputs":[]}]}