{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-17T13:58:08.021319Z","iopub.execute_input":"2022-07-17T13:58:08.022285Z","iopub.status.idle":"2022-07-17T13:58:08.030122Z","shell.execute_reply.started":"2022-07-17T13:58:08.022249Z","shell.execute_reply":"2022-07-17T13:58:08.028976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"#\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer,KNNImputer,SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder,OneHotEncoder\nfrom category_encoders import MEstimateEncoder,PolynomialEncoder,BackwardDifferenceEncoder,LeaveOneOutEncoder,QuantileEncoder\n\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom sklearn.feature_selection import mutual_info_regression\nfrom sklearn.model_selection import KFold, cross_val_score\nimport xgboost as xgb\n\n\n\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport missingno as mno\n\n\n# Set Matplotlib defaults\nplt.style.use(\"seaborn-whitegrid\")\nplt.rc(\"figure\", autolayout=True)\nplt.rc(\n    \"axes\",\n    labelweight=\"bold\",\n    labelsize=\"large\",\n    titleweight=\"bold\",\n    titlesize=14,\n    titlepad=10,\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.031731Z","iopub.execute_input":"2022-07-17T13:58:08.032280Z","iopub.status.idle":"2022-07-17T13:58:08.040792Z","shell.execute_reply.started":"2022-07-17T13:58:08.032246Z","shell.execute_reply":"2022-07-17T13:58:08.039758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\ntest = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\n\ndf = pd.concat([train, test])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.042058Z","iopub.execute_input":"2022-07-17T13:58:08.043895Z","iopub.status.idle":"2022-07-17T13:58:08.098969Z","shell.execute_reply.started":"2022-07-17T13:58:08.043837Z","shell.execute_reply":"2022-07-17T13:58:08.097571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summarize","metadata":{}},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.100309Z","iopub.execute_input":"2022-07-17T13:58:08.101569Z","iopub.status.idle":"2022-07-17T13:58:08.108779Z","shell.execute_reply.started":"2022-07-17T13:58:08.101506Z","shell.execute_reply":"2022-07-17T13:58:08.107875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.110048Z","iopub.execute_input":"2022-07-17T13:58:08.111105Z","iopub.status.idle":"2022-07-17T13:58:08.122494Z","shell.execute_reply.started":"2022-07-17T13:58:08.111055Z","shell.execute_reply":"2022-07-17T13:58:08.120758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.124035Z","iopub.execute_input":"2022-07-17T13:58:08.124894Z","iopub.status.idle":"2022-07-17T13:58:08.153904Z","shell.execute_reply.started":"2022-07-17T13:58:08.124841Z","shell.execute_reply":"2022-07-17T13:58:08.152799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.155214Z","iopub.execute_input":"2022-07-17T13:58:08.155613Z","iopub.status.idle":"2022-07-17T13:58:08.176626Z","shell.execute_reply.started":"2022-07-17T13:58:08.155567Z","shell.execute_reply":"2022-07-17T13:58:08.175470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.178196Z","iopub.execute_input":"2022-07-17T13:58:08.178631Z","iopub.status.idle":"2022-07-17T13:58:08.216085Z","shell.execute_reply.started":"2022-07-17T13:58:08.178589Z","shell.execute_reply":"2022-07-17T13:58:08.214865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clean","metadata":{}},{"cell_type":"markdown","source":"## Parse Passenger ID\nWe need to parse passenger ID to 'group num' and 'personal num'.\nIt means that we need to create new features.","metadata":{}},{"cell_type":"code","source":"group_num = []\npersonal_num = []\nfor str in df.PassengerId:\n    group_num.append(int(str.split(\"_\")[0]))\n    personal_num.append(int(str.split(\"_\")[1]))\ndf[\"Group\"] = pd.Series(group_num)\n\ntest_passengerId = df.iloc[train.shape[0]+test.index].PassengerId\n\ndf.pop(\"PassengerId\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.217495Z","iopub.execute_input":"2022-07-17T13:58:08.218725Z","iopub.status.idle":"2022-07-17T13:58:08.265431Z","shell.execute_reply.started":"2022-07-17T13:58:08.218655Z","shell.execute_reply":"2022-07-17T13:58:08.263920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parse Cabin\nWe need to parse cabin also.","metadata":{}},{"cell_type":"code","source":"cabin_deck = []\ncabin_num = []\ncabin_side = []\nfor str in df.Cabin:\n    if(pd.isna(str)):\n        cabin_deck.append(str)\n        cabin_num.append(str)\n        cabin_side.append(str)\n    else:\n        cabin_deck.append(str.split(\"/\")[0])\n        cabin_num.append(int(str.split(\"/\")[1]))\n        cabin_side.append(str.split(\"/\")[2])\n    \ndf[\"Cabin_Deck\"] = pd.Series(cabin_deck)\ndf[\"Cabin_Num\"] = pd.Series(cabin_num)\ndf[\"Cabin_Side\"] = pd.Series(cabin_side)\ndf.pop(\"Cabin\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.267250Z","iopub.execute_input":"2022-07-17T13:58:08.267844Z","iopub.status.idle":"2022-07-17T13:58:08.319665Z","shell.execute_reply.started":"2022-07-17T13:58:08.267805Z","shell.execute_reply":"2022-07-17T13:58:08.318794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"train = df.iloc[train.index]\ntest = df.iloc[train.shape[0]+test.index]\ntest.pop(\"Transported\")\ntest.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.320988Z","iopub.execute_input":"2022-07-17T13:58:08.321560Z","iopub.status.idle":"2022-07-17T13:58:08.343840Z","shell.execute_reply.started":"2022-07-17T13:58:08.321529Z","shell.execute_reply":"2022-07-17T13:58:08.342280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Univariate Plots for Numerical Features","metadata":{}},{"cell_type":"code","source":"train.plot(kind=\"box\", subplots=True, layout=(2,5), figsize=(10,10))\ntrain.plot(kind=\"hist\", subplots=True, layout=(2,5), figsize=(10,10))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:08.346777Z","iopub.execute_input":"2022-07-17T13:58:08.347940Z","iopub.status.idle":"2022-07-17T13:58:10.528124Z","shell.execute_reply.started":"2022-07-17T13:58:08.347667Z","shell.execute_reply":"2022-07-17T13:58:10.526912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Univariate Plot for Categorical Features","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2,2)\nnames = ['HomePlanet', 'Destination', 'Cabin_Deck', 'Cabin_Side']\n\nfor name, ax in zip(names, axes.flatten()):\n    sns.countplot(x=name, data=train, ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:10.529512Z","iopub.execute_input":"2022-07-17T13:58:10.530542Z","iopub.status.idle":"2022-07-17T13:58:11.001482Z","shell.execute_reply.started":"2022-07-17T13:58:10.530497Z","shell.execute_reply":"2022-07-17T13:58:11.000402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multivariate Plots for Numerical Data","metadata":{}},{"cell_type":"code","source":"train = train.astype({\"Transported\":bool})\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:11.002927Z","iopub.execute_input":"2022-07-17T13:58:11.003366Z","iopub.status.idle":"2022-07-17T13:58:11.027510Z","shell.execute_reply.started":"2022-07-17T13:58:11.003331Z","shell.execute_reply":"2022-07-17T13:58:11.026372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.pairplot(data=train.select_dtypes([\"number\", \"bool\"]), hue=\"Transported\",palette='CMRmap')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:11.029716Z","iopub.execute_input":"2022-07-17T13:58:11.030230Z","iopub.status.idle":"2022-07-17T13:58:11.035332Z","shell.execute_reply.started":"2022-07-17T13:58:11.030183Z","shell.execute_reply":"2022-07-17T13:58:11.034293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(3, 3,figsize=(18,15))\nnames=train.select_dtypes(\"number\").columns\n\nfor name, ax in zip(names, axes.flatten()):\n    sns.stripplot(y=name,x='Transported',data=train,ax=ax)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:11.036373Z","iopub.execute_input":"2022-07-17T13:58:11.037033Z","iopub.status.idle":"2022-07-17T13:58:12.517638Z","shell.execute_reply.started":"2022-07-17T13:58:11.036996Z","shell.execute_reply":"2022-07-17T13:58:12.516790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Multivariate Plots for Categorical Data","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2)\nnames=train.select_dtypes(\"object\").columns\n\nfor name, ax in zip(names, axes.flatten()):\n    sns.barplot(x=name,y='Transported',data=train,ax=ax)\n    ax.set(ylabel=\"Transportation Probability\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:12.518905Z","iopub.execute_input":"2022-07-17T13:58:12.519679Z","iopub.status.idle":"2022-07-17T13:58:13.592252Z","shell.execute_reply.started":"2022-07-17T13:58:12.519636Z","shell.execute_reply":"2022-07-17T13:58:13.590968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encode","metadata":{}},{"cell_type":"code","source":"num_features = train.select_dtypes(\"number\").columns\nint_features = [\"Age\", \"Group\", \"Cabin_Num\"]\nfloat_features = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Spa\", \"VRDeck\"]\n\ncat_features = train.select_dtypes([\"object\"]).columns","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:13.593471Z","iopub.execute_input":"2022-07-17T13:58:13.593804Z","iopub.status.idle":"2022-07-17T13:58:13.603619Z","shell.execute_reply.started":"2022-07-17T13:58:13.593775Z","shell.execute_reply":"2022-07-17T13:58:13.602611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Numerical Values","metadata":{}},{"cell_type":"markdown","source":"First, we need to handle missing values.","metadata":{}},{"cell_type":"code","source":"train.isna().sum() / train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:13.605785Z","iopub.execute_input":"2022-07-17T13:58:13.606768Z","iopub.status.idle":"2022-07-17T13:58:13.627565Z","shell.execute_reply.started":"2022-07-17T13:58:13.606715Z","shell.execute_reply":"2022-07-17T13:58:13.625778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mno.matrix(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:13.629276Z","iopub.execute_input":"2022-07-17T13:58:13.630207Z","iopub.status.idle":"2022-07-17T13:58:14.195245Z","shell.execute_reply.started":"2022-07-17T13:58:13.630162Z","shell.execute_reply":"2022-07-17T13:58:14.193967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.196746Z","iopub.execute_input":"2022-07-17T13:58:14.197505Z","iopub.status.idle":"2022-07-17T13:58:14.215603Z","shell.execute_reply.started":"2022-07-17T13:58:14.197457Z","shell.execute_reply":"2022-07-17T13:58:14.214040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use simple imputer\nfloat_simp = SimpleImputer(missing_values = np.nan, strategy = 'mean')\nint_simp = SimpleImputer(missing_values = np.nan, strategy = 'median')\n\ntrain[float_features] = float_simp.fit_transform(train[float_features])\ntrain[int_features] = int_simp.fit_transform(train[int_features])\n\ntest[float_features] = float_simp.transform(test[float_features])\ntest[int_features] = int_simp.transform(test[int_features])\n\ntrain.head(30)\n# Use IterativeImputer","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.219573Z","iopub.execute_input":"2022-07-17T13:58:14.220740Z","iopub.status.idle":"2022-07-17T13:58:14.283118Z","shell.execute_reply.started":"2022-07-17T13:58:14.220664Z","shell.execute_reply":"2022-07-17T13:58:14.281958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, let's change dtype.","metadata":{}},{"cell_type":"code","source":"train[int_features] = train[int_features].astype(\"int\")\ntest[int_features] = test[int_features].astype(\"int\")\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.284321Z","iopub.execute_input":"2022-07-17T13:58:14.284640Z","iopub.status.idle":"2022-07-17T13:58:14.312280Z","shell.execute_reply.started":"2022-07-17T13:58:14.284610Z","shell.execute_reply":"2022-07-17T13:58:14.310976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical Values","metadata":{}},{"cell_type":"code","source":"for feature in cat_features:\n    print('{0}: {1}'.format(feature, len(df[feature].unique())))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.313727Z","iopub.execute_input":"2022-07-17T13:58:14.314071Z","iopub.status.idle":"2022-07-17T13:58:14.328766Z","shell.execute_reply.started":"2022-07-17T13:58:14.314039Z","shell.execute_reply":"2022-07-17T13:58:14.327767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We know that we don't need \"Name\" column.","metadata":{}},{"cell_type":"code","source":"train.pop(\"Name\")\ntest.pop(\"Name\")\ncat_features = train.select_dtypes(\"object\").columns\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.330272Z","iopub.execute_input":"2022-07-17T13:58:14.330600Z","iopub.status.idle":"2022-07-17T13:58:14.365819Z","shell.execute_reply.started":"2022-07-17T13:58:14.330572Z","shell.execute_reply":"2022-07-17T13:58:14.364428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have two options of imputing categorical values: 'most_frequent', 'classifier' ","metadata":{}},{"cell_type":"code","source":"# Use 'most_frequent' method.\ncat_median_simp = SimpleImputer(strategy = \"most_frequent\")\ntrain[cat_features] = cat_median_simp.fit_transform(train[cat_features])\ntest[cat_features] = cat_median_simp.transform(test[cat_features])\n\ntrain.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.367448Z","iopub.execute_input":"2022-07-17T13:58:14.367814Z","iopub.status.idle":"2022-07-17T13:58:14.404887Z","shell.execute_reply.started":"2022-07-17T13:58:14.367782Z","shell.execute_reply":"2022-07-17T13:58:14.403429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Encoding method:\n* label encoding\n* one-hot encoding","metadata":{}},{"cell_type":"code","source":"# Use one-hot encoding\noh_enc = OneHotEncoder(drop='first', handle_unknown='ignore', sparse=False)\noh_train = pd.DataFrame(oh_enc.fit_transform(train[cat_features]))\noh_test = pd.DataFrame(oh_enc.transform(test[cat_features]))\n\noh_train.index = train.index\noh_test.index = test.index\n\ntrain = train.drop(cat_features, axis=1)\ntest = test.drop(cat_features, axis=1)\n\ntrain = pd.concat([train, oh_train], axis=1)\ntest = pd.concat([test, oh_test], axis=1)\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.406802Z","iopub.execute_input":"2022-07-17T13:58:14.407425Z","iopub.status.idle":"2022-07-17T13:58:14.482967Z","shell.execute_reply.started":"2022-07-17T13:58:14.407369Z","shell.execute_reply":"2022-07-17T13:58:14.481604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.484602Z","iopub.execute_input":"2022-07-17T13:58:14.484998Z","iopub.status.idle":"2022-07-17T13:58:14.520203Z","shell.execute_reply.started":"2022-07-17T13:58:14.484964Z","shell.execute_reply":"2022-07-17T13:58:14.518627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Exploring","metadata":{}},{"cell_type":"markdown","source":"## Mutual Information","metadata":{}},{"cell_type":"code","source":"# descrete -> int\nfor feature in range(14):\n    train[feature] = train[feature].astype(\"int\")\n    test[feature] = test[feature].astype(\"int\")\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.524133Z","iopub.execute_input":"2022-07-17T13:58:14.524503Z","iopub.status.idle":"2022-07-17T13:58:14.562326Z","shell.execute_reply.started":"2022-07-17T13:58:14.524474Z","shell.execute_reply":"2022-07-17T13:58:14.560761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"discrete_features = train.select_dtypes(\"number\").dtypes == \"int\"\ndiscrete_features","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.563973Z","iopub.execute_input":"2022-07-17T13:58:14.564333Z","iopub.status.idle":"2022-07-17T13:58:14.577172Z","shell.execute_reply.started":"2022-07-17T13:58:14.564304Z","shell.execute_reply":"2022-07-17T13:58:14.575939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mi_scores = mutual_info_regression(train.select_dtypes(\"number\"), train[\"Transported\"], discrete_features=discrete_features, random_state=0)\nmi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=train.select_dtypes(\"number\").columns)\nmi_scores = mi_scores.sort_values(ascending=False)\nmi_scores","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:14.578946Z","iopub.execute_input":"2022-07-17T13:58:14.579423Z","iopub.status.idle":"2022-07-17T13:58:17.586727Z","shell.execute_reply.started":"2022-07-17T13:58:14.579375Z","shell.execute_reply":"2022-07-17T13:58:17.585455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mi_scores = mi_scores.sort_values(ascending=True)\nwidth = np.arange(len(mi_scores))\nticks = list(mi_scores.index)\nplt.barh(width, mi_scores)\nplt.yticks(width, ticks)\nplt.title(\"Mutual Information Scores\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.588037Z","iopub.execute_input":"2022-07-17T13:58:17.588497Z","iopub.status.idle":"2022-07-17T13:58:17.830881Z","shell.execute_reply.started":"2022-07-17T13:58:17.588461Z","shell.execute_reply":"2022-07-17T13:58:17.829693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can conclude that '10' ~ '3' column could be droped.\n'Group', 'Cabin_Num' have high MI scores, however, that result could be caused by nearly perfect linearity of them.","metadata":{}},{"cell_type":"code","source":"# Drop features whose mi score less than 0.011.\ndrop_features = mi_scores[mi_scores < 0.011].index\ntrain1 = train.drop(drop_features, axis=1)\ntest1 = test.drop(drop_features, axis=1)\ntrain1","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.832418Z","iopub.execute_input":"2022-07-17T13:58:17.832827Z","iopub.status.idle":"2022-07-17T13:58:17.859266Z","shell.execute_reply.started":"2022-07-17T13:58:17.832796Z","shell.execute_reply":"2022-07-17T13:58:17.858007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1.pop(\"Group\")\ntrain1.pop(\"Cabin_Num\")\ntest1.pop(\"Group\")\ntest1.pop(\"Cabin_Num\")","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.861101Z","iopub.execute_input":"2022-07-17T13:58:17.861554Z","iopub.status.idle":"2022-07-17T13:58:17.873008Z","shell.execute_reply.started":"2022-07-17T13:58:17.861493Z","shell.execute_reply":"2022-07-17T13:58:17.871691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models and Scoring","metadata":{}},{"cell_type":"code","source":"def score_dataset(X, y, model=xgb.XGBClassifier()):\n    score = cross_val_score(\n        model, X, y, cv=5, scoring=\"accuracy\",\n    )\n    return [score, score.mean()]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.875088Z","iopub.execute_input":"2022-07-17T13:58:17.875900Z","iopub.status.idle":"2022-07-17T13:58:17.883907Z","shell.execute_reply.started":"2022-07-17T13:58:17.875855Z","shell.execute_reply":"2022-07-17T13:58:17.882625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y = train.pop(\"Transported\")\ntrain_X = train\ntrain1_y = train1.pop(\"Transported\")\ntrain1_X = train1","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.885825Z","iopub.execute_input":"2022-07-17T13:58:17.886679Z","iopub.status.idle":"2022-07-17T13:58:17.899083Z","shell.execute_reply.started":"2022-07-17T13:58:17.886628Z","shell.execute_reply":"2022-07-17T13:58:17.897961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(score_dataset(train_X, train_y))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:17.902123Z","iopub.execute_input":"2022-07-17T13:58:17.902554Z","iopub.status.idle":"2022-07-17T13:58:22.394367Z","shell.execute_reply.started":"2022-07-17T13:58:17.902519Z","shell.execute_reply":"2022-07-17T13:58:22.393394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(score_dataset(train1_X, train1_y))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:22.395737Z","iopub.execute_input":"2022-07-17T13:58:22.397350Z","iopub.status.idle":"2022-07-17T13:58:25.279737Z","shell.execute_reply.started":"2022-07-17T13:58:22.397291Z","shell.execute_reply":"2022-07-17T13:58:25.278748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb.XGBClassifier()\nmodel.fit(train1_X, train1_y)\npredictions = model.predict(test1) == 1\noutput = pd.DataFrame(predictions, columns=[\"Transported\"])\noutput = pd.concat([test_passengerId, output], axis=1)\noutput","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:25.280922Z","iopub.execute_input":"2022-07-17T13:58:25.281387Z","iopub.status.idle":"2022-07-17T13:58:26.918942Z","shell.execute_reply.started":"2022-07-17T13:58:25.281357Z","shell.execute_reply":"2022-07-17T13:58:26.917737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"output.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T13:58:26.920330Z","iopub.execute_input":"2022-07-17T13:58:26.920795Z","iopub.status.idle":"2022-07-17T13:58:26.933526Z","shell.execute_reply.started":"2022-07-17T13:58:26.920749Z","shell.execute_reply":"2022-07-17T13:58:26.932561Z"},"trusted":true},"execution_count":null,"outputs":[]}]}