{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\n%config InlineBackend.figure_format = 'retina'\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nplt.style.use('ggplot')\n\nimport seaborn as sns\n\nparams = {\n    'text.color': (0.25, 0.25, 0.25),\n    'figure.figsize': [18, 6],\n   }\nplt.rcParams.update(params)\n\nimport pandas as pd\npd.options.display.max_rows = 100\npd.options.display.max_seq_items = 100\n\nimport numpy as np\nnp.random.seed(42)\n\nfrom tqdm.notebook import tqdm\nfrom tabulate import tabulate\n\nTITLE_SIZE = 24\nTITLE_PAD = 20\n\nDEFAULT_COLORS = plt.rcParams['axes.prop_cycle'].by_key()['color']","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:48:21.805794Z","iopub.execute_input":"2022-07-24T13:48:21.806166Z","iopub.status.idle":"2022-07-24T13:48:21.832591Z","shell.execute_reply.started":"2022-07-24T13:48:21.806136Z","shell.execute_reply":"2022-07-24T13:48:21.831522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, QuantileTransformer, RobustScaler, PowerTransformer \nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\n\nfrom sklearn.dummy import DummyClassifier\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression, RidgeClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.gaussian_process.kernels import RBF\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis, QuadraticDiscriminantAnalysis\nfrom sklearn.tree import DecisionTreeClassifier\n\nfrom sklearn.metrics import confusion_matrix, accuracy_score, auc, classification_report\nfrom sklearn.inspection import permutation_importance\n\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:48:24.445837Z","iopub.execute_input":"2022-07-24T13:48:24.446272Z","iopub.status.idle":"2022-07-24T13:48:25.758661Z","shell.execute_reply.started":"2022-07-24T13:48:24.446240Z","shell.execute_reply":"2022-07-24T13:48:25.757295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"../input/spaceship-titanic/train.csv\")\ntest = pd.read_csv(\"../input/spaceship-titanic/test.csv\")\n\n# concatenate to one dataframe to analyze the whole data set\ndf = pd.concat([train, test]).reset_index(drop=True)\n\n# set column names to lowercase for easier handling of the dataframe\ndf.columns = df.columns.str.lower()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:03.736883Z","iopub.execute_input":"2022-07-24T13:49:03.737285Z","iopub.status.idle":"2022-07-24T13:49:03.833924Z","shell.execute_reply.started":"2022-07-24T13:49:03.737254Z","shell.execute_reply":"2022-07-24T13:49:03.832869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- We have a total of **12,970 samples with 14 features**.\n- **We have training data for 8'693 passengers** and **test data for 4'277 passengers**. \n- **6 features are numerical** and **8 are categorical** including the target variable `transported`. \n- There are **no exact duplicates**.\n- We can see that **data is missing for several features**.","metadata":{}},{"cell_type":"code","source":"print(f\"The combined data set contains a total of {df.shape[0]:,.0f} samples with {df.shape[1]} features.\")\nprint(f\"There are {df.select_dtypes('number').shape[1]} numerical and {df.select_dtypes('object').shape[1]} categorical features\")\n\nexact_duplicates = df.duplicated().sum()\nprint(f\"We have {exact_duplicates} exact duplicates in the data set.\\n\")\n\ndisplay(train.info())\nprint()\ndisplay(test.info())\nprint()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:07.336281Z","iopub.execute_input":"2022-07-24T13:49:07.336711Z","iopub.status.idle":"2022-07-24T13:49:07.412354Z","shell.execute_reply.started":"2022-07-24T13:49:07.336678Z","shell.execute_reply":"2022-07-24T13:49:07.411063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **Between 2% to 2.4% of data is missing** in 12 of the 13 features in the combined data (train and test).\n- **The difference in the amount of missing data between test and training data seems similar.**","metadata":{}},{"cell_type":"code","source":"missing = [(c, df[c].isna().mean()*100) for c in df.drop(\"transported\", axis=1)]\nmissing_train = [(c, train[c].isna().mean()*100) for c in train]\nmissing_test = [(c, test[c].isna().mean()*100) for c in test]\n\nmissing = pd.DataFrame(missing, columns=[\"column_name\", \"full_data\"])\nmissing_train = pd.DataFrame(missing_train, columns=[\"column_name\", \"train_data\"])\nmissing_test = pd.DataFrame(missing_test, columns=[\"column_name\", \"test_data\"])\nmissing = pd.concat([missing, missing_train.train_data, missing_test.test_data], axis=1)\n\nmissing = missing[missing[\"full_data\"] > 0].sort_values(\"full_data\", ascending=False)\nprint(\"Percentage of missing values\")\nprint(tabulate(missing, floatfmt=\".1f\", showindex=False, headers=\"keys\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:12.757686Z","iopub.execute_input":"2022-07-24T13:49:12.758083Z","iopub.status.idle":"2022-07-24T13:49:12.809305Z","shell.execute_reply.started":"2022-07-24T13:49:12.758039Z","shell.execute_reply":"2022-07-24T13:49:12.808243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's have a closer look at the data description:**\n\n- `passengerid` - A unique Id for each passenger. Each Id takes the form gggg_pp where gggg indicates a group the passenger is travelling with and pp is their number within the group. People in a group are often family members, but not always.\n- `homeplanet` - The planet the passenger departed from, typically their planet of permanent residence.\n- `cryosleep` - Indicates whether the passenger elected to be put into suspended animation for the duration of the voyage. Passengers in cryosleep are confined to their cabins.\n- `cabin` - The cabin number where the passenger is staying. Takes the form deck/num/side, where side can be either P for Port or S for Starboard.\n- `destination` - The planet the passenger will be debarking to.\n- `age` - The age of the passenger.\n- `vip` - Whether the passenger has paid for VIP service during the voyage.\n- `roomservice`, `foodcourt`, `shoppingmall`, `spa`, `vrdeck` - The amount of money the passenger has billed at each of the amenities.\n- `name` - First and last names of the passenger.\n- `transported` - Whether the passenger was transported to another dimension. This is the target variable that we need to predict.","metadata":{}},{"cell_type":"markdown","source":"It makes sense to prepare some of the variables for EDA and modelling. ","metadata":{}},{"cell_type":"code","source":"# create feature of total spendings per passenger\nexpenses = [\"foodcourt\", \"shoppingmall\", \"roomservice\", \"spa\", \"vrdeck\"]\ndf[\"spent_total\"] = df[expenses].sum(axis=1)\n\n# create feature to indicate passengers age 13 and above who can spend\nage_limit_for_spending = 13\ndf[\"can_spend\"] = np.where(df.age >= age_limit_for_spending, True, False)\n\n# split passenger IDs into group ID and number of passenger in group\ndf[[\"group\", \"group_no\"]] = df.passengerid.str.split(\"_\", expand=True)\ndf.group_no = df.group_no.apply(lambda x: int(x))\n\n# create a feature for the size of each group\ntmp = df.groupby(\"group\").group.count()\ndf[\"group_size\"] = df.group.map(dict(tmp))\n\n# calculate total of group spending\ntmp = df.groupby(\"group\").spent_total.sum()\ndf[\"group_spent_total\"] = df.group.map(tmp)\n\n# split cabin string into deck, cabin numer and side of ship\ndf[[\"cabin_deck\", \"cabin_num\", \"cabin_side\"]] = df.cabin.str.split(\"/\", expand=True)\ndf.cabin_num = df.cabin_num.apply(lambda x: int(x) if type(x)==str else np.nan)\n\n# create a feature for cabin size\ntmp = df.cabin.value_counts()\ndf[\"cabin_size\"] = df.cabin.map(tmp)\n\n# bin ages of passengers\nrng = range(0, 85, 5)\ndf[\"age_bins\"] = pd.cut(df.age, rng)\ndf.age_bins = df.age_bins.apply(lambda x: x.left)\n\n# create feature of total spending of passenger\nexpenses = [\"roomservice\", \"foodcourt\", \"shoppingmall\", \"spa\", \"vrdeck\"]\ndf[\"spent_total\"] = df[expenses].sum(axis=1)\n\n# split names in first and last name to identify family relationships\ndf[[\"name_first\", \"name_last\"]] = df.name.str.split(\" \", expand=True)\n\ntrain = df[df.transported.notna()]\ntest = df[df.transported.isna()]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:20.338127Z","iopub.execute_input":"2022-07-24T13:49:20.338543Z","iopub.status.idle":"2022-07-24T13:49:20.707587Z","shell.execute_reply.started":"2022-07-24T13:49:20.338494Z","shell.execute_reply":"2022-07-24T13:49:20.706192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's have a **first check of the data by printing out value counts for all variables.**","metadata":{}},{"cell_type":"code","source":"for c in df.columns:\n    print(c)\n    display(df[c].value_counts())\n    print()","metadata":{"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-07-24T13:49:27.127610Z","iopub.execute_input":"2022-07-24T13:49:27.128397Z","iopub.status.idle":"2022-07-24T13:49:27.318221Z","shell.execute_reply.started":"2022-07-24T13:49:27.128347Z","shell.execute_reply":"2022-07-24T13:49:27.316985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Examining all features one by one","metadata":{}},{"cell_type":"markdown","source":"### `passengerid` and derived group features","metadata":{}},{"cell_type":"code","source":"# Are all passenger IDs unique? Yes, they are...\nassert df.passengerid.value_counts().unique()[0] == 1","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:32.076139Z","iopub.execute_input":"2022-07-24T13:49:32.076788Z","iopub.status.idle":"2022-07-24T13:49:32.089944Z","shell.execute_reply.started":"2022-07-24T13:49:32.076755Z","shell.execute_reply":"2022-07-24T13:49:32.088590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"There are {df.group.nunique():,.0f} distinct passenger groups.\")\nprint(f\"There are {df.passengerid.isna().sum():.0f} missing values.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:33.126495Z","iopub.execute_input":"2022-07-24T13:49:33.127455Z","iopub.status.idle":"2022-07-24T13:49:33.138129Z","shell.execute_reply.started":"2022-07-24T13:49:33.127417Z","shell.execute_reply":"2022-07-24T13:49:33.136587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Groups do not overlap between train and test data.**","metadata":{}},{"cell_type":"code","source":"in_test = set(df[df.transported.isna()].group.unique().tolist())\nnot_in_test = set(df[df.transported.notna()].group.unique().tolist())\nprint(f\"{len(set.intersection(in_test, not_in_test))} groups are both in the train and the test set.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:35.187029Z","iopub.execute_input":"2022-07-24T13:49:35.187863Z","iopub.status.idle":"2022-07-24T13:49:35.204350Z","shell.execute_reply.started":"2022-07-24T13:49:35.187827Z","shell.execute_reply":"2022-07-24T13:49:35.202922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **Passenger groups have sizes between 1 and 8 passengers.**\n- **Groups occupy between 1 and 3 cabins.**\n- More than half of the passengers (55.1%) travel alone. \n- There are 20% of passengers travelling with one companion, 11.6% are in a group of three. ","metadata":{}},{"cell_type":"code","source":"tmp = df.dropna(subset=[\"cabin\"]).groupby(\"group\").cabin.nunique().unique().tolist()\nprint(f\"Groups can occupy between {np.min(tmp)} and {np.max(tmp)} cabins.\\n\")\n\ntmp = df.group_size.value_counts(normalize=True).sort_index()\nprint(\"Percentage of passenger group sizes\")\nprint(tabulate(pd.DataFrame(tmp*100), floatfmt=\".1f\"))\n\ntmp = df.group_size.value_counts()\nplt.figure(figsize=(16, 5))\nsns.barplot(y=tmp.values, x=tmp.index, color=DEFAULT_COLORS[0])\nplt.title(\"Count of passenger group sizes\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.ylabel(\"Count of groups\")\nplt.xlabel(\"Passenger group size\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:49:55.225930Z","iopub.execute_input":"2022-07-24T13:49:55.226313Z","iopub.status.idle":"2022-07-24T13:49:55.699823Z","shell.execute_reply.started":"2022-07-24T13:49:55.226282Z","shell.execute_reply":"2022-07-24T13:49:55.698466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The anomaly has affected single travellers as well as partial and entire groups.","metadata":{}},{"cell_type":"code","source":"single_passengers_affected = df[(df.group_size==1) & (df.transported==1)].shape[0]\n\ng = df[df.transported.notna()].groupby(\"group\")\n\nentire_groups_affected = []\npartial_groups_affected = []\n\nfor n, data in g:\n    if data.group_size.mean()>1:\n        if data.transported.sum() == data.group_size.mean():\n            entire_groups_affected.append((data.group.unique()[0], data.transported.sum()))\n        else:\n            partial_groups_affected.append((data.group.unique()[0], data.transported.sum()))\n            \nentire_groups_affected = pd.DataFrame(entire_groups_affected)\npartial_groups_affected = pd.DataFrame(partial_groups_affected)\n\nprint(\"The anomaly has affected:\")\nprint(f\"- {single_passengers_affected:,.0f} passengers travelling alone.\")\nprint(f\"-   {entire_groups_affected.sum()[1]} passengers in   {len(entire_groups_affected)} groups that were affected entirely.\")\nprint(f\"- {partial_groups_affected.sum()[1]:,.0f} passengers in {len(partial_groups_affected):,.0f} groups that were affected only partially.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:00.325405Z","iopub.execute_input":"2022-07-24T13:50:00.325843Z","iopub.status.idle":"2022-07-24T13:50:02.888192Z","shell.execute_reply.started":"2022-07-24T13:50:00.325810Z","shell.execute_reply":"2022-07-24T13:50:02.886983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assert that we have found all affected passengers\nassert train.transported.sum() == single_passengers_affected + entire_groups_affected.sum()[1] + partial_groups_affected.sum()[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:05.296914Z","iopub.execute_input":"2022-07-24T13:50:05.297642Z","iopub.status.idle":"2022-07-24T13:50:05.306744Z","shell.execute_reply.started":"2022-07-24T13:50:05.297602Z","shell.execute_reply":"2022-07-24T13:50:05.305686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at larger groups we notice that groups can consist of people with several different last names. ","metadata":{}},{"cell_type":"code","source":"tmp = df.group.value_counts() > 6\ndf[df.group.isin(tmp[tmp==True].index)].groupby(\"group\").name_last.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:07.027386Z","iopub.execute_input":"2022-07-24T13:50:07.028622Z","iopub.status.idle":"2022-07-24T13:50:07.051883Z","shell.execute_reply.started":"2022-07-24T13:50:07.028586Z","shell.execute_reply":"2022-07-24T13:50:07.050734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `cabin` and derived features","metadata":{}},{"cell_type":"code","source":"feature = \"cabin\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:09.496894Z","iopub.execute_input":"2022-07-24T13:50:09.497295Z","iopub.status.idle":"2022-07-24T13:50:09.502597Z","shell.execute_reply.started":"2022-07-24T13:50:09.497263Z","shell.execute_reply":"2022-07-24T13:50:09.501121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_missing(feature):\n    print(f\"{df[feature].isna().sum()} missing values for feature «{feature}»\\\n (train: {train[feature].isna().mean()*100:.1f}%, test: {test[feature].isna().mean()*100:.1f}%)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:10.276751Z","iopub.execute_input":"2022-07-24T13:50:10.277810Z","iopub.status.idle":"2022-07-24T13:50:10.283750Z","shell.execute_reply.started":"2022-07-24T13:50:10.277757Z","shell.execute_reply":"2022-07-24T13:50:10.282717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:13.376860Z","iopub.execute_input":"2022-07-24T13:50:13.377247Z","iopub.status.idle":"2022-07-24T13:50:13.387233Z","shell.execute_reply.started":"2022-07-24T13:50:13.377216Z","shell.execute_reply":"2022-07-24T13:50:13.385841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"The ship has {df.cabin_num.nunique():,.0f} unique cabins on {df.cabin_deck.nunique()} unique decks.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:14.765851Z","iopub.execute_input":"2022-07-24T13:50:14.766233Z","iopub.status.idle":"2022-07-24T13:50:14.774481Z","shell.execute_reply.started":"2022-07-24T13:50:14.766201Z","shell.execute_reply":"2022-07-24T13:50:14.773357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Decks range from `A` to `T`. \n- `T` is the smallest deck and `F` is the biggest ain terms of cabin count as well as in passenger count. ","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").cabin_num.nunique().sort_index().to_frame()\ntmp.columns = [\"cabin_count\"]\ntmp[\"passenger_count\"] = df.groupby(\"cabin_deck\").cabin_num.count().sort_index()\n\nprint(tabulate(tmp, headers=\"keys\", floatfmt=\".1f\"))\n\ntmp.plot.bar(figsize=(16,5))\nplt.xticks(rotation=0)\nplt.title(\"Full data: Cabin and passenger count per deck\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xlabel(\"Deck\")\nplt.ylabel(\"Count\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:18.426146Z","iopub.execute_input":"2022-07-24T13:50:18.427144Z","iopub.status.idle":"2022-07-24T13:50:18.939837Z","shell.execute_reply.started":"2022-07-24T13:50:18.427081Z","shell.execute_reply":"2022-07-24T13:50:18.938630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Decks `B` and `C` overindex on transported passengers.**","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").transported.value_counts(normalize=True).unstack()\ntmp = tmp.iloc[:, ::-1]\n\ntmp.plot.bar(stacked=True, figsize=(16,5), color=(DEFAULT_COLORS))\nplt.title(\"Transported vs. not transported passengers on cabin decks\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xticks(rotation=0)\nplt.ylabel(\"Ratio transported vs not transported\")\nplt.xlabel(\"Cabin decks\")\nplt.legend([\"transported\", \"not transported\"], bbox_to_anchor=(1, 1), loc=\"upper left\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:22.326701Z","iopub.execute_input":"2022-07-24T13:50:22.327088Z","iopub.status.idle":"2022-07-24T13:50:22.791621Z","shell.execute_reply.started":"2022-07-24T13:50:22.327056Z","shell.execute_reply":"2022-07-24T13:50:22.790474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By splitting the data between the two sides `P` and `S` we can see that **decks B and C on Side S have a particularly high ratio of affected passengers.**","metadata":{}},{"cell_type":"code","source":"g = df[df.transported.notna()].groupby(\"cabin_deck\")\ntmp = g.transported.mean().sort_index().to_frame()\ntmp[\"Side P\"] = df[(df.transported.notna()) & (df.cabin_side==\"P\")].groupby(\"cabin_deck\").transported.mean().sort_index()\ntmp[\"Side S\"] = df[(df.transported.notna()) & (df.cabin_side==\"S\")].groupby(\"cabin_deck\").transported.mean().sort_index()\n\n# print(tabulate(tmp*100, headers=\"keys\", floatfmt=\".1f\"))\nplt.figure(figsize=(6, 6))\nsns.heatmap(tmp.iloc[:, 1:], vmin=0, vmax=1, cmap=\"Reds\")\nplt.title(\"Ratio of transported passengers per deck and side\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.ylabel(\"Deck\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:26.416969Z","iopub.execute_input":"2022-07-24T13:50:26.417355Z","iopub.status.idle":"2022-07-24T13:50:26.880902Z","shell.execute_reply.started":"2022-07-24T13:50:26.417323Z","shell.execute_reply":"2022-07-24T13:50:26.879606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cabin numbers on all decks start with zero and continuously increase up to the highest cabin number on the respective deck and side.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").cabin_num.unique()\ntmp = tmp.apply(lambda x: sorted(x))\nprint(tmp)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:30.417129Z","iopub.execute_input":"2022-07-24T13:50:30.417714Z","iopub.status.idle":"2022-07-24T13:50:30.437611Z","shell.execute_reply.started":"2022-07-24T13:50:30.417662Z","shell.execute_reply":"2022-07-24T13:50:30.436545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The cabins aren't distributed symetrically. The cabin counts and cabin sizes differ significantly between port and starboard.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby([\"cabin_deck\", \"cabin_side\"]).agg({'cabin_num': ['count', 'max'], \n                                             \"cabin_size\": \"sum\"})\ndisplay(tmp)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:32.048290Z","iopub.execute_input":"2022-07-24T13:50:32.048728Z","iopub.status.idle":"2022-07-24T13:50:32.082207Z","shell.execute_reply.started":"2022-07-24T13:50:32.048696Z","shell.execute_reply":"2022-07-24T13:50:32.081232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There seems to be no obvious pattern in the distribution of the cabins.","metadata":{}},{"cell_type":"code","source":"for deck in np.sort(df.cabin_deck.dropna().unique()):\n    fig, ax = plt.subplots(nrows=2, sharex=True, sharey=True)\n    df[(df.cabin_deck==deck) & (df.cabin_side==\"S\")].sort_values(\"cabin_num\")[[\"cabin_size\"]].T.plot.barh(\n        stacked=True, color=DEFAULT_COLORS[:2], figsize=(16, 2), ax=ax[0])\n    ax[0].legend([])\n    ax[0].set_ylabel(\"Side S\", rotation=0, labelpad=25)\n    ax[0].set_yticks([])\n    df[(df.cabin_deck==deck) & (df.cabin_side==\"P\")].sort_values(\"cabin_num\")[[\"cabin_size\"]].T.plot.barh(\n        stacked=True, color=DEFAULT_COLORS[:2], figsize=(16, 2), ax=ax[1])\n    ax[1].legend([])\n    ax[1].set_ylabel(\"Side P\", rotation=0, labelpad=25)\n    ax[1].set_yticks([])\n    ax[1].set_xlabel(\"Passenger count\")\n    plt.suptitle(f\"Deck {deck}\", size=16, y=1.05)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:50:36.467920Z","iopub.execute_input":"2022-07-24T13:50:36.468314Z","iopub.status.idle":"2022-07-24T13:51:32.284245Z","shell.execute_reply.started":"2022-07-24T13:50:36.468281Z","shell.execute_reply":"2022-07-24T13:51:32.282930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There also don't seem to be any clear patterns present in regard to the effects of the anomaly. Again we can observe that Decks `B` and `C` on side `S` as well as Deck `G` on side `S` were more heavily affected than the rest of the ship.","metadata":{}},{"cell_type":"code","source":"sns.heatmap(df.drop(df[df.cabin_deck==\"T\"].index).groupby([\"cabin_deck\", \"cabin_side\", \"cabin_num\"]).transported.mean().unstack(), cmap=\"Reds\")\nplt.xticks([])\nplt.ylabel(\"Cabin deck – cabin side\")\nplt.xlabel(\"Cabin numbers sorted from 0 to maximum\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:32.287437Z","iopub.execute_input":"2022-07-24T13:51:32.287923Z","iopub.status.idle":"2022-07-24T13:51:34.974948Z","shell.execute_reply.started":"2022-07-24T13:51:32.287877Z","shell.execute_reply":"2022-07-24T13:51:34.971637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Passengers from decks `A`, `B`, `C` and `T` entirely come from Europa. \n- No passengers from Europa inhabit deck `F` and `G`.\n- Passengers on deck `G` all come from Earth.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").homeplanet.value_counts(normalize=True).to_frame()\nprint(tmp)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:40.546800Z","iopub.execute_input":"2022-07-24T13:51:40.547175Z","iopub.status.idle":"2022-07-24T13:51:40.561317Z","shell.execute_reply.started":"2022-07-24T13:51:40.547143Z","shell.execute_reply":"2022-07-24T13:51:40.560026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No passengers cryosleeping are on deck `T`.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").cryosleep.value_counts(normalize=True).to_frame()\nprint(tmp)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:42.198464Z","iopub.execute_input":"2022-07-24T13:51:42.198888Z","iopub.status.idle":"2022-07-24T13:51:42.214287Z","shell.execute_reply.started":"2022-07-24T13:51:42.198856Z","shell.execute_reply":"2022-07-24T13:51:42.213000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No VIP passengers inhabit decks `G` and `T`.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"cabin_deck\").vip.value_counts(normalize=True).to_frame()\nprint(tmp)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:43.535792Z","iopub.execute_input":"2022-07-24T13:51:43.536944Z","iopub.status.idle":"2022-07-24T13:51:43.551482Z","shell.execute_reply.started":"2022-07-24T13:51:43.536886Z","shell.execute_reply":"2022-07-24T13:51:43.550493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again we see that passengers on side starboard were affected more heavily by the anomaly.","metadata":{}},{"cell_type":"code","source":"def compare_transportation(feature):\n    tmp = train.groupby(feature).transported.value_counts(normalize=True).unstack()\n    tmp.columns = [\"not transported\", \"transported\"]\n    print(\"Percentage of passengers\\n\")\n    print(tabulate(tmp*100, floatfmt=\".1f\", headers=\"keys\"))\n    \ncompare_transportation(\"cabin_side\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:44.956774Z","iopub.execute_input":"2022-07-24T13:51:44.957476Z","iopub.status.idle":"2022-07-24T13:51:44.969823Z","shell.execute_reply.started":"2022-07-24T13:51:44.957440Z","shell.execute_reply":"2022-07-24T13:51:44.968728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of the cabin side is almost half/half and similar between train and test set.","metadata":{}},{"cell_type":"code","source":"def compare_distribution(feature):\n    not_in_test = df[df.transported.notna()][feature]\n    in_test = df[df.transported.isna()][feature]\n    tmp = pd.concat([not_in_test.value_counts(normalize=True).sort_index(), \n                   in_test.value_counts(normalize=True).sort_index()], axis=1)\n    tmp.columns = [\"train_data\", \"test_data\"]\n    print(\"Percentage of passengers\")\n    print(tabulate(tmp*100, headers=\"keys\", floatfmt=\".1f\"))\n    \ncompare_distribution(\"cabin_side\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:46.586980Z","iopub.execute_input":"2022-07-24T13:51:46.587401Z","iopub.status.idle":"2022-07-24T13:51:46.611155Z","shell.execute_reply.started":"2022-07-24T13:51:46.587368Z","shell.execute_reply":"2022-07-24T13:51:46.609763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `homeplanet`","metadata":{}},{"cell_type":"code","source":"feature = \"homeplanet\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:48.358139Z","iopub.execute_input":"2022-07-24T13:51:48.358565Z","iopub.status.idle":"2022-07-24T13:51:48.364092Z","shell.execute_reply.started":"2022-07-24T13:51:48.358505Z","shell.execute_reply":"2022-07-24T13:51:48.362584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- More than half of the passengers come from **Earth**. \n- The rest comes from **[Europa](https://en.wikipedia.org/wiki/Europa_(moon))** (which is one of Jupiter's moons) or **Mars**. ","metadata":{}},{"cell_type":"code","source":"print(\"Percentage of passengers\")\nprint(tabulate(df[feature].value_counts(normalize=True).to_frame()*100, floatfmt=\".1f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:49.296401Z","iopub.execute_input":"2022-07-24T13:51:49.297614Z","iopub.status.idle":"2022-07-24T13:51:49.308275Z","shell.execute_reply.started":"2022-07-24T13:51:49.297564Z","shell.execute_reply":"2022-07-24T13:51:49.306903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:49.857471Z","iopub.execute_input":"2022-07-24T13:51:49.858153Z","iopub.status.idle":"2022-07-24T13:51:49.867492Z","shell.execute_reply.started":"2022-07-24T13:51:49.858118Z","shell.execute_reply":"2022-07-24T13:51:49.866433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The anomaly has affected passengers unevenly in regard to their homeplanet.** A lot more travelers from Europa were transported.","metadata":{}},{"cell_type":"code","source":"compare_transportation(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:50.696545Z","iopub.execute_input":"2022-07-24T13:51:50.697230Z","iopub.status.idle":"2022-07-24T13:51:50.708848Z","shell.execute_reply.started":"2022-07-24T13:51:50.697187Z","shell.execute_reply":"2022-07-24T13:51:50.707744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of the feature is similar in the train and the test set.","metadata":{}},{"cell_type":"code","source":"compare_distribution(\"homeplanet\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:51.596535Z","iopub.execute_input":"2022-07-24T13:51:51.597242Z","iopub.status.idle":"2022-07-24T13:51:51.614655Z","shell.execute_reply.started":"2022-07-24T13:51:51.597206Z","shell.execute_reply":"2022-07-24T13:51:51.613693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `destination`","metadata":{}},{"cell_type":"markdown","source":"All three destination planets actually exist! 😃\n\n- Around 70% of all passengers travel **[TRAPPIST-1e](https://en.wikipedia.org/wiki/TRAPPIST-1e)**. The planet is around 40 lightyears away from Earth and potentially inhabitable.\n- Around 21% of the passengers are on their way to **[55 Cancri e](http://www.sci-news.com/astronomy/article00649.html)** which is also around 40 lightyears away. \n- The rest of the passengers – 9.3% – want to travel to **[PSO J318.5-22](https://en.wikipedia.org/wiki/PSO_J318.5%E2%88%9222)** (~80 lightyears away).","metadata":{}},{"cell_type":"code","source":"feature = \"destination\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:52.796445Z","iopub.execute_input":"2022-07-24T13:51:52.797165Z","iopub.status.idle":"2022-07-24T13:51:52.801774Z","shell.execute_reply.started":"2022-07-24T13:51:52.797132Z","shell.execute_reply":"2022-07-24T13:51:52.800538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Percentage of passengers in regard to {feature}\")\nprint(tabulate(df[feature].value_counts(normalize=True).to_frame()*100, floatfmt=\".1f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:52.957068Z","iopub.execute_input":"2022-07-24T13:51:52.957796Z","iopub.status.idle":"2022-07-24T13:51:52.967557Z","shell.execute_reply.started":"2022-07-24T13:51:52.957763Z","shell.execute_reply":"2022-07-24T13:51:52.966588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(\"destination\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:53.836029Z","iopub.execute_input":"2022-07-24T13:51:53.837056Z","iopub.status.idle":"2022-07-24T13:51:53.847190Z","shell.execute_reply.started":"2022-07-24T13:51:53.837020Z","shell.execute_reply":"2022-07-24T13:51:53.845924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The anomaly has affected passengers unevenly in regard to their destination.** Mostly passengers travelling to 55 Cancri e were transported.","metadata":{}},{"cell_type":"code","source":"compare_transportation(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:54.977319Z","iopub.execute_input":"2022-07-24T13:51:54.977974Z","iopub.status.idle":"2022-07-24T13:51:54.989919Z","shell.execute_reply.started":"2022-07-24T13:51:54.977930Z","shell.execute_reply":"2022-07-24T13:51:54.988759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of the feature is similar in the train and the test set.","metadata":{}},{"cell_type":"code","source":"compare_distribution(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:55.495911Z","iopub.execute_input":"2022-07-24T13:51:55.496605Z","iopub.status.idle":"2022-07-24T13:51:55.515780Z","shell.execute_reply.started":"2022-07-24T13:51:55.496556Z","shell.execute_reply":"2022-07-24T13:51:55.514807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `cryosleep`","metadata":{}},{"cell_type":"code","source":"feature = \"cryosleep\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:56.587631Z","iopub.execute_input":"2022-07-24T13:51:56.588231Z","iopub.status.idle":"2022-07-24T13:51:56.592505Z","shell.execute_reply.started":"2022-07-24T13:51:56.588199Z","shell.execute_reply":"2022-07-24T13:51:56.591585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:57.627741Z","iopub.execute_input":"2022-07-24T13:51:57.628434Z","iopub.status.idle":"2022-07-24T13:51:57.639356Z","shell.execute_reply.started":"2022-07-24T13:51:57.628384Z","shell.execute_reply":"2022-07-24T13:51:57.638233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"More than a third of the passengers have been in cryosleep.","metadata":{}},{"cell_type":"code","source":"print(f\"Percentage of passengers in {feature}\")\nprint(tabulate(df[feature].value_counts(normalize=True).to_frame()*100, floatfmt=\".1f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:51:59.317030Z","iopub.execute_input":"2022-07-24T13:51:59.318000Z","iopub.status.idle":"2022-07-24T13:51:59.329456Z","shell.execute_reply.started":"2022-07-24T13:51:59.317950Z","shell.execute_reply":"2022-07-24T13:51:59.328433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"More passengers heading for PSO J318.5-22 are in cryosleep than those heading for the other destinations. This makes sense since it is the planet which is farthest away. ","metadata":{}},{"cell_type":"code","source":"print(tabulate(df.groupby(\"destination\")[feature].value_counts(normalize=True).unstack().sort_values(True, ascending=False)*100, headers=\"keys\", floatfmt=\".1f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:00.096023Z","iopub.execute_input":"2022-07-24T13:52:00.096842Z","iopub.status.idle":"2022-07-24T13:52:00.112388Z","shell.execute_reply.started":"2022-07-24T13:52:00.096791Z","shell.execute_reply":"2022-07-24T13:52:00.111070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Much more cryosleeping passengers were affected by the anomaly.**","metadata":{}},{"cell_type":"code","source":"compare_transportation(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:01.516921Z","iopub.execute_input":"2022-07-24T13:52:01.517673Z","iopub.status.idle":"2022-07-24T13:52:01.529351Z","shell.execute_reply.started":"2022-07-24T13:52:01.517637Z","shell.execute_reply":"2022-07-24T13:52:01.527772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution is similar in the train and the test set.","metadata":{}},{"cell_type":"code","source":"compare_distribution(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:01.886395Z","iopub.execute_input":"2022-07-24T13:52:01.886833Z","iopub.status.idle":"2022-07-24T13:52:01.904561Z","shell.execute_reply.started":"2022-07-24T13:52:01.886801Z","shell.execute_reply":"2022-07-24T13:52:01.903392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `vip`","metadata":{}},{"cell_type":"code","source":"feature = \"vip\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:02.965997Z","iopub.execute_input":"2022-07-24T13:52:02.966782Z","iopub.status.idle":"2022-07-24T13:52:02.971073Z","shell.execute_reply.started":"2022-07-24T13:52:02.966747Z","shell.execute_reply":"2022-07-24T13:52:02.970059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:03.556748Z","iopub.execute_input":"2022-07-24T13:52:03.557619Z","iopub.status.idle":"2022-07-24T13:52:03.569072Z","shell.execute_reply.started":"2022-07-24T13:52:03.557568Z","shell.execute_reply":"2022-07-24T13:52:03.567582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just 2.2% of passengers classify as VIPs.","metadata":{}},{"cell_type":"code","source":"tmp = df.vip.value_counts(normalize=True).to_frame()\nprint(f\"Percentage of passengers labeled as {feature}\")\nprint(tabulate(tmp*100, floatfmt=\".1f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:04.506721Z","iopub.execute_input":"2022-07-24T13:52:04.507535Z","iopub.status.idle":"2022-07-24T13:52:04.518955Z","shell.execute_reply.started":"2022-07-24T13:52:04.507467Z","shell.execute_reply":"2022-07-24T13:52:04.517451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VIP passengers are mainly older than non VIP passengers.","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(data=df, x=\"age\", hue=\"vip\", common_norm=False, palette=DEFAULT_COLORS[:2])\nplt.title(\"Age distribution of normal passengers and VIPs\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:05.766004Z","iopub.execute_input":"2022-07-24T13:52:05.766914Z","iopub.status.idle":"2022-07-24T13:52:06.359262Z","shell.execute_reply.started":"2022-07-24T13:52:05.766874Z","shell.execute_reply":"2022-07-24T13:52:06.357886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The anomaly has affected the VIPs less than the non VIPs.**","metadata":{}},{"cell_type":"code","source":"compare_transportation(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:09.037099Z","iopub.execute_input":"2022-07-24T13:52:09.037548Z","iopub.status.idle":"2022-07-24T13:52:09.049252Z","shell.execute_reply.started":"2022-07-24T13:52:09.037495Z","shell.execute_reply":"2022-07-24T13:52:09.047937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The test data has a little less VIPs than the train data.","metadata":{}},{"cell_type":"code","source":"compare_distribution(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:10.326746Z","iopub.execute_input":"2022-07-24T13:52:10.327622Z","iopub.status.idle":"2022-07-24T13:52:10.347082Z","shell.execute_reply.started":"2022-07-24T13:52:10.327572Z","shell.execute_reply":"2022-07-24T13:52:10.345906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `age`","metadata":{}},{"cell_type":"code","source":"feature = \"age\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:11.826870Z","iopub.execute_input":"2022-07-24T13:52:11.827280Z","iopub.status.idle":"2022-07-24T13:52:11.832664Z","shell.execute_reply.started":"2022-07-24T13:52:11.827246Z","shell.execute_reply":"2022-07-24T13:52:11.831270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:12.316274Z","iopub.execute_input":"2022-07-24T13:52:12.317374Z","iopub.status.idle":"2022-07-24T13:52:12.324452Z","shell.execute_reply.started":"2022-07-24T13:52:12.317335Z","shell.execute_reply":"2022-07-24T13:52:12.323539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Passenger ages range from 0 to 79 years. The average age is 28.8 with a median of 27. ","metadata":{}},{"cell_type":"code","source":"display(df.age.describe().to_frame().style.format(\"{:.1f}\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:13.028938Z","iopub.execute_input":"2022-07-24T13:52:13.029579Z","iopub.status.idle":"2022-07-24T13:52:13.125475Z","shell.execute_reply.started":"2022-07-24T13:52:13.029532Z","shell.execute_reply":"2022-07-24T13:52:13.124293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We seem to have some outliers for `age`.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,3))\nsns.boxplot(data=df.age, orient=\"horizontal\")\nplt.title(\"Age distribution of passengers\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xlabel(\"Age of passengers\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:13.975818Z","iopub.execute_input":"2022-07-24T13:52:13.976230Z","iopub.status.idle":"2022-07-24T13:52:14.271587Z","shell.execute_reply.started":"2022-07-24T13:52:13.976198Z","shell.execute_reply":"2022-07-24T13:52:14.268600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature `age` is not normally distributed. There seem to be quite a few young children on board and a long tail of older passengers.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,5))\nsns.kdeplot(df.age, palette=DEFAULT_COLORS)\nplt.title(\"Age distribution of passengers\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.ylabel(\"Age of passengers\")\nplt.xlabel(\"Count of passengers\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:15.094182Z","iopub.execute_input":"2022-07-24T13:52:15.094606Z","iopub.status.idle":"2022-07-24T13:52:15.527465Z","shell.execute_reply.started":"2022-07-24T13:52:15.094574Z","shell.execute_reply":"2022-07-24T13:52:15.526280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The anomaly has affected particularly younger passengers.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=2, figsize=(16,4), sharex=True)\nsns.boxplot(data=df[df.transported==True], x=\"age\", orient=\"horizontal\", ax=ax[0])\nax[0].set_xlabel(\"Age of transported passengers\")\nsns.boxplot(data=df[df.transported==False], x=\"age\", orient=\"horizontal\", ax=ax[1])\nax[1].set_xlabel(\"Age of not transported passengers\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:17.247650Z","iopub.execute_input":"2022-07-24T13:52:17.248079Z","iopub.status.idle":"2022-07-24T13:52:17.648083Z","shell.execute_reply.started":"2022-07-24T13:52:17.248046Z","shell.execute_reply":"2022-07-24T13:52:17.646620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(data=df, x=\"age\", hue=\"transported\", palette=[\"grey\", \"red\"])\nplt.title(\"Age distribution of passengers\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xlim(0, 80)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:18.917932Z","iopub.execute_input":"2022-07-24T13:52:18.918335Z","iopub.status.idle":"2022-07-24T13:52:19.403928Z","shell.execute_reply.started":"2022-07-24T13:52:18.918303Z","shell.execute_reply":"2022-07-24T13:52:19.402538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By binning in age groups we observe the same phenomenon: Younger passengers were affected more heavily by the anomaly and the oldest passengers less so.","metadata":{}},{"cell_type":"code","source":"tmp = df.groupby(\"age_bins\").transported.value_counts(normalize=True).unstack()\ntmp = tmp.iloc[:, ::-1]\n\ntmp.plot.bar(stacked=True, figsize=(16,6))\nplt.title(\"Transported vs not transported in age groups\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xticks(rotation=0)\nplt.ylabel(\"Ratio transported vs not transported\")\nplt.xlabel(\"Age groups\")\nplt.legend([\"transported\", \"not transported\"], bbox_to_anchor=(1, 1), loc=\"upper left\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:19.876305Z","iopub.execute_input":"2022-07-24T13:52:19.877557Z","iopub.status.idle":"2022-07-24T13:52:20.347206Z","shell.execute_reply.started":"2022-07-24T13:52:19.877499Z","shell.execute_reply":"2022-07-24T13:52:20.345908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Passengers from Earth are younger than those form Mars and Europa.","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(data=df, x=\"age\", hue=\"homeplanet\", \n            common_norm=False, \n            palette=DEFAULT_COLORS[:3])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:22.605766Z","iopub.execute_input":"2022-07-24T13:52:22.606153Z","iopub.status.idle":"2022-07-24T13:52:23.008919Z","shell.execute_reply.started":"2022-07-24T13:52:22.606122Z","shell.execute_reply":"2022-07-24T13:52:23.007361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of `age` is similar between train and test data.","metadata":{}},{"cell_type":"code","source":"sns.kdeplot(data=train, x=\"age\", common_norm=False, palette=DEFAULT_COLORS[0])\nsns.kdeplot(data=test, x=\"age\", common_norm=False, palette=DEFAULT_COLORS[1])\nplt.legend([\"train\", \"test\"])\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:25.977126Z","iopub.execute_input":"2022-07-24T13:52:25.977575Z","iopub.status.idle":"2022-07-24T13:52:26.391936Z","shell.execute_reply.started":"2022-07-24T13:52:25.977542Z","shell.execute_reply":"2022-07-24T13:52:26.390590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Expenses","metadata":{}},{"cell_type":"code","source":"for feature in expenses:\n    get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:26.896368Z","iopub.execute_input":"2022-07-24T13:52:26.897219Z","iopub.status.idle":"2022-07-24T13:52:26.911940Z","shell.execute_reply.started":"2022-07-24T13:52:26.897176Z","shell.execute_reply":"2022-07-24T13:52:26.910663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution of expenses is quite similar to each other. However, the values aren't normally distributed but rather skewed.","metadata":{}},{"cell_type":"code","source":"results = []\nfor expense in expenses:\n    results.append(df[expense].describe())\ntmp = pd.concat(results, axis=1)\nprint(tabulate(tmp, headers=\"keys\", floatfmt=\",.0f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:28.476985Z","iopub.execute_input":"2022-07-24T13:52:28.477380Z","iopub.status.idle":"2022-07-24T13:52:28.505390Z","shell.execute_reply.started":"2022-07-24T13:52:28.477348Z","shell.execute_reply":"2022-07-24T13:52:28.504262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Passengers have spent most of their money for food and the least amount in the shopping mall.","metadata":{}},{"cell_type":"code","source":"print(tabulate(df[expenses].sum().sort_values(ascending=False).to_frame(), floatfmt=\",.0f\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:29.376317Z","iopub.execute_input":"2022-07-24T13:52:29.376721Z","iopub.status.idle":"2022-07-24T13:52:29.385861Z","shell.execute_reply.started":"2022-07-24T13:52:29.376690Z","shell.execute_reply":"2022-07-24T13:52:29.384957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distributions of spendings in the various categories appear similar (all samples with a value of zero removed).","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows=len(expenses), sharex=True, figsize=(16, 8))\n\nfor idx, expense in enumerate(expenses):\n    tmp = df[df[expense]>0]\n    sns.boxplot(data=tmp, x=expense, ax=ax[idx])\nplt.suptitle(\"Distribution of spendings, log scaled\", size=TITLE_SIZE)\nplt.xscale('log') \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:30.206377Z","iopub.execute_input":"2022-07-24T13:52:30.207330Z","iopub.status.idle":"2022-07-24T13:52:33.872086Z","shell.execute_reply.started":"2022-07-24T13:52:30.207295Z","shell.execute_reply":"2022-07-24T13:52:33.871018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From age 15 onward the distribution of total spendings is quite similar without any obvious pattern.","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=df, y=\"spent_total\", x=\"age_bins\", color=DEFAULT_COLORS[0])\nplt.yscale(\"log\")\nplt.title(\"Distribution of total spendings per age group of passengers, log scaled\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xlabel(\"Passenger age range\")\nplt.ylabel(\"Total spending\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:37.387457Z","iopub.execute_input":"2022-07-24T13:52:37.388007Z","iopub.status.idle":"2022-07-24T13:52:38.362193Z","shell.execute_reply.started":"2022-07-24T13:52:37.387970Z","shell.execute_reply":"2022-07-24T13:52:38.360877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"However, if we look at the total spending in the age group we see that the oldest passengers have spent the most money.","metadata":{}},{"cell_type":"code","source":"df.groupby(\"age_bins\").spent_total.mean().plot.bar()\nplt.xticks(rotation=0)\nplt.title(\"Total sum of spendings per age group\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.xlabel(\"Passenger age groups\")\nplt.ylabel(\"Total spending\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:38.546224Z","iopub.execute_input":"2022-07-24T13:52:38.546765Z","iopub.status.idle":"2022-07-24T13:52:39.061677Z","shell.execute_reply.started":"2022-07-24T13:52:38.546723Z","shell.execute_reply":"2022-07-24T13:52:39.060419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `name` and derived features","metadata":{}},{"cell_type":"code","source":"feature = \"name\"","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:39.866978Z","iopub.execute_input":"2022-07-24T13:52:39.867805Z","iopub.status.idle":"2022-07-24T13:52:39.873047Z","shell.execute_reply.started":"2022-07-24T13:52:39.867760Z","shell.execute_reply":"2022-07-24T13:52:39.871935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_missing(feature)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:40.576241Z","iopub.execute_input":"2022-07-24T13:52:40.577054Z","iopub.status.idle":"2022-07-24T13:52:40.587743Z","shell.execute_reply.started":"2022-07-24T13:52:40.577006Z","shell.execute_reply":"2022-07-24T13:52:40.586720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"same_name = np.sum(df.name.value_counts()==2)\nprint(f\"There are {same_name} passengers with the same name in the data.\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:40.766560Z","iopub.execute_input":"2022-07-24T13:52:40.767686Z","iopub.status.idle":"2022-07-24T13:52:40.781954Z","shell.execute_reply.started":"2022-07-24T13:52:40.767648Z","shell.execute_reply":"2022-07-24T13:52:40.780552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Several last names are very frequent in the data.","metadata":{}},{"cell_type":"code","source":"print(df.name_last.value_counts()[:15])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:41.825880Z","iopub.execute_input":"2022-07-24T13:52:41.826631Z","iopub.status.idle":"2022-07-24T13:52:41.836813Z","shell.execute_reply.started":"2022-07-24T13:52:41.826597Z","shell.execute_reply":"2022-07-24T13:52:41.835700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### `transported`","metadata":{}},{"cell_type":"markdown","source":"We almost have class balance in our target variable.","metadata":{}},{"cell_type":"code","source":"print(df.transported.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:42.956346Z","iopub.execute_input":"2022-07-24T13:52:42.957309Z","iopub.status.idle":"2022-07-24T13:52:42.964990Z","shell.execute_reply.started":"2022-07-24T13:52:42.957270Z","shell.execute_reply":"2022-07-24T13:52:42.964030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlations to the target variable**","metadata":{}},{"cell_type":"code","source":"tmp = df.transported.dropna().astype(int)\ncorr = df.corrwith(tmp, drop=True)\ncorr.sort_values(inplace=True, ascending=False)\ncorrelation_treshold = 0.3 # lower bound for moderate correlation\n\ncorr.plot.barh(figsize=(12,8))\nplt.title(f\"Correlation of features to target variable «transported»\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.axvline(-correlation_treshold, color=\"grey\")\nplt.axvline(correlation_treshold, color=\"grey\")\nplt.xlabel(f\"Correlation to target (grey lines indicate +-{correlation_treshold} correlation threshold)\")\nplt.ylabel(\"Features\")\nplt.xlim(-0.6, 0.6)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:52:43.905778Z","iopub.execute_input":"2022-07-24T13:52:43.906460Z","iopub.status.idle":"2022-07-24T13:52:44.381952Z","shell.execute_reply.started":"2022-07-24T13:52:43.906427Z","shell.execute_reply":"2022-07-24T13:52:44.380707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification","metadata":{}},{"cell_type":"markdown","source":"Prepare data and setup preprocessing pipeline","metadata":{}},{"cell_type":"code","source":"def create_preprocess(nums, cats_ohe, cats_bool):\n    return ColumnTransformer(\n                        [(\"imputer_num\", make_pipeline(SimpleImputer(strategy=\"mean\"),), \n                                                       nums),\n                         (\"imputer_cat_ohe\", make_pipeline(SimpleImputer(strategy=\"most_frequent\"), \n                                                       OneHotEncoder(handle_unknown=\"ignore\")), \n                                                       cats_ohe),\n                         (\"imputer_cat_bool\", make_pipeline(SimpleImputer(strategy=\"most_frequent\")), \n                                                       cats_bool),\n                        ], remainder=\"passthrough\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:00.086735Z","iopub.execute_input":"2022-07-24T13:53:00.087165Z","iopub.status.idle":"2022-07-24T13:53:00.094642Z","shell.execute_reply.started":"2022-07-24T13:53:00.087124Z","shell.execute_reply":"2022-07-24T13:53:00.093047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nums =['age', \n       'roomservice', 'foodcourt', 'shoppingmall', 'spa', 'vrdeck', \n       'spent_total', \"group_size\"]\n\ncats_ohe = ['homeplanet', 'destination', 'cabin_deck', 'cabin_side']\ncats_bool = ['cryosleep', \"vip\"]\nall_cols = [*nums, *cats_ohe, *cats_bool]\n\nX_train = train[all_cols]\ny_train = train.transported.astype(int)\nX_test = test[all_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:01.696344Z","iopub.execute_input":"2022-07-24T13:53:01.697342Z","iopub.status.idle":"2022-07-24T13:53:01.708246Z","shell.execute_reply.started":"2022-07-24T13:53:01.697304Z","shell.execute_reply":"2022-07-24T13:53:01.707049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Baseline","metadata":{}},{"cell_type":"code","source":"classifiers = [\n    DummyClassifier(),\n    LogisticRegression(max_iter=1e4, n_jobs=-1),\n    RidgeClassifier(max_iter=1e4),\n    LinearDiscriminantAnalysis(),\n    KNeighborsClassifier(n_jobs=-1),\n    SVC(kernel=\"rbf\"),\n    RandomForestClassifier(n_jobs=-1),\n    lgb.LGBMClassifier(n_jobs=-1)\n]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:04.688110Z","iopub.execute_input":"2022-07-24T13:53:04.688531Z","iopub.status.idle":"2022-07-24T13:53:04.694927Z","shell.execute_reply.started":"2022-07-24T13:53:04.688485Z","shell.execute_reply":"2022-07-24T13:53:04.693912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nbaseline_results = []\nfor clf in tqdm(classifiers):\n    pipe = make_pipeline(create_preprocess(nums, cats_ohe, cats_bool), clf)\n    score = np.mean(cross_val_score(pipe, \n                                    X_train, \n                                    y_train, \n                                    scoring=\"accuracy\", \n                                    cv=5,\n                                    n_jobs=-1,\n                                   ))\n    baseline_results.append((clf, np.round(score, 4)))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-24T13:53:12.186332Z","iopub.execute_input":"2022-07-24T13:53:12.187168Z","iopub.status.idle":"2022-07-24T13:53:36.540164Z","shell.execute_reply.started":"2022-07-24T13:53:12.187127Z","shell.execute_reply":"2022-07-24T13:53:36.538406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame(baseline_results, columns=[\"clf\", \"accuracy\"])\nresults.clf = results.clf.apply(lambda x: str(x).split(\"(\")[0])\ndisplay(results.sort_values(\"accuracy\", ascending=False).style.hide_index())","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:51.825816Z","iopub.execute_input":"2022-07-24T13:53:51.826229Z","iopub.status.idle":"2022-07-24T13:53:51.842368Z","shell.execute_reply.started":"2022-07-24T13:53:51.826195Z","shell.execute_reply":"2022-07-24T13:53:51.841488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = lgb.LGBMClassifier(n_jobs=-1)\npipe = make_pipeline(create_preprocess(nums, cats_ohe, cats_bool), clf)\n_ = pipe.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:55.956527Z","iopub.execute_input":"2022-07-24T13:53:55.956976Z","iopub.status.idle":"2022-07-24T13:53:56.236563Z","shell.execute_reply.started":"2022-07-24T13:53:55.956942Z","shell.execute_reply":"2022-07-24T13:53:56.235142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_submission(pipe=pipe, file_name=\"submission.csv\"):\n    labels = pipe.predict(X_test)\n    labels = [True if x==1 else False for x in labels]\n    submission = df[df.transported.isna()][\"passengerid\"].to_frame()\n    submission[\"labels\"] = labels\n    submission.columns = [\"PassengerId\", \"Transported\"]\n    submission.to_csv(file_name, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:53:59.656016Z","iopub.execute_input":"2022-07-24T13:53:59.656411Z","iopub.status.idle":"2022-07-24T13:53:59.664696Z","shell.execute_reply.started":"2022-07-24T13:53:59.656378Z","shell.execute_reply":"2022-07-24T13:53:59.663392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first baseline submission with LGBM yields **0.80219** on the leaderboard.","metadata":{}},{"cell_type":"code","source":"make_submission(pipe)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:54:02.859257Z","iopub.execute_input":"2022-07-24T13:54:02.860100Z","iopub.status.idle":"2022-07-24T13:54:02.926286Z","shell.execute_reply.started":"2022-07-24T13:54:02.860064Z","shell.execute_reply":"2022-07-24T13:54:02.925083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Optimize hyperparameters","metadata":{}},{"cell_type":"markdown","source":"Reduce to best set of features for modelling.","metadata":{}},{"cell_type":"code","source":"nums =['age', \"group_size\", \"cabin_size\",\n       'roomservice', 'foodcourt', 'shoppingmall', 'spa', 'vrdeck', \n       \"spent_total\", ]\n\ncats_ohe = ['homeplanet', 'destination', \"cabin_deck\", \"cabin_side\", \"age_bins\"]\ncats_bool = [\"cryosleep\", \"vip\", \"can_spend\"]\n\nall_cols = [*nums, *cats_ohe, *cats_bool]\n\nX_train = df[df.transported.notna()][all_cols]\ny_train = df[df.transported.notna()].transported.astype(int)\nX_test = df[df.transported.isna()][all_cols]","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:54:22.890658Z","iopub.execute_input":"2022-07-24T13:54:22.891078Z","iopub.status.idle":"2022-07-24T13:54:22.916986Z","shell.execute_reply.started":"2022-07-24T13:54:22.891046Z","shell.execute_reply":"2022-07-24T13:54:22.915774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Grid search optimal hyperparameters","metadata":{}},{"cell_type":"code","source":"clf = lgb.LGBMClassifier(n_jobs=-1)\npipe = make_pipeline(create_preprocess(nums, cats_ohe, cats_bool), clf)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T13:54:25.655976Z","iopub.execute_input":"2022-07-24T13:54:25.656404Z","iopub.status.idle":"2022-07-24T13:54:25.662687Z","shell.execute_reply.started":"2022-07-24T13:54:25.656369Z","shell.execute_reply":"2022-07-24T13:54:25.661689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n%%time\nsearch_space = dict(lgbmclassifier__learning_rate=np.logspace(-2.0, -1.2, 10),\n                    lgbmclassifier__n_estimators=np.linspace(350, 500, 11).astype(int), \n                    lgbmclassifier__num_leaves=np.linspace(20, 30, 11).astype(int),\n                   )\n\ngrd_search = GridSearchCV(pipe, \n                          search_space, \n                          scoring=\"accuracy\",\n                          cv=5,\n                          n_jobs=-1)\n\n_ = grd_search.fit(X_train, y_train)\n'''","metadata":{"tags":[],"execution":{"iopub.status.busy":"2022-07-24T14:01:14.936121Z","iopub.status.idle":"2022-07-24T14:01:14.936533Z","shell.execute_reply.started":"2022-07-24T14:01:14.936315Z","shell.execute_reply":"2022-07-24T14:01:14.936335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nprint(f\"Best accuracy from grid search: {grd_search.best_score_:.3f}\")\nprint(\"Best found hyperparameters:\")\nprint(grd_search.best_params_)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-24T14:01:14.938505Z","iopub.status.idle":"2022-07-24T14:01:14.939161Z","shell.execute_reply.started":"2022-07-24T14:01:14.938820Z","shell.execute_reply":"2022-07-24T14:01:14.938851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some feature engineering and optimized hyperparameters yield **0.80313** on the leader board.","metadata":{}},{"cell_type":"code","source":"clf = lgb.LGBMClassifier(num_leaves=24, n_estimators=395, learning_rate=0.015, n_jobs=-1)\n\npipe = make_pipeline(create_preprocess(nums, cats_ohe, cats_bool), clf)\nprint(f\"{np.mean(cross_val_score(pipe, X_train, y_train)):.4f}\")\n_ = pipe.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T14:06:29.158744Z","iopub.execute_input":"2022-07-24T14:06:29.161072Z","iopub.status.idle":"2022-07-24T14:06:36.219036Z","shell.execute_reply.started":"2022-07-24T14:06:29.161021Z","shell.execute_reply":"2022-07-24T14:06:36.217618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"make_submission(pipe, file_name=\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-24T14:06:36.221963Z","iopub.execute_input":"2022-07-24T14:06:36.222554Z","iopub.status.idle":"2022-07-24T14:06:36.344127Z","shell.execute_reply.started":"2022-07-24T14:06:36.222470Z","shell.execute_reply":"2022-07-24T14:06:36.343014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Permutation importance","metadata":{}},{"cell_type":"code","source":"%%time\nresult = permutation_importance(pipe, \n                                X_train, y_train,\n                                scoring=\"accuracy\", \n                                n_repeats=10,\n                                random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T14:01:22.669148Z","iopub.execute_input":"2022-07-24T14:01:22.669572Z","iopub.status.idle":"2022-07-24T14:01:53.638082Z","shell.execute_reply.started":"2022-07-24T14:01:22.669538Z","shell.execute_reply":"2022-07-24T14:01:53.636586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"By analysing the feature importance with permutation we see that `spent_total`, `foodcourt`, `cryosleep`, `spa` and `vrdeck` rank high as salient features.","metadata":{}},{"cell_type":"code","source":"df_results = pd.DataFrame(result[\"importances_mean\"])\ndf_results.columns = [\"importance_mean\"]\ndf_results.sort_values(by=\"importance_mean\", ascending=True, inplace=True)\nfeature_labels = X_train.columns[df_results.index]\n\ndf_results.plot.barh(figsize=(12,8))\nplt.yticks(ticks=range(0,len(df_results)), labels=feature_labels)\nplt.title(\"Permutation importance of features\", size=TITLE_SIZE, pad=TITLE_PAD)\nplt.ylabel(\"Features, importance mean > 0.005\")\nplt.xlabel(\"Contribution of feature to accuracy\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T14:01:53.640938Z","iopub.execute_input":"2022-07-24T14:01:53.641464Z","iopub.status.idle":"2022-07-24T14:01:54.215058Z","shell.execute_reply.started":"2022-07-24T14:01:53.641414Z","shell.execute_reply":"2022-07-24T14:01:54.213860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}