{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# this notebook is improved before \nhttps://www.kaggle.com/code/sinjeongyeol/spaceshiptitanic-binaryclassification-1/notebook\nnotebook","metadata":{}},{"cell_type":"markdown","source":"## plus skill\n### Target encoding with smoothing\n### Add feature\n### optimize hyperparameter\n### stack","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T03:41:36.234184Z","iopub.execute_input":"2022-07-12T03:41:36.234742Z","iopub.status.idle":"2022-07-12T03:41:36.276295Z","shell.execute_reply.started":"2022-07-12T03:41:36.234570Z","shell.execute_reply":"2022-07-12T03:41:36.274986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntest = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\nsubmission = pd.read_csv('/kaggle/input/spaceship-titanic/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.278646Z","iopub.execute_input":"2022-07-12T03:41:36.279112Z","iopub.status.idle":"2022-07-12T03:41:36.371491Z","shell.execute_reply.started":"2022-07-12T03:41:36.279076Z","shell.execute_reply":"2022-07-12T03:41:36.370248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.373628Z","iopub.execute_input":"2022-07-12T03:41:36.374074Z","iopub.status.idle":"2022-07-12T03:41:36.415648Z","shell.execute_reply.started":"2022-07-12T03:41:36.374030Z","shell.execute_reply":"2022-07-12T03:41:36.414166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_cat = []\nfeats_num = []\nfor col in train.columns:\n    if col in ['PassengerId', 'Transported']:\n        continue\n    if train[col].dtype == 'object':\n        feats_cat += [col]\n    else:\n        feats_num += [col]\n        \nfeats_cat, feats_num","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.417828Z","iopub.execute_input":"2022-07-12T03:41:36.418395Z","iopub.status.idle":"2022-07-12T03:41:36.428646Z","shell.execute_reply.started":"2022-07-12T03:41:36.418333Z","shell.execute_reply":"2022-07-12T03:41:36.427836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_len = len(train)\ndataset = pd.concat([train, test], axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.429890Z","iopub.execute_input":"2022-07-12T03:41:36.430419Z","iopub.status.idle":"2022-07-12T03:41:36.458031Z","shell.execute_reply.started":"2022-07-12T03:41:36.430384Z","shell.execute_reply":"2022-07-12T03:41:36.456901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.shape, train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.459459Z","iopub.execute_input":"2022-07-12T03:41:36.459926Z","iopub.status.idle":"2022-07-12T03:41:36.466823Z","shell.execute_reply.started":"2022-07-12T03:41:36.459894Z","shell.execute_reply":"2022-07-12T03:41:36.465606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Transported'] = pd.to_numeric(dataset['Transported'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.468218Z","iopub.execute_input":"2022-07-12T03:41:36.469292Z","iopub.status.idle":"2022-07-12T03:41:36.480242Z","shell.execute_reply.started":"2022-07-12T03:41:36.469235Z","shell.execute_reply":"2022-07-12T03:41:36.479398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### PassengerId\n### Add group_cnt, group_num","metadata":{}},{"cell_type":"code","source":"train['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.481597Z","iopub.execute_input":"2022-07-12T03:41:36.482859Z","iopub.status.idle":"2022-07-12T03:41:36.491260Z","shell.execute_reply.started":"2022-07-12T03:41:36.482820Z","shell.execute_reply":"2022-07-12T03:41:36.490446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['group_id'] = 0\ndataset['group_num'] = 0\nfor i in range(len(dataset)):\n    group_id, id_at_group = map(int, dataset.iloc[i]['PassengerId'].split('_'))\n    dataset.loc[i, 'group_id'] = group_id\n    dataset.loc[i, 'group_num'] = id_at_group","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:36.492721Z","iopub.execute_input":"2022-07-12T03:41:36.494026Z","iopub.status.idle":"2022-07-12T03:41:44.735376Z","shell.execute_reply.started":"2022-07-12T03:41:36.493973Z","shell.execute_reply":"2022-07-12T03:41:44.734284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = pd.merge(dataset, dataset[['group_id', 'PassengerId']].groupby('group_id').count().reset_index().rename(columns={'PassengerId': 'group_cnt'}), how='left', on='group_id')\ndataset = dataset.drop('group_id', axis=1)\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:44.740155Z","iopub.execute_input":"2022-07-12T03:41:44.740957Z","iopub.status.idle":"2022-07-12T03:41:44.801307Z","shell.execute_reply.started":"2022-07-12T03:41:44.740908Z","shell.execute_reply":"2022-07-12T03:41:44.800258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_num += ['group_num', 'group_cnt']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:44.802949Z","iopub.execute_input":"2022-07-12T03:41:44.803602Z","iopub.status.idle":"2022-07-12T03:41:44.809062Z","shell.execute_reply.started":"2022-07-12T03:41:44.803559Z","shell.execute_reply":"2022-07-12T03:41:44.808275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\ncorr = dataset[feats_num].corr()\nsns.heatmap(corr, xticklabels=corr.columns, yticklabels=corr.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:44.810215Z","iopub.execute_input":"2022-07-12T03:41:44.810712Z","iopub.status.idle":"2022-07-12T03:41:45.798153Z","shell.execute_reply.started":"2022-07-12T03:41:44.810681Z","shell.execute_reply":"2022-07-12T03:41:45.797116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig, ax = plt.subplots(1, 2, figsize=(10, 6))\ndataset[dataset['Transported'].notnull()][['group_cnt', 'Transported']].groupby('group_cnt').mean().plot.bar(ax=ax[0])\nsns.countplot(x='group_cnt', data=dataset[dataset['Transported'].notnull()], hue='Transported', ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:45.799582Z","iopub.execute_input":"2022-07-12T03:41:45.799909Z","iopub.status.idle":"2022-07-12T03:41:46.178069Z","shell.execute_reply.started":"2022-07-12T03:41:45.799879Z","shell.execute_reply":"2022-07-12T03:41:46.176867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(10, 6))\ndataset[dataset['Transported'].notnull()][['group_num', 'Transported']].groupby('group_num').mean().plot.bar(ax=ax[0])\nsns.countplot(x='group_num', data=dataset[dataset['Transported'].notnull()], hue='Transported', ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:46.179779Z","iopub.execute_input":"2022-07-12T03:41:46.180079Z","iopub.status.idle":"2022-07-12T03:41:46.550300Z","shell.execute_reply.started":"2022-07-12T03:41:46.180052Z","shell.execute_reply":"2022-07-12T03:41:46.549166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### I'll add this features and decide whether to add it or not after see feature importance ","metadata":{}},{"cell_type":"code","source":"feats_num, feats_cat","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:46.551820Z","iopub.execute_input":"2022-07-12T03:41:46.552246Z","iopub.status.idle":"2022-07-12T03:41:46.562706Z","shell.execute_reply.started":"2022-07-12T03:41:46.552204Z","shell.execute_reply":"2022-07-12T03:41:46.561423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cabin\n### The cabin number where the passenger is staying. Takes the form deck/num/side, where side can be either P for Port or S for Starboard.","metadata":{}},{"cell_type":"code","source":"dataset['Cabin']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:46.564645Z","iopub.execute_input":"2022-07-12T03:41:46.565102Z","iopub.status.idle":"2022-07-12T03:41:46.578605Z","shell.execute_reply.started":"2022-07-12T03:41:46.565058Z","shell.execute_reply":"2022-07-12T03:41:46.577425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Cabin'].isnull().mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:46.579896Z","iopub.execute_input":"2022-07-12T03:41:46.580702Z","iopub.status.idle":"2022-07-12T03:41:46.589464Z","shell.execute_reply.started":"2022-07-12T03:41:46.580671Z","shell.execute_reply":"2022-07-12T03:41:46.588595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Cabin_deck'] = np.NaN\ndataset['Cabin_num'] = np.NaN\ndataset['Cabin_side'] = np.NaN\nfor i in range(len(dataset)):\n    if pd.isna(dataset.iloc[i]['Cabin']):\n        continue\n    deck, num, side = dataset.iloc[i]['Cabin'].split('/')\n    dataset.loc[i, 'Cabin_deck'] = deck\n    dataset.loc[i, 'Cabin_num'] = int(num)\n    dataset.loc[i, 'Cabin_side'] = side","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:41:46.590790Z","iopub.execute_input":"2022-07-12T03:41:46.591096Z","iopub.status.idle":"2022-07-12T03:42:02.727533Z","shell.execute_reply.started":"2022-07-12T03:41:46.591068Z","shell.execute_reply":"2022-07-12T03:42:02.726353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:02.728780Z","iopub.execute_input":"2022-07-12T03:42:02.729086Z","iopub.status.idle":"2022-07-12T03:42:02.752193Z","shell.execute_reply.started":"2022-07-12T03:42:02.729058Z","shell.execute_reply":"2022-07-12T03:42:02.751501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Cabin_deck'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:02.753239Z","iopub.execute_input":"2022-07-12T03:42:02.753544Z","iopub.status.idle":"2022-07-12T03:42:02.761500Z","shell.execute_reply.started":"2022-07-12T03:42:02.753516Z","shell.execute_reply":"2022-07-12T03:42:02.760624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(10, 6))\ndataset[dataset['Transported'].notnull()][['Cabin_deck', 'Transported']].groupby('Cabin_deck').mean().plot.bar(ax=ax[0])\nsns.countplot(x='Cabin_deck', data=dataset[dataset['Transported'].notnull()], hue='Transported', ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:02.762582Z","iopub.execute_input":"2022-07-12T03:42:02.763055Z","iopub.status.idle":"2022-07-12T03:42:03.134564Z","shell.execute_reply.started":"2022-07-12T03:42:02.763023Z","shell.execute_reply":"2022-07-12T03:42:03.133315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### There is very little 'T' value of 'Cabin_deck' feature\n### I'll decide whether to add this or not later","metadata":{"execution":{"iopub.status.busy":"2022-07-01T07:42:10.422796Z","iopub.execute_input":"2022-07-01T07:42:10.423137Z","iopub.status.idle":"2022-07-01T07:42:10.432778Z","shell.execute_reply.started":"2022-07-01T07:42:10.423108Z","shell.execute_reply":"2022-07-01T07:42:10.43191Z"}}},{"cell_type":"code","source":"dataset['Cabin_num'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.136150Z","iopub.execute_input":"2022-07-12T03:42:03.136933Z","iopub.status.idle":"2022-07-12T03:42:03.148120Z","shell.execute_reply.started":"2022-07-12T03:42:03.136891Z","shell.execute_reply":"2022-07-12T03:42:03.146932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(x='Cabin_num', data=dataset)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.153613Z","iopub.execute_input":"2022-07-12T03:42:03.155657Z","iopub.status.idle":"2022-07-12T03:42:03.421376Z","shell.execute_reply.started":"2022-07-12T03:42:03.155608Z","shell.execute_reply":"2022-07-12T03:42:03.420563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(x='Cabin_num', data=dataset[dataset['Transported'].notnull()], hue='Transported')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.422612Z","iopub.execute_input":"2022-07-12T03:42:03.423334Z","iopub.status.idle":"2022-07-12T03:42:03.665749Z","shell.execute_reply.started":"2022-07-12T03:42:03.423291Z","shell.execute_reply":"2022-07-12T03:42:03.664567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset['Cabin_side'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.667247Z","iopub.execute_input":"2022-07-12T03:42:03.667689Z","iopub.status.idle":"2022-07-12T03:42:03.678207Z","shell.execute_reply.started":"2022-07-12T03:42:03.667647Z","shell.execute_reply":"2022-07-12T03:42:03.676893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(10, 6))\ndataset[dataset['Transported'].notnull()][['Cabin_side', 'Transported']].groupby('Cabin_side').mean().plot.bar(ax=ax[0])\nsns.countplot(x='Cabin_side', data=dataset[dataset['Transported'].notnull()], hue='Transported', ax=ax[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.679547Z","iopub.execute_input":"2022-07-12T03:42:03.679865Z","iopub.status.idle":"2022-07-12T03:42:03.965796Z","shell.execute_reply.started":"2022-07-12T03:42:03.679837Z","shell.execute_reply":"2022-07-12T03:42:03.964693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.drop('Cabin', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.967597Z","iopub.execute_input":"2022-07-12T03:42:03.968035Z","iopub.status.idle":"2022-07-12T03:42:03.975753Z","shell.execute_reply.started":"2022-07-12T03:42:03.967993Z","shell.execute_reply":"2022-07-12T03:42:03.974902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:03.977110Z","iopub.execute_input":"2022-07-12T03:42:03.977590Z","iopub.status.idle":"2022-07-12T03:42:04.006505Z","shell.execute_reply.started":"2022-07-12T03:42:03.977561Z","shell.execute_reply":"2022-07-12T03:42:04.005645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'Cabin' in feats_cat:\n    feats_cat.remove('Cabin')\nfeats_cat.append('Cabin_deck')\nfeats_cat.append('Cabin_side')\nfeats_num.append('Cabin_num')\nfeats_cat, feats_num","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.007447Z","iopub.execute_input":"2022-07-12T03:42:04.008085Z","iopub.status.idle":"2022-07-12T03:42:04.016857Z","shell.execute_reply.started":"2022-07-12T03:42:04.008052Z","shell.execute_reply":"2022-07-12T03:42:04.015647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Mean Encoding with Smoothing, CV loop, Expanding mean","metadata":{}},{"cell_type":"markdown","source":"### testing","metadata":{}},{"cell_type":"code","source":"train_ = dataset[dataset['Transported'].notnull()]\ntest_ = dataset[dataset['Transported'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.018020Z","iopub.execute_input":"2022-07-12T03:42:04.018328Z","iopub.status.idle":"2022-07-12T03:42:04.031441Z","shell.execute_reply.started":"2022-07-12T03:42:04.018299Z","shell.execute_reply":"2022-07-12T03:42:04.030396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.032898Z","iopub.execute_input":"2022-07-12T03:42:04.033906Z","iopub.status.idle":"2022-07-12T03:42:04.053932Z","shell.execute_reply.started":"2022-07-12T03:42:04.033861Z","shell.execute_reply":"2022-07-12T03:42:04.052745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_series = train_['HomePlanet']\ntst_series = test_['HomePlanet']\ntarget = train_['Transported']\ntemp = pd.concat([trn_series, target], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.055680Z","iopub.execute_input":"2022-07-12T03:42:04.056375Z","iopub.status.idle":"2022-07-12T03:42:04.064382Z","shell.execute_reply.started":"2022-07-12T03:42:04.056300Z","shell.execute_reply":"2022-07-12T03:42:04.063254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cumsum = temp.groupby(trn_series.name)[target.name].cumsum() - target\ncumcnt = temp.groupby(trn_series.name).cumcount() + 1\ntrn_series_new = cumsum / cumcnt","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.066068Z","iopub.execute_input":"2022-07-12T03:42:04.067290Z","iopub.status.idle":"2022-07-12T03:42:04.084228Z","shell.execute_reply.started":"2022-07-12T03:42:04.067243Z","shell.execute_reply":"2022-07-12T03:42:04.083397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp['trn_series_new'] = trn_series_new","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.085704Z","iopub.execute_input":"2022-07-12T03:42:04.086802Z","iopub.status.idle":"2022-07-12T03:42:04.092736Z","shell.execute_reply.started":"2022-07-12T03:42:04.086758Z","shell.execute_reply":"2022-07-12T03:42:04.091682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.kdeplot(x='trn_series_new', data=temp, hue=trn_series.name)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.094235Z","iopub.execute_input":"2022-07-12T03:42:04.094594Z","iopub.status.idle":"2022-07-12T03:42:04.356496Z","shell.execute_reply.started":"2022-07-12T03:42:04.094562Z","shell.execute_reply":"2022-07-12T03:42:04.355633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smoothing(n_rows, target_mean, global_mean, alpha):\n    return (target_mean * n_rows + global_mean * alpha) / (n_rows + alpha)\nglobal_mean = target.mean()\nalpha = 0.7\n\nmean_cnt = temp.groupby(trn_series.name)[target.name].agg(['mean', 'count'])\nmean_cnt['mean_smoothing'] = mean_cnt.apply(lambda x: smoothing(x['count'], x['mean'], global_mean, alpha), axis=1)\nmean_cnt.drop(['mean', 'count'], axis=1, inplace=True)\nmean_cnt","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.357658Z","iopub.execute_input":"2022-07-12T03:42:04.358310Z","iopub.status.idle":"2022-07-12T03:42:04.375937Z","shell.execute_reply.started":"2022-07-12T03:42:04.358271Z","shell.execute_reply":"2022-07-12T03:42:04.374549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_df = pd.merge(\n    tst_series.to_frame(tst_series.name), \n    mean_cnt.reset_index(), on=tst_series.name, how='left')\ntst_df.index = tst_series.index\ntst_df","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.377526Z","iopub.execute_input":"2022-07-12T03:42:04.377843Z","iopub.status.idle":"2022-07-12T03:42:04.398356Z","shell.execute_reply.started":"2022-07-12T03:42:04.377813Z","shell.execute_reply":"2022-07-12T03:42:04.397290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tst_df[tst_df[tst_series.name].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.399467Z","iopub.execute_input":"2022-07-12T03:42:04.399749Z","iopub.status.idle":"2022-07-12T03:42:04.414740Z","shell.execute_reply.started":"2022-07-12T03:42:04.399722Z","shell.execute_reply":"2022-07-12T03:42:04.413415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### mean encode function","metadata":{}},{"cell_type":"code","source":"def smoothing(n_rows, target_mean, global_mean, alpha):\n    return (target_mean * n_rows + global_mean * alpha) / (n_rows + alpha)\n\ndef mean_encode(trn_series, val_series, tst_series, target, alpha):\n    temp = pd.concat([trn_series, target], axis=1)\n    mean_cnt = temp.groupby(trn_series.name)[target.name].agg(['mean', 'count'])\n    global_mean = target.mean()\n    \n    mean_cnt['mean_smoothing'] = mean_cnt.apply(lambda x: smoothing(x['count'], x['mean'], global_mean, alpha), axis=1)\n    mean_cnt.drop(['mean', 'count'], axis=1, inplace=True)\n    \n    # expanding mean\n    cumsum = temp.groupby(trn_series.name)[target.name].cumsum() - target\n    cumcnt = temp.groupby(trn_series.name).cumcount() + 1\n    trn_series_new = cumsum / cumcnt\n    \n    val_df = pd.merge(\n        val_series.to_frame(val_series.name), \n        mean_cnt.reset_index(), on=val_series.name, how='left'\n        )\n    val_df.index = val_series.index\n    \n    tst_df = pd.merge(\n        tst_series.to_frame(tst_series.name), \n        mean_cnt.reset_index(), on=tst_series.name, how='left'\n        )\n    tst_df.index = tst_series.index\n    \n    return trn_series_new, val_df['mean_smoothing'], tst_df['mean_smoothing']","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.415856Z","iopub.execute_input":"2022-07-12T03:42:04.416710Z","iopub.status.idle":"2022-07-12T03:42:04.426326Z","shell.execute_reply.started":"2022-07-12T03:42:04.416675Z","shell.execute_reply":"2022-07-12T03:42:04.425316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.428096Z","iopub.execute_input":"2022-07-12T03:42:04.428546Z","iopub.status.idle":"2022-07-12T03:42:04.457489Z","shell.execute_reply.started":"2022-07-12T03:42:04.428506Z","shell.execute_reply":"2022-07-12T03:42:04.456300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_cat","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.458988Z","iopub.execute_input":"2022-07-12T03:42:04.459543Z","iopub.status.idle":"2022-07-12T03:42:04.469081Z","shell.execute_reply.started":"2022-07-12T03:42:04.459395Z","shell.execute_reply":"2022-07-12T03:42:04.468299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(dataset.columns) - set(feats_cat + feats_num)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.470551Z","iopub.execute_input":"2022-07-12T03:42:04.471182Z","iopub.status.idle":"2022-07-12T03:42:04.480482Z","shell.execute_reply.started":"2022-07-12T03:42:04.471141Z","shell.execute_reply":"2022-07-12T03:42:04.479609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_bin = []\nfor col in feats_cat:\n    val_arr = np.array(dataset[col].unique())\n    val_arr = val_arr[~pd.isnull(val_arr)]\n    if len(val_arr) == 2:\n        feats_bin += [col]\n        feats_cat.remove(col)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.482045Z","iopub.execute_input":"2022-07-12T03:42:04.482948Z","iopub.status.idle":"2022-07-12T03:42:04.494313Z","shell.execute_reply.started":"2022-07-12T03:42:04.482912Z","shell.execute_reply":"2022-07-12T03:42:04.493201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_bin","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.495746Z","iopub.execute_input":"2022-07-12T03:42:04.496874Z","iopub.status.idle":"2022-07-12T03:42:04.507506Z","shell.execute_reply.started":"2022-07-12T03:42:04.496836Z","shell.execute_reply":"2022-07-12T03:42:04.506333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_cat","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.509458Z","iopub.execute_input":"2022-07-12T03:42:04.510190Z","iopub.status.idle":"2022-07-12T03:42:04.517039Z","shell.execute_reply.started":"2022-07-12T03:42:04.510145Z","shell.execute_reply":"2022-07-12T03:42:04.516280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_num","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.518062Z","iopub.execute_input":"2022-07-12T03:42:04.518778Z","iopub.status.idle":"2022-07-12T03:42:04.529182Z","shell.execute_reply.started":"2022-07-12T03:42:04.518748Z","shell.execute_reply":"2022-07-12T03:42:04.528395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feats_cat.remove('Name')\ndataset.drop('Name', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.532213Z","iopub.execute_input":"2022-07-12T03:42:04.532751Z","iopub.status.idle":"2022-07-12T03:42:04.541558Z","shell.execute_reply.started":"2022-07-12T03:42:04.532707Z","shell.execute_reply":"2022-07-12T03:42:04.540638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.542750Z","iopub.execute_input":"2022-07-12T03:42:04.543199Z","iopub.status.idle":"2022-07-12T03:42:04.570620Z","shell.execute_reply.started":"2022-07-12T03:42:04.543160Z","shell.execute_reply":"2022-07-12T03:42:04.569722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## label encoding (binary feature)","metadata":{}},{"cell_type":"markdown","source":"### testing","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ncry = dataset['CryoSleep'].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.572080Z","iopub.execute_input":"2022-07-12T03:42:04.572370Z","iopub.status.idle":"2022-07-12T03:42:04.643775Z","shell.execute_reply.started":"2022-07-12T03:42:04.572333Z","shell.execute_reply":"2022-07-12T03:42:04.642730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cry[cry.notnull()] = le.fit_transform(cry[cry.notnull()])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.644998Z","iopub.execute_input":"2022-07-12T03:42:04.645351Z","iopub.status.idle":"2022-07-12T03:42:04.657505Z","shell.execute_reply.started":"2022-07-12T03:42:04.645307Z","shell.execute_reply":"2022-07-12T03:42:04.656664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cry.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.659058Z","iopub.execute_input":"2022-07-12T03:42:04.659774Z","iopub.status.idle":"2022-07-12T03:42:04.668909Z","shell.execute_reply.started":"2022-07-12T03:42:04.659731Z","shell.execute_reply":"2022-07-12T03:42:04.667923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CV training","metadata":{}},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.672266Z","iopub.execute_input":"2022-07-12T03:42:04.672813Z","iopub.status.idle":"2022-07-12T03:42:04.697321Z","shell.execute_reply.started":"2022-07-12T03:42:04.672777Z","shell.execute_reply":"2022-07-12T03:42:04.696170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.drop('PassengerId', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.698763Z","iopub.execute_input":"2022-07-12T03:42:04.699711Z","iopub.status.idle":"2022-07-12T03:42:04.710145Z","shell.execute_reply.started":"2022-07-12T03:42:04.699665Z","shell.execute_reply":"2022-07-12T03:42:04.708819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_val = dataset[dataset['Transported'].notnull()].copy()\ntest = dataset[dataset['Transported'].isnull()].copy()\ntest.drop('Transported', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.711949Z","iopub.execute_input":"2022-07-12T03:42:04.712410Z","iopub.status.idle":"2022-07-12T03:42:04.728855Z","shell.execute_reply.started":"2022-07-12T03:42:04.712365Z","shell.execute_reply":"2022-07-12T03:42:04.727714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_val['Transported']\nX = train_val.drop('Transported', axis=1)\nX = X.fillna(-1)\ntest = test.fillna(-1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.730887Z","iopub.execute_input":"2022-07-12T03:42:04.731659Z","iopub.status.idle":"2022-07-12T03:42:04.749360Z","shell.execute_reply.started":"2022-07-12T03:42:04.731615Z","shell.execute_reply":"2022-07-12T03:42:04.748012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[feats_bin] = X[feats_bin].astype('str')\ntest[feats_bin] = test[feats_bin].astype('str')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.751459Z","iopub.execute_input":"2022-07-12T03:42:04.752210Z","iopub.status.idle":"2022-07-12T03:42:04.769022Z","shell.execute_reply.started":"2022-07-12T03:42:04.752167Z","shell.execute_reply":"2022-07-12T03:42:04.767776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\nkfold = KFold(n_splits=5, shuffle=True, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T03:42:04.770810Z","iopub.execute_input":"2022-07-12T03:42:04.771229Z","iopub.status.idle":"2022-07-12T03:42:04.838755Z","shell.execute_reply.started":"2022-07-12T03:42:04.771188Z","shell.execute_reply":"2022-07-12T03:42:04.837847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from xgboost import plot_importance\n# from xgboost import XGBClassifier\n\n# model = XGBClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.240911Z","iopub.status.idle":"2022-07-12T02:32:51.241471Z","shell.execute_reply.started":"2022-07-12T02:32:51.241276Z","shell.execute_reply":"2022-07-12T02:32:51.241297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.set_option('mode.chained_assignment',  None)\n# y_test_pred = 0\n\n\n# for cv, (train_idx, val_idx) in enumerate(kfold.split(X)):\n#     X_test = test.copy()\n#     X_train, y_train = X.iloc[train_idx].copy(), y.iloc[train_idx].copy()\n#     X_val, y_val = X.iloc[val_idx].copy(), y.iloc[val_idx].copy()\n    \n#     for col in feats_bin:\n#         X_train_bin = X_train[col].copy()\n#         le = LabelEncoder()\n#         le.fit(X_train_bin)\n#         X_train[col] = le.transform(X_train[col][X_train[col].notnull()]).astype('int')\n#         X_val[col] = le.transform(X_val[col][X_val[col].notnull()]).astype('int')\n#         X_test[col] = le.transform(X_test[col][X_test[col].notnull()]).astype('int')\n    \n#     for col in feats_cat:\n#         X_train[col], X_val[col], X_test[col] = mean_encode(X_train[col], X_val[col], X_test[col], y_train, 1)\n    \n#     eval_set = [(X_val, y_val)]\n#     fit_model = model.fit(X_train, y_train, eval_set=eval_set, \n#                          eval_metric='error', verbose=False)\n#     print(\"cv {} accuracy: \".format(cv+1), np.mean(fit_model.predict(X_val) == y_val))\n#     print(\"-------------------------------------\")\n    \n#     y_test_pred += fit_model.predict_proba(X_test)[:, 1]\n\n# y_test_pred /= 5","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.245654Z","iopub.status.idle":"2022-07-12T02:32:51.246258Z","shell.execute_reply.started":"2022-07-12T02:32:51.245927Z","shell.execute_reply":"2022-07-12T02:32:51.245948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission['Transported'] = y_test_pred > 0.5\n# submission.to_csv('sampleSubmission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.247755Z","iopub.status.idle":"2022-07-12T02:32:51.248209Z","shell.execute_reply.started":"2022-07-12T02:32:51.248011Z","shell.execute_reply":"2022-07-12T02:32:51.248031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test score: 0.80149","metadata":{}},{"cell_type":"code","source":"# plot_importance(fit_model","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.249606Z","iopub.status.idle":"2022-07-12T02:32:51.250022Z","shell.execute_reply.started":"2022-07-12T02:32:51.249804Z","shell.execute_reply":"2022-07-12T02:32:51.249821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bayesian Optimization","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:30:13.694847Z","iopub.execute_input":"2022-07-08T07:30:13.695366Z","iopub.status.idle":"2022-07-08T07:30:13.701078Z","shell.execute_reply.started":"2022-07-08T07:30:13.695324Z","shell.execute_reply":"2022-07-08T07:30:13.699907Z"}}},{"cell_type":"code","source":"# from bayes_opt import BayesianOptimization\n\n# pbounds = {'max_depth': (3, 7),\n#                 'learning_rate': (0.01, 0.2),\n#                 'n_estimators': (5000, 10000),\n#                 'gamma': (0, 100),\n#                 'min_child_weight': (0, 3),\n#                 'subsample': (0.5, 1),\n#                 'colsample_bytree' :(0.2, 1)\n#                 }\n\n# def xgb_cv(max_depth, learning_rate, n_estimators, gamma, min_child_weight,\n#           subsample, colsample_bytree, silent=True, nthread=-1):\n#     model = XGBClassifier(\n#         max_depth=int(max_depth),\n#         learning_rate=learning_rate,\n#         n_estimators=int(n_estimators),\n#         gamma=gamma,\n#         min_child_weight=min_child_weight,\n#         subsample=subsample,\n#         colsample_bytree=colsample_bytree, \n#         nthread=nthread)\n    \n#     val_acc_mean = 0\n#     for cv, (train_idx, val_idx) in enumerate(kfold.split(X)):\n#         X_test = test.copy()\n#         X_train, y_train = X.iloc[train_idx].copy(), y.iloc[train_idx].copy()\n#         X_val, y_val = X.iloc[val_idx].copy(), y.iloc[val_idx].copy()\n\n#         for col in feats_bin:\n#             X_train_bin = X_train[col].copy()\n#             le = LabelEncoder()\n#             le.fit(X_train_bin)\n#             X_train[col] = le.transform(X_train[col][X_train[col].notnull()]).astype('int')\n#             X_val[col] = le.transform(X_val[col][X_val[col].notnull()]).astype('int')\n#             X_test[col] = le.transform(X_test[col][X_test[col].notnull()]).astype('int')\n\n#         for col in feats_cat:\n#             X_train[col], X_val[col], X_test[col] = mean_encode(X_train[col], X_val[col], X_test[col], y_train, 1)\n\n#         eval_set = [(X_val, y_val)]\n#         fit_model = model.fit(X_train, y_train, eval_set=eval_set, \n#                              eval_metric='error', verbose=False)\n#         val_acc_mean += np.mean(fit_model.predict(X_val) == y_val)\n#     return val_acc_mean / 5\n\n# bo = BayesianOptimization(f=xgb_cv, pbounds=pbounds, verbose=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.251418Z","iopub.status.idle":"2022-07-12T02:32:51.252051Z","shell.execute_reply.started":"2022-07-12T02:32:51.251805Z","shell.execute_reply":"2022-07-12T02:32:51.251835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bo.maximize(init_points=2, n_iter=10, acq='ei', xi=0.01)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.253309Z","iopub.status.idle":"2022-07-12T02:32:51.253925Z","shell.execute_reply.started":"2022-07-12T02:32:51.253703Z","shell.execute_reply":"2022-07-12T02:32:51.253732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(bo.max)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:32:51.255132Z","iopub.status.idle":"2022-07-12T02:32:51.255767Z","shell.execute_reply.started":"2022-07-12T02:32:51.255566Z","shell.execute_reply":"2022-07-12T02:32:51.255591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### find best hyperparameter","metadata":{}},{"cell_type":"code","source":"# best_hyper = {'target': 0.807202200118979, 'params': {'colsample_bytree': 0.8297967555914176, 'gamma': 7.978488769638847, 'learning_rate': 0.022961714375493332, 'max_depth': 5, 'min_child_weight': 1.2887579454791558, 'n_estimators': 5662, 'subsample': 0.640835814340262}}","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:49:17.194252Z","iopub.execute_input":"2022-07-12T02:49:17.194712Z","iopub.status.idle":"2022-07-12T02:49:17.201871Z","shell.execute_reply.started":"2022-07-12T02:49:17.194680Z","shell.execute_reply":"2022-07-12T02:49:17.200463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from xgboost import plot_importance\n# from xgboost import XGBClassifier\n# model = XGBClassifier(**best_hyper['params'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:49:18.207528Z","iopub.execute_input":"2022-07-12T02:49:18.207909Z","iopub.status.idle":"2022-07-12T02:49:18.215288Z","shell.execute_reply.started":"2022-07-12T02:49:18.207879Z","shell.execute_reply":"2022-07-12T02:49:18.213662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.set_option('mode.chained_assignment',  None)\n# y_test_pred = 0\n\n\n# for cv, (train_idx, val_idx) in enumerate(kfold.split(X)):\n#     X_test = test.copy()\n#     X_train, y_train = X.iloc[train_idx].copy(), y.iloc[train_idx].copy()\n#     X_val, y_val = X.iloc[val_idx].copy(), y.iloc[val_idx].copy()\n    \n#     for col in feats_bin:\n#         X_train_bin = X_train[col].copy()\n#         le = LabelEncoder()\n#         le.fit(X_train_bin)\n#         X_train[col] = le.transform(X_train[col][X_train[col].notnull()]).astype('int')\n#         X_val[col] = le.transform(X_val[col][X_val[col].notnull()]).astype('int')\n#         X_test[col] = le.transform(X_test[col][X_test[col].notnull()]).astype('int')\n    \n#     for col in feats_cat:\n#         X_train[col], X_val[col], X_test[col] = mean_encode(X_train[col], X_val[col], X_test[col], y_train, 1)\n    \n#     eval_set = [(X_val, y_val)]\n#     fit_model = model.fit(X_train, y_train, eval_set=eval_set, \n#                          eval_metric='error', verbose=False)\n#     print(\"cv {} accuracy: \".format(cv+1), np.mean(fit_model.predict(X_val) == y_val))\n#     print(\"-------------------------------------\")\n    \n#     y_test_pred += fit_model.predict_proba(X_test)[:, 1]\n\n# y_test_pred /= 5","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:49:19.885883Z","iopub.execute_input":"2022-07-12T02:49:19.886247Z","iopub.status.idle":"2022-07-12T02:52:56.626951Z","shell.execute_reply.started":"2022-07-12T02:49:19.886220Z","shell.execute_reply":"2022-07-12T02:52:56.626224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission['Transported'] = y_test_pred > 0.5\n# submission.to_csv('sampleSubmission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T02:53:31.117892Z","iopub.execute_input":"2022-07-12T02:53:31.118293Z","iopub.status.idle":"2022-07-12T02:53:31.137251Z","shell.execute_reply.started":"2022-07-12T02:53:31.118267Z","shell.execute_reply":"2022-07-12T02:53:31.136151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stacking\n### base model: Xgb, Lightgbm, RandomForest, ExtraTree, CatBoost\n### final model: Logistic ","metadata":{}},{"cell_type":"code","source":"def acc(pred, target):\n    return (pred == target).mean()\n\ndef get_oof(name, model):\n    val_predict = np.zeros((len(y),))\n    test_predict = np.zeros((len(test), 5))\n    print(\"TRAIN \", name)\n    for cv, (train_idx, val_idx) in enumerate(kfold.split(X)):\n        X_test = test.copy()\n        X_train, y_train = X.iloc[train_idx].copy(), y.iloc[train_idx].copy()\n        X_val, y_val = X.iloc[val_idx].copy(), y.iloc[val_idx].copy()\n        \n        for col in feats_bin:\n            X_train_bin = X_train[col].copy()\n            le = LabelEncoder()\n            le.fit(X_train_bin)\n            X_train[col] = le.transform(X_train[col][X_train[col].notnull()]).astype('int')\n            X_val[col] = le.transform(X_val[col][X_val[col].notnull()]).astype('int')\n            X_test[col] = le.transform(X_test[col][X_test[col].notnull()]).astype('int')\n\n        for col in feats_cat:\n            X_train[col], X_val[col], X_test[col] = mean_encode(X_train[col], X_val[col], X_test[col], y_train, 1)\n        \n        model.fit(X_train, y_train)\n        print(\"cv{} val_acc: {:.4f}\".format(cv+1, acc(model.predict(X_val), y_val)))\n        \n        val_predict[val_idx] = model.predict_proba(X_val)[:, 1]\n        test_predict[:, cv] = model.predict_proba(X_test)[:, 1]\n        \n    print(\"-------------------------\")\n    return val_predict, test_predict.mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T06:40:02.879116Z","iopub.execute_input":"2022-07-12T06:40:02.880022Z","iopub.status.idle":"2022-07-12T06:40:02.891434Z","shell.execute_reply.started":"2022-07-12T06:40:02.879981Z","shell.execute_reply":"2022-07-12T06:40:02.890283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier\nfrom catboost import CatBoostClassifier\n\nmodel_lst = [('Xgb', XGBClassifier()), ('Lightgbm', LGBMClassifier()), ('RandomForest', RandomForestClassifier()), ('ExtraTree', ExtraTreesClassifier()), ('CatBoost', CatBoostClassifier(verbose=False))]\n# new_train_x1, new_test_x = get_oof(*model_lst[0])\nnew_train_x = np.zeros((len(X), 5))\nnew_test_x = np.zeros((len(test), 5))\nfor i, model in enumerate(model_lst):\n    train_x_, test_x_ = get_oof(*model)\n    new_train_x[:, i] = train_x_\n    new_test_x[:, i] = test_x_","metadata":{"execution":{"iopub.status.busy":"2022-07-12T06:42:03.022022Z","iopub.execute_input":"2022-07-12T06:42:03.022512Z","iopub.status.idle":"2022-07-12T06:42:39.571651Z","shell.execute_reply.started":"2022-07-12T06:42:03.022470Z","shell.execute_reply":"2022-07-12T06:42:39.570738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nfinal_model = LogisticRegression()\nfinal_model.fit(new_train_x, y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T06:44:43.265135Z","iopub.execute_input":"2022-07-12T06:44:43.265601Z","iopub.status.idle":"2022-07-12T06:44:43.341101Z","shell.execute_reply.started":"2022-07-12T06:44:43.265566Z","shell.execute_reply":"2022-07-12T06:44:43.339406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['Transported'] = final_model.predict(new_test_x) > 0.5\nsubmission.to_csv('sampleSubmission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T06:46:14.764520Z","iopub.execute_input":"2022-07-12T06:46:14.765408Z","iopub.status.idle":"2022-07-12T06:46:14.791820Z","shell.execute_reply.started":"2022-07-12T06:46:14.765361Z","shell.execute_reply":"2022-07-12T06:46:14.790136Z"},"trusted":true},"execution_count":null,"outputs":[]}]}