{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## TPS Aug 22 - Bayesian Neural Network","metadata":{}},{"cell_type":"markdown","source":"# References\n\n[desalegngeb notebook](https://www.kaggle.com/code/desalegngeb/tps08-logisticregression-and-some-fe/notebook?scriptVersionId=102691691)<br>\n[aidenfoster notebook](https://www.kaggle.com/code/aidenfoster/tps-aug-22-probabilistic-regression)","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install feature_engine\n!pip install missingno\n# all may not be needed\n\nimport os\nimport sys\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pylab as plt\nfrom IPython.display import clear_output\n\nfrom scipy.stats import uniform\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier,ExtraTreesClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.ensemble import StackingClassifier,VotingClassifier,StackingClassifier\n\nfrom catboost import CatBoostClassifier\n\nfrom sklearn.model_selection import cross_validate, StratifiedKFold, RepeatedStratifiedKFold, RandomizedSearchCV, GridSearchCV\nfrom sklearn.metrics import confusion_matrix, plot_confusion_matrix, classification_report, roc_auc_score, accuracy_score\nfrom sklearn import metrics\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, KNNImputer, IterativeImputer\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import OneHotEncoder, RobustScaler, PowerTransformer, LabelEncoder, StandardScaler, MinMaxScaler\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2022-08-08T15:56:13.504036Z","iopub.execute_input":"2022-08-08T15:56:13.505511Z","iopub.status.idle":"2022-08-08T15:56:36.283605Z","shell.execute_reply.started":"2022-08-08T15:56:13.505455Z","shell.execute_reply":"2022-08-08T15:56:36.281746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsubmission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:24.520900Z","iopub.execute_input":"2022-08-08T15:46:24.521446Z","iopub.status.idle":"2022-08-08T15:46:24.715238Z","shell.execute_reply.started":"2022-08-08T15:46:24.521397Z","shell.execute_reply":"2022-08-08T15:46:24.713768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Null values\n- Around 3% of the data (cells) is missing in both train and test datset.\n- We will need to impute.","metadata":{}},{"cell_type":"code","source":"train_na_cols = [col for col in train.columns if train[col].isnull().sum()!=0]\nprint('Train data cols with missing values ares: \\n', train_na_cols)\n\nprint('\\n')\n\ntest_na_cols = [col for col in test.columns if test[col].isnull().sum()!=0]\nprint('Train data cols with missing values ares: \\n', test_na_cols)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-08T15:46:24.717054Z","iopub.execute_input":"2022-08-08T15:46:24.718587Z","iopub.status.idle":"2022-08-08T15:46:24.756637Z","shell.execute_reply.started":"2022-08-08T15:46:24.718539Z","shell.execute_reply":"2022-08-08T15:46:24.755360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = train.pop('failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:24.761563Z","iopub.execute_input":"2022-08-08T15:46:24.761954Z","iopub.status.idle":"2022-08-08T15:46:24.768025Z","shell.execute_reply.started":"2022-08-08T15:46:24.761920Z","shell.execute_reply":"2022-08-08T15:46:24.766723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_cols = [col for col in train.columns if train[col].dtypes == 'float64']\nobject_cols = [col for col in train.columns if train[col].dtypes == 'object']\nint_object_cols = [col for col in train.columns[1:-1] if (train[col].dtypes == 'object' or train[col].dtypes == 'int64')]\nnullValue_cols = [col for col in train.columns if train[col].isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:24.769880Z","iopub.execute_input":"2022-08-08T15:46:24.770652Z","iopub.status.idle":"2022-08-08T15:46:24.792388Z","shell.execute_reply.started":"2022-08-08T15:46:24.770605Z","shell.execute_reply":"2022-08-08T15:46:24.791071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ambros' idea of adding missing values as extra columns\n# https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/342319\ntrain['m_3_missing'] = train.measurement_3.isna()\ntrain['m_5_missing'] = train.measurement_5.isna()\n\ntest['m_3_missing'] = test.measurement_3.isna()\ntest['m_5_missing'] = test.measurement_5.isna()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:24.794483Z","iopub.execute_input":"2022-08-08T15:46:24.795050Z","iopub.status.idle":"2022-08-08T15:46:24.804395Z","shell.execute_reply.started":"2022-08-08T15:46:24.794993Z","shell.execute_reply":"2022-08-08T15:46:24.803426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Missing Value Imputation\n- Impute based on product code. We first group the data based on similar product code and impute the missing values using the same group data. This way of imputing missing values was discussed here in [Ambros' notebook](https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense). \n- One option we can use to impute the missing values is LGBMImputer. ","metadata":{}},{"cell_type":"code","source":"# !rm -r kuma_utils\n!git clone https://github.com/analokmaus/kuma_utils.git","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:24.805826Z","iopub.execute_input":"2022-08-08T15:46:24.806621Z","iopub.status.idle":"2022-08-08T15:46:25.967174Z","shell.execute_reply.started":"2022-08-08T15:46:24.806586Z","shell.execute_reply":"2022-08-08T15:46:25.965736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append(\"kuma_utils/\")\nfrom kuma_utils.preprocessing.imputer import LGBMImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:25.969153Z","iopub.execute_input":"2022-08-08T15:46:25.969575Z","iopub.status.idle":"2022-08-08T15:46:25.975718Z","shell.execute_reply.started":"2022-08-08T15:46:25.969539Z","shell.execute_reply":"2022-08-08T15:46:25.974607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train['product_code'].unique())\ndf_A = train[train['product_code']=='A']\ndf_B = train[train['product_code']=='B']\ndf_C = train[train['product_code']=='C']\ndf_D = train[train['product_code']=='D']\ndf_E = train[train['product_code']=='E']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:25.977208Z","iopub.execute_input":"2022-08-08T15:46:25.977574Z","iopub.status.idle":"2022-08-08T15:46:26.023856Z","shell.execute_reply.started":"2022-08-08T15:46:25.977544Z","shell.execute_reply":"2022-08-08T15:46:26.022646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test['product_code'].unique())\ndf_F_t = test[test['product_code']=='F']\ndf_G_t = test[test['product_code']=='G']\ndf_H_t = test[test['product_code']=='H']\ndf_I_t = test[test['product_code']=='I']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:26.028716Z","iopub.execute_input":"2022-08-08T15:46:26.029079Z","iopub.status.idle":"2022-08-08T15:46:26.055177Z","shell.execute_reply.started":"2022-08-08T15:46:26.029048Z","shell.execute_reply":"2022-08-08T15:46:26.054287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_imtr = LGBMImputer(cat_features=object_cols, n_iter=50)\n\n# train dataset\ntrain_iterimp_A = lgbm_imtr.fit_transform(df_A[nullValue_cols])\ntrain_iterimp_B = lgbm_imtr.fit_transform(df_B[nullValue_cols])\ntrain_iterimp_C = lgbm_imtr.fit_transform(df_C[nullValue_cols])\ntrain_iterimp_D = lgbm_imtr.fit_transform(df_D[nullValue_cols])\ntrain_iterimp_E = lgbm_imtr.fit_transform(df_E[nullValue_cols])\n\n# test dataset\ntest_iterimp_F = lgbm_imtr.fit_transform(df_F_t[nullValue_cols])\ntest_iterimp_G = lgbm_imtr.fit_transform(df_G_t[nullValue_cols])\ntest_iterimp_H = lgbm_imtr.fit_transform(df_H_t[nullValue_cols])\ntest_iterimp_I = lgbm_imtr.fit_transform(df_I_t[nullValue_cols])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T15:46:26.056366Z","iopub.execute_input":"2022-08-08T15:46:26.057203Z","iopub.status.idle":"2022-08-08T15:46:54.438491Z","shell.execute_reply.started":"2022-08-08T15:46:26.057172Z","shell.execute_reply":"2022-08-08T15:46:54.437536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"none_na_cols = [col for col in train.columns if col not in nullValue_cols]\ndf_train = train[none_na_cols]\ndf_test = test[none_na_cols]\n\ntrain_ = pd.concat([train_iterimp_A, train_iterimp_B,train_iterimp_C,train_iterimp_D,train_iterimp_E], axis=0)\ntrain = pd.concat([df_train, train_], axis=1)\n\ntest_ = pd.concat([test_iterimp_F, test_iterimp_G,test_iterimp_H,test_iterimp_I], axis=0)\ntest = pd.concat([df_test, test_], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.442245Z","iopub.execute_input":"2022-08-08T15:46:54.444043Z","iopub.status.idle":"2022-08-08T15:46:54.473073Z","shell.execute_reply.started":"2022-08-08T15:46:54.444003Z","shell.execute_reply":"2022-08-08T15:46:54.471592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Missing values in train dataset after pre-peocessing is: \", format(train.isna().sum().sum()))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.474798Z","iopub.execute_input":"2022-08-08T15:46:54.475149Z","iopub.status.idle":"2022-08-08T15:46:54.489700Z","shell.execute_reply.started":"2022-08-08T15:46:54.475118Z","shell.execute_reply":"2022-08-08T15:46:54.488220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Feature Engineering Ideas\n- This dataset, being not too obscure what the columns maight be, gives a good opportunity to engineer new features.\n- From initial observations, some of the features seem to be `dimension` measurements and we may be able to combine and derive, area or volume features for example. `attribute_2` and `attribute_3` for example seem to be `width` and `length` dimensions?.\n- Looking at the values of `measurement_3` to `measurement_16`, they all look different variants of the same type of measurement. So we may aggregate into one or two features (mean, std) for example.","metadata":{}},{"cell_type":"code","source":"display(train['attribute_2'].unique())\ndisplay(train['attribute_3'].unique())\nprint()\ndisplay(train['measurement_0'].unique())\ndisplay(train['measurement_1'].unique())\ndisplay(train['measurement_2'].unique())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-08T15:46:54.491576Z","iopub.execute_input":"2022-08-08T15:46:54.491936Z","iopub.status.idle":"2022-08-08T15:46:54.516811Z","shell.execute_reply.started":"2022-08-08T15:46:54.491897Z","shell.execute_reply":"2022-08-08T15:46:54.515448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 4.1 Combine features and create new \n- Multiply `attribute_2` and `attribute_2` and create a new feature\n- The idea here is that these two seem to me that they are dimensions of the material used for cleaning i.e, width and length for example. We could multiply and get area as new feature instead. Then drop the them.","metadata":{}},{"cell_type":"code","source":"train['attribute_2*3'] = train['attribute_2'] * train['attribute_3']\ntest['attribute_2*3'] = test['attribute_2'] * test['attribute_3']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.518781Z","iopub.execute_input":"2022-08-08T15:46:54.519244Z","iopub.status.idle":"2022-08-08T15:46:54.529045Z","shell.execute_reply.started":"2022-08-08T15:46:54.519201Z","shell.execute_reply":"2022-08-08T15:46:54.527661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 4.2 Use aggregation\n- We see from the data that features `measuremsnt_3` to `measurement_16` belong to the same family. More like different variants of the same measurement type. So we could aggregate them into  their averages and may be standard deviations of them.","metadata":{}},{"cell_type":"code","source":"meas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain['meas_gr1_avg'] = np.mean(train[meas_gr1_cols], axis=1)\ntrain['meas_gr1_std'] = np.std(train[meas_gr1_cols], axis=1)\n\ntest['meas_gr1_avg'] = np.mean(test[meas_gr1_cols], axis=1)\ntest['meas_gr1_std'] = np.std(test[meas_gr1_cols], axis=1) \n\nmeas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\ntrain['meas_gr2_avg'] = np.mean(train[meas_gr2_cols], axis=1)\ntest['meas_gr2_avg'] = np.mean(test[meas_gr2_cols], axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.530855Z","iopub.execute_input":"2022-08-08T15:46:54.531282Z","iopub.status.idle":"2022-08-08T15:46:54.591491Z","shell.execute_reply.started":"2022-08-08T15:46:54.531231Z","shell.execute_reply":"2022-08-08T15:46:54.589842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['meas17/meas_gr2_avg'] = train['measurement_17'] / train['meas_gr2_avg']\ntest['meas17/meas_gr2_avg'] = test['measurement_17'] / test['meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.593576Z","iopub.execute_input":"2022-08-08T15:46:54.593933Z","iopub.status.idle":"2022-08-08T15:46:54.602757Z","shell.execute_reply.started":"2022-08-08T15:46:54.593902Z","shell.execute_reply":"2022-08-08T15:46:54.601406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.Using Weight of Evidence\nThat is my only new idea for the notebook. It helps a bit. I would suggest some more feature engineering using both feature engine (https://feature-engine.readthedocs.io/en/latest/index.html) and feature tools (https://www.featuretools.com/)","metadata":{}},{"cell_type":"code","source":"train.set_index('id',inplace=True)\ntest.set_index('id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.604122Z","iopub.execute_input":"2022-08-08T15:46:54.604600Z","iopub.status.idle":"2022-08-08T15:46:54.616347Z","shell.execute_reply.started":"2022-08-08T15:46:54.604565Z","shell.execute_reply":"2022-08-08T15:46:54.615420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_data = pd.concat([train,test],axis = 0).reset_index(drop=True)\ndef LableEncoder(df_org, df_comb, cats):\n    #https://www.kaggle.com/code/pourchot/update-for-keras-optuna-for-lr/notebook\n    for col in cats :\n        le = LabelEncoder()\n        df_comb[col] = le.fit_transform(df_comb[col])\n    return train, test","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.619822Z","iopub.execute_input":"2022-08-08T15:46:54.620196Z","iopub.status.idle":"2022-08-08T15:46:54.651807Z","shell.execute_reply.started":"2022-08-08T15:46:54.620165Z","shell.execute_reply":"2022-08-08T15:46:54.650562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cats = ['attribute_0', 'attribute_1', 'm_3_missing', 'm_5_missing']\ntrain, test = LableEncoder(train,combined_data, cats)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.654180Z","iopub.execute_input":"2022-08-08T15:46:54.654767Z","iopub.status.idle":"2022-08-08T15:46:54.692785Z","shell.execute_reply.started":"2022-08-08T15:46:54.654717Z","shell.execute_reply":"2022-08-08T15:46:54.691524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- # from sklearn.feature_selection import mutual_info_regression\n\n# features = train.dtypes != object\n\n# def make_mi_scores(train, y, discrete_features):\n#     mi_scores = mutual_info_regression(train, y, discrete_features=features)\n#     mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=train.columns)\n#     mi_scores = mi_scores.sort_values(ascending=False)\n#     return mi_scores\n\n# mi_scores = make_mi_scores(train, y, features)\n# mi_scores -->","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:10:11.116692Z","iopub.execute_input":"2022-08-04T14:10:11.117643Z","iopub.status.idle":"2022-08-04T14:10:11.122842Z","shell.execute_reply.started":"2022-08-04T14:10:11.117607Z","shell.execute_reply":"2022-08-04T14:10:11.121219Z"}}},{"cell_type":"code","source":"from feature_engine.encoding import WoEEncoder, RareLabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.694091Z","iopub.execute_input":"2022-08-08T15:46:54.694430Z","iopub.status.idle":"2022-08-08T15:46:54.699086Z","shell.execute_reply.started":"2022-08-08T15:46:54.694399Z","shell.execute_reply":"2022-08-08T15:46:54.698228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here we could try some more features and some more targeting encoding, but I am focused on other competitions right now. I would suggest mean encoding and treatement of nulls in the test set\nwoe_encoder = WoEEncoder(variables=['attribute_0'])\n                                ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.700685Z","iopub.execute_input":"2022-08-08T15:46:54.701029Z","iopub.status.idle":"2022-08-08T15:46:54.713041Z","shell.execute_reply.started":"2022-08-08T15:46:54.700999Z","shell.execute_reply":"2022-08-08T15:46:54.711630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = train, target \n\nseed = 0\nfold = 5","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.715130Z","iopub.execute_input":"2022-08-08T15:46:54.715706Z","iopub.status.idle":"2022-08-08T15:46:54.726495Z","shell.execute_reply.started":"2022-08-08T15:46:54.715668Z","shell.execute_reply":"2022-08-08T15:46:54.725363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.drop(['product_code'],axis=1,inplace=True)\ntest.drop(['product_code'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.728575Z","iopub.execute_input":"2022-08-08T15:46:54.729487Z","iopub.status.idle":"2022-08-08T15:46:54.749864Z","shell.execute_reply.started":"2022-08-08T15:46:54.729440Z","shell.execute_reply":"2022-08-08T15:46:54.749004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"woe_encoder.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.751258Z","iopub.execute_input":"2022-08-08T15:46:54.751934Z","iopub.status.idle":"2022-08-08T15:46:54.789752Z","shell.execute_reply.started":"2022-08-08T15:46:54.751893Z","shell.execute_reply":"2022-08-08T15:46:54.788318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_t = woe_encoder.transform(X)\ntest_t = woe_encoder.transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.791672Z","iopub.execute_input":"2022-08-08T15:46:54.792505Z","iopub.status.idle":"2022-08-08T15:46:54.830371Z","shell.execute_reply.started":"2022-08-08T15:46:54.792456Z","shell.execute_reply":"2022-08-08T15:46:54.828889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_use = ['attribute_0','measurement_0', 'measurement_1', 'measurement_2', 'attribute_0','m_3_missing', 'm_5_missing',\n               'meas_gr1_avg', 'meas_gr1_std', 'attribute_2*3', 'loading', 'measurement_17', 'meas17/meas_gr2_avg']\ntrain = train_t[cols_to_use]\ntest = test_t[cols_to_use]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.831844Z","iopub.execute_input":"2022-08-08T15:46:54.832196Z","iopub.status.idle":"2022-08-08T15:46:54.846824Z","shell.execute_reply.started":"2022-08-08T15:46:54.832165Z","shell.execute_reply":"2022-08-08T15:46:54.844867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- model = LogisticRegression(tol = 1e-4, max_iter=500,random_state=seed)\n\nsearch_space = dict(C=[0.0001, 0.01, 0.1, 1],\n                     penalty=['l2', 'l1'],\n                     solver= ['saga', 'liblinear', 'newton-cg'])\n\nsearch = RandomizedSearchCV(model,\n                            search_space, \n                            random_state=seed,\n                            cv = 5, \n                            scoring='roc_auc')\n\nrand_search = search.fit(X, y)\n\nprint('Best Hyperparameters: %s' % rand_search.best_params_)\nprint(\"Best Estimator: \\n{}\\n\".format(rand_search.best_estimator_))\nprint(\"Best Score: \\n{}\\n\".format(rand_search.best_score_)) -->","metadata":{"execution":{"iopub.status.busy":"2022-08-05T22:45:55.190524Z","iopub.execute_input":"2022-08-05T22:45:55.190921Z","iopub.status.idle":"2022-08-05T22:46:19.212786Z","shell.execute_reply.started":"2022-08-05T22:45:55.19089Z","shell.execute_reply":"2022-08-05T22:46:19.211702Z"}}},{"cell_type":"markdown","source":"### 4. Models\n### 4.1 TFP Logistic Regression","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_probability as tfp\n\ntfk = tf.keras\ntfd = tfp.distributions\ntfpl = tfp.layers","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.853385Z","iopub.execute_input":"2022-08-08T15:46:54.854070Z","iopub.status.idle":"2022-08-08T15:46:54.861974Z","shell.execute_reply.started":"2022-08-08T15:46:54.854022Z","shell.execute_reply":"2022-08-08T15:46:54.860349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y, test = train.values.astype('float32'), y.values.astype('float32'), test.values.astype('float32')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:46:54.863850Z","iopub.execute_input":"2022-08-08T15:46:54.864359Z","iopub.status.idle":"2022-08-08T15:46:54.929378Z","shell.execute_reply.started":"2022-08-08T15:46:54.864319Z","shell.execute_reply":"2022-08-08T15:46:54.927405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hidden_units = [64, 32, 16, 8]\nlearning_rate = 0.001\n\ndef prior(kernel_size, bias_size, dtype=None):\n    n = kernel_size + bias_size\n    prior_model = tf.keras.Sequential(\n        [\n            tfp.layers.DistributionLambda(\n                lambda t: tfp.distributions.MultivariateNormalDiag(\n                    loc=tf.zeros(n), scale_diag=tf.ones(n)\n                )\n            )\n        ]\n    )\n    return prior_model\n\ndef posterior(kernel_size, bias_size, dtype=None):\n    n = kernel_size + bias_size\n    posterior_model = tf.keras.Sequential(\n        [\n            tfp.layers.VariableLayer(\n                tfp.layers.MultivariateNormalTriL.params_size(n), dtype=dtype\n            ),\n            tfp.layers.MultivariateNormalTriL(n),\n        ]\n    )\n    return posterior_model\n\n\ndef create_bnn_model(train_size):\n    inputs = tf.keras.layers.Input(shape=(13,), dtype=tf.float32)\n    features = tf.keras.layers.BatchNormalization()(inputs)\n\n    # Create hidden layers with weight uncertainty using the DenseVariational layer.\n    for units in hidden_units:\n        features = tfp.layers.DenseVariational(\n            units=units,\n            make_prior_fn=prior,\n            make_posterior_fn=posterior,\n            kl_weight=1 / train_size,\n            activation=\"sigmoid\",\n        )(features)\n\n    # The output is deterministic: a single point estimate.\n    outputs = tf.keras.layers.Dense(units=1)(features)\n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:54:39.453133Z","iopub.execute_input":"2022-08-08T15:54:39.453607Z","iopub.status.idle":"2022-08-08T15:54:39.467758Z","shell.execute_reply.started":"2022-08-08T15:54:39.453573Z","shell.execute_reply":"2022-08-08T15:54:39.466308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_size = int(len(X) * 0.8)\nmse_loss = tf.keras.losses.MeanSquaredError()\n\nskf = StratifiedKFold(n_splits=5)\noof_preds = np.zeros((5, len(test)))\n\nfor i,(train_index, test_index) in enumerate(skf.split(X, y)):\n    X_train, X_test = X[train_index], X[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n    \n    batch_size = 32\n    model = create_bnn_model(train_size)\n    model.compile(\n        optimizer=tf.keras.optimizers.RMSprop(learning_rate=learning_rate),\n        loss=mse_loss,\n        metrics=[tf.keras.metrics.RootMeanSquaredError()],\n    )\n        \n    print('Training fold:%2d' % (i+1))\n    \n    callback = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=7, restore_best_weights=True)\n    model.fit(X_train, y_train, validation_data=(X_test, y_test), \n              callbacks=[callback],\n              batch_size=batch_size, epochs=300, verbose=True)\n    clear_output(wait=True)\n    \n    oof_preds[i] = model.predict(test).reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T15:56:47.626149Z","iopub.execute_input":"2022-08-08T15:56:47.626697Z","iopub.status.idle":"2022-08-08T16:02:48.700586Z","shell.execute_reply.started":"2022-08-08T15:56:47.626657Z","shell.execute_reply":"2022-08-08T16:02:48.698954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5. Submissions\n","metadata":{}},{"cell_type":"code","source":"sub = pd.DataFrame({'id': submission.id, 'failure': np.mean(oof_preds, axis=0)})\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T16:06:14.789722Z","iopub.execute_input":"2022-08-08T16:06:14.790177Z","iopub.status.idle":"2022-08-08T16:06:14.853276Z","shell.execute_reply.started":"2022-08-08T16:06:14.790143Z","shell.execute_reply":"2022-08-08T16:06:14.851671Z"},"trusted":true},"execution_count":null,"outputs":[]}]}