{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is heavily inspired and draws its main strengths from delai50's approach (https://www.kaggle.com/code/delai50/standing-on-the-head-of-giants-more-fe/notebook), who in turn derives from other notebooks. \n\nBasically, my approach is to create some new features and use cross-validation (more precisely, sklearn.feature_selection.RFECV with leave-one-group-out cross validation, i.e., four product codes are used to predict the observations of the fifth code) to select the features that can actually predict the target variable. Then, hyperparameters of the logistic regression are optimized (again using LOGO).","metadata":{}},{"cell_type":"code","source":"# import stuff\nimport numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import cross_validate, LeaveOneGroupOut, GridSearchCV\nfrom sklearn.preprocessing import StandardScaler, SplineTransformer, OneHotEncoder\nimport matplotlib.pyplot as plt\n!pip install feature-engine\nfrom feature_engine.encoding import WoEEncoder\nfrom sklearn.feature_selection import RFECV\n\nimport sys, os\nif not os.path.isdir(\"kuma_utils/\"):\n    !git clone https://github.com/analokmaus/kuma_utils.git\n\nsys.path.append(\"kuma_utils/\")\nfrom kuma_utils.preprocessing.imputer import LGBMImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:24.497103Z","iopub.execute_input":"2022-08-11T14:05:24.498066Z","iopub.status.idle":"2022-08-11T14:05:36.046091Z","shell.execute_reply.started":"2022-08-11T14:05:24.497993Z","shell.execute_reply":"2022-08-11T14:05:36.044678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data\ntrain = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ntest = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\")\n\ntest_ids = test['id']\npcodes = train['product_code']\npcodes_test = test['product_code']\ntrain_failure = train['failure']\ntrain = train.drop(['id', 'product_code', 'failure'], axis=1)\ntest = test.drop(['id', 'product_code'], axis=1)\ntrain_loading = train['loading']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:36.049676Z","iopub.execute_input":"2022-08-11T14:05:36.050232Z","iopub.status.idle":"2022-08-11T14:05:36.270884Z","shell.execute_reply.started":"2022-08-11T14:05:36.050163Z","shell.execute_reply":"2022-08-11T14:05:36.269591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check that train and test have the same structure \n[c for c in train.columns] == [c for c in test.columns]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:36.272559Z","iopub.execute_input":"2022-08-11T14:05:36.273077Z","iopub.status.idle":"2022-08-11T14:05:36.283442Z","shell.execute_reply.started":"2022-08-11T14:05:36.273027Z","shell.execute_reply":"2022-08-11T14:05:36.281712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get NA features  \ntrain['m3_na'] = 1*train['measurement_3'].isna()\ntest['m3_na'] = 1*test['measurement_3'].isna()\n\ntrain['m5_na'] = 1*train['measurement_5'].isna()\ntest['m5_na'] = 1*test['measurement_5'].isna()\n\n# count of NAs across columns\nmeas_names = [\"measurement_\" + format(x) for x in range(3, 16+1)]\n\ntrain_meas_isna = np.array(train[meas_names].isna())\n\ntrain['na_count'] = np.sum(np.array(train[meas_names].isna()), axis=1)\ntest['na_count'] = np.sum(np.array(test[meas_names].isna()), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:36.286822Z","iopub.execute_input":"2022-08-11T14:05:36.287278Z","iopub.status.idle":"2022-08-11T14:05:36.315637Z","shell.execute_reply.started":"2022-08-11T14:05:36.287236Z","shell.execute_reply":"2022-08-11T14:05:36.314389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# impute values with LGBM (https://www.kaggle.com/code/themikejones/tps-aug-22-votingclassifier/notebook?scriptVersionId=102761965)\nfloat_cols = [col for col in train.columns if train[col].dtypes == 'float64']\nobject_cols = [col for col in train.columns if train[col].dtypes == 'object']\nint_object_cols = [col for col in train.columns[1:-1] if (train[col].dtypes == 'object' or train[col].dtypes == 'int64')]\nnullValue_cols = [col for col in train.columns if train[col].isnull().sum()!=0]\n\n\ndf_A = train[pcodes=='A']\ndf_B = train[pcodes=='B']\ndf_C = train[pcodes=='C']\ndf_D = train[pcodes=='D']\ndf_E = train[pcodes=='E']\n\ndf_F_t = test[pcodes_test=='F']\ndf_G_t = test[pcodes_test=='G']\ndf_H_t = test[pcodes_test=='H']\ndf_I_t = test[pcodes_test=='I']\n\nlgbm_imtr = LGBMImputer(cat_features=object_cols, n_iter=50)\n\ntrain_iterimp_A = lgbm_imtr.fit_transform(df_A[nullValue_cols])\ntrain_iterimp_B = lgbm_imtr.fit_transform(df_B[nullValue_cols])\ntrain_iterimp_C = lgbm_imtr.fit_transform(df_C[nullValue_cols])\ntrain_iterimp_D = lgbm_imtr.fit_transform(df_D[nullValue_cols])\ntrain_iterimp_E = lgbm_imtr.fit_transform(df_E[nullValue_cols])\n\ntest_iterimp_F = lgbm_imtr.fit_transform(df_F_t[nullValue_cols])\ntest_iterimp_G = lgbm_imtr.fit_transform(df_G_t[nullValue_cols])\ntest_iterimp_H = lgbm_imtr.fit_transform(df_H_t[nullValue_cols])\ntest_iterimp_I = lgbm_imtr.fit_transform(df_I_t[nullValue_cols])\n\nnone_na_cols = [col for col in train.columns if col not in nullValue_cols]\ndf_train = train[none_na_cols]\ndf_test = test[none_na_cols]\n\ntrain_ = pd.concat([train_iterimp_A, train_iterimp_B,train_iterimp_C,train_iterimp_D,train_iterimp_E], axis=0)\ntrain = pd.concat([df_train, train_], axis=1)\n\ntest_ = pd.concat([test_iterimp_F, test_iterimp_G,test_iterimp_H,test_iterimp_I], axis=0)\ntest = pd.concat([df_test, test_], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:36.317307Z","iopub.execute_input":"2022-08-11T14:05:36.317645Z","iopub.status.idle":"2022-08-11T14:05:59.354885Z","shell.execute_reply.started":"2022-08-11T14:05:36.317614Z","shell.execute_reply":"2022-08-11T14:05:59.353603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generate features\nmeas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain['meas_gr1_avg'] = np.mean(train[meas_gr1_cols], axis=1)\ntrain['meas_gr1_std'] = np.std(train[meas_gr1_cols], axis=1)\n\ntest['meas_gr1_avg'] = np.mean(test[meas_gr1_cols], axis=1)\ntest['meas_gr1_std'] = np.std(test[meas_gr1_cols], axis=1) \n\nmeas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\ntrain['meas_gr2_avg'] = np.mean(train[meas_gr2_cols], axis=1)\ntest['meas_gr2_avg'] = np.mean(test[meas_gr2_cols], axis=1)\n\ntrain['meas17/meas_gr2_avg'] = train['measurement_17'] / train['meas_gr2_avg']\ntest['meas17/meas_gr2_avg'] = test['measurement_17'] / test['meas_gr2_avg']\n\ntrain['logload'] = np.log(train['loading'])\ntest['logload'] = np.log(test['loading'])\n\ntrain['loading_measmean1'] = train['loading'] / train['meas_gr1_avg']\ntest['loading_measmean1'] = test['loading'] / test['meas_gr1_avg']\n\ntrain['loading_measmean2'] = train['loading'] / train['meas_gr2_avg']\ntest['loading_measmean2'] = test['loading'] / test['meas_gr2_avg']\n\ntrain['area'] = train['attribute_2'] * train['attribute_3']\ntest['area'] = test['attribute_2'] * test['attribute_3']\n\ntrain['loading_area'] = train['loading'] / train['area']\ntest['loading_area'] = test['loading'] / test['area']\n\ntrain['same_material'] = np.array( train['attribute_0'] == train['attribute_1'], dtype='float')\ntest['same_material'] = np.array(test['attribute_0'] == test['attribute_1'], dtype='float')\n\ntrain['m17/load'] = train['measurement_17'] / train['loading']\ntest['m17/load'] = test['measurement_17'] / test['loading']\n\ntrain['m17/area'] = train['measurement_17'] / train['area']\ntest['m17/area'] = test['measurement_17'] / test['area']\n\ntrain['loading_a2'] = train['loading'] / train['attribute_2']\ntest['loading_a2'] = test['loading'] / test['attribute_2']\n\ntrain['loading_a3'] = train['loading'] / train['attribute_3']\ntest['loading_a3'] = test['loading'] / test['attribute_3']\n\ntrain['m0_load'] = train['measurement_0'] / train['loading']\ntest['m0_load'] = test['measurement_0'] / test['loading']\n\ntrain['m1_load'] = train['measurement_1'] / train['loading']\ntest['m1_load'] = test['measurement_1'] / test['loading']\n\ntrain['m2_load'] = train['measurement_2'] / train['loading']\ntest['m2_load'] = test['measurement_2'] / test['loading']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.356928Z","iopub.execute_input":"2022-08-11T14:05:59.357305Z","iopub.status.idle":"2022-08-11T14:05:59.445553Z","shell.execute_reply.started":"2022-08-11T14:05:59.357273Z","shell.execute_reply":"2022-08-11T14:05:59.444315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# WoE encoder (whatever that is) for attribute 0\n\nwoe_encoder = WoEEncoder(variables=['attribute_0'])\nwoe_encoder.fit(train, train_failure)\ntrain = woe_encoder.transform(train)\ntest = woe_encoder.transform(test)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.447350Z","iopub.execute_input":"2022-08-11T14:05:59.447750Z","iopub.status.idle":"2022-08-11T14:05:59.532703Z","shell.execute_reply.started":"2022-08-11T14:05:59.447708Z","shell.execute_reply":"2022-08-11T14:05:59.531124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove non-numeric classes at this step\ntrain.drop('attribute_1', axis=1, inplace=True)\ntest.drop('attribute_1', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.534334Z","iopub.execute_input":"2022-08-11T14:05:59.534716Z","iopub.status.idle":"2022-08-11T14:05:59.548257Z","shell.execute_reply.started":"2022-08-11T14:05:59.534683Z","shell.execute_reply":"2022-08-11T14:05:59.547006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scale everything to mean=0 and sd=1\nsts = StandardScaler()\nsts.fit(train)\ntrain = pd.DataFrame(sts.transform(train), columns=train.columns)\ntest = pd.DataFrame(sts.transform(test), columns=test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.550813Z","iopub.execute_input":"2022-08-11T14:05:59.551432Z","iopub.status.idle":"2022-08-11T14:05:59.606545Z","shell.execute_reply.started":"2022-08-11T14:05:59.551378Z","shell.execute_reply":"2022-08-11T14:05:59.605028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select columns \ncols_to_use = [\n    'attribute_0',\n    'attribute_2',\n    'attribute_3',\n    'measurement_0',\n    'measurement_1',\n    'measurement_2',\n    'na_count',\n    'm3_na',\n    'm5_na',\n    'area',\n    'same_material',\n    'loading', \n    'logload', \n    'measurement_17',\n    'loading_area',\n    'meas_gr1_avg', \n    'meas_gr1_std',\n    'meas_gr2_avg',\n    'meas17/meas_gr2_avg',  \n    'loading_measmean1',\n    'loading_measmean2',\n    'm17/load',\n    'm17/area',\n    'loading_a2',\n    'loading_a3',\n    'm0_load',\n    'm1_load',\n    'm2_load'\n    #*meas_names # pure measurements 3-16 excluded here (CV would throw them out anyway)\n    ]\n\ntrain = train[cols_to_use]\ntest = test[cols_to_use]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.612337Z","iopub.execute_input":"2022-08-11T14:05:59.613640Z","iopub.status.idle":"2022-08-11T14:05:59.626435Z","shell.execute_reply.started":"2022-08-11T14:05:59.613573Z","shell.execute_reply":"2022-08-11T14:05:59.624768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CV for feature selection \n\nlogo = LeaveOneGroupOut()\ncv_features = logo.split(train, train_failure, pcodes)\nlr_features = LogisticRegression(penalty='l2', C=0.0003, solver='saga', max_iter=1000)\nfeature_select = RFECV(lr_features, cv=cv_features, scoring='roc_auc').fit(train, train_failure)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:59.628718Z","iopub.execute_input":"2022-08-11T14:05:59.629239Z","iopub.status.idle":"2022-08-11T14:06:33.242255Z","shell.execute_reply.started":"2022-08-11T14:05:59.629172Z","shell.execute_reply":"2022-08-11T14:06:33.240994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check results of feature selection and save dfs\ncved_features = feature_select.get_feature_names_out()\n\ntrain_cved = train[cved_features]\ntest_cved = test[cved_features]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:33.243737Z","iopub.execute_input":"2022-08-11T14:06:33.244068Z","iopub.status.idle":"2022-08-11T14:06:33.254324Z","shell.execute_reply.started":"2022-08-11T14:06:33.244038Z","shell.execute_reply":"2022-08-11T14:06:33.252912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show which features are selected\npd.DataFrame({'vars': train.columns, 'support': feature_select.support_})\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:33.255918Z","iopub.execute_input":"2022-08-11T14:06:33.256327Z","iopub.status.idle":"2022-08-11T14:06:33.275284Z","shell.execute_reply.started":"2022-08-11T14:06:33.256293Z","shell.execute_reply":"2022-08-11T14:06:33.274091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hyperparameter optimization\ncv_params = logo.split(train_cved, train_failure, pcodes)\n\nlr_model_search = LogisticRegression(penalty='elasticnet', solver='saga', max_iter=1000)\nparam_grid = {'l1_ratio': np.arange(0.0, 0.5, 0.1), 'C': [5e-5, 8e-5, 0.0001, 0.00015, 0.0002, 0.0003]}\ngscv = GridSearchCV(lr_model_search, param_grid, scoring='roc_auc', cv=cv_params).fit(train_cved, train_failure)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:33.276983Z","iopub.execute_input":"2022-08-11T14:06:33.277388Z","iopub.status.idle":"2022-08-11T14:06:58.343364Z","shell.execute_reply.started":"2022-08-11T14:06:33.277353Z","shell.execute_reply":"2022-08-11T14:06:58.341832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best estimator\ngscv.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.345080Z","iopub.execute_input":"2022-08-11T14:06:58.345500Z","iopub.status.idle":"2022-08-11T14:06:58.353647Z","shell.execute_reply.started":"2022-08-11T14:06:58.345462Z","shell.execute_reply":"2022-08-11T14:06:58.352288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# score in LOGO CV \ngscv.best_score_\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.355647Z","iopub.execute_input":"2022-08-11T14:06:58.356170Z","iopub.status.idle":"2022-08-11T14:06:58.367471Z","shell.execute_reply.started":"2022-08-11T14:06:58.356122Z","shell.execute_reply":"2022-08-11T14:06:58.366298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at regression coefficients \ngscv.best_estimator_.coef_","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.368958Z","iopub.execute_input":"2022-08-11T14:06:58.369391Z","iopub.status.idle":"2022-08-11T14:06:58.381861Z","shell.execute_reply.started":"2022-08-11T14:06:58.369356Z","shell.execute_reply":"2022-08-11T14:06:58.380813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict failure in the test set\nprediction = gscv.predict_proba(test_cved)\nprediction","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.383708Z","iopub.execute_input":"2022-08-11T14:06:58.384073Z","iopub.status.idle":"2022-08-11T14:06:58.402910Z","shell.execute_reply.started":"2022-08-11T14:06:58.384040Z","shell.execute_reply":"2022-08-11T14:06:58.401081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# histogram of failure predictions\nn, bins, patches = plt.hist(prediction[:,1], density=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.404842Z","iopub.execute_input":"2022-08-11T14:06:58.407351Z","iopub.status.idle":"2022-08-11T14:06:58.664801Z","shell.execute_reply.started":"2022-08-11T14:06:58.407278Z","shell.execute_reply":"2022-08-11T14:06:58.663479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scatterplot: loading and predictions (loading is clearly the dominant variable)\n\nfitted = gscv.predict_proba(train_cved)\nplt.scatter(train_loading, fitted[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:06:58.667458Z","iopub.execute_input":"2022-08-11T14:06:58.667837Z","iopub.status.idle":"2022-08-11T14:06:59.153942Z","shell.execute_reply.started":"2022-08-11T14:06:58.667805Z","shell.execute_reply":"2022-08-11T14:06:59.152786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save submission\nsub = pd.DataFrame({'id': test_ids, 'failure': prediction[:,1]})\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:08:07.098326Z","iopub.execute_input":"2022-08-11T14:08:07.098931Z","iopub.status.idle":"2022-08-11T14:08:07.158671Z","shell.execute_reply.started":"2022-08-11T14:08:07.098883Z","shell.execute_reply":"2022-08-11T14:08:07.157184Z"},"trusted":true},"execution_count":null,"outputs":[]}]}