{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Copied from @DES with more feature engineering\n\n- measurement_4, measurement_9 and loading missing values were added. ","metadata":{}},{"cell_type":"markdown","source":"# Summary\n\n- Tried several classifiers (such as LogisticRegression, ExtraTrees and kNN, VotingClassifiers, etc..) and blended submission at older versions of the NB. However, my current best score (**LB 0.58952 version 9**) is a single Logistic regression model.\n- Also tried feature engineering by creating new features.\n- Only used 11 features (7 original and 4 derived). **This might change with further experiments**.\n- Imputed missing values using `product_code` grouping.\n- In version of this notebook, I will try [Ambros' suggestion of adding missing values of `measuremsnt_3` and `measuremsnt_5`](https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/342319). Any change in LB from the previous score is a result of that.\n\n\n#### Import Libararies","metadata":{}},{"cell_type":"code","source":"# all may not be needed\n\nimport os\nimport sys\nimport numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pylab as plt\n\n\nfrom scipy.stats import uniform\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier,ExtraTreesClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.ensemble import StackingClassifier,VotingClassifier,StackingClassifier\n\nfrom catboost import CatBoostClassifier\n\nfrom sklearn.model_selection import cross_validate, StratifiedKFold, RepeatedStratifiedKFold, RandomizedSearchCV, GridSearchCV\nfrom sklearn.metrics import confusion_matrix, plot_confusion_matrix, classification_report, roc_auc_score, accuracy_score\nfrom sklearn import metrics\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import SimpleImputer, KNNImputer, IterativeImputer\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import OneHotEncoder, RobustScaler, PowerTransformer, LabelEncoder, StandardScaler, MinMaxScaler\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-07T00:40:56.331649Z","iopub.execute_input":"2022-08-07T00:40:56.332081Z","iopub.status.idle":"2022-08-07T00:40:56.347078Z","shell.execute_reply.started":"2022-08-07T00:40:56.332047Z","shell.execute_reply":"2022-08-07T00:40:56.345766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsubmission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:40:56.349891Z","iopub.execute_input":"2022-08-07T00:40:56.350796Z","iopub.status.idle":"2022-08-07T00:40:56.552760Z","shell.execute_reply.started":"2022-08-07T00:40:56.350748Z","shell.execute_reply":"2022-08-07T00:40:56.551386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. Dataset Overview","metadata":{}},{"cell_type":"code","source":"train.drop('id', axis=1).head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:40:56.556315Z","iopub.execute_input":"2022-08-07T00:40:56.557351Z","iopub.status.idle":"2022-08-07T00:40:56.594583Z","shell.execute_reply.started":"2022-08-07T00:40:56.557295Z","shell.execute_reply":"2022-08-07T00:40:56.593219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop('id', axis=1).head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:40:56.597488Z","iopub.execute_input":"2022-08-07T00:40:56.598768Z","iopub.status.idle":"2022-08-07T00:40:56.631534Z","shell.execute_reply.started":"2022-08-07T00:40:56.598711Z","shell.execute_reply":"2022-08-07T00:40:56.630330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Data size\n- Train dataset  has 26570 rows and 25 columns including the target column (failure).\n- Test dataset has 20775 rows and 24 columsn.","metadata":{}},{"cell_type":"code","source":"display(train.shape)\ndisplay(test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:40:56.633358Z","iopub.execute_input":"2022-08-07T00:40:56.634110Z","iopub.status.idle":"2022-08-07T00:40:56.645072Z","shell.execute_reply.started":"2022-08-07T00:40:56.634062Z","shell.execute_reply":"2022-08-07T00:40:56.643548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Null values\n- Around 3% of the data (cells) is missing in both train and test datset.\n- We will need to impute.","metadata":{}},{"cell_type":"code","source":"print('Train data missing value is = {} %'.format(100* train.isna().sum().sum()/(len(train)*25)))\nprint('Test data missing value is  = {} %'.format(100* test.isna().sum().sum()/(len(test)*25)))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:40:56.647322Z","iopub.execute_input":"2022-08-07T00:40:56.648156Z","iopub.status.idle":"2022-08-07T00:40:56.671100Z","shell.execute_reply.started":"2022-08-07T00:40:56.648110Z","shell.execute_reply":"2022-08-07T00:40:56.669877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_na_cols = [col for col in train.columns if train[col].isnull().sum()!=0]\nprint('Train data cols with missing values ares: \\n', train_na_cols)\n\nprint('\\n')\n\ntest_na_cols = [col for col in test.columns if test[col].isnull().sum()!=0]\nprint('Train data cols with missing values ares: \\n', test_na_cols)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-07T00:40:56.672898Z","iopub.execute_input":"2022-08-07T00:40:56.673644Z","iopub.status.idle":"2022-08-07T00:40:56.706414Z","shell.execute_reply.started":"2022-08-07T00:40:56.673589Z","shell.execute_reply":"2022-08-07T00:40:56.705451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install missingno","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-07T00:40:56.707960Z","iopub.execute_input":"2022-08-07T00:40:56.708989Z","iopub.status.idle":"2022-08-07T00:41:07.838021Z","shell.execute_reply.started":"2022-08-07T00:40:56.708952Z","shell.execute_reply":"2022-08-07T00:41:07.836327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as msno\nmsno.matrix(train.drop('id', axis=1), color=(0.55, 0.75, 0.85), figsize=(16, 8))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:07.844579Z","iopub.execute_input":"2022-08-07T00:41:07.845856Z","iopub.status.idle":"2022-08-07T00:41:08.626109Z","shell.execute_reply.started":"2022-08-07T00:41:07.845808Z","shell.execute_reply":"2022-08-07T00:41:08.625219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2. Correlation plots\n- Plotted the correlation heatmap to get a global idea of what features might be related to each other and may lead us to some feature engineering.\n- We see two groups where feature-to-feature correlation might exist: \n\n> `measuremsnt_17` seems to be correlated with `measurements_5 and 8`\n\n> There seem to be also some sort of correlation between `attributes_2, 3` and `measuremsnt_0, 1`\n\n- There could be an opportunity for a new feature to be derived from these.\n","metadata":{}},{"cell_type":"code","source":"corr = train.drop('id', axis=1).corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nf, ax = plt.subplots(figsize=(12, 12), facecolor='#EAECEE')\ncmap = sns.color_palette(\"rainbow\", as_cmap=True)\nsns.heatmap(corr, mask=mask, cmap=cmap, vmax=1, vmin=-1., center=0, annot=False,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": 0.75})\n\nax.set_title('Correlation heatmap', fontsize=24, y= 1.05)\ncolorbar = ax.collections[0].colorbar","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:08.627501Z","iopub.execute_input":"2022-08-07T00:41:08.628037Z","iopub.status.idle":"2022-08-07T00:41:09.370055Z","shell.execute_reply.started":"2022-08-07T00:41:08.628004Z","shell.execute_reply":"2022-08-07T00:41:09.368874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = train.pop('failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:09.371706Z","iopub.execute_input":"2022-08-07T00:41:09.372066Z","iopub.status.idle":"2022-08-07T00:41:09.377698Z","shell.execute_reply.started":"2022-08-07T00:41:09.372033Z","shell.execute_reply":"2022-08-07T00:41:09.376597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_cols = [col for col in train.columns if train[col].dtypes == 'float64']\nobject_cols = [col for col in train.columns if train[col].dtypes == 'object']\nint_object_cols = [col for col in train.columns[1:-1] if (train[col].dtypes == 'object' or train[col].dtypes == 'int64')]\nnullValue_cols = [col for col in train.columns if train[col].isnull().sum()!=0]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:09.379093Z","iopub.execute_input":"2022-08-07T00:41:09.379456Z","iopub.status.idle":"2022-08-07T00:41:09.400926Z","shell.execute_reply.started":"2022-08-07T00:41:09.379424Z","shell.execute_reply":"2022-08-07T00:41:09.399798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ambros' idea of adding missing values as extra columns\n# https://www.kaggle.com/competitions/tabular-playground-series-aug-2022/discussion/342319\ntrain['m_3_missing'] = train.measurement_3.isna()\ntrain['m_5_missing'] = train.measurement_5.isna()\ntrain['m_4_missing'] = train.measurement_3.isna()\ntrain['m_9_missing'] = train.measurement_5.isna()\ntrain['loading_missing'] = train.loading.isna()\n\ntest['m_3_missing'] = test.measurement_3.isna()\ntest['m_5_missing'] = test.measurement_5.isna()\ntest['m_4_missing'] = test.measurement_3.isna()\ntest['m_9_missing'] = test.measurement_5.isna()\ntest['loading_missing'] = test.loading.isna()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:09.402891Z","iopub.execute_input":"2022-08-07T00:41:09.403972Z","iopub.status.idle":"2022-08-07T00:41:09.419029Z","shell.execute_reply.started":"2022-08-07T00:41:09.403932Z","shell.execute_reply":"2022-08-07T00:41:09.417743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Missing Value Imputation\n- Impute based on product code. We first group the data based on similar product code and impute the missing values using the same group data. This way of imputing missing values was discussed here in [Ambros' notebook](https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense). \n- One option we can use to impute the missing values is LGBMImputer. ","metadata":{}},{"cell_type":"code","source":"# !rm -r kuma_utils\n!git clone https://github.com/analokmaus/kuma_utils.git","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:09.421145Z","iopub.execute_input":"2022-08-07T00:41:09.421555Z","iopub.status.idle":"2022-08-07T00:41:10.604311Z","shell.execute_reply.started":"2022-08-07T00:41:09.421520Z","shell.execute_reply":"2022-08-07T00:41:10.602540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append(\"kuma_utils/\")\nfrom kuma_utils.preprocessing.imputer import LGBMImputer","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:10.606992Z","iopub.execute_input":"2022-08-07T00:41:10.607720Z","iopub.status.idle":"2022-08-07T00:41:10.614763Z","shell.execute_reply.started":"2022-08-07T00:41:10.607670Z","shell.execute_reply":"2022-08-07T00:41:10.613665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train['product_code'].unique())\ndf_A = train[train['product_code']=='A']\ndf_B = train[train['product_code']=='B']\ndf_C = train[train['product_code']=='C']\ndf_D = train[train['product_code']=='D']\ndf_E = train[train['product_code']=='E']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:10.616482Z","iopub.execute_input":"2022-08-07T00:41:10.617622Z","iopub.status.idle":"2022-08-07T00:41:10.666750Z","shell.execute_reply.started":"2022-08-07T00:41:10.617570Z","shell.execute_reply":"2022-08-07T00:41:10.664948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test['product_code'].unique())\ndf_F_t = test[test['product_code']=='F']\ndf_G_t = test[test['product_code']=='G']\ndf_H_t = test[test['product_code']=='H']\ndf_I_t = test[test['product_code']=='I']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:10.668754Z","iopub.execute_input":"2022-08-07T00:41:10.669431Z","iopub.status.idle":"2022-08-07T00:41:10.697865Z","shell.execute_reply.started":"2022-08-07T00:41:10.669385Z","shell.execute_reply":"2022-08-07T00:41:10.696317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_imtr = LGBMImputer(cat_features=object_cols, n_iter=50)\n\n# train dataset\ntrain_iterimp_A = lgbm_imtr.fit_transform(df_A[nullValue_cols])\ntrain_iterimp_B = lgbm_imtr.fit_transform(df_B[nullValue_cols])\ntrain_iterimp_C = lgbm_imtr.fit_transform(df_C[nullValue_cols])\ntrain_iterimp_D = lgbm_imtr.fit_transform(df_D[nullValue_cols])\ntrain_iterimp_E = lgbm_imtr.fit_transform(df_E[nullValue_cols])\n\n# test dataset\ntest_iterimp_F = lgbm_imtr.fit_transform(df_F_t[nullValue_cols])\ntest_iterimp_G = lgbm_imtr.fit_transform(df_G_t[nullValue_cols])\ntest_iterimp_H = lgbm_imtr.fit_transform(df_H_t[nullValue_cols])\ntest_iterimp_I = lgbm_imtr.fit_transform(df_I_t[nullValue_cols])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-07T00:41:10.699715Z","iopub.execute_input":"2022-08-07T00:41:10.700260Z","iopub.status.idle":"2022-08-07T00:41:34.055685Z","shell.execute_reply.started":"2022-08-07T00:41:10.700179Z","shell.execute_reply":"2022-08-07T00:41:34.054652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"none_na_cols = [col for col in train.columns if col not in nullValue_cols]\ndf_train = train[none_na_cols]\ndf_test = test[none_na_cols]\n\ntrain_ = pd.concat([train_iterimp_A, train_iterimp_B,train_iterimp_C,train_iterimp_D,train_iterimp_E], axis=0)\ntrain = pd.concat([df_train, train_], axis=1)\n\ntest_ = pd.concat([test_iterimp_F, test_iterimp_G,test_iterimp_H,test_iterimp_I], axis=0)\ntest = pd.concat([df_test, test_], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.056974Z","iopub.execute_input":"2022-08-07T00:41:34.060914Z","iopub.status.idle":"2022-08-07T00:41:34.085942Z","shell.execute_reply.started":"2022-08-07T00:41:34.060861Z","shell.execute_reply":"2022-08-07T00:41:34.084940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Missing values in train dataset after pre-peocessing is: \", format(train.isna().sum().sum()))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.087437Z","iopub.execute_input":"2022-08-07T00:41:34.088134Z","iopub.status.idle":"2022-08-07T00:41:34.102030Z","shell.execute_reply.started":"2022-08-07T00:41:34.088097Z","shell.execute_reply":"2022-08-07T00:41:34.100852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Feature Engineering Ideas\n- This dataset, being not too obscure what the columns maight be, gives a good opportunity to engineer new features.\n- From initial observations, some of the features seem to be `dimension` measurements and we may be able to combine and derive, area or volume features for example. `attribute_2` and `attribute_3` for example seem to be `width` and `length` dimensions?.\n- Looking at the values of `measurement_3` to `measurement_16`, they all look different variants of the same type of measurement. So we may aggregate into one or two features (mean, std) for example.","metadata":{}},{"cell_type":"code","source":"display(train['attribute_2'].unique())\ndisplay(train['attribute_3'].unique())\nprint()\ndisplay(train['measurement_0'].unique())\ndisplay(train['measurement_1'].unique())\ndisplay(train['measurement_2'].unique())","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-07T00:41:34.103832Z","iopub.execute_input":"2022-08-07T00:41:34.104713Z","iopub.status.idle":"2022-08-07T00:41:34.121631Z","shell.execute_reply.started":"2022-08-07T00:41:34.104678Z","shell.execute_reply":"2022-08-07T00:41:34.120796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 4.1 Combine features and create new \n- Multiply `attribute_2` and `attribute_2` and create a new feature\n- The idea here is that these two seem to me that they are dimensions of the material used for cleaning i.e, width and length for example. We could multiply and get area as new feature instead. Then drop the them.","metadata":{}},{"cell_type":"code","source":"train['attribute_2*3'] = train['attribute_2'] * train['attribute_3']\ntest['attribute_2*3'] = test['attribute_2'] * test['attribute_3']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.122773Z","iopub.execute_input":"2022-08-07T00:41:34.123488Z","iopub.status.idle":"2022-08-07T00:41:34.130354Z","shell.execute_reply.started":"2022-08-07T00:41:34.123454Z","shell.execute_reply":"2022-08-07T00:41:34.129497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- train['meas17_loading_ratio'] = train['measurement_17']*train['attribute_2*3']\ntest['meas17_loading_ratio'] = test['measurement_17']*test['attribute_2*3']\n\ntrain['measurement_012_avg'] = np.mean(train['measurement_0']+train['measurement_1']+train['measurement_2'])\ntest['measurement_012_avg'] = np.mean(test['measurement_0']+test['measurement_1']+test['measurement_2']) -->","metadata":{"execution":{"iopub.status.busy":"2022-08-06T18:54:42.603214Z","iopub.execute_input":"2022-08-06T18:54:42.603843Z","iopub.status.idle":"2022-08-06T18:54:42.619142Z","shell.execute_reply.started":"2022-08-06T18:54:42.603807Z","shell.execute_reply":"2022-08-06T18:54:42.617776Z"}}},{"cell_type":"markdown","source":"#### 4.2 Use aggregation\n- We see from the data that features `measuremsnt_3` to `measurement_16` belong to the same family. More like different variants of the same measurement type. So we could aggregate them into  their averages and may be standard deviations of them.","metadata":{}},{"cell_type":"code","source":"meas_gr1_cols = [f\"measurement_{i:d}\" for i in list(range(3, 5)) + list(range(9, 17))]\ntrain['meas_gr1_avg'] = np.mean(train[meas_gr1_cols], axis=1)\ntrain['meas_gr1_std'] = np.std(train[meas_gr1_cols], axis=1)\n\ntest['meas_gr1_avg'] = np.mean(test[meas_gr1_cols], axis=1)\ntest['meas_gr1_std'] = np.std(test[meas_gr1_cols], axis=1) \n\nmeas_gr2_cols = [f\"measurement_{i:d}\" for i in list(range(5, 9))]\ntrain['meas_gr2_avg'] = np.mean(train[meas_gr2_cols], axis=1)\ntest['meas_gr2_avg'] = np.mean(test[meas_gr2_cols], axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.131441Z","iopub.execute_input":"2022-08-07T00:41:34.132376Z","iopub.status.idle":"2022-08-07T00:41:34.179745Z","shell.execute_reply.started":"2022-08-07T00:41:34.132343Z","shell.execute_reply":"2022-08-07T00:41:34.178531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['meas17/meas_gr2_avg'] = train['measurement_17'] / train['meas_gr2_avg']\ntest['meas17/meas_gr2_avg'] = test['measurement_17'] / test['meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.185032Z","iopub.execute_input":"2022-08-07T00:41:34.185420Z","iopub.status.idle":"2022-08-07T00:41:34.193550Z","shell.execute_reply.started":"2022-08-07T00:41:34.185386Z","shell.execute_reply":"2022-08-07T00:41:34.192192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_to_use = ['measurement_0', 'measurement_1', 'measurement_2', 'attribute_0', 'attribute_1', 'id', 'm_3_missing', 'm_5_missing',\n               'm_4_missing','m_9_missing', 'loading_missing','meas_gr1_avg', 'meas_gr1_std', 'attribute_2*3', 'loading', 'measurement_17', 'meas17/meas_gr2_avg']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.195519Z","iopub.execute_input":"2022-08-07T00:41:34.196775Z","iopub.status.idle":"2022-08-07T00:41:34.205453Z","shell.execute_reply.started":"2022-08-07T00:41:34.196725Z","shell.execute_reply":"2022-08-07T00:41:34.204460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- # train['meas_17/8'] = np.sqrt(train['measurement_17']) /train['measurement_8']\n# train['meas_17/5'] = np.sqrt(train['measurement_17']) /train['measurement_5']\n\ntrain['area*meas_17'] = 1/(train['attribute_2*3']*train['measurement_17'])\ntest['area*meas_17'] = 1/(test['attribute_2*3']*test['measurement_17'])\n\n# test['meas_17/8'] = np.sqrt(test['measurement_17']) +test['measurement_8']\n# test['meas_17/5'] = np.sqrt(test['measurement_17']) +test['measurement_5']\n -->","metadata":{"execution":{"iopub.status.busy":"2022-08-04T19:01:38.557144Z","iopub.execute_input":"2022-08-04T19:01:38.557484Z","iopub.status.idle":"2022-08-04T19:01:38.563605Z","shell.execute_reply.started":"2022-08-04T19:01:38.557445Z","shell.execute_reply":"2022-08-04T19:01:38.562458Z"}}},{"cell_type":"code","source":"train = train[cols_to_use]\ntest = test[cols_to_use]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.206761Z","iopub.execute_input":"2022-08-07T00:41:34.207559Z","iopub.status.idle":"2022-08-07T00:41:34.227979Z","shell.execute_reply.started":"2022-08-07T00:41:34.207526Z","shell.execute_reply":"2022-08-07T00:41:34.227040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.229210Z","iopub.execute_input":"2022-08-07T00:41:34.229838Z","iopub.status.idle":"2022-08-07T00:41:34.250740Z","shell.execute_reply.started":"2022-08-07T00:41:34.229799Z","shell.execute_reply":"2022-08-07T00:41:34.249566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- # # train['meas_1_2'] = train['measurement_1'] + train['measurement_2']\n# # test['meas_1_2'] = test['measurement_1'] + test['measurement_2']\n\n# train['attribute_2+3'] = train['attribute_2'] * train['attribute_3']\n# test['attribute_2+3'] = test['attribute_2'] * test['attribute_3']\n\n# train['att2_meas3'] = train['attribute_2'] * train['measurement_3']\n# test['att2_meas3'] = test['attribute_2'] * test['measurement_3']\n\n# # train['att3_meas0'] = train['attribute_3'] + train['measurement_0']\n# # test['att3_meas0'] = test['attribute_3'] + test['measurement_0']\n\n\n# train.drop(['attribute_2', 'attribute_3'], axis=1, inplace=True)\n# test.drop(['attribute_2', 'attribute_3'], axis=1, inplace=True)\n\n# train.drop(meas_cols, axis=1, inplace=True)\n# test.drop(meas_cols, axis=1, inplace=True) -->","metadata":{"execution":{"iopub.status.busy":"2022-08-04T19:01:38.564931Z","iopub.execute_input":"2022-08-04T19:01:38.565293Z","iopub.status.idle":"2022-08-04T19:01:38.574089Z","shell.execute_reply.started":"2022-08-04T19:01:38.565261Z","shell.execute_reply":"2022-08-04T19:01:38.573082Z"}}},{"cell_type":"markdown","source":"### 5. Label Encoding the `object` columns","metadata":{}},{"cell_type":"code","source":"combined_data = pd.concat([train,test],axis = 0).reset_index(drop=True)\ndef LableEncoder(df_org, df_comb, cats):\n    #https://www.kaggle.com/code/pourchot/update-for-keras-optuna-for-lr/notebook\n    for col in cats :\n        le = LabelEncoder()\n        df_comb[col] = le.fit_transform(df_comb[col])\n\n    train = df_comb[df_comb['id'] < df_org.shape[0]].drop('id',axis=1).copy()\n    test = df_comb[df_comb['id'] > df_org.shape[0]-1].drop('id',axis=1).copy()\n    return train, test","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.252159Z","iopub.execute_input":"2022-08-07T00:41:34.252527Z","iopub.status.idle":"2022-08-07T00:41:34.271257Z","shell.execute_reply.started":"2022-08-07T00:41:34.252496Z","shell.execute_reply":"2022-08-07T00:41:34.269797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cats = ['attribute_0', 'attribute_1', 'm_3_missing', 'm_5_missing', 'm_4_missing','m_9_missing','loading_missing']\ntrain, test = LableEncoder(train,combined_data, cats)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.272970Z","iopub.execute_input":"2022-08-07T00:41:34.273343Z","iopub.status.idle":"2022-08-07T00:41:34.325648Z","shell.execute_reply.started":"2022-08-07T00:41:34.273310Z","shell.execute_reply":"2022-08-07T00:41:34.324198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- # from sklearn.feature_selection import mutual_info_regression\n\n# features = train.dtypes != object\n\n# def make_mi_scores(train, y, discrete_features):\n#     mi_scores = mutual_info_regression(train, y, discrete_features=features)\n#     mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=train.columns)\n#     mi_scores = mi_scores.sort_values(ascending=False)\n#     return mi_scores\n\n# mi_scores = make_mi_scores(train, y, features)\n# mi_scores -->","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:10:11.116692Z","iopub.execute_input":"2022-08-04T14:10:11.117643Z","iopub.status.idle":"2022-08-04T14:10:11.122842Z","shell.execute_reply.started":"2022-08-04T14:10:11.117607Z","shell.execute_reply":"2022-08-04T14:10:11.121219Z"}}},{"cell_type":"code","source":"X, y = train, target \n\nseed = 0\nfold = 5","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.327300Z","iopub.execute_input":"2022-08-07T00:41:34.327718Z","iopub.status.idle":"2022-08-07T00:41:34.332843Z","shell.execute_reply.started":"2022-08-07T00:41:34.327682Z","shell.execute_reply":"2022-08-07T00:41:34.331982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def score(X, y, model, cv):\n    scoring = [\"roc_auc\"]\n    scores = cross_validate(\n        model, X, y, scoring=scoring, cv=cv, return_train_score=True,\n    )\n    scores = pd.DataFrame(scores).T\n    return scores.assign(\n        mean = lambda x: x.mean(axis=1),\n        std = lambda x: x.std(axis=1),\n    )\n\nskf = StratifiedKFold(n_splits=fold, shuffle=True, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.334119Z","iopub.execute_input":"2022-08-07T00:41:34.335351Z","iopub.status.idle":"2022-08-07T00:41:34.348292Z","shell.execute_reply.started":"2022-08-07T00:41:34.335312Z","shell.execute_reply":"2022-08-07T00:41:34.347180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- model = LogisticRegression(tol = 1e-4, max_iter=500,random_state=seed)\n\nsearch_space = dict(C=[0.0001, 0.01, 0.1, 1],\n                     penalty=['l2', 'l1'],\n                     solver= ['saga', 'liblinear', 'newton-cg'])\n\nsearch = RandomizedSearchCV(model,\n                            search_space, \n                            random_state=seed,\n                            cv = 5, \n                            scoring='roc_auc')\n\nrand_search = search.fit(X, y)\n\nprint('Best Hyperparameters: %s' % rand_search.best_params_)\nprint(\"Best Estimator: \\n{}\\n\".format(rand_search.best_estimator_))\nprint(\"Best Score: \\n{}\\n\".format(rand_search.best_score_)) -->","metadata":{"execution":{"iopub.status.busy":"2022-08-05T22:45:55.190524Z","iopub.execute_input":"2022-08-05T22:45:55.190921Z","iopub.status.idle":"2022-08-05T22:46:19.212786Z","shell.execute_reply.started":"2022-08-05T22:45:55.19089Z","shell.execute_reply":"2022-08-05T22:46:19.211702Z"}}},{"cell_type":"markdown","source":"### 4. Models\n### 4.1 LogisticRegression","metadata":{}},{"cell_type":"code","source":"model_lr = LogisticRegression(max_iter = 500, C=0.0001, penalty='l2', solver='newton-cg')\nscores = score(X, y, model_lr, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:34.350316Z","iopub.execute_input":"2022-08-07T00:41:34.351393Z","iopub.status.idle":"2022-08-07T00:41:39.015081Z","shell.execute_reply.started":"2022-08-07T00:41:34.351339Z","shell.execute_reply":"2022-08-07T00:41:39.013452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lr.fit(X, y);\ny_pred_lr = pd.Series(\n    model_lr.predict_proba(test)[:, 1],\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:39.017648Z","iopub.execute_input":"2022-08-07T00:41:39.018293Z","iopub.status.idle":"2022-08-07T00:41:40.319802Z","shell.execute_reply.started":"2022-08-07T00:41:39.018213Z","shell.execute_reply":"2022-08-07T00:41:40.318013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5. Submissions\n","metadata":{}},{"cell_type":"code","source":"sub = pd.DataFrame({'id': submission.id, 'failure': y_pred_lr})\nsub.to_csv(\"submission_LR.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T00:41:40.329916Z","iopub.execute_input":"2022-08-07T00:41:40.335983Z","iopub.status.idle":"2022-08-07T00:41:40.459713Z","shell.execute_reply.started":"2022-08-07T00:41:40.335886Z","shell.execute_reply":"2022-08-07T00:41:40.458594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}