{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# logistic regression & feature selection is all u need\n\nThis is a simple notebook fitting logistic regression Grid Search. \n\nSpecial thanks to Devastator for his [clean notebook ](https://www.kaggle.com/code/thedevastator/tps-aug-simple-baseline)\nand to Lucas See for his [great EDA](https://www.kaggle.com/code/pinstripezebra/eda-baseline-model). Make sure to give them an upvote! ","metadata":{}},{"cell_type":"markdown","source":"![](https://media.slid.es/uploads/1047697/images/6229366/pasted-from-clipboard.png)","metadata":{}},{"cell_type":"code","source":"# Viz\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Data Processing\nimport numpy as np\nimport pandas as pd\nfrom sklearn import preprocessing\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.preprocessing import LabelEncoder\n\n# Modeling \nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cluster, accuracy_score, roc_auc_score\nfrom sklearn.model_selection import cross_validate, GridSearchCV, cross_val_score, StratifiedKFold\n\n# Other \nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T13:43:49.143963Z","iopub.execute_input":"2022-08-06T13:43:49.144434Z","iopub.status.idle":"2022-08-06T13:43:49.154625Z","shell.execute_reply.started":"2022-08-06T13:43:49.144400Z","shell.execute_reply":"2022-08-06T13:43:49.153066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Data\ntrain = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsample_submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:43:49.405869Z","iopub.execute_input":"2022-08-06T13:43:49.406407Z","iopub.status.idle":"2022-08-06T13:43:49.627473Z","shell.execute_reply.started":"2022-08-06T13:43:49.406357Z","shell.execute_reply":"2022-08-06T13:43:49.626536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variable lists for easy manipulation\nid_var = ['id']\ntarget= ['failure']\ncat_vars = ['product_code','attribute_0','attribute_1']\nnum_vars = [v for v in test.columns if v not in id_var and v not in cat_vars]\npredictors = cat_vars + num_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:43:49.773892Z","iopub.execute_input":"2022-08-06T13:43:49.775019Z","iopub.status.idle":"2022-08-06T13:43:49.781985Z","shell.execute_reply.started":"2022-08-06T13:43:49.774971Z","shell.execute_reply":"2022-08-06T13:43:49.780695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-Processing ","metadata":{}},{"cell_type":"code","source":"!rm -r kuma_utils\n!git clone https://github.com/analokmaus/kuma_utils.git","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:43:50.038126Z","iopub.execute_input":"2022-08-06T13:43:50.038910Z","iopub.status.idle":"2022-08-06T13:44:12.349643Z","shell.execute_reply.started":"2022-08-06T13:43:50.038867Z","shell.execute_reply":"2022-08-06T13:44:12.348205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Track missing values \n#for v in num_vars: \n#    if train[v].isna().sum() > 0: \n#        train[f'na_{v}']= np.where(train[v].isna()==True, 1,0)\n#        test[f'na_{v}']= np.where(test[v].isna()==True, 1,0)\n        \n#Imputing missing values \nmulti_imp = IterativeImputer(max_iter = 9, random_state = 42, verbose = 0, skip_complete = True, n_nearest_features = 10, tol = 0.001)\nmulti_imp.fit(train[num_vars])\ntrain[num_vars] = multi_imp.transform(train[num_vars])\ntest[num_vars] = multi_imp.transform(test[num_vars])\n\nprint('Train NA:', train.isna().sum().sum())\nprint('Test NA:', train.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:12.352964Z","iopub.execute_input":"2022-08-06T13:44:12.353405Z","iopub.status.idle":"2022-08-06T13:44:18.504602Z","shell.execute_reply.started":"2022-08-06T13:44:12.353367Z","shell.execute_reply":"2022-08-06T13:44:18.503314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize columns\n#attributes = ['attribute_2', 'attribute_3', 'measurement_4', 'measurement_5', 'measurement_6']\n#train[attributes] = preprocessing.normalize(train[attributes])\n#test[attributes] = preprocessing.normalize(test[attributes])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.506329Z","iopub.execute_input":"2022-08-06T13:44:18.507381Z","iopub.status.idle":"2022-08-06T13:44:18.515273Z","shell.execute_reply.started":"2022-08-06T13:44:18.507323Z","shell.execute_reply":"2022-08-06T13:44:18.513382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop product code\ntest = test.drop(['product_code'], axis = 1)\ntrain = train.drop(['product_code'], axis = 1)\ncat_vars.remove('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.520308Z","iopub.execute_input":"2022-08-06T13:44:18.528396Z","iopub.status.idle":"2022-08-06T13:44:18.557709Z","shell.execute_reply.started":"2022-08-06T13:44:18.528308Z","shell.execute_reply":"2022-08-06T13:44:18.556514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"I would like to thank DES. for his great feature engineering ideas on [this notebook.](https://www.kaggle.com/code/desalegngeb/tps08-logisticregression-and-some-fe) Give him an upvote! \n\nI added max and min features but they don't seem to be relevant later on. the loading ratio is the most important from all of them according to fisher scores.","metadata":{}},{"cell_type":"code","source":"train['attribute_2*3'] = train['attribute_2'] * train['attribute_3']\ntest['attribute_2*3'] = test['attribute_2'] * test['attribute_3']\n\ntrain['meas17_loading_ratio'] = train['measurement_17']/train['loading']\ntest['meas17_loading_ratio'] = test['measurement_17']/test['loading']\n\nmeas_cols = [f\"measurement_{i:d}\" for i in list(range(3, 17))]\ntrain['meas_avg'] = np.mean(train[meas_cols], axis=1)\ntrain['meas_std'] = np.std(train[meas_cols], axis=1)\ntrain['meas_max'] = np.max(train[meas_cols], axis=1)\ntrain['meas_min'] = np.min(train[meas_cols], axis=1)\n\ntest['meas_avg'] = np.mean(test[meas_cols], axis=1)\ntest['meas_std'] = np.std(test[meas_cols], axis=1) \ntest['meas_max'] = np.max(test[meas_cols], axis=1)\ntest['meas_min'] = np.min(test[meas_cols], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.560151Z","iopub.execute_input":"2022-08-06T13:44:18.560654Z","iopub.status.idle":"2022-08-06T13:44:18.649987Z","shell.execute_reply.started":"2022-08-06T13:44:18.560619Z","shell.execute_reply":"2022-08-06T13:44:18.648949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label Encoding\nlabel_encoder = LabelEncoder()\ntrain_le = train.copy()\ntest_le = test.copy()\n\nfor col in ['attribute_0', 'attribute_1']:\n    train_le[col] = label_encoder.fit_transform(train[col])\n    test_le[col] = label_encoder.fit_transform(test[col]) \n        \ntrain = train_le\ntest = test_le","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.651397Z","iopub.execute_input":"2022-08-06T13:44:18.652165Z","iopub.status.idle":"2022-08-06T13:44:18.699233Z","shell.execute_reply.started":"2022-08-06T13:44:18.652126Z","shell.execute_reply":"2022-08-06T13:44:18.697565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# update predictor list\npredictors = [v for v in train.columns if v not in id_var and v not in target]\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.701076Z","iopub.execute_input":"2022-08-06T13:44:18.701616Z","iopub.status.idle":"2022-08-06T13:44:18.711571Z","shell.execute_reply.started":"2022-08-06T13:44:18.701569Z","shell.execute_reply":"2022-08-06T13:44:18.710360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Selection with Fisher's Scores","metadata":{}},{"cell_type":"code","source":"def FisherScore(bt, y_train, predictors):\n    \"\"\"\n    Verbeke, W., Dejaeger, K., Martens, D., Hur, J., & Baesens, B. (2012). New insights\n    into churn prediction in the telecommunication sector: A profit driven data mining\n    approach. European Journal of Operational Research, 218(1), 211-229.\n    \"\"\"\n    \n    # Get the unique values of dependent variable\n    target_var_val = y_train.unique()\n    \n    # Calculate FisherScore for each predictor\n    predictor_FisherScore = []\n    for v in predictors:\n        fs = np.abs(np.mean(bt.loc[y_train == target_var_val[0], v]) - np.mean(bt.loc[y_train == target_var_val[1], v])) / \\\n             np.sqrt(np.var(bt.loc[y_train == target_var_val[0], v]) + np.var(bt.loc[y_train == target_var_val[1], v]))\n        predictor_FisherScore.append(fs)\n    return predictor_FisherScore","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.713350Z","iopub.execute_input":"2022-08-06T13:44:18.714208Z","iopub.status.idle":"2022-08-06T13:44:18.722829Z","shell.execute_reply.started":"2022-08-06T13:44:18.714152Z","shell.execute_reply":"2022-08-06T13:44:18.721811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate Fisher Score for all variables\nfs = FisherScore(train, train['failure'], predictors)\nfs_df = pd.DataFrame({\"predictor\":predictors, \"fisherscore\":fs})\nfs_df = fs_df.sort_values('fisherscore', ascending=False)\nfs_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:18.724325Z","iopub.execute_input":"2022-08-06T13:44:18.724965Z","iopub.status.idle":"2022-08-06T13:44:18.851305Z","shell.execute_reply.started":"2022-08-06T13:44:18.724929Z","shell.execute_reply":"2022-08-06T13:44:18.849685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how AUC changes when adding more variables sorted by fisher score importance\nfs_scores = []\ntop_n_vars = 20\nfor i in range(1, top_n_vars+1):\n    if i % 10 == 0: print('Added # top vars :', i)\n    top_n_predictors = fs_df['predictor'][:i]\n    clf = LogisticRegression(max_iter = 2500)\n    fs_scores.append(cross_validate(clf, train[top_n_predictors], train['failure'],\n                                    scoring='roc_auc', cv=5, verbose=0, n_jobs=-1, return_train_score=True))\n\n# Plot\nplt.plot([s['train_score'].mean() for s in fs_scores], color='blue')\nplt.plot([s['test_score'].mean() for s in fs_scores], color='red')\nplt.xlabel('# vars')\nplt.ylabel('AUC')\nplt.legend(['train', 'test'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:44:48.582682Z","iopub.execute_input":"2022-08-06T13:44:48.583598Z","iopub.status.idle":"2022-08-06T13:45:21.912153Z","shell.execute_reply.started":"2022-08-06T13:44:48.583553Z","shell.execute_reply":"2022-08-06T13:45:21.910539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Even though the test score is better with 4 variables, it seems to be more stable with 14 variables. The AUC actually improves by using 14 instead of 4. Also, for some reason by adding the 'meas_std' variable our score improves a bit. Perhaps we could do a different feature selection method since this one doesn't seem to give us the best results on it's own. This is still a work in progress","metadata":{}},{"cell_type":"code","source":"# Select the top variables based on Fisher Score\nn_top_fs_vars = 13  # Top FS vars\ntop_fs_vars = fs_df['predictor'].values[:n_top_fs_vars]\ntop_fs_vars = top_fs_vars.tolist() ","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.914916Z","iopub.execute_input":"2022-08-06T13:45:21.915340Z","iopub.status.idle":"2022-08-06T13:45:21.922886Z","shell.execute_reply.started":"2022-08-06T13:45:21.915303Z","shell.execute_reply":"2022-08-06T13:45:21.921098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# Train test split\nX_train, X_test, y_train, y_test = train_test_split(train[predictors], train[target], test_size=0.20, random_state=42)\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.927800Z","iopub.execute_input":"2022-08-06T13:45:21.928253Z","iopub.status.idle":"2022-08-06T13:45:21.978441Z","shell.execute_reply.started":"2022-08-06T13:45:21.928216Z","shell.execute_reply":"2022-08-06T13:45:21.976980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T10:21:19.770777Z","iopub.execute_input":"2022-08-05T10:21:19.771354Z","iopub.status.idle":"2022-08-05T10:21:19.777448Z","shell.execute_reply.started":"2022-08-05T10:21:19.771312Z","shell.execute_reply":"2022-08-05T10:21:19.775839Z"}}},{"cell_type":"code","source":"seed = 0\nfold = 5\n\ndef score(X, y, model, cv):\n    scoring = [\"roc_auc\"]\n    scores = cross_validate(\n        model, X, y, scoring=scoring, cv=cv, return_train_score=True,\n    )\n    scores = pd.DataFrame(scores).T\n    return scores.assign(\n        mean = lambda x: x.mean(axis=1),\n        std = lambda x: x.std(axis=1),\n    )\n\nskf = StratifiedKFold(n_splits=fold, shuffle=True, random_state=seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:21.981378Z","iopub.execute_input":"2022-08-06T13:45:21.981814Z","iopub.status.idle":"2022-08-06T13:45:21.989500Z","shell.execute_reply.started":"2022-08-06T13:45:21.981749Z","shell.execute_reply":"2022-08-06T13:45:21.988570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression all predictors\nlr = LogisticRegression()\n\nscores = score(train[predictors], train['failure'], lr, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:45:31.493990Z","iopub.execute_input":"2022-08-06T13:45:31.495094Z","iopub.status.idle":"2022-08-06T13:45:34.148436Z","shell.execute_reply.started":"2022-08-06T13:45:31.495036Z","shell.execute_reply":"2022-08-06T13:45:34.146423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression selected vars\nnew_predictors = ['measurement_0', 'measurement_1', 'measurement_2', 'attribute_0', 'attribute_1', \n               'meas_avg', 'meas_std','attribute_2*3', 'loading', 'measurement_17']\n\nlr = LogisticRegression(random_state = 0)\nlr.fit(train[new_predictors], train['failure'])\n\nscores = score(train[new_predictors], train['failure'], lr, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:48:03.177332Z","iopub.execute_input":"2022-08-06T13:48:03.177886Z","iopub.status.idle":"2022-08-06T13:48:05.792558Z","shell.execute_reply.started":"2022-08-06T13:48:03.177844Z","shell.execute_reply":"2022-08-06T13:48:05.790832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Logistic Regression top fs vars\nlr = LogisticRegression(random_state = 0)\nlr.fit(train[top_fs_vars], train['failure'])\n\nscores = score(train[top_fs_vars], train['failure'], lr, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:46:19.606128Z","iopub.execute_input":"2022-08-06T13:46:19.606663Z","iopub.status.idle":"2022-08-06T13:46:22.430706Z","shell.execute_reply.started":"2022-08-06T13:46:19.606628Z","shell.execute_reply":"2022-08-06T13:46:22.428338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LR Grid Search\nThis is commented out so the notebook runs faster. params are saved on the next cell","metadata":{}},{"cell_type":"code","source":"# Logistic Regression Grid Search\n#space = {\"C\":np.logspace(-4, 4, 50),\"penalty\":[\"l1\", \"l2\"], \"solver\":['liblinear', 'lbfgs', 'newton-cg']}\n#gsLR = GridSearchCV(LogisticRegression(max_iter = 200), space ,scoring='roc_auc', cv = 3, verbose = 1)\n#gsLR.fit(X_train[top_fs_vars], y_train.values.squeeze())\n#print(gsLR.best_params_)\n\n#LR_best = gsLR.best_estimator_\n#print('Validation AUC:', roc_auc_score(y_test, LR_best.predict_proba(X_test[top_fs_vars])[:,1]))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T13:08:54.381425Z","iopub.execute_input":"2022-08-06T13:08:54.386563Z","iopub.status.idle":"2022-08-06T13:08:54.399132Z","shell.execute_reply.started":"2022-08-06T13:08:54.386467Z","shell.execute_reply":"2022-08-06T13:08:54.397045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best Params from Grid Search\nlrgs = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs.fit(train[predictors], train['failure'])\n\nscores = score(train[predictors], train['failure'], lrgs, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:46:23.370215Z","iopub.execute_input":"2022-08-06T13:46:23.371290Z","iopub.status.idle":"2022-08-06T13:46:33.239836Z","shell.execute_reply.started":"2022-08-06T13:46:23.371240Z","shell.execute_reply":"2022-08-06T13:46:33.237935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best Params from Grid Search\nlrgs = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs.fit(train[new_predictors], train['failure'])\n\nscores = score(train[new_predictors], train['failure'], lrgs, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:46:54.205607Z","iopub.execute_input":"2022-08-06T13:46:54.207034Z","iopub.status.idle":"2022-08-06T13:47:00.016441Z","shell.execute_reply.started":"2022-08-06T13:46:54.206972Z","shell.execute_reply":"2022-08-06T13:47:00.014840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LDA\nLDA = LinearDiscriminantAnalysis()\nLDA.fit(train[predictors], train['failure'])\n\nscores = score(train[predictors], train['failure'], LDA, cv=skf)\ndisplay(scores)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:47:18.428164Z","iopub.execute_input":"2022-08-06T13:47:18.428657Z","iopub.status.idle":"2022-08-06T13:47:19.880470Z","shell.execute_reply.started":"2022-08-06T13:47:18.428621Z","shell.execute_reply":"2022-08-06T13:47:19.879297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Public Leaderboard scores:\n- Logistic top fs vars = 0.58192\n- Logistic Regression GS = 0.58786 \n- LDA = 0.58133\n\nLDA drops a lot on the public LB, but tbh everything about this LB is weird. I still need to do some in-depth cross validation to make sure the score is stable and we don't drop fro the LB on private.","metadata":{}},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Remove id\n#test = test.drop('id', axis = 1)\n\nsub1 = sample_submission.copy()\nlrgs.fit(train[new_predictors], train['failure'])\nsub1.failure = lrgs.predict_proba(test[new_predictors])[:,1]\nsub1.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:49:10.349421Z","iopub.execute_input":"2022-08-06T13:49:10.350671Z","iopub.status.idle":"2022-08-06T13:49:11.583169Z","shell.execute_reply.started":"2022-08-06T13:49:10.350620Z","shell.execute_reply":"2022-08-06T13:49:11.581402Z"},"trusted":true},"execution_count":null,"outputs":[]}]}